diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 3aee141..ce5cda6 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -40,7 +40,6 @@ jobs: test "$parent_count" -eq 1 candidate_commit="$(git rev-parse "$GITHUB_SHA^")" evidence="docs/releases/v${package_version}-evidence.md" - test "$evidence" = "docs/releases/v0.1.0-evidence.md" test -f "$evidence" mapfile -t changed_paths < <(git diff-tree --no-commit-id --name-only -r "$GITHUB_SHA") test "${#changed_paths[@]}" -eq 1 @@ -55,9 +54,12 @@ jobs: uv run --frozen ty check src tests benchmarks scripts examples git diff --check - name: Build and inspect once + shell: bash run: | + package_version="${GITHUB_REF_NAME#v}" uv build - uv run --frozen python scripts/check_release_artifacts.py dist + uv run --frozen python scripts/check_release_artifacts.py dist \ + --version "$package_version" - name: Upload immutable release distributions uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: @@ -69,9 +71,39 @@ jobs: if-no-files-found: error retention-days: 7 - testpypi: + wheel-smoke: runs-on: ubuntu-latest needs: build + strategy: + fail-fast: false + matrix: + python-version: ["3.11", "3.12", "3.13"] + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + - name: Install Python + uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0 + with: + python-version: ${{ matrix.python-version }} + - name: Download checked distributions + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: release-dists + path: dist + - name: Install the built wheel and run the native smoke + shell: bash + run: | + package_version="${GITHUB_REF_NAME#v}" + wheel="$(find dist -name '*.whl' -print -quit)" + python scripts/check_release_artifacts.py dist \ + --version "$package_version" --verify-checksums + python -m venv "$RUNNER_TEMP/wheel-env" + "$RUNNER_TEMP/wheel-env/bin/pip" install "$wheel" + "$RUNNER_TEMP/wheel-env/bin/python" scripts/release_smoke.py \ + "$RUNNER_TEMP/wheel-smoke" --version "$package_version" + + testpypi: + runs-on: ubuntu-latest + needs: [build, wheel-smoke] environment: testpypi permissions: contents: read @@ -88,7 +120,11 @@ jobs: name: release-dists path: dist - name: Verify downloaded distributions - run: python scripts/check_release_artifacts.py dist --verify-checksums + shell: bash + run: | + package_version="${GITHUB_REF_NAME#v}" + python scripts/check_release_artifacts.py dist \ + --version "$package_version" --verify-checksums - name: Stage only publishable files run: | mkdir publish @@ -101,13 +137,14 @@ jobs: - name: Smoke exact TestPyPI version shell: bash run: | + package_version="${GITHUB_REF_NAME#v}" python -m venv "$RUNNER_TEMP/testpypi-env" for attempt in {1..12}; do if "$RUNNER_TEMP/testpypi-env/bin/pip" download \ --no-deps --only-binary=:all: \ --index-url https://test.pypi.org/simple/ \ --dest "$RUNNER_TEMP/testpypi-dist" \ - opendocs-sdk==0.1.0; then + opendocs-sdk=="$package_version"; then break fi test "$attempt" -lt 12 @@ -128,7 +165,7 @@ jobs: PY "$RUNNER_TEMP/testpypi-env/bin/pip" install "$testpypi_wheel" "$RUNNER_TEMP/testpypi-env/bin/python" scripts/release_smoke.py \ - "$RUNNER_TEMP/testpypi-smoke" --version 0.1.0 + "$RUNNER_TEMP/testpypi-smoke" --version "$package_version" pypi: runs-on: ubuntu-latest @@ -145,7 +182,11 @@ jobs: name: release-dists path: dist - name: Verify downloaded distributions - run: python scripts/check_release_artifacts.py dist --verify-checksums + shell: bash + run: | + package_version="${GITHUB_REF_NAME#v}" + python scripts/check_release_artifacts.py dist \ + --version "$package_version" --verify-checksums - name: Stage only publishable files run: | mkdir publish @@ -180,17 +221,22 @@ jobs: name: release-dists path: dist - name: Verify downloaded distributions - run: python scripts/check_release_artifacts.py dist --verify-checksums + shell: bash + run: | + package_version="${GITHUB_REF_NAME#v}" + python scripts/check_release_artifacts.py dist \ + --version "$package_version" --verify-checksums - name: Download exact public wheel and verify identity shell: bash run: | + package_version="${GITHUB_REF_NAME#v}" python -m venv "$RUNNER_TEMP/public-env" for attempt in {1..12}; do if "$RUNNER_TEMP/public-env/bin/pip" download \ --no-deps --only-binary=:all: \ --index-url https://pypi.org/simple \ --dest "$RUNNER_TEMP/public-dist" \ - opendocs-sdk==0.1.0; then + opendocs-sdk=="$package_version"; then break fi test "$attempt" -lt 12 @@ -211,7 +257,7 @@ jobs: PY "$RUNNER_TEMP/public-env/bin/pip" install "$public_wheel" "$RUNNER_TEMP/public-env/bin/python" scripts/release_smoke.py \ - "$RUNNER_TEMP/public-smoke" --version 0.1.0 + "$RUNNER_TEMP/public-smoke" --version "$package_version" github-release: runs-on: ubuntu-latest @@ -226,13 +272,21 @@ jobs: name: release-dists path: dist - name: Verify downloaded distributions - run: python scripts/check_release_artifacts.py dist --verify-checksums + shell: bash + run: | + package_version="${GITHUB_REF_NAME#v}" + python scripts/check_release_artifacts.py dist \ + --version "$package_version" --verify-checksums - name: Create GitHub Release for the verified public package env: GH_TOKEN: ${{ github.token }} + shell: bash run: | + package_version="${GITHUB_REF_NAME#v}" + notes="docs/releases/v${package_version}-notes.md" + test -f "$notes" gh release create "$GITHUB_REF_NAME" \ dist/*.whl dist/*.tar.gz dist/SHA256SUMS \ --verify-tag \ --title "OpenDocs $GITHUB_REF_NAME" \ - --notes-file docs/releases/v0.1.0-notes.md + --notes-file "$notes" diff --git a/CHANGELOG.md b/CHANGELOG.md index e2d50ed..97f17b0 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,21 @@ ## 未发布 +### 新增 + +- 新增标准 `.xlsx` 工作簿解析:按源顺序保留全部 worksheet/chartsheet(含 hidden、 + very hidden 与空 sheet)、常见保存值和货币/日期格式、Excel 表格、相离区域、合并跨度、 + 公式缓存缺失回退、标准文本对象以及原生图表事实。 +- XLSX 内嵌图片与图表支持可选视觉语义补充;视觉只解释趋势、标注、关系与含义,任一模型 + 不可用、超时或失败均保留原生结果并产生可定位 warning。 + +### 兼容性与限制 + +- XLSX 不使用 `max_pages` 限制工作表数量,而由私有结构预算在昂贵加载前约束资源;公共 + `ParseOptions` 与 Markdown 返回契约保持不变。 +- 仅支持 `.xlsx`,不支持 `.xls`、`.xlsm` 或 `.xlsb`;不重算公式,不访问外部 URL、链接 + 工作簿或数据连接,也不承诺字体、颜色、边框、尺寸或像素级 Excel 外观保真。 + ### 修复 - 限制单页 PDF 视觉候选数量,避免异常重叠对象触发高复杂度区域合并。 @@ -13,7 +28,7 @@ ## 0.1.0 - Alpha -OpenDocs 的首个公开 Alpha 将本地文档转换为 Markdown。 +OpenDocs 的首个公开 Alpha `opendocs-sdk==0.1.0` 将本地文档转换为 Markdown。 ### 支持范围 diff --git a/README.md b/README.md index a1f6fbc..0f7b19e 100644 --- a/README.md +++ b/README.md @@ -6,7 +6,7 @@ [![Downloads](https://img.shields.io/pypi/dm/opendocs-sdk.svg)](https://pypistats.org/packages/opendocs-sdk) OpenDocs is a Python SDK that converts local documents into clean Markdown — -TXT, Markdown, images, PDF (native / hybrid / vision), DOCX, and PPTX — through a unified +TXT, Markdown, images, PDF (native / hybrid / vision), DOCX, PPTX, and XLSX — through a unified sync/async API. > **Package name**: `opendocs-sdk`  |  **Import name**: `opendocs`  |  **Python**: 3.11+ @@ -69,6 +69,10 @@ downloaded before calling OpenDocs. | PNG / JPEG / WebP | ✅ | Static images only; sanitized before the configured vision model sees them | | DOCX | ✅ | Continuous authored body flow with structured text, lists, links, tables, explicit breaks, and inline images | | PPTX | ✅ | Slide and shape-tree order with text, tables, accessible charts, groups, and inline images | +| XLSX (`.xlsx`) | ✅ | All sheet-like entries in source order, saved values, tables/regions, merges, standard text objects, native chart facts, and optional visual interpretation | + +Only standard `.xlsx` workbooks are supported. Legacy `.xls`, macro-enabled `.xlsm`, binary +`.xlsb`, and other spreadsheet formats are not accepted. ## Vision parsing @@ -87,8 +91,9 @@ apt-get install poppler-utils ``` Standalone images require `VisionConfig`. PDFs and Office documents without vision configuration -preserve usable native content and emit deterministic warnings for visual regions; a document with -no usable native content raises `VisionRequiredError`. +preserve usable native content and emit deterministic warnings for visual regions; XLSX always +keeps its native sheet and chart facts when visual enrichment is unavailable. A document with no +usable native content raises `VisionRequiredError` where that format requires vision. ```python from opendocs import ParseOptions, VisionConfig, parse @@ -111,7 +116,7 @@ response — use distinct typed exceptions for precise error handling. cross-document concurrency themselves (e.g. with an `asyncio.Semaphore`); see the [independent consumer example](examples/basic_consumer/README.md). -### DOCX & PPTX details +### DOCX, PPTX & XLSX details DOCX extraction preserves authored body paragraphs, headings, lists, safe links, tables, merged cells, explicit page breaks, and inline raster-image positions. A DOCX remains one continuous @@ -121,6 +126,20 @@ PPTX extraction emits every slide boundary and traverses each slide's shape tree including recursive groups, text, tables, accessible chart data, and raster pictures. Exact duplicate embedded images are analyzed once per parse and replayed at every authored slot. +XLSX extraction emits every worksheet and chartsheet in workbook order, including visible, hidden, +very hidden, and empty sheets. It preserves non-empty regions, Excel tables, merged-cell spans, +standard comments/text boxes/links/header-footer text, and common saved display semantics such as +`$`, `€`, `£`, and `¥` currency, grouping, decimals, percentages, dates, and times. Saved formula +caches are preferred; when a cache is missing, the formula text is returned with a warning. OpenDocs +does not recalculate formulas or fetch linked workbooks, data connections, or URLs—the reference is +preserved as text only. + +Chart titles, labels, series, categories, and accessible values come from native workbook data. +When vision is configured, normalized chart fact cards and embedded images may add trend, label, +relationship, or meaning interpretation. This enrichment is fail-open and never replaces native +facts. XLSX output does not promise Excel pixel appearance, fonts, colors, borders, dimensions, or +other visual styling fidelity. + ### How OpenDocs compares | Feature | OpenDocs | marker | docling | unstructured | pypdf | @@ -128,6 +147,7 @@ duplicate embedded images are analyzed once per parse and replayed at every auth | PDF → Markdown | ✅ | ✅ | ✅ | ✅ | ❌ | | DOCX → Markdown | ✅ | ❌ | ✅ | ✅ | N/A | | PPTX → Markdown | ✅ | ❌ | ✅ | ✅ | N/A | +| XLSX → Markdown | ✅ | ❌ | ✅ | ✅ | N/A | | LLM vision integration | ✅ | ❌ | ❌ | ❌ | ❌ | | Sync + Async API | ✅ | ❌ | ❌ | ❌ | ❌ | | No external service required | ✅ | ✅ | ✅ | ⚠️ | ✅ | diff --git a/docs/plans/2026-08-07-001-feat-xlsx-parsing-plan.md b/docs/plans/2026-08-07-001-feat-xlsx-parsing-plan.md new file mode 100644 index 0000000..eb81947 --- /dev/null +++ b/docs/plans/2026-08-07-001-feat-xlsx-parsing-plan.md @@ -0,0 +1,651 @@ +--- +title: XLSX 解析 - Plan +type: feat +date: 2026-08-07 +topic: xlsx-parsing +artifact_contract: ce-unified-plan/v1 +artifact_readiness: implementation-ready +product_contract_source: ce-brainstorm +execution: code +--- + +# XLSX 解析 - Plan + +## Goal Capsule + +- **Objective:** 为 OpenDocs v0.2.0 交付 XLSX 到 Markdown 的文本语义保真解析,并让视觉增强失败时仍可获得确定性的原生结果。 +- **Authority order:** Product Contract 定义产品行为,Planning Contract 定义实现方式,Implementation Units 不得覆盖前两者。本计划仅在 XLSX 范围内取代 `docs/plans/2026-08-07-v0.2.0-release-plan.md` 的 A4 视觉禁用约束;v0.2.0 的其他工作仍由父计划管理。 +- **Execution profile:** 代码实施,测试先行;先锁定容器、wire 和输出契约,再接入第三方解析与视觉增强。 +- **Stop conditions:** 若实现需要新增公共返回类型、把工作表解释为页面、引入 Excel/LibreOffice 运行时,或无法在预检阶段约束峰值资源,则停止并回到计划评审。 +- **Tail ownership:** 实施需完成公开测试、静态检查、构建和独立 wheel smoke;真实 XLSX 只进入忽略的私有探索流程。 +- **Open blockers:** 无发布阻塞型规划问题。 + +--- + +## Product Contract + +### Summary + +OpenDocs 将从 path、bytes 和 binary stream 解析标准 XLSX 工作簿,并通过 `parse()` 与 `aparse()` 返回确定性的 Markdown。 +原生解析是工作表、文本、显示值、合并关系和图表数值的事实来源,视觉模型只补充图片与图表的视觉含义。 + +### Problem Frame + +OpenDocs 当前没有 XLSX 文档类型、检测、注册或解析能力,因此 Excel 工作簿仍是明确的不支持格式。 +XLSX 不只是二维单元格集合:一份工作簿还可能包含多个可见或隐藏工作表、合并与稀疏区域、公式缓存、数字格式、批注、文本框、超链接、页眉页脚、图片、图表、外部关系和声明范围异常。 +没有真实工作簿可以作为当前基线,因此本任务需要先用合成样本锁定公开行为,再以维护者提供的私有真实工作簿做发布前探索性验证。 + +### Risk Map + +| 风险 | 可能造成的问题 | 契约响应 | +| --- | --- | --- | +| 公式缓存缺失或过期 | 输出为空或与重新计算结果不同 | 由 R5 和公式不重算边界约束 | +| 货币、日期和自定义格式 | 原始数值可读但用户看到的文本错误 | 由 R5-R6 约束 | +| 隐藏、空白和多工作表 | 内容被遗漏或顺序改变 | 由 R3 约束 | +| 合并单元格和相离区域 | 表格关系丢失或被错误拼接 | 由 R4 约束 | +| 稀疏但声明范围巨大的工作表 | 无界时间或内存消耗 | 由 R11-R12 约束 | +| 浮动图片、图表和文本框 | 内容脱离原始工作表位置 | 由 R4、R8-R10 约束 | +| 外部链接和数据关系 | 发生未授权网络访问或结果不可复现 | 由 R7 和 Scope Boundaries 约束 | +| 恶意或异常 OOXML 容器 | ZIP bomb、路径逃逸或不受控解压 | 由 R2、R11-R12 约束 | +| 厂商扩展和嵌入对象 | 静默遗漏或模型猜测内容 | 由 R14 约束 | + +### Key Decisions + +- **Markdown 语义保真。** (session-settled: user-directed — chosen over pixel-level fidelity or dual output: the SDK keeps its Markdown result contract and only text-bearing semantics are required.) Governs R1, R4-R7. +- **保存的显示值优先。** (session-settled: user-directed — chosen over always emitting formulas or attaching every formula: readable workbook text is the primary result.) Governs R5. +- **全部工作表进入结果。** (session-settled: user-directed — chosen over skipping hidden sheets or adding an inclusion option: sheet content must not disappear because of visibility state.) Governs R3. +- **全部标准文本对象属于核心内容。** (session-settled: user-directed — chosen over cell-only or content-only extraction: text outside cells is still document content.) Governs R7. +- **图表原生数据优先。** (session-settled: user-directed — chosen over vision-only or native-only chart handling: numeric accuracy and visual meaning are both useful, but native values remain authoritative.) Governs R8-R9. +- **视觉增强失败不阻断原生结果。** (session-settled: user-directed — chosen over failing the document or adding a strict mode: deterministic native content remains useful without a model.) Governs R9-R10. +- **可靠文本核心作为发布硬门。** (session-settled: user-directed — chosen over broad text completeness or full-sheet visual review: the first XLSX release needs a bounded contract without silently losing supported content.) Governs R1-R15. +- **真实工作簿是探索性验证。** (session-settled: user-directed — chosen over a mandatory release gate or post-release-only validation: real evidence should inform the release without making every rare gap blocking.) Governs R15. + + +### How This Work Fits Together + +本计划只负责 v0.2.0 的 XLSX 解析产品契约;下面是当前理解的相邻工作关系,不构成已承诺路线图。 + +- **XLSX 资源边界** — Shares 本计划的安全与有界处理要求;具体阈值在本任务的技术规划中确定。 + - **跨格式 `max_pages` 语义** — Can proceed independently of XLSX 内容解析;不得把工作表伪装成页面。 +- **取消清理与对抗性 PDF 回归** — Can proceed independently of 本计划,仅共享 v0.2.0 发布门。 +- **v0.2.0 发布准备** — Depends on 本计划完成公开文档、构建产物和独立安装 smoke 的 XLSX 覆盖。 +- **Windows 探索性 smoke** — Can proceed independently of 本计划,且不是 XLSX 产品契约的一部分。 + +### Actors + +- A1. SDK 使用者向 OpenDocs 提交 XLSX,并消费 Markdown 与 warning。 +- A2. 视觉模型提供方在已配置且可用时补充图片和图表含义,不拥有原生数值的解释权。 +- A3. 维护者在发布前审阅私有真实工作簿的探索性结果,并判断发现是否违反核心契约。 + +### Requirements + +**格式识别与公共契约** + +- R1. OpenDocs 必须从 path、bytes 和 binary stream 接受合法 XLSX,并让 `parse()` 与 `aparse()` 返回相同契约的确定性 Markdown。 +- R2. OpenDocs 必须在无界处理前拒绝损坏、加密、伪装或超过资源边界的工作簿,并沿用现有类型化错误语义。 + +**工作簿结构与文本语义** + +- R3. 输出必须按源顺序识别全部工作表,以稳定标题标明工作表名称及 Visible、Hidden 或 Very Hidden 状态,空工作表不得使其他工作表失败。 +- R4. 每个工作表必须按行列顺序输出全部非空区域,保留合并跨度,并为区域与浮动对象提供稳定锚点;只有源工作簿提供表格语义时才能标记表头,不得仅凭首行位置猜测。 +- R5. 单元格必须输出保存的显示文本及常见货币、百分比、日期时间、千分位和小数位语义;公式缓存缺失时输出公式文本,不支持的自定义格式输出可读值并产生 warning。 +- R6. 解析器不得仅因样式存在而输出空单元格,也不得把字体、颜色、边框、尺寸或条件格式外观解释为必须还原的内容。 +- R7. 标准批注、备注、文本框、图表文字、超链接文字与 URL、页眉和页脚必须进入结果;外部 URL 与数据关系只保留引用,不发起访问。 + +**图表与视觉内容** + +- R8. 图表必须优先原生输出可确定获取的标题、分类、系列和数值,并保留其工作表位置。 +- R9. 视觉模型已配置时,内嵌图片必须进入视觉解析,需要解释趋势、标注或含义的图表必须获得视觉补充,但视觉结果不得覆盖或改写原生文本和数值。 +- R10. 视觉模型未配置、超时或失败时必须返回已完成的原生结果,并为每个未解析对象产生包含工作表和位置的 warning。 + +**安全、有界处理与确定性** + +- R11. 解析必须限制工作表数量、声明维度、访问与非空单元格数量、合并区域数量、对象与媒体数量、媒体大小及总输出字符数。 +- R12. 稀疏但声明范围巨大的工作表不得触发无界遍历,超限必须在昂贵解析或视觉调用前失败。 +- R13. 相同工作簿在等价输入形式和同步、异步 API 下必须保持工作表顺序、内容顺序、warning 分类和失败类型一致。 +- R14. 对无法可靠理解的厂商扩展、复杂绘图、SmartArt 或嵌入对象必须确定性跳过并产生可定位 warning,禁止静默丢失或猜测内容。 + +**验收证据** + +- R15. 自动化发布门必须使用合成 XLSX 覆盖全部核心契约;私有真实工作簿作为发布前探索性验证,仅当发现违反 R1-R14 的核心缺陷时阻断发布,罕见结构和视觉增强缺口可以记录后发布。 + +### Key Flows + +```mermaid +flowchart TB + A[XLSX input] --> B{Safe and within limits?} + B -->|no| C[Typed failure] + B -->|yes| D[Native sheets and text] + D --> E[Native chart data] + D --> F[Images and visual chart regions] + F --> G{Vision available?} + G -->|yes| H[Visual enrichment] + G -->|no or failed| I[Anchored warnings] + E --> J[Deterministic Markdown] + H --> J + I --> J +``` + +- F1. 正常解析 + - **Trigger:** A1 提交合法且未超限的 XLSX。 + - **Actors:** A1、A2。 + - **Steps:** 按 R1-R9 验证工作簿、提取全部工作表与原生内容、补充可用的视觉结果,并按源顺序合并。 + - **Outcome:** 返回确定性 Markdown 和必要 warning。 + - **Covers:** R1-R9、R11、R13。 +- F2. 视觉能力不可用 + - **Trigger:** 工作簿包含图片或需要视觉补充的图表,但模型未配置、超时或失败。 + - **Actors:** A1、A2。 + - **Steps:** 保留原生结果并按 R10 标记每个未完成的视觉对象。 + - **Outcome:** 解析成功返回,调用方可以定位缺失的视觉补充。 + - **Covers:** R9-R10、R13-R14。 +- F3. 非法或超限工作簿 + - **Trigger:** 输入损坏、加密、伪装或触发任一资源边界。 + - **Actors:** A1。 + - **Steps:** 在无界遍历和视觉调用前终止处理。 + - **Outcome:** 返回稳定的类型化错误,不返回误导性的部分成功结果。 + - **Covers:** R2、R11-R12。 +- F4. 私有真实工作簿探索 + - **Trigger:** A3 在发布前提供并审阅一份真实 XLSX 及其 Markdown。 + - **Actors:** A3。 + - **Steps:** 判断发现是否违反核心要求,并将非核心缺口记录为后续候选。 + - **Outcome:** 核心缺陷阻断发布,罕见结构或视觉增强缺口不自动阻断。 + - **Covers:** R15。 + +### Acceptance Examples + +- AE1. 全部工作表与状态 + - **Covers R3-R4.** + - **Given:** 工作簿依次包含可见、Hidden、Very Hidden 和空工作表。 + - **When:** 通过任一受支持输入形式解析。 + - **Then:** Markdown 按相同顺序输出四个工作表标题并标注状态,空表不影响其他内容。 +- AE2. 公式与显示格式 + - **Covers R5-R6.** + - **Given:** 单元格包含货币、百分比、日期、带缓存的公式、无缓存公式和仅样式空单元格。 + - **When:** 解析工作表。 + - **Then:** 输出保存的可读文本,缺少缓存的公式降级为公式文本,仅样式空单元格不制造内容。 +- AE3. 合并与相离区域 + - **Covers R4.** + - **Given:** 工作表包含横向和纵向合并单元格,以及被全空行列分隔的多个非空区域。 + - **When:** 解析工作表。 + - **Then:** 合并跨度和所有非空内容均保留,各区域以确定的行列顺序出现并可定位。 +- AE4. 工作表外围文本与链接 + - **Covers R7、R14.** + - **Given:** 工作表包含批注、备注、文本框、图表标签、超链接和页眉页脚,并引用外部 URL。 + - **When:** 解析工作簿。 + - **Then:** 可支持的文本和 URL 进入结果,OpenDocs 不访问外部地址,无法支持的对象产生可定位 warning。 +- AE5. 图表原生与视觉结果 + - **Covers R8-R10.** + - **Given:** 图表具有标题、分类、系列、数值和可由视觉模型识别的趋势标注。 + - **When:** 原生与视觉解析均成功。 + - **Then:** 原生数据按源值输出,视觉结果只补充趋势与含义,并保留图表锚点。 +- AE6. 视觉失败降级 + - **Covers R9-R10.** + - **Given:** 工作簿包含图片和图表,但视觉模型不可用或调用失败。 + - **When:** 解析工作簿。 + - **Then:** 原生文本和图表数据正常返回,每个未解析视觉对象都有工作表与位置 warning。 +- AE7. 对抗性和稀疏工作簿 + - **Covers R2、R11-R12.** + - **Given:** 工作簿包含异常容器关系、超量媒体或极大的声明范围但只有少量非空单元格。 + - **When:** 解析工作簿。 + - **Then:** 在无界遍历或视觉调用前以稳定类型化错误终止。 +- AE8. 等价 API 行为 + - **Covers R1、R13.** + - **Given:** 同一合成 XLSX 以 path、bytes 和 binary stream 输入。 + - **When:** 分别调用 `parse()` 与 `aparse()`。 + - **Then:** Markdown、warning 分类和失败类型满足等价契约。 +- AE9. 私有探索性验证 + - **Covers R15.** + - **Given:** 维护者提供真实 XLSX,并人工对照源文件审阅结果。 + - **When:** 发现差异。 + - **Then:** 违反核心要求的差异阻断发布,罕见结构或视觉增强缺口被记录但不自动阻断。 + +### Success Criteria + +- R1-R15 均有自动化合成样本或确定性检查覆盖,且现有支持格式没有输出顺序回归。 +- 每一次内容降级都能由返回结果或 warning 观察,不能以“解析成功”掩盖静默丢失。 +- 维护者可以只依赖本 Product Contract 判断一个真实工作簿发现属于核心缺陷还是非阻断增强缺口。 + +### Scope Boundaries + +- 不还原字体、字号、颜色、背景、边框、列宽、行高、条件格式外观或像素级版式。 +- 不访问外部 URL、链接工作簿、外部数据连接或远程资源。 +- 不支持旧版 `.xls`、宏工作簿 `.xlsm`、二进制 `.xlsb` 或其他表格格式。 +- 不重新计算公式,也不承诺缓存值反映工作簿最后保存之后的外部变化。 +- 不把整张工作表的视觉解析设为默认结果或发布硬门。 +- 不在本计划中改变结构化返回类型、依赖 extras、全局并发、CLI、Node.js SDK 或跨语言 schema。 +- 不把跨格式 `max_pages`、DOCX 结构上限、取消清理、PDF 防护或 Windows 支持纳入 XLSX 主动范围。 + +### Dependencies and Assumptions + +- 当前公共 API 的成功结果是 Markdown 字符串,XLSX 必须遵循相同契约。 +- 当前 Office 安全层和视觉管线可以提供复用基础,但 XLSX 仍需自己的容器识别、关系约束和资源边界。 +- 公式显示值来自工作簿保存的缓存,OpenDocs 不承担电子表格计算引擎职责。 +- 视觉模型是可选增强依赖,任何视觉结论都不能成为原生数值的替代来源。 +- 当前没有真实 XLSX 基线;维护者将在发布前提供私有工作簿做探索性审阅。 + +### Sources and Research + +- `docs/plans/2026-08-07-v0.2.0-release-plan.md`:v0.2.0 总体边界、XLSX 候选范围与相邻工作。 +- `docs/roadmap.md`:已发布能力和 v0.2.0 在路线图中的位置。 +- `src/opendocs/api.py`:当前 Markdown 返回契约与同步、异步共享路径。 +- `src/opendocs/parsers/office/package.py`:现有 Office 容器安全边界及 XLSX 需要扩展的基础。 +- `src/opendocs/markdown.py`:合并单元格可使用现有跨度表格语义表达。 +- `src/opendocs/parsers/office/parser.py` 与 `src/opendocs/parsers/office/pptx.py`:现有 Office 视觉合并和图表原生数据行为。 + +--- + +## Planning Contract + +Product Contract unchanged. + +### Key Technical Decisions + +- KTD1. **使用独立 XLSX 解析域和混合读取路径。** 新增私有 `XlsxParser`、XLSX 文档模型和严格 wire;不把工作表塞入基于页面的 `OfficeDocument`。容器与补充对象由受限 OOXML 流式读取,单元格、合并和 Excel 表格由 `openpyxl` 完整模式读取。该方案复用公共 parser/runtime/vision 契约,但不扩大 DOCX/PPTX 的模型。Governs R1-R4, R7-R14. (session-settled: user-approved — chosen over extending the page-oriented OfficeParser or using only one parser: XLSX needs sheet semantics plus low-level coverage that neither alternative provides.) +- KTD2. **先预检,后加载工作簿。** `openpyxl.load_workbook()` 之前必须完成 ZIP 成员、关系、XML 安全、工作表、维度、单元格、合并、字符串、对象和媒体预算检查。XLSX 扩展现有 Office 包验证白名单,但不绕过已有 2,048 成员、32 MiB 声明总量、4 MiB 单 XML、256 个媒体、16 MiB 单媒体、24 MiB 媒体总量和 100:1 压缩比限制。直接 OOXML 解析显式使用 `defusedxml` 并拒绝 DTD/实体;ZIP bomb 仍由包级预算处理。Governs R2, R11-R12. (session-settled: user-approved — chosen over loader-first validation: malformed dimensions and XML can consume resources before a high-level library returns control.) +- KTD3. **固定依赖为 `openpyxl>=3.1.5,<3.2` 与 `defusedxml>=0.7.1,<1`。** 使用 `read_only=False`、`data_only=False`、`rich_text=False`、`keep_links=False`。公式工作簿只加载一次;直接 OOXML sidecar 同时保留 `` 和保存的 `` 缓存,避免第二次完整加载。禁止依赖 `openpyxl` 私有 `_charts` 或 `_images` 作为核心事实来源。Governs R2, R5, R7-R9, R11. (session-settled: user-approved — chosen over read-only loading or dependency-free OOXML reimplementation: read-only omits required objects, while a full spreadsheet reader is beyond v0.2.0.) +- KTD4. **“保存的显示文本”采用有界格式子集。** 原生事实是保存的标量或公式缓存,解析器只对 `General`、布尔/错误值、整数、定点小数、千分位、百分比、常用直接货币符号、日期、时间、日期时间和 elapsed-time 执行确定性格式化,并尊重 1900/1904 日期系统。颜色、填充、条件段、会计占位、科学计数、分数和复杂本地化自定义格式不进入 v0.2.0 保真承诺;它们输出稳定原始值并产生 `xlsx_unsupported_number_format`。Governs R5-R6. +- KTD5. **公式缓存优先,缺失时回退公式。** 缓存节点存在时按 KTD4 输出;节点不存在时输出公式文本并产生 `xlsx_formula_cache_missing`。解析器不计算公式、不判断缓存是否过期,也不访问外部工作簿。Governs R5, R7. +- KTD6. **按语义坐标生成区域和对象槽位。** Excel 原生表格范围先占位,且只有它设置 `header_rows=1`。其余非空语义单元格与合并矩形按上下左右连通分量拆分,按左上角、右下角排序;无跨度使用 `TableBlock(header_rows=0)`,有跨度使用 `SpannedTableBlock`。每个 sheet、区域和浮动对象使用仅含工作表序号、A1 范围和对象序号的受控 Markdown 注释锚点,禁止把用户文本写入注释。Governs R3-R4, R13. +- KTD7. **所有 sheet-like 条目按工作簿关系顺序处理。** `workbook.xml` 与关系文件是 worksheet、chartsheet、名称、状态和目标 part 的权威来源;不依赖 `wb.worksheets` 推断全量顺序。空 sheet 仍输出标题。页眉页脚使用 odd/even/first、header/footer、left/center/right 的固定次序;普通对象按 `(row, column, kind_rank, source_ordinal)` 插入。Governs R3-R4, R7-R8, R13. +- KTD8. **文本对象走低层 OOXML 补充读取。** 经典 comments/notes、threaded comments 与 person 映射、DrawingML 文本框、图表标题/轴/数据标签、超链接显示文本与目标、页眉页脚均进入原生槽位。SmartArt、OLE、控件、VML 绘图文本和厂商扩展无法可靠读取时,按每个对象产生 `xlsx_unsupported_object`。URL 只允许安全转义后的 `http`、`https`、`mailto` 和工作簿内锚点形成链接;其他 scheme 与外部工作簿引用保留为纯文本并 warning,绝不访问。Governs R7, R14. +- KTD9. **图表视觉使用原生事实生成的语义预览。** ChartML 直接提取标题、轴/标签文本、系列名、分类、X/Y/数值、缓存和锚点;简单本地引用可从已提取单元格解析,外部或不支持引用保留文字并 warning。使用 Pillow 把这些权威事实渲染成规范化语义卡片,再让视觉模型补充趋势、关系和含义。该结果必须标记为视觉解释,不宣称还原 Excel 图表外观。Governs R8-R9. (session-settled: user-approved — chosen over Excel/LibreOffice pixel rendering or native-only chart output: external rendering adds a cross-platform runtime, while native-only output cannot provide the approved visual interpretation.) +- KTD10. **图片复用原始媒体,视觉按内容去重、按出现位置回放。** 图片使用包内原始 bytes、锚点和 alt text,并复用现有图片安全准备逻辑。同一 SHA-256 只调用一次模型,但结果或失败 warning 必须回放到每个出现位置。图表语义预览采用相同调度模型。Governs R9-R10, R13. +- KTD11. **XLSX 视觉错误采用 fail-open。** 模型未配置、认证/权限错误、无效请求、单对象超时、提供方失败和模型输出无效都返回原生结果,并按对象产生 `xlsx_vision_unavailable`、`xlsx_vision_timeout` 或 `xlsx_vision_failed`。只有调用方取消和整份文档超时沿用公共 API 的中断语义。Governs R9-R10, R13. (session-settled: user-approved — chosen over inheriting every Office fatal-vision branch: the Product Contract requires useful native output for all visual-provider failures.) +- KTD12. **资源限制保持私有,不改变 `ParseOptions`。** `max_pages` 对 XLSX 无效,且不得限制工作表。v0.2.0 使用下面的内部常量;资源特征测试只能在不突破包预算和 wire 预算的前提下收紧它们。Governs R1-R2, R11-R13. (session-settled: user-approved — chosen over adding public spreadsheet options or mapping sheets to pages: the release should preserve the public API and correct semantics.) + +### Resource Budget + +| 资源 | v0.2.0 上限 | 执行点 | 超限结果 | +| --- | ---: | --- | --- | +| Sheet-like 条目 | 128 | `workbook.xml` 预检 | `LimitExceededError` | +| 单表声明矩形 | 2,000,000 个坐标 | worksheet dimension 与实际坐标预检 | `LimitExceededError` | +| 全工作簿序列化 `` | 200,000 | worksheet XML 流式计数 | `LimitExceededError` | +| 非空语义单元格 | 50,000 | 类型与值解码时 | `LimitExceededError` | +| 最终物化网格坐标 | 200,000 | 区域、表格与合并布局前 | `LimitExceededError` | +| 合并范围 | 10,000 个且总 footprint 50,000 | mergeCells 预检 | `LimitExceededError` | +| Shared strings | 100,000 项且解码文本 1,000,000 字符 | sharedStrings 流式读取 | `LimitExceededError` | +| Excel 表格 | 1,024 个且总 footprint 200,000 | table part 预检 | `LimitExceededError` | +| 超链接与批注 | 合计 20,000 个 | 关系和 comments 预检 | `LimitExceededError` | +| 浮动绘图、图表、图片、文本框 | 合计 256 个 | drawing/chart 关系预检 | `LimitExceededError` | +| 图表缓存点 | 200,000 个 | ChartML 预检 | `LimitExceededError` | +| 原生解码文本 | 1,000,000 字符 | block 构建前累计 | `LimitExceededError` | +| Native worker inline / frame | 沿用 8 MiB / 12 MiB | 严格 wire 编解码 | 现有协议错误映射 | +| 最终 Markdown | 沿用 `max_output_chars`,默认 400,000 | 公共 renderer | 现有截断 warning | + +不支持格式类 warning 每个 code 保留前 20 条并附确定性汇总;视觉对象与无法支持的浮动对象因总量已受 256 限制,必须逐个保留可定位 warning。 + +### Warning and Failure Taxonomy + +| 场景 | 结果 | +| --- | --- | +| ZIP 损坏、必需 part/关系缺失、DTD/实体、非法 XML | `CorruptDocumentError` | +| 加密 ZIP 成员或 Office 加密容器 | 沿用检测层的类型化拒绝;进入 XLSX 包层后映射为 `CorruptDocumentError` | +| 任一包、结构、对象、字符或 wire 预算超限 | `LimitExceededError` | +| 公式无缓存 | 公式文本 + `xlsx_formula_cache_missing` | +| 数字格式超出 KTD4 | 稳定原始值 + `xlsx_unsupported_number_format` | +| 外部引用、危险 URL scheme 或无法解析的本地引用 | 纯文本引用 + `xlsx_external_reference` | +| 不支持的标准/厂商对象 | 跳过该对象 + `xlsx_unsupported_object` | +| 视觉未配置、超时或失败 | 原生结果 + KTD11 对应 warning | + +### High-Level Technical Design + +#### Component and Data Flow + +```mermaid +flowchart TB + S[Resolved XLSX source] --> D[Detection and package preflight] + D --> O[Bounded OOXML index] + D --> W[Full-mode openpyxl load] + O --> X[XLSX extractor] + W --> X + X --> N[Strict XLSX native document wire] + N --> M[Deterministic slot merge] + X --> V[Image and chart visual slots] + V --> P[Shared image preparation and vision dispatch] + P --> M + M --> B[Existing core blocks] + B --> R[Existing Markdown renderer] +``` + +OOXML index 负责信任边界、sheet 关系、公式缓存和不支持对象发现。`openpyxl` 只在包已证明有界后负责受支持的工作簿值对象。XLSX merge layer 只输出现有 core blocks 和受控 Markdown anchors。 + +#### Parse Sequence + +```mermaid +sequenceDiagram + participant API as parse/aparse + participant PKG as XLSX preflight + participant RT as Native worker + participant EXT as XLSX extractor + participant VIS as Vision dispatcher + participant MD as Markdown renderer + API->>PKG: detect and validate bounded OOXML + PKG->>RT: validated path and options + RT->>EXT: extract native sheets, values, objects, chart facts + EXT-->>API: strict native document plus visual slots + API->>VIS: deduplicated images and semantic chart previews + VIS-->>API: result or per-object failure + API->>MD: merged existing blocks and warnings + MD-->>API: deterministic Markdown +``` + +预检发生在第三方工作簿加载之前。视觉调用发生在全部原生事实已经成功提取之后,因此 KTD11 的 fail-open 不会产生部分原生文档。 + +#### Decision and Failure Flow + +```mermaid +flowchart TB + A[Input] --> B{XLSX identity matches?} + B -->|No| E[Existing typed detection error] + B -->|Yes| C{Package and structure within budgets?} + C -->|No| F[CorruptDocumentError or LimitExceededError] + C -->|Yes| D[Build complete native document] + D --> G{Visual slots exist?} + G -->|No| J[Render native Markdown] + G -->|Yes| H{Vision configured and succeeds?} + H -->|Yes| I[Append grounded visual interpretation] + H -->|No| K[Append anchored warning] + I --> J + K --> J +``` + +### Output Structure + +```text +src/opendocs/parsers/xlsx/ +├── __init__.py +├── extract.py +├── merge.py +├── models.py +├── parser.py +├── preflight.py +└── values.py +``` + +共享图片准备若需要抽取,只新增一个中立的私有 helper,例如 `src/opendocs/parsers/embedded_vision.py`。不得借 XLSX 引入 DOCX/PPTX 模型重构。 + +### System-Wide Impact + +| 表面 | 影响 | 约束 | +| --- | --- | --- | +| 公共 API | 新增可识别格式,不改签名与返回类型 | `parse()`/`aparse()`、输入形态和 warning 对等 | +| Detection / registry | 新增 `DocumentType.XLSX`、ZIP 身份和默认 parser | 扩展名不可信;容器身份优先 | +| Native runtime | 新增 XLSX 严格 wire 与 worker 路径 | 8 MiB inline、12 MiB frame 和取消清理不放宽 | +| Markdown | 复用 Heading、Table、SpannedTable、Paragraph、InlineLink、MarkdownBlock | 不新增公开 XLSX Markdown 方言 | +| Vision | 新增图片与语义图表预览请求 | 原生事实优先;按 digest 去重、按位置回放 | +| Packaging | 增加两个直接运行依赖 | wheel metadata、锁文件和隔离安装必须一致 | +| Release evidence | 增加公开合成门与私有真实工作簿协议 | 私有文件、输出和检查表不得提交 | + +### Sequencing + +```mermaid +flowchart LR + U1[U1 Public wiring and dependencies] --> U2[U2 Models and preflight] + U2 --> U3[U3 Values and regions] + U3 --> U4[U4 Text objects] + U3 --> U5[U5 Charts and media] + U4 --> U6[U6 Parser, vision and merge] + U5 --> U6 + U6 --> U7[U7 API, lifecycle and adversarial proof] + U7 --> U8[U8 Release integration and private validation] +``` + +每个 feature-bearing unit 先增加失败测试,再修改生产代码。U1-U2 固定输入、错误和 wire 边界;U3-U5 构建原生事实;U6 才接视觉;U7-U8 完成跨层与发布证据。 + +### Alternative Approaches Considered + +- **纯 `openpyxl`。** 拒绝,因为 read-only 模式缺失图表、图片和批注,完整模式也不覆盖 threaded comments、DrawingML 文本框和所有图表关系;私有对象字段不能承担稳定核心契约。 +- **纯手写 OOXML。** 拒绝,因为 v0.2.0 不应重写 Excel 单元格类型、样式索引、日期系统、表格和合并兼容层;低层解析只负责安全预检和高层库缺失的对象。 +- **调用 Excel 或 LibreOffice 渲染原图。** 拒绝,因为它扩大跨平台运行时、沙箱、进程清理和发布体积;KTD9 已用有界语义预览满足趋势补充。 +- **整张工作表视觉解析。** 拒绝,因为成本、确定性和大表资源风险与原生数据优先相冲突。 +- **只做单元格文本,不做外围对象。** 拒绝,因为它违反 R7,并会把正文之外的标准文本静默丢失。 + +### Risks and Mitigations + +| 风险 | 缓解 | 验证证据 | +| --- | --- | --- | +| `openpyxl` 完整模式内存放大 | loader 前执行 KTD2 与 Resource Budget;worker wire 保持既有上限 | 对抗性结构测试与 U7 资源特征测试 | +| Excel 显示格式复杂且本地化 | KTD4 固定硬支持子集;其余可读降级并 warning | 精确字符串断言覆盖货币、百分比、日期和负值 | +| 公式缓存不存在或过期 | 保留缓存节点存在性;缺失回退公式;不宣称新鲜度 | 直接 patch OOXML 的缓存有/无/空值测试 | +| 图表 API 与厂商扩展不稳定 | ChartML 作为事实来源;私有 `openpyxl` 字段仅可作非契约辅助 | 原生图表 fixture 与未知扩展 warning | +| 图表语义预览被误读为原图 | 输出明确标为视觉解释,模型 prompt 禁止改写原生值 | mock vision 断言 prompt、锚点与合并优先级 | +| URL 或外部关系触发访问 | `keep_links=False`,直接关系只保留文本,危险 scheme 不形成链接 | 网络调用禁用测试和链接转义测试 | +| warning 风暴掩盖输出 | 非对象 warning 有界聚合;对象总数先硬限 | 超量格式与 256 对象边界测试 | +| 无真实工作簿导致合成盲区 | 公开合成门覆盖契约;发布前维护者提供私有 XLSX 做探索 | U8 私有审阅清单,不生成可提交 baseline | + +### Dependencies and Prerequisites + +- `openpyxl` 与 `defusedxml` 进入直接 runtime dependencies 和 `uv.lock`;不新增 extra。 +- Pillow、native worker、Markdown renderer、warning 发射与 vision dispatcher 沿用现有依赖和生命周期。 +- 真实 XLSX 在实施完成后由维护者提供;缺少该文件不阻止公开合成开发,但阻止把真实兼容性写成已验证事实。 + +### Sources and Research + +- [`openpyxl` PyPI](https://pypi.org/project/openpyxl/):当前稳定版 3.1.5、Python 版本要求和官方 XML 安全提示。 +- [`openpyxl` tutorial](https://openpyxl.readthedocs.io/en/stable/tutorial.html):`data_only`、`read_only`、`rich_text`、`keep_links` 行为,以及 shapes 和 read-only 特性缺口。 +- [`openpyxl` optimized modes](https://openpyxl.readthedocs.io/en/stable/optimized.html):read-only 对声明维度的依赖和显式 close 责任。 +- [`openpyxl` comments](https://openpyxl.readthedocs.io/en/3.0/comments.html):经典批注仅保留文本/作者、格式与容器信息丢失,read-only 不支持批注。 +- [`defusedxml` PyPI](https://pypi.org/project/defusedxml/):XML entity/DTD/DoS 防护范围;它不替代 ZIP bomb 预算。 +- `src/opendocs/parsers/office/package.py`:现有 OOXML 包预算、关系和路径验证模式。 +- `src/opendocs/parsers/office/models.py`、`src/opendocs/parsers/office/parser.py`、`src/opendocs/parsers/office/merge.py`:严格 worker wire、视觉去重和按出现位置合并模式。 +- `src/opendocs/parsers/office/pptx.py`:原生图表标题、分类、系列和值的相邻实现模式。 +- `src/opendocs/markdown.py`、`tests/test_markdown.py`:合并跨度与受控 Markdown 注释的渲染模式。 +- AGENTS.md 指定的本地 `43x-agent` checkout 中,Office parser 仅作为行为参考;OpenDocs 必须独立实现,且不得复制私有文件、模型载荷或应用依赖。 + +--- + +## Implementation Units + +### U1. Public XLSX Identity, Dependencies, and Registration + +- **Goal:** 让合法 XLSX 进入现有公共解析主路径,并在第三方加载前复用 OOXML 包安全边界。 +- **Requirements:** R1-R2, R11-R13;KTD2-KTD3、KTD12;Covers F3 / AE7-AE8. +- **Dependencies:** 无。 +- **Files:** `pyproject.toml`, `uv.lock`, `src/opendocs/_models.py`, `src/opendocs/detection.py`, `src/opendocs/parsers/registry.py`, `src/opendocs/parsers/office/package.py`, `tests/test_models.py`, `tests/test_detection.py`, `tests/test_registry.py`, `tests/test_office_package.py`, `tests/xlsx_fixtures.py`. +- **Approach:** + 1. 先以合成 ZIP fixture 锁定 `.xlsx`、无名 bytes/stream、后缀不匹配、缺失 workbook part、重复/加密成员和关系逃逸行为。 + 2. 新增 `DocumentType.XLSX`,以 `[Content_Types].xml`、根 relationship 和 `xl/workbook.xml` 共同确认身份。 + 3. 扩展 Office package validator 的 XLSX required parts 与允许关系根,不改变 DOCX/PPTX 预算和错误行为。 + 4. 加入 KTD3 依赖与默认 registry;parser 在 U6 前可用明确的未完成测试替身,不合入不能解析的默认注册状态。 +- **Execution note:** 从失败的 detection、package 和 registry 契约测试开始;该 unit 必须以完整可调用的最小 parser seam 收尾,避免中间提交破坏默认 registry。 +- **Patterns to follow:** `src/opendocs/detection.py` 的容器身份匹配,`src/opendocs/parsers/office/package.py` 的路径/关系验证,`src/opendocs/parsers/registry.py` 的默认注册。 +- **Test scenarios:** + 1. `.xlsx` path、无扩展 bytes、命名与无名 binary stream 包含合法 workbook 关系时都识别为 XLSX。 + 2. `.xlsx` 实际为 DOCX/PPTX、ZIP 缺少 `xl/workbook.xml`、content type 或根关系不匹配时返回现有类型化 mismatch/corrupt 语义。 + 3. XLSX 包含路径穿越、重复成员、加密 flag、悬空关系、外部 required root relationship、超限成员或压缩比时,在 parser 调用前失败。 + 4. DOCX/PPTX 的 required part 和包预算回归测试保持不变。 + 5. 构建默认 registry 时 XLSX 与现有格式各注册一次,缺少 runtime 的错误契约不变。 +- **Verification:** 所有合法输入到达 XLSX parser seam;所有伪装、损坏与包超限输入在第三方加载前以稳定类型失败;现有 Office 格式无注册或验证回归。 + +### U2. XLSX Wire Models and Structural Preflight + +- **Goal:** 建立不使用 page 语义的严格 XLSX native document,并在 `openpyxl` 前执行完整结构预算。 +- **Requirements:** R2-R4, R7-R8, R11-R14;KTD1-KTD2、KTD7、KTD12;Covers F3 / AE1、AE7. +- **Dependencies:** U1. +- **Files:** `src/opendocs/parsers/xlsx/__init__.py`, `src/opendocs/parsers/xlsx/models.py`, `src/opendocs/parsers/xlsx/preflight.py`, `tests/test_xlsx_models.py`, `tests/test_xlsx_preflight.py`, `tests/xlsx_fixtures.py`, `tests/test_runtime.py`. +- **Approach:** + 1. 定义不可变 `XlsxDocument`、`XlsxSheet`、原生槽位、图片/图表视觉槽位、数字 sheet index 和 A1 anchor;wire 只接受已知字段、tuple、受限 basename、SHA-256 和现有 Block。 + 2. 用 `defusedxml` 流式建立 OOXML index,解析 workbook/sheet/chartsheet 顺序、状态、part 目标、日期系统、shared strings、worksheet 元数据、drawing/chart/comment/table 关系。 + 3. 在返回 index 前执行 Resource Budget;维度、实际 cell 坐标、merge/table footprint、对象和字符计数都必须交叉验证。 + 4. 把恶意 XML、非法命名空间/关系和超限分别映射到 Planning Contract 的失败分类。 +- **Execution note:** 在引入完整工作簿加载前用 ZIP/XML patch fixture 穷举失败面;不要创建或提交二进制样本。 +- **Patterns to follow:** `src/opendocs/parsers/office/models.py` 的 dataclass/wire 校验,`src/opendocs/_native_protocol.py` 的 frame 预算,`src/opendocs/parsers/office/package.py` 的安全路径。 +- **Test scenarios:** + 1. worksheet、chartsheet、hidden、veryHidden 和空 sheet 按 relationship 顺序进入严格 wire,重复 source index、非法 anchor 或未知字段被拒绝。 + 2. 128 个 sheet 成功,129 个失败;声明矩形、序列化 cell、非空 cell、materialized grid、merge、shared string、table、object、chart cache 和文本预算逐项验证边界值与超一值。 + 3. 声明 `A1:XFD1048576` 但只有一个 cell 的稀疏表在 `openpyxl` 前失败,不触发矩形遍历。 + 4. DTD、内部/外部实体、畸形 XML、Zip Slip、悬空 drawing/chart/comment 关系返回 `CorruptDocumentError`。 + 5. 编码后的最大合法 wire 小于 12 MiB;超出 inline/container/frame 预算时沿用现有 runtime 失败映射。 +- **Verification:** 任一第三方加载都只接收已通过预检的包;XLSX wire 不含 page 概念,且所有可放大结构有明确边界测试。 + +### U3. Saved Values, Sheets, Tables, Regions, and Merges + +- **Goal:** 输出全部 sheet 的稳定原生网格内容,并准确处理常见显示格式、公式缓存、表格和合并跨度。 +- **Requirements:** R3-R6, R11-R13;KTD3-KTD7、KTD12;Covers F1 / AE1-AE3. +- **Dependencies:** U2. +- **Files:** `src/opendocs/parsers/xlsx/values.py`, `src/opendocs/parsers/xlsx/extract.py`, `tests/test_xlsx_values.py`, `tests/test_xlsx_extract.py`, `tests/xlsx_fixtures.py`. +- **Approach:** + 1. 以 full-mode `openpyxl` 读取已预检 workbook,并由 OOXML formula/cache sidecar 覆盖公式显示选择。 + 2. 在 `values.py` 集中实现 KTD4-KTD5;格式 warning 按 code 和 sheet/coordinate 稳定聚合。 + 3. 先占用 Excel table 精确范围,再对剩余语义单元格和 merge footprint 做连通分量;物化 component bounding box 前再次扣减 grid 预算。 + 4. 空 sheet 只产生标题与 sheet anchor;style-only 空 cell 计入安全访问但不成为语义坐标。 + 5. 将无 merge 区域转换为 `TableBlock(header_rows=0)`,有 merge 区域转换为 `SpannedTableBlock`,并按 KTD6 排序。 +- **Execution note:** 公式 fixture 必须直接 patch worksheet XML 的 ``/``,不能假装 `openpyxl` 会计算公式。 +- **Patterns to follow:** `src/opendocs/parsers/office/pptx.py` 和 `src/opendocs/parsers/office/docx.py` 的 table block 构造,`src/opendocs/markdown.py` 的 span 渲染。 +- **Test scenarios:** + 1. Covers AE1. visible、hidden、veryHidden、空 worksheet 和 chartsheet 按源顺序输出固定标题、状态和安全 anchor。 + 2. Covers AE2. `$1,234.50`、`¥1,234`、`12.50%`、千分位、负数、1900/1904 日期、时间和 elapsed-time 产生精确稳定文本。 + 3. 带缓存公式输出格式化缓存;缓存缺失输出公式并 warning;缓存节点存在但值为空与真正缺失可区分;外部公式不发起访问。 + 4. scientific、fraction、accounting、条件/颜色段和复杂本地化格式输出稳定原始值并按坐标 warning;同 code 超过 20 条时保留确定性汇总。 + 5. Covers AE3. 横向/纵向合并、Excel table、无表头普通区域、多个相离区域和内部空 cell 保留全部非空内容与跨度,且顺序可重复。 + 6. 仅字体/颜色/边框/条件格式的空 cell 不产生 Markdown 内容;全空工作簿仍成功输出所有 sheet 标题。 + 7. component bounding box、table 或 merge footprint 在预检后因组合超预算时,在物化前返回 `LimitExceededError`。 +- **Verification:** 合成工作簿的 sheet、值、公式、格式、区域和 merge Markdown 与黄金字符串一致;重复解析及输入形态不改变顺序。 + +### U4. Comments, Text Boxes, Links, and Headers or Footers + +- **Goal:** 把单元格之外的标准文本对象纳入原生结果,并对外部关系和不支持对象提供可定位降级。 +- **Requirements:** R4, R7, R11, R13-R14;KTD7-KTD8、KTD12;Covers F1 / AE4. +- **Dependencies:** U3. +- **Files:** `src/opendocs/parsers/xlsx/extract.py`, `src/opendocs/parsers/xlsx/models.py`, `tests/test_xlsx_extract.py`, `tests/test_xlsx_models.py`, `tests/xlsx_fixtures.py`. +- **Approach:** + 1. 从 OOXML index 读取 classic comments/notes、threaded comments/person、`xdr:sp/a:txBody` 文本框和 shape alt text,不依赖 `openpyxl` 的有限 comment/drawing 映射。 + 2. 单元格 hyperlink 用 `InlineLink` 表达安全目标;内部 anchor、外部 URL、外部 workbook 和危险 scheme 按 KTD8 分类。 + 3. 解析页眉页脚文本并去除字体/颜色控制码;页码、日期、时间、文件名和 sheet 名字段保留为命名占位符,图片字段 warning。 + 4. 生成对象槽位并按 KTD7 与 cell regions 合并;SmartArt、OLE、controls、VML 文本和 vendor extensions 逐对象 warning。 +- **Patterns to follow:** `src/opendocs/parsers/office/docx.py` 的 `InlineLink` 与关系处理,`src/opendocs/parsers/office/models.py` 的 source index 和 warning 模型。 +- **Test scenarios:** + 1. Covers AE4. 经典批注文本/作者、threaded comment/person、DrawingML 文本框、alt text 和页眉页脚都在正确 sheet/anchor 下出现。 + 2. safe HTTP/HTTPS/mailto 与工作簿内链接转义为 Markdown 链接;`javascript:`、`file:`、相对外部工作簿和远程数据关系只输出文本并 warning。 + 3. 测试期间禁止网络调用,包含外部 URL、externalLinks、data connections 的 workbook 仍零访问完成解析。 + 4. odd/even/first × header/footer × left/center/right 按固定顺序输出;格式控制码不泄漏,动态字段使用稳定占位符。 + 5. SmartArt、OLE、ActiveX/control、VML-only textbox 和未知 extension 各产生包含 sheet index、A1 anchor 或 object ordinal 的 warning。 + 6. 20,000 个 comment/hyperlink 在边界成功,超一值在对象解码前失败;原生文本预算仍限制总内容。 +- **Verification:** 所有承诺的非 cell 文本均可定位;任何外部引用无 I/O;不支持对象不会静默消失。 + +### U5. Native Chart Facts, Embedded Images, and Semantic Previews + +- **Goal:** 以原生 ChartML 和媒体为事实来源,构建可去重的图片与图表视觉任务。 +- **Requirements:** R4, R8-R10, R11, R13-R14;KTD8-KTD10、KTD12;Covers F1-F2 / AE5-AE6. +- **Dependencies:** U2-U4. +- **Files:** `src/opendocs/parsers/xlsx/extract.py`, `src/opendocs/parsers/xlsx/models.py`, `src/opendocs/parsers/xlsx/parser.py`, `src/opendocs/parsers/embedded_vision.py`, `tests/test_xlsx_extract.py`, `tests/test_xlsx_parser.py`, `tests/test_office_parser.py`, `tests/xlsx_fixtures.py`. +- **Approach:** + 1. 从 drawing anchors 和 ChartML 提取 chart title、axis/data-label text、series names、categories、X/Y/values、cache、local formulas、alt text 和位置。 + 2. 解析已在 workbook 内的简单引用;外部、动态或不支持公式保留引用文本并产生 `xlsx_external_reference`,不得调用计算引擎或网络。 + 3. 把原生图表事实输出为标题与无表头数据 block,再用 Pillow 生成无样式保真承诺的语义卡片;视觉 prompt 只允许补充趋势、关系、标注和含义。 + 4. 提取原始图片 part、anchor 和 alt text,复用现有安全解码、尺寸、像素和 tile 预算。若抽取共享 helper,保持 Office 现有结果与 warning 不变。 + 5. 以内容 digest 构建 visual slot;相同图片或语义卡片只准备一次。 +- **Execution note:** 先锁定 native-only 图表与图片槽位,再加入语义卡片;视觉测试只使用 fake client,不调用真实模型。 +- **Patterns to follow:** `src/opendocs/parsers/office/pptx.py` 的原生图表 block,`src/opendocs/parsers/office/parser.py` 的 digest 去重,`src/opendocs/vision/images.py` 的图片防护。 +- **Test scenarios:** + 1. Covers AE5. line、bar、pie/doughnut 和 scatter 合成图表的标题、系列、分类、X/Y/值及 anchor 由 ChartML 稳定输出,视觉文本不能覆盖原生表。 + 2. chart cache、简单本地 range、缺少 cache、动态 named formula 和 external reference 按 KTD9 解析或 warning,无网络和公式计算。 + 3. 语义预览只包含原生事实与明确“视觉解释”标签;fake vision 返回趋势时结果位于对应 chart anchor 后。 + 4. 同一媒体在多个 sheet/anchor 出现时只产生一次 vision request,但结果或失败状态可回放到全部位置。 + 5. 图片格式伪装、解压 bomb、超尺寸、超像素、超 tile 和损坏媒体沿用图片安全错误或 warning,不扩大既有预算。 + 6. 256 个浮动对象在边界成功,257 个失败;200,000 个 chart cache point 在边界成功,超一值在 preview 前失败。 + 7. Office parser 回归测试证明共享 helper 抽取不改变 DOCX/PPTX request、结果顺序和 warning。 +- **Verification:** 原生图表事实可独立构成完整 native-only 输出;视觉任务均有受控输入、digest 和出现位置;不需要 Excel/LibreOffice。 + +### U6. Parser Orchestration, Fail-Open Vision, and Deterministic Merge + +- **Goal:** 把 native worker、视觉调度、warning 和 Markdown 合并成完整 `XlsxParser`,同时保持取消、超时和清理语义。 +- **Requirements:** R1, R3-R4, R8-R10, R13-R14;KTD1、KTD6-KTD12;Covers F1-F2 / AE5-AE6、AE8. +- **Dependencies:** U3-U5. +- **Files:** `src/opendocs/parsers/xlsx/parser.py`, `src/opendocs/parsers/xlsx/merge.py`, `src/opendocs/parsers/xlsx/models.py`, `src/opendocs/parsers/registry.py`, `tests/test_xlsx_parser.py`, `tests/test_xlsx_merge.py`, `tests/test_registry.py`. +- **Approach:** + 1. 在 native worker 中执行 KTD2-KTD8 的完整原生提取并返回 strict wire;主进程只处理受限视觉 artifact 和 block merge。 + 2. 以 digest 去重请求并按 source ordinal 排序;把每个 request 的 outcome 回放到 occurrence 集合。 + 3. 单独实现 KTD11 的 XLSX fail-open 分类;不修改 Office parser 对现有格式的异常策略。 + 4. `merge.py` 先输出 sheet heading/anchor,再按 KTD6-KTD7 排序原生和视觉槽位;原生 block 永远在同一对象的视觉解释之前。 + 5. 只有 native extraction 不完整、调用方取消或文档 deadline 才中断成功路径;模型失败不得制造 `NoUsableContentError`。 +- **Execution note:** 用 fake runtime/vision 先锁定异常矩阵,再接真实 native worker;同步超时和异步取消必须使用现有工作区生命周期。 +- **Patterns to follow:** `src/opendocs/parsers/office/parser.py` 的 worker/vision 编排,`src/opendocs/parsers/office/merge.py` 的 occurrence 回放和 warning,`src/opendocs/api.py` 的 deadline/cancellation。 +- **Test scenarios:** + 1. 无视觉对象、视觉未配置、全部视觉成功、部分成功和全部失败都返回相同原生 Markdown 主干。 + 2. Covers AE6. 未配置、认证、权限、无效请求、provider error、坏模型输出和单 request timeout 按对象产生 KTD11 warning,且不抛出文档失败。 + 3. 调用方 async cancellation 及时传播;sync document timeout 返回现有超时错误;两者不被 fail-open 捕获。 + 4. 重复 digest 只调用一次,结果与 warning 按每个 sheet/anchor occurrence 回放,顺序不受 future 完成顺序影响。 + 5. 原生 chart/table 与视觉解释同 anchor 时原生先出现;视觉不得删除、替换或重排原生 block。 + 6. 全空 workbook 仍有 sheet heading,只有无法支持对象的 workbook 返回 headings 与 warnings,而不是空成功或模型猜测。 + 7. XLSX registry 使用完整 parser;DOCX/PPTX/Image/PDF 视觉异常行为不受 KTD11 影响。 +- **Verification:** `XlsxParser` 在所有视觉状态下满足原生优先和确定顺序;用户取消、文档超时和 parser 失败边界清晰且无资源泄漏。 + +### U7. Public API Parity, Resource Characterization, and Adversarial Regression + +- **Goal:** 从公共 API 证明输入、同步/异步、warning、失败、资源和生命周期契约,并保护所有现有格式。 +- **Requirements:** R1-R2, R10-R15;KTD2、KTD11-KTD12;Covers F1-F3 / AE7-AE8. +- **Dependencies:** U6. +- **Files:** `tests/test_api_xlsx.py`, `tests/test_xlsx_preflight.py`, `tests/test_xlsx_parser.py`, `tests/test_api.py`, `tests/test_api_m2.py`, `tests/test_runtime.py`, `tests/test_markdown.py`, `tests/xlsx_fixtures.py`. +- **Approach:** + 1. 用同一合成工作簿参数化 path、bytes、named/unnamed binary stream 和 `parse()`/`aparse()`,对比 Markdown、warning code/顺序和错误类型。 + 2. 覆盖 repeated parse、`max_output_chars`、`max_pages=1`、vision 配置和输入后缀组合;明确 `max_pages` 不减少 sheet。 + 3. 对 Resource Budget 每项执行边界/超一测试,并验证失败发生在 `openpyxl` load、grid materialization 或 vision 前。 + 4. 用受控 worker fixture 验证正常、异常、async cancellation、sync timeout 和 hard termination 后 workspace/artifact 清理。 + 5. 运行现有全格式回归,确认新增 DocumentType、共享 helper 和 renderer 使用不改变旧输出。 +- **Execution note:** 先写公共集成失败测试,再补资源特征;不要把环境相关峰值 RSS 写成未经证实的发布承诺。 +- **Patterns to follow:** `tests/test_api_m2.py` 的 path/bytes/stream 对等,`tests/test_runtime.py` 的 worker 边界,现有 parser cancellation/cleanup 测试。 +- **Test scenarios:** + 1. Covers AE8. 六组输入/API 组合的 Markdown 字符串完全一致,warning code 与失败类型对等;warning 的 Python 发射位置保持公共契约。 + 2. 同一 workbook 连续解析至少三次,sheet、region、object 和 warning 顺序完全一致。 + 3. `max_pages=1` 的多 sheet workbook 仍输出全部 sheet;`max_output_chars` 继续按现有 renderer 截断并 warning。 + 4. ZIP bomb、巨维度、巨 merge、巨 shared strings、对象风暴、chart cache 风暴和 XML entity 在昂贵阶段前失败。 + 5. sync timeout、async cancellation、native crash、vision timeout 后,OpenDocs 临时目录与测试前基线一致;清理失败保留主异常并 warning。 + 6. TXT、Markdown、PDF、Image、DOCX 和 PPTX 的选定黄金输出、warning 和 registry 行为不变。 +- **Verification:** 公共 XLSX 契约在所有等价入口可复现;资源限制执行点有可观察证据;全套现有测试无回归。 + +### U8. Packaging, Documentation, Release Smoke, and Private Workbook Protocol + +- **Goal:** 让 wheel、发布文档和验收流程准确包含 XLSX,并建立不提交真实内容的私有探索门。 +- **Requirements:** R1-R2, R9-R15;KTD3、KTD9、KTD12;Covers F4 / AE9. +- **Dependencies:** U7. +- **Files:** `README.md`, `CHANGELOG.md`, `docs/roadmap.md`, `docs/plans/2026-08-07-v0.2.0-release-plan.md`, `scripts/release_smoke.py`, `scripts/check_release_artifacts.py`, `tests/test_package.py`, `tests/test_release_scripts.py`, `tests/test_acceptance_corpus.py`. +- **Approach:** + 1. 更新 README、roadmap、CHANGELOG 和父 v0.2.0 计划,明确支持 `.xlsx`、不支持 `.xls/.xlsm/.xlsb`、不下载 URL、公式/格式/视觉边界,以及本计划取代父计划 A4 的范围。 + 2. 更新 release artifact dependency set、wheel metadata 断言和隔离安装 smoke,验证 `openpyxl`、`defusedxml` 与 parser 都来自构建产物环境。 + 3. release smoke 动态生成最小 XLSX,覆盖多 sheet、货币/日期、merge、公式 fallback 和 native-only 输出;不把视觉 provider 作为安装 smoke 前提。 + 4. 维护者提供真实 XLSX 后,在 `tests/corpus/` 与 `tests/corpus.local.toml` 建立忽略的探索条目,人工对照内容、sheet、格式、对象、图表和 warning。 + 5. 只有明确违反 R1-R14 的核心缺陷阻断 v0.2.0;罕见厂商扩展或视觉增强不足记录为 follow-up。禁止提交真实工作簿、hash、模型输出或完成的检查表。 +- **Execution note:** 该 unit 以构建后隔离安装为首要证明;私有探索结果必须区分“已运行”和“待提供样本”。 +- **Patterns to follow:** `scripts/release_smoke.py`、`scripts/check_release_artifacts.py`、`tests/test_package.py` 的 v0.1.0 发布验证;AGENTS.md 的私有 corpus 边界。 +- **Test scenarios:** + 1. wheel/sdist metadata 的直接依赖与 `pyproject.toml`、`uv.lock`、release checker 期望完全一致。 + 2. 隔离环境只安装构建 wheel 后可以导入 XLSX parser,并解析动态生成的 native-only workbook。 + 3. release smoke 的多 sheet、货币、日期、merge 和公式 fallback 产生预期 Markdown,且不需要网络、LibreOffice、Excel 或视觉凭据。 + 4. README/roadmap/release plan 不再声称 XLSX 禁用视觉,也不把语义图表预览描述为源图像素保真。 + 5. Covers AE9. 私有工作簿未提供时门明确为 `not_run`;提供后人工结论只记录核心缺陷或非阻断 follow-up,不自动生成可提交 baseline。 +- **Verification:** 构建产物可独立解析 XLSX;依赖、文档和父计划无冲突;私有验证不泄漏内容且证据等级准确。 + +--- + +## Verification Contract + +| Gate | Command | Proves | Applies after | +| --- | --- | --- | --- | +| Targeted XLSX suite | `uv run --frozen pytest tests/test_xlsx_models.py tests/test_xlsx_preflight.py tests/test_xlsx_values.py tests/test_xlsx_extract.py tests/test_xlsx_merge.py tests/test_xlsx_parser.py tests/test_api_xlsx.py -q` | XLSX core behavior、limits、vision fallback 和 API parity | U2-U7 | +| Detection and Office regression | `uv run --frozen pytest tests/test_detection.py tests/test_registry.py tests/test_office_package.py tests/test_office_parser.py tests/test_markdown.py -q` | Shared detection/package/vision/renderer 无回归 | U1、U5-U7 | +| Package and release checks | `uv run --frozen pytest tests/test_package.py tests/test_release_scripts.py -q` | Dependency metadata、isolated import 和 release smoke | U8 | +| Public suite | `uv run --frozen pytest -q` | 全格式公开回归 | U7-U8 | +| Lint | `uv run --frozen ruff check .` | Imports、style 和 common defects | 每个 unit 收尾 | +| Format | `uv run --frozen ruff format --check .` | 100 字符等格式契约 | 每个 unit 收尾 | +| Type check | `uv run --frozen ty check src tests` | Strict models、wire 和 parser type consistency | U2-U8 | +| Build | `uv build` | Wheel 与 sdist 可生成 | U8 | +| Fresh lock install | `uv sync --all-groups --frozen` | 锁文件完整且依赖可安装 | U1、U8 | +| Private corpus | `uv run --frozen pytest tests/test_acceptance_corpus.py -q --corpus-dir=@local` | 维护者提供真实 XLSX 后的私有探索与既有 corpus 回归 | U8,可选且证据需单列 | + +Python 3.11、3.12 和 3.13 的 CI/隔离 smoke 必须覆盖依赖安装与最小 XLSX 解析。不得用 focused suite 代替完整公开 suite,也不得把未运行的私有或真实模型验证写成通过。 + +--- + +## Definition of Done + +- R1-R15 均由至少一个 U-ID 和一个自动化场景覆盖;AE1-AE8 是公开合成硬门,AE9 按真实样本是否提供报告 `not_run` 或审阅结论。 +- `DocumentType.XLSX`、检测、包安全、严格 wire、解析、registry 和 Markdown 主路径完整连通,path/bytes/stream 与 `parse()`/`aparse()` 对等。 +- 全部 worksheet/chartsheet、状态、空 sheet、区域、Excel tables、merge、支持格式、公式 fallback、标准文本对象、超链接和页眉页脚满足 Product Contract。 +- 原生图表事实和图片均可定位;视觉成功只追加解释,任一视觉失败返回原生结果和逐对象 warning。 +- 所有 Resource Budget 在昂贵加载、物化或视觉调用前执行,`max_pages` 不改变 sheet 数量,`max_output_chars` 沿用公共语义。 +- `openpyxl` 和 `defusedxml` 的依赖范围、锁文件、wheel metadata、release checker 与隔离安装一致。 +- Verification Contract 中除条件式 private corpus 外的 gate 全部通过;private gate 的未运行/通过/缺陷证据单独报告。 +- README、CHANGELOG、roadmap 与父 v0.2.0 计划准确描述 XLSX 支持和限制,不再保留相互冲突的视觉范围。 +- 未提交任何真实 XLSX、私有 hash、模型 payload、生成 Markdown 或完成检查表。 +- 最终 diff 不包含试验性 parser、废弃 wire、临时预览文件、调试输出或被放弃方案的死代码。 diff --git a/docs/plans/2026-08-07-v0.2.0-release-plan.md b/docs/plans/2026-08-07-v0.2.0-release-plan.md new file mode 100644 index 0000000..6e5e4b8 --- /dev/null +++ b/docs/plans/2026-08-07-v0.2.0-release-plan.md @@ -0,0 +1,278 @@ +# OpenDocs v0.2.0 版本规划 + +状态:详细设计已批准并完成实施,待发布切版与正式发布。 + +XLSX 的产品契约、技术决策、资源预算与实现单元由 +[XLSX 解析详细计划](2026-08-07-001-feat-xlsx-parsing-plan.md) 承载;该计划已获批准, +并在 XLSX 视觉范围内取代本文恢复阶段的 A4 约束。 + +## 一、恢复结论 + +v0.2.0 不承接完整的 M4。该版本采用一个克制的切片: + +1. 交付一个高需求的新格式:Excel/XLSX。 +2. 在 `0.x` 的 minor 版本窗口内处理资源限制语义和必要的行为变更。 +3. 补齐少量已识别的稳健性边界。 +4. 不同时引入 Node.js SDK、跨语言 schema、全局并发或 CLI。 + +历史上曾讨论把结构化 `ParseResult`、依赖 extras 和 HTML 一并纳入 v0.2.0, +后续讨论用更窄的方案取代了它们;这些内容不属于本版本承诺。 + +历史上也曾把 30/30 PDF/图片质量基准和独立 holdout 作为最高优先级证据项。 +维护者随后明确决定延期该工作,并删除私有下载数据、标注和解析输出;提交 +`b12ed51` 只记录了仓库内陈旧文档引用的清理。因此,30/30 标注和基准不属于 +v0.2.0 的范围,benchmark 基础设施继续保留供未来版本复用。 + +## 二、版本定位 + +### 目标 + +- 在不扩大到完整 M4 的前提下增加 XLSX 支持。 +- 保持 `parse()` / `aparse()` 对等,所有成功结果仍为 Markdown。 +- 为 XLSX 和现有格式建立明确、可测试的资源边界。 +- 解决或验证历史讨论中识别的 `max_pages`、取消清理和对抗性 PDF 风险。 +- 为后续 M4 和跨平台工作保留清晰边界。 + +### 发布级别 + +v0.2.0 继续按 Alpha 发布,不以本版本宣称 Beta 或生产就绪。 + +原因:原先用于讨论 Alpha → Beta 的 30/30 PDF/图片质量证据已明确延期。 +Beta 的目标与证据门槛由未来版本另行设计和批准,不在本计划中预先承诺。 + +### 非目标 + +- Node.js SDK。 +- 公共跨语言中间 schema。 +- 结构化 `ParseResult` / `parse_result()` / `aparse_result()`。 +- 依赖 extras 重构和按安装依赖动态注册 parser。 +- HTML、email、旧版 `.xls`、宏工作簿 `.xlsm`、二进制工作簿 `.xlsb`。 +- CLI、服务端 API、HTTP/OSS/S3 下载。 +- 全局跨文档并发控制。 +- 重新开展或恢复 30/30 PDF/图片人工标注。 + +上述项目可作为后续版本候选,但不得为了 v0.2.0 的实现方便而隐式引入公共承诺。 + +## 三、当前基线 + +v0.1.0 已支持 TXT、Markdown、PDF、PNG/JPEG/WebP、DOCX 和 PPTX,并提供 +同步与异步 API、类型化错误、资源边界、原生 worker、共享视觉管线和 +Ubuntu/macOS 发布验证。 + +与本计划相关的当前状态: + +| 项目 | 当前状态 | v0.2.0 动作 | +| --- | --- | --- | +| XLSX | 类型、检测、安全预检、registry、原生/视觉 parser 与 Markdown 主路径已实现 | 待发布切版与真实样本探索 | +| `max_pages` | PDF 和 PPTX 执行;DOCX/XLSX 不伪装成页面 | 决议已实现并由独立内部结构限制保护 | +| 取消和临时文件清理 | source/workspace/worker 及 XLSX 正常、失败、取消、超时、硬终止路径已有回归 | 已核验;未做无证据重写 | +| PDF 视觉候选边界 | 已有页对象、reading-order 候选和 `build_visual_regions` 上限 | 作为回归门槛验证,不重复实现 | +| Windows | 有路径名防御;每 PR CI 仅 Ubuntu,发布 smoke 仍含 macOS | 不加入每个 PR 的阻断矩阵;保留手动/周期性 smoke 候选 | + +## 四、工作流 A:XLSX 支持 + +### A1. 格式识别与注册 + +- 增加 `DocumentType.XLSX`。 +- 通过 ZIP 容器中的 `xl/workbook.xml` 识别 XLSX,不能只依赖扩展名。 +- `.xlsx` 扩展名与实际容器类型不一致时沿用现有类型化错误行为。 +- 默认 registry 注册 XLSX parser。 +- 保持 path、bytes、binary stream 三种输入的一致行为。 + +### A2. 包安全与依赖 + +- 复用 Office ZIP 包验证:成员数量、单个 XML 大小、总解压大小、压缩比、加密成员、重复成员、路径穿越和关系目标检查。 +- 将 XLSX 主部件和关系约束加入 Office package 层,而不是绕过验证直接交给第三方库。 +- 原生解析依赖已固定为 `openpyxl>=3.1.5,<3.2`,低层 XML 依赖固定为 + `defusedxml>=0.7.1,<1`,并已进入锁文件与发布元数据。 +- v0.2.0 不同时做 dependency extras,因此 XLSX 依赖按当前单包安装模型处理。 + +### A3. 原生内容提取 + +最小交付能力: + +- 按工作簿中的工作表顺序输出。 +- 每个工作表生成稳定、可定位的 Markdown 标题。 +- 提取非空单元格区域,保持行列顺序。 +- 表格输出复用现有 block 和 Markdown renderer,不建立 XLSX 专用 Markdown 方言。 +- 处理合并单元格,并复用现有 `SpannedTableBlock` 能力;具体降级规则必须有固定测试。 +- 空工作表不得导致整份文档失败,仍按源顺序输出工作表标题。 +- 单个工作表存在多个相离数据区域时,输出顺序必须确定。 + +以下恢复阶段的语义审批门已由 XLSX 详细计划批准并实现: + +1. 公式采用保存缓存优先、缓存缺失时输出公式文本并 warning,不重新计算。 +2. visible、hidden、very hidden、空 worksheet 与 chartsheet 全部按源顺序输出。 +3. 常见货币、百分比、日期时间、千分位、小数位、布尔和错误值走确定性子集; + 不支持的复杂自定义格式保留稳定原始值并 warning。 +4. Excel table 优先,其余非空语义单元格按连通区域拆分;样式空单元格不制造内容。 +5. 合并关系通过现有 `SpannedTableBlock` 保留,区域与对象使用受控 A1 锚点。 + +完整依据和边界见 XLSX 详细计划的 KTD4-KTD8。 + +### A4. 图片和图表 + +- XLSX 图表先从 ChartML 提取标题、标签、系列、分类和可访问数值;原生事实始终是 + 权威来源,缺少缓存或无法解析的引用保留文字并 warning。 +- 内嵌图片使用原始媒体;图表视觉输入使用原生事实生成的规范化语义预览,不调用 + Excel 或 LibreOffice,也不宣称还原源图的像素外观。 +- 配置视觉模型时,可为图片和图表补充趋势、标注、关系与含义;模型结果不得覆盖 + 原生文本或数值。 +- 视觉未配置、超时、认证/权限错误、提供方失败或输出无效均 fail-open:返回原生结果, + 并为每个出现位置产生可定位 warning。只有调用方取消与整份文档超时保留中断语义。 +- 相同媒体或语义预览按 SHA-256 去重调用,再按工作表和锚点回放结果或 warning。 + +### A5. XLSX 资源边界 + +必须在打开和遍历工作簿时执行可测试的硬限制: + +- 工作表数量。 +- 单表最大行数和列数。 +- 总访问单元格数和非空单元格数。 +- 合并区域数量。 +- 内嵌媒体数量和总大小。 +- 输出字符数继续受 `max_output_chars` 约束。 +- 稀疏但声明巨大 used range 的工作簿不得导致无界遍历。 + +超限统一抛出 `LimitExceededError`,损坏或不合法容器统一映射为 `CorruptDocumentError`,缺失运行依赖映射为 `RuntimeDependencyError`。 + +上述限制已按 XLSX 详细计划的 Resource Budget 在第三方加载、网格物化、wire 编码或 +视觉调用前实现;具体数值与执行点以该详细计划为准。 + +## 五、工作流 B:资源限制语义收尾 + +### B1. `max_pages` 决策门 + +当前 `ParseOptions.max_pages` 对 PDF 页和 PPTX 幻灯片生效,但 DOCX 没有页面概念, +也没有使用该选项。强行把 DOCX 顶层元素称为“页”会制造错误 API 语义。 + +0.1.0 无下游用户,因此 0.2.0 可以自由选择正确语义,无需保留旧行为。 + +详细设计已选择并实现第一条方向: + +- `max_pages` 只约束真正分页或类分页格式;DOCX 保持连续 authored flow,XLSX 使用 + 独立内部结构上限,sheet 数量不受 `max_pages` 影响。 +- 未引入名称中立的新公共资源选项,`ParseOptions` 的字段与默认值保持不变。 + +恢复阶段提出的原则已落实:不把 DOCX body element 或 XLSX worksheet 解释成 page, +并通过格式私有的硬限制保护其真实结构单位。 + +### B2. 设计记录 + +- `parse()` 和 `aparse()` 的签名及返回 `str` 的契约保持不变。 +- 最终决议不引入 DOCX 的伪页面计数;DOCX 沿用既有包、媒体、XML、wire 与输出限制, + XLSX 使用详细计划记录的 sheet/cell/merge/object/text/wire 预算和执行点。 +- path、bytes、binary stream 以及 `parse()`、`aparse()` 的资源限制结果必须一致。 +- 不允许只在一个 parser 中静默忽略公开选项。 + +## 六、工作流 C:稳健性核验 + +### C1. 取消和工作区清理 + +当前代码已覆盖 source-owned 临时文件、`ParseWorkspace`、外部取消、同步超时和清理失败保留主异常等场景。本版本不做无证据重写,只执行以下核验: + +- native worker 在工作区写入额外文件后被取消或硬终止,工作区仍能清理。 +- 清理失败时保留原始取消/超时异常,并通过异常 note 记录清理错误。 +- 同步 `parse()` 返回前完成其承诺的清理;异步取消保持及时传播。 +- 详细实现计划必须固定取消、超时与硬终止场景;回归测试在有界清理等待后断言 + 受控 source、workspace、worker 和视觉 artifact 均已清理。 + +只有发现可复现缺口时才修改实现,并先提交失败回归测试。 + +本次 XLSX 已增加正常/失败、外部取消、文档超时、native worker 硬终止、视觉对象超时 +和清理失败保留主异常的回归;既有 source/workspace 清理实现满足契约,因此未做广泛重写。 + +### C2. 对抗性 PDF + +当前实现已经具备: + +- 单类和整页 PDF 对象数量上限。 +- reading-order 比较与视觉候选数量上限。 +- `build_visual_regions()` 在 O(n²) 合并前的候选数量上限。 +- native wire 大小预算。 + +因此该项从“待实现”调整为“必须保留的发布回归门槛”。需要确认超大 +image/line/rect/curve 列表、重叠文字和视觉候选均在昂贵合并或无界内存增长前 +以 `LimitExceededError` 终止。 + +### C3. Windows 探索性验证 + +近期每 PR CI 已为节省 GitHub Actions 额度主动缩减到 Ubuntu 3.11/3.13; +发布 smoke 仍覆盖 Ubuntu/macOS。因此 v0.2.0 不重新扩张每个 PR 的平台矩阵。 + +可选动作: + +- 增加手动触发或低频 Windows smoke。 +- 验证路径保留名、临时文件、ZIP、纯原生 TXT/Markdown/Office 行为。 +- Poppler/vision 完整 Windows 支持仍不作为 v0.2.0 发布阻断条件。 + +## 七、实施顺序 + +以下顺序已经完成;当前仅余发布切版与条件式私有探索。 + +1. **决策记录**:详细设计固定 XLSX 语义、`max_pages` 与 DOCX 结构上限,并在 + 维护者批准后进入实现。 +2. **检测和安全容器**:增加类型、ZIP 识别、主部件/关系验证和类型化错误。 +3. **原生 XLSX 提取**:工作表、区域、表格、合并单元格、基础值规范化。 +4. **非文本内容**:提取图片媒体与原生图表事实,可选视觉语义补充按对象 fail-open。 +5. **资源边界**:sheet/row/column/cell/merge/media 限制和对抗性 fixtures。 +6. **现有稳健性门槛**:取消清理与 PDF 候选回归验证。 +7. **发布准备**:文档、CHANGELOG、构建和独立安装 smoke。 + +不要并行修改共享 Office package、registry、renderer 和 public options。实现期保持一个 +writer,先完成底层契约再接 parser。 + +## 八、验收标准 + +### 功能 + +- XLSX 对等支持 path、bytes、binary stream。 +- `parse()` 与 `aparse()` 对等并返回确定性 Markdown。 +- 多工作表、空工作表、合并单元格、稀疏区域、公式/缓存值、图片、原生图表事实与 + 可选 fail-open 视觉补充均有明确测试。 +- DOCX、PPTX 和已有格式无输出顺序回归。 + +### 安全与稳健性 + +- ZIP bomb、路径穿越、重复成员、加密成员和超大 XML/媒体被类型化拒绝。 +- XLSX 的 sheet/cell/merge/media 限制在昂贵处理前执行。 +- PDF 视觉候选和页对象上限回归测试通过。 +- 取消、超时和 worker 硬终止的固定压力回归通过,临时目录残留数在有界等待后 + 回到测试前基线。 + +### 兼容与发布 + +- 最终 `max_pages` 跨格式决议、DOCX 计数单位、hard limit 和执行点均已记录并测试。 +- README、roadmap、CHANGELOG、支持格式列表和安装依赖同步更新。 +- 运行并通过: + +```bash +uv sync --all-groups --frozen +uv run --frozen pytest -q +uv run --frozen ruff check . +uv run --frozen ruff format --check . +uv run --frozen ty check src tests +uv build +``` + +- wheel 和 sdist 独立安装 smoke 覆盖新增 XLSX 依赖与至少一个合成 XLSX fixture。 +- 私有真实工作簿验收仍遵守本仓库的私有语料与人工基线规则,不提交真实内容、 + 解析输出或标注。 + +### 私有 XLSX 探索状态与协议 + +XLSX 私有探索门状态: `not_run`。维护者尚未提供真实工作簿,因此当前不得把该门报告为 +通过、失败或普通公开测试 skip,也不得生成占位 baseline。 + +维护者未来提供样本后,只能把工作簿放入忽略的 `tests/corpus/`,并通过忽略的 +`tests/corpus.local.toml` 选择本地目录。首次结果必须由维护者对照源工作簿人工审阅,再按 +核心契约缺陷或非阻断 follow-up 分级;不得提交真实工作簿、hash、模型输出或完成检查表。 + +## 九、已解决事项与保留候选 + +1. 公式、隐藏工作表、日期/数字格式和合并单元格语义已由详细计划批准并实现。 +2. `max_pages` 不限制 XLSX sheet;XLSX 私有 hard limit 已由详细计划固定并测试。 +3. Windows 手动/周期性 smoke 仍是非阻断候选;每 PR CI 保持 Ubuntu 3.11/3.13 精简矩阵。 + +发布专属 Ubuntu wheel smoke 覆盖 Python 3.11、3.12 和 3.13;真实 XLSX 私有探索按样本 +是否提供单独报告,不改变公开合成发布门。 diff --git a/docs/roadmap.md b/docs/roadmap.md index 02a4cf3..50a2517 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -117,16 +117,21 @@ smoke 以及 GitHub tag/Release 均针对同一不可变产物集通过。 ## v0.2.0 - XLSX 与资源边界加固 -状态:计划已确认;无向下兼容负担(0.1.0 无下游用户);尚未开始实现。 +状态:已实现,待完成发布切版与正式发布;继续按 Alpha 定位。 详细发布计划: [v0.2.0 发布计划](plans/2026-08-07-v0.2.0-release-plan.md)。 +XLSX 的已批准产品与技术契约: +[XLSX 解析详细计划](plans/2026-08-07-001-feat-xlsx-parsing-plan.md)。 + 概述: -- 新增 Excel/XLSX 作为一个需求驱动的新格式,不承接完整 M4 范围。 -- 在实现前固定 XLSX 提取、合并单元格、非文本降级和资源限制行为。 -- 解决公开 `max_pages` 语义,不伪造 DOCX 物理页。 +- 已新增标准 `.xlsx` 作为一个需求驱动的新格式,不承接完整 M4 范围;旧版 `.xls`、 + `.xlsm` 与 `.xlsb` 仍不支持。 +- 已固定并实现全部 sheet、保存值、常见格式、合并单元格、标准文本对象、原生图表事实、 + 可选 fail-open 视觉语义补充和预加载资源限制。 +- `max_pages` 继续只约束真正分页或类分页格式,不把 DOCX body 或 XLSX sheet 伪装成页面。 - 保留已完成的取消清理和对抗性 PDF 防护作为发布回归门槛。 - Windows 保持探索性非阻断。 @@ -144,7 +149,7 @@ smoke 以及 GitHub tag/Release 均针对同一不可变产物集通过。 概述: -- 基于实际需求增加更多格式,如 Excel、HTML 或 email。 +- 基于实际需求增加更多格式,如 HTML、email 或其他表格格式。 - 决定是否稳定一个跨语言的中间 schema。 - 发布 Node.js SDK,匹配输入语义、配置名称、错误码和 Markdown 固定件。 diff --git a/pyproject.toml b/pyproject.toml index 423cacf..032650a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -43,7 +43,9 @@ classifiers = [ "Topic :: Scientific/Engineering :: Artificial Intelligence", ] dependencies = [ + "defusedxml>=0.7.1,<1", "litellm>=1.93,<2", + "openpyxl>=3.1.5,<3.2", "pdfplumber>=0.11.10,<0.12", "pillow>=12.3,<13", "python-docx>=1.1.2,<2", diff --git a/scripts/check_release_artifacts.py b/scripts/check_release_artifacts.py index c5e0218..5cac124 100644 --- a/scripts/check_release_artifacts.py +++ b/scripts/check_release_artifacts.py @@ -3,6 +3,7 @@ import argparse import hashlib import tarfile +import tomllib from dataclasses import dataclass from email.message import Message from email.parser import BytesParser @@ -11,7 +12,9 @@ from zipfile import ZipFile EXPECTED_DEPENDENCIES = { + "defusedxml<1,>=0.7.1", "litellm<2,>=1.93", + "openpyxl<3.2,>=3.1.5", "pdfplumber<0.12,>=0.11.10", "pillow<13,>=12.3", "python-docx<2,>=1.1.2", @@ -140,6 +143,9 @@ def _inspect_wheel(path: Path, expected_version: str) -> Message: raise ArtifactError("wheel contains an unexpected top-level package") required = { "opendocs/__init__.py", + "opendocs/parsers/xlsx/__init__.py", + "opendocs/parsers/xlsx/parser.py", + "opendocs/parsers/xlsx/preflight.py", "opendocs/py.typed", f"{dist_info}/WHEEL", f"{dist_info}/RECORD", @@ -168,6 +174,9 @@ def _inspect_sdist(path: Path, expected_version: str) -> Message: f"{expected_root}/README.md", f"{expected_root}/pyproject.toml", f"{expected_root}/src/opendocs/__init__.py", + f"{expected_root}/src/opendocs/parsers/xlsx/__init__.py", + f"{expected_root}/src/opendocs/parsers/xlsx/parser.py", + f"{expected_root}/src/opendocs/parsers/xlsx/preflight.py", f"{expected_root}/src/opendocs/py.typed", f"{expected_root}/PKG-INFO", } @@ -260,7 +269,7 @@ def _sha256(path: Path) -> str: def _parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser(description="Validate OpenDocs release distributions.") parser.add_argument("directory", type=Path) - parser.add_argument("--version", default="0.1.0") + parser.add_argument("--version") parser.add_argument( "--verify-checksums", action="store_true", @@ -271,9 +280,10 @@ def _parser() -> argparse.ArgumentParser: def main(argv: list[str] | None = None) -> int: args = _parser().parse_args(argv) + expected_version = args.version or _project_version() result = inspect_release_artifacts( args.directory, - expected_version=args.version, + expected_version=expected_version, verify_existing_checksums=args.verify_checksums, ) print(f"validated {result.wheel.name} and {result.sdist.name}") @@ -281,5 +291,12 @@ def main(argv: list[str] | None = None) -> int: return 0 +def _project_version() -> str: + pyproject = Path(__file__).resolve().parents[1] / "pyproject.toml" + with pyproject.open("rb") as source: + payload = tomllib.load(source) + return str(payload["project"]["version"]) + + if __name__ == "__main__": raise SystemExit(main()) diff --git a/scripts/release_smoke.py b/scripts/release_smoke.py index 58fd6ae..ed96c39 100644 --- a/scripts/release_smoke.py +++ b/scripts/release_smoke.py @@ -3,14 +3,18 @@ import argparse import asyncio import io +import warnings +from datetime import date from importlib.metadata import version from pathlib import Path +from zipfile import ZIP_DEFLATED, ZipFile from docx import Document +from openpyxl import Workbook from pptx import Presentation from pptx.util import Inches -from opendocs import aparse, parse +from opendocs import OpenDocsWarning, aparse, parse def run_smoke(directory: Path, *, expected_version: str) -> dict[str, bool]: @@ -37,8 +41,16 @@ def run_smoke(directory: Path, *, expected_version: str) -> dict[str, bool]: textbox = slide.shapes.add_textbox(Inches(1), Inches(1), Inches(6), Inches(1)) textbox.text = "OpenDocs PPTX smoke" presentation.save(str(pptx_path)) + xlsx_path = directory / "smoke.xlsx" + _write_xlsx_smoke(xlsx_path) pdf_markdown = " ".join(parse(pdf_path).split()) + with warnings.catch_warnings(record=True) as caught: + warnings.simplefilter("always", OpenDocsWarning) + xlsx_markdown = parse(xlsx_path) + xlsx_warning_codes = [ + warning.message.code for warning in caught if isinstance(warning.message, OpenDocsWarning) + ] results = { "text": "OpenDocs text smoke" in parse(text_path), "markdown": "# OpenDocs Markdown smoke" in parse(markdown_path), @@ -46,6 +58,20 @@ def run_smoke(directory: Path, *, expected_version: str) -> dict[str, bool]: "docx": "OpenDocs DOCX smoke" in parse(docx_path), "pptx": "OpenDocs PPTX smoke" in parse(pptx_path), "async": "OpenDocs text smoke" in asyncio.run(aparse(text_path)), + "xlsx": all( + anchor in xlsx_markdown + for anchor in ( + "# Summary (Visible)", + "# Hidden (Hidden)", + "# Very Hidden (Very Hidden)", + "# Empty (Visible)", + "$1,234.50", + "2026-08-14", + 'Merged smoke', + "=SUM(B1,1)", + ) + ) + and xlsx_warning_codes == ["xlsx_formula_cache_missing"], } if not all(results.values()): failed = sorted(name for name, passed in results.items() if not passed) @@ -53,6 +79,45 @@ def run_smoke(directory: Path, *, expected_version: str) -> dict[str, bool]: return dict(sorted(results.items())) +def _write_xlsx_smoke(path: Path) -> None: + workbook = Workbook() + summary = workbook.active + summary.title = "Summary" + summary["A1"] = "Amount" + summary["B1"] = 1234.5 + summary["B1"].number_format = "$#,##0.00" + summary["A2"] = "Date" + summary["B2"] = date(2026, 8, 14) + summary["B2"].number_format = "yyyy-mm-dd" + summary.merge_cells("A3:B3") + summary["A3"] = "Merged smoke" + summary["A4"] = "=SUM(B1,1)" + + hidden = workbook.create_sheet("Hidden") + hidden.sheet_state = "hidden" + hidden["A1"] = "Hidden smoke" + very_hidden = workbook.create_sheet("Very Hidden") + very_hidden.sheet_state = "veryHidden" + very_hidden["A1"] = "Very hidden smoke" + workbook.create_sheet("Empty") + workbook.save(path) + workbook.close() + + rewritten = path.with_suffix(".rewritten.xlsx") + formula_with_empty_cache = b"SUM(B1,1)" + replacement_count = 0 + with ZipFile(path) as source, ZipFile(rewritten, "w", ZIP_DEFLATED) as target: + for member in source.infolist(): + payload = source.read(member.filename) + if member.filename == "xl/worksheets/sheet1.xml": + replacement_count = payload.count(formula_with_empty_cache) + payload = payload.replace(formula_with_empty_cache, b"SUM(B1,1)") + target.writestr(member, payload) + if replacement_count != 1: + raise RuntimeError("XLSX smoke fixture did not contain the expected empty formula cache") + rewritten.replace(path) + + def _native_pdf(text: str) -> bytes: escaped = text.replace("\\", "\\\\").replace("(", "\\(").replace(")", "\\)") stream = f"BT /F1 14 Tf 72 720 Td ({escaped}) Tj ET".encode("ascii") @@ -93,13 +158,13 @@ def _parser() -> argparse.ArgumentParser: description="Run native-only smoke checks against an installed OpenDocs artifact." ) parser.add_argument("directory", type=Path) - parser.add_argument("--version", default="0.1.0") + parser.add_argument("--version") return parser def main(argv: list[str] | None = None) -> int: args = _parser().parse_args(argv) - results = run_smoke(args.directory, expected_version=args.version) + results = run_smoke(args.directory, expected_version=args.version or version("opendocs-sdk")) print(", ".join(f"{name}=pass" for name in results)) return 0 diff --git a/src/opendocs/_models.py b/src/opendocs/_models.py index cc98bfd..b8a83a1 100644 --- a/src/opendocs/_models.py +++ b/src/opendocs/_models.py @@ -65,6 +65,7 @@ class DocumentType(StrEnum): IMAGE = "image" DOCX = "docx" PPTX = "pptx" + XLSX = "xlsx" class ListKind(StrEnum): diff --git a/src/opendocs/detection.py b/src/opendocs/detection.py index 6959b86..2142a3e 100644 --- a/src/opendocs/detection.py +++ b/src/opendocs/detection.py @@ -10,6 +10,7 @@ DocumentTypeMismatchError, UnsupportedDocumentError, ) +from opendocs.parsers.office.package import validate_office_package from opendocs.source import ResolvedSource _SUFFIX_TYPES = { @@ -23,6 +24,7 @@ ".webp": DocumentType.IMAGE, ".docx": DocumentType.DOCX, ".pptx": DocumentType.PPTX, + ".xlsx": DocumentType.XLSX, } _ZIP_PREFIXES = (b"PK\x03\x04", b"PK\x05\x06", b"PK\x07\x08") _UTF8_CHUNK_SIZE = 4096 @@ -43,17 +45,26 @@ def _signature_type(signature: bytes) -> DocumentType | None: def _container_type(path: Path) -> DocumentType: try: with ZipFile(path) as archive: + found_docx = False found_pptx = False + found_xlsx = False for info in archive.infolist(): if info.filename == "word/document.xml": - return DocumentType.DOCX - if info.filename == "ppt/presentation.xml": + found_docx = True + elif info.filename == "ppt/presentation.xml": found_pptx = True + elif info.filename == "xl/workbook.xml": + found_xlsx = True except BadZipFile as error: raise CorruptDocumentError("ZIP-based document is corrupt") from error + if found_docx: + return DocumentType.DOCX if found_pptx: return DocumentType.PPTX + if found_xlsx: + validate_office_package(path, document_type=DocumentType.XLSX) + return DocumentType.XLSX raise UnsupportedDocumentError("ZIP container is neither DOCX nor PPTX") @@ -81,11 +92,17 @@ def detect_document_type(source: ResolvedSource) -> DocumentType: if suffix and suffix not in _SUFFIX_TYPES: raise UnsupportedDocumentError(f"unsupported document extension: {suffix}") + declared = _SUFFIX_TYPES.get(suffix) detected = _signature_type(signature) if detected is None and signature.startswith(_ZIP_PREFIXES): - detected = _container_type(source.path) + try: + detected = _container_type(source.path) + except UnsupportedDocumentError: + if declared is not DocumentType.XLSX: + raise + validate_office_package(source.path, document_type=DocumentType.XLSX) + detected = DocumentType.XLSX - declared = _SUFFIX_TYPES.get(suffix) if declared is DocumentType.TEXT or declared is DocumentType.MARKDOWN: if detected is not None: raise _mismatch(suffix, detected) diff --git a/src/opendocs/markdown.py b/src/opendocs/markdown.py index 47dd78f..f3cb109 100644 --- a/src/opendocs/markdown.py +++ b/src/opendocs/markdown.py @@ -51,7 +51,12 @@ def _normalize_cell_newlines(value: str) -> str: def _escape_pipe_cell(value: str) -> str: value = _normalize_cell_newlines(value) - return value.replace("\\", "\\\\").replace("|", "\\|").replace("\n", "
") + return ( + html.escape(value, quote=False) + .replace("\\", "\\\\") + .replace("|", "\\|") + .replace("\n", "
") + ) def _escape_html_cell(value: str) -> str: diff --git a/src/opendocs/parsers/office/package.py b/src/opendocs/parsers/office/package.py index 4b0b523..10a423e 100644 --- a/src/opendocs/parsers/office/package.py +++ b/src/opendocs/parsers/office/package.py @@ -9,6 +9,9 @@ from typing import TypeVar from zipfile import BadZipFile, ZipFile, ZipInfo +from defusedxml import ElementTree as DefusedET +from defusedxml.common import DefusedXmlException + from opendocs._models import DocumentType from opendocs.errors import CorruptDocumentError, LimitExceededError from opendocs.source import ParseWorkspace @@ -23,9 +26,18 @@ _REL_NS = "{http://schemas.openxmlformats.org/package/2006/relationships}" _RELATIONSHIP_TAG = f"{_REL_NS}Relationship" +_CONTENT_TYPES_NS = "{http://schemas.openxmlformats.org/package/2006/content-types}" +_CONTENT_TYPE_OVERRIDE_TAG = f"{_CONTENT_TYPES_NS}Override" +_OFFICE_DOCUMENT_RELATIONSHIP = ( + "http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" +) +_XLSX_WORKBOOK_CONTENT_TYPE = ( + "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml" +) _REQUIRED_PARTS = { DocumentType.DOCX: "word/document.xml", DocumentType.PPTX: "ppt/presentation.xml", + DocumentType.XLSX: "xl/workbook.xml", } _MEDIA_SEGMENT = "media" _BINARY_SUFFIXES = { @@ -109,29 +121,49 @@ def _rels_source_base(name: str) -> str: def _normalize_target(base_dir: str, target: str) -> str: - if not target or target.startswith(("/", "\\")) or "\\" in target: + if not target or target.startswith(("//", "\\")) or "\\" in target: raise CorruptDocumentError("Office package relationship target is invalid") - if ":" in PurePosixPath(target).parts[:1]: + package_absolute = target.startswith("/") + candidate = target[1:] if package_absolute else target + first_parts = PurePosixPath(candidate).parts[:1] + if not candidate or (first_parts and ":" in first_parts[0]): raise CorruptDocumentError("Office package relationship target is invalid") joined = ( - posixpath.normpath(posixpath.join(base_dir, target)) - if base_dir - else posixpath.normpath(target) + posixpath.normpath(candidate) + if package_absolute or not base_dir + else posixpath.normpath(posixpath.join(base_dir, candidate)) ) - if joined.startswith("../") or joined == ".." or joined.startswith("/"): + if joined in {"", ".", ".."} or joined.startswith(("../", "/")): raise CorruptDocumentError("Office package relationship target is invalid") return joined +def _parse_package_xml(data: bytes, *, message: str) -> ET.Element: + try: + return DefusedET.fromstring( + data, + forbid_dtd=True, + forbid_entities=True, + forbid_external=True, + ) + except (DefusedXmlException, ET.ParseError) as error: + raise CorruptDocumentError(message) from error + + +def _read_relationships(archive: ZipFile, rels_name: str) -> ET.Element: + try: + data = archive.read(rels_name) + except (KeyError, OSError) as error: + raise CorruptDocumentError("Office package relationships are corrupt") from error + return _parse_package_xml(data, message="Office package relationships are corrupt") + + def _parse_relationship_targets( archive: ZipFile, infos_by_name: dict[str, ZipInfo], rels_name: str, ) -> None: - try: - root = ET.fromstring(archive.read(rels_name)) - except (KeyError, OSError, ET.ParseError) as error: - raise CorruptDocumentError("Office package relationships are corrupt") from error + root = _read_relationships(archive, rels_name) base_dir = _rels_source_base(rels_name) for node in root.iter(_RELATIONSHIP_TAG): target = node.get("Target") @@ -152,23 +184,37 @@ def _required_root_target( if "_rels/.rels" not in infos_by_name: raise CorruptDocumentError("Office package root relationships are missing") required = _REQUIRED_PARTS[document_type] - try: - root = ET.fromstring(archive.read("_rels/.rels")) - except (OSError, ET.ParseError) as error: - raise CorruptDocumentError("Office package relationships are corrupt") from error + root = _read_relationships(archive, "_rels/.rels") for node in root.iter(_RELATIONSHIP_TAG): target = node.get("Target") if target is None or node.get("TargetMode") == "External": continue + if document_type is DocumentType.XLSX and node.get("Type") != _OFFICE_DOCUMENT_RELATIONSHIP: + continue if _normalize_target("", target) == required: return raise CorruptDocumentError("Office package required root relationship is missing") +def _required_xlsx_content_type(archive: ZipFile) -> None: + try: + data = archive.read("[Content_Types].xml") + except (KeyError, OSError) as error: + raise CorruptDocumentError("Office package content types part is corrupt") from error + root = _parse_package_xml(data, message="Office package content types part is corrupt") + for node in root.iter(_CONTENT_TYPE_OVERRIDE_TAG): + if ( + node.get("PartName") == "/xl/workbook.xml" + and node.get("ContentType") == _XLSX_WORKBOOK_CONTENT_TYPE + ): + return + raise CorruptDocumentError("Office package required main part content type is missing") + + def validate_office_package(path: Path, *, document_type: DocumentType) -> OfficePackageLayout: required_main_part = _REQUIRED_PARTS.get(document_type) if required_main_part is None: - raise ValueError("document_type must be DOCX or PPTX") + raise ValueError("document_type must be DOCX, PPTX, or XLSX") try: with ZipFile(path) as archive: infos = archive.infolist() @@ -210,6 +256,8 @@ def validate_office_package(path: Path, *, document_type: DocumentType) -> Offic raise CorruptDocumentError("Office package content types part is missing") if required_main_part not in infos_by_name: raise CorruptDocumentError("Office package required main part is missing") + if document_type is DocumentType.XLSX: + _required_xlsx_content_type(archive) _required_root_target(archive, infos_by_name, document_type) for rels_name in tuple(name for name in infos_by_name if name.endswith(".rels")): _parse_relationship_targets(archive, infos_by_name, rels_name) diff --git a/src/opendocs/parsers/registry.py b/src/opendocs/parsers/registry.py index 7e7ccb0..05f4581 100644 --- a/src/opendocs/parsers/registry.py +++ b/src/opendocs/parsers/registry.py @@ -69,6 +69,7 @@ def build_default_registry( from opendocs.parsers.image import ImageParser from opendocs.parsers.office.parser import OfficeParser from opendocs.parsers.pdf.parser import PDFParser + from opendocs.parsers.xlsx import XlsxParser registry.register(DocumentType.IMAGE, ImageParser(runtime, vision, vision_config)) registry.register( @@ -83,4 +84,8 @@ def build_default_registry( DocumentType.PPTX, OfficeParser(DocumentType.PPTX, runtime, vision, vision_config, deadline=deadline), ) + registry.register( + DocumentType.XLSX, + XlsxParser(runtime, vision, vision_config, deadline=deadline), + ) return registry diff --git a/src/opendocs/parsers/xlsx/__init__.py b/src/opendocs/parsers/xlsx/__init__.py new file mode 100644 index 0000000..9833ca9 --- /dev/null +++ b/src/opendocs/parsers/xlsx/__init__.py @@ -0,0 +1,3 @@ +from opendocs.parsers.xlsx.parser import XlsxParser + +__all__ = ["XlsxParser"] diff --git a/src/opendocs/parsers/xlsx/extract.py b/src/opendocs/parsers/xlsx/extract.py new file mode 100644 index 0000000..4e88ba1 --- /dev/null +++ b/src/opendocs/parsers/xlsx/extract.py @@ -0,0 +1,917 @@ +from __future__ import annotations + +import re +from dataclasses import dataclass +from datetime import datetime +from decimal import Decimal, InvalidOperation +from pathlib import Path +from typing import Any +from zipfile import BadZipFile, ZipFile + +import openpyxl +from defusedxml import ElementTree as DefusedET +from defusedxml.common import DefusedXmlException +from openpyxl.utils.cell import coordinate_to_tuple, get_column_letter, range_boundaries + +from opendocs._models import ( + Block, + HeadingBlock, + InlineText, + MarkdownBlock, + ParagraphBlock, + SpannedTableBlock, + SpannedTableCell, + TableBlock, + WarningRecord, +) +from opendocs.errors import CorruptDocumentError, LimitExceededError +from opendocs.parsers.xlsx.media import XlsxVisualOccurrence, read_xlsx_visual_objects +from opendocs.parsers.xlsx.models import ( + XlsxChartSlot, + XlsxDocument, + XlsxImageSlot, + XlsxNativeSlot, + XlsxSheet, + XlsxSheetKind, +) +from opendocs.parsers.xlsx.preflight import ( + MAX_MATERIALIZED_GRID_CELLS as PREFLIGHT_MAX_MATERIALIZED_GRID_CELLS, +) +from opendocs.parsers.xlsx.preflight import XlsxPreflight, XlsxPreflightSheet +from opendocs.parsers.xlsx.text_objects import ( + XlsxTextObject, + read_xlsx_text_objects, + text_object_blocks, +) +from opendocs.parsers.xlsx.values import format_saved_value + +MAX_MATERIALIZED_GRID_CELLS = PREFLIGHT_MAX_MATERIALIZED_GRID_CELLS + +_SPREADSHEET_NS = "http://schemas.openxmlformats.org/spreadsheetml/2006/main" +_CELL_TAG = f"{{{_SPREADSHEET_NS}}}c" +_FORMULA_TAG = f"{{{_SPREADSHEET_NS}}}f" +_VALUE_TAG = f"{{{_SPREADSHEET_NS}}}v" +_WARNING_LIMIT_PER_CODE = 20 +_MAX_EXCEL_COLUMN = 16_384 +_EXTERNAL_FORMULA_REFERENCE = re.compile(r"\[[^\]\r\n]+\][^!\r\n]*!") + +_Coordinate = tuple[int, int] +_Bounds = tuple[int, int, int, int] + + +@dataclass(frozen=True, slots=True) +class _FormulaRecord: + formula_type: str + text: str | None + reference: str | None + attributes: tuple[tuple[str, str], ...] + + +@dataclass(frozen=True, slots=True) +class _CellRecord: + coordinate: str + cell_type: str | None + formula: _FormulaRecord | None + cache_present: bool + cached_value: str | None + + +@dataclass(frozen=True, slots=True, order=True) +class _WarningEvent: + code: str + sheet_index: int + row: int + column: int + sheet_name: str + detail: str + + +class _WarningCollector: + def __init__(self) -> None: + self._events: set[_WarningEvent] = set() + + def add( + self, + code: str, + *, + sheet: XlsxPreflightSheet, + coordinate: str, + detail: str, + ) -> None: + row, column = coordinate_to_tuple(coordinate) + self._events.add( + _WarningEvent( + code=code, + sheet_index=sheet.sheet_index, + row=row, + column=column, + sheet_name=sheet.name, + detail=detail, + ) + ) + + def freeze(self) -> tuple[WarningRecord, ...]: + grouped: dict[str, list[_WarningEvent]] = {} + for event in sorted(self._events): + grouped.setdefault(event.code, []).append(event) + warnings: list[WarningRecord] = [] + for code in sorted(grouped): + events = grouped[code] + for event in events[:_WARNING_LIMIT_PER_CODE]: + coordinate = f"{get_column_letter(event.column)}{event.row}" + warnings.append( + WarningRecord( + code=code, + message=f"{event.sheet_name}!{coordinate}: {event.detail}", + ) + ) + suppressed = len(events) - _WARNING_LIMIT_PER_CODE + if suppressed > 0: + warnings.append( + WarningRecord( + code=code, + message=f"{suppressed} additional {code} warnings suppressed", + ) + ) + return tuple(warnings) + + +@dataclass(frozen=True, slots=True) +class _MergeSpec: + bounds: _Bounds + + +@dataclass(frozen=True, slots=True) +class _RegionSpec: + bounds: _Bounds + header_rows: int + merges: tuple[_MergeSpec, ...] + kind: str + + +@dataclass(frozen=True, slots=True) +class _SlotCandidate: + row: int + column: int + kind_rank: int + source_ordinal: int + anchor: str + region: _RegionSpec | None = None + text_object: XlsxTextObject | None = None + + +@dataclass(slots=True) +class _MaterializationBudget: + used: int = 0 + + def consume(self, cells: int) -> None: + self.used += cells + if self.used > MAX_MATERIALIZED_GRID_CELLS: + raise LimitExceededError("XLSX exceeds the materialized grid limit") + + +def _safe_worksheet_xml(data: bytes, *, part_name: str) -> Any: + try: + root = DefusedET.fromstring( + data, + forbid_dtd=True, + forbid_entities=True, + forbid_external=True, + ) + except (DefusedXmlException, DefusedET.ParseError) as error: + raise CorruptDocumentError(f"XLSX worksheet part is corrupt: {part_name}") from error + if root.tag != f"{{{_SPREADSHEET_NS}}}worksheet": + raise CorruptDocumentError(f"XLSX worksheet part is corrupt: {part_name}") + return root + + +def _formula_record(element: Any) -> _FormulaRecord: + attributes = tuple(sorted((str(name), str(value)) for name, value in element.attrib.items())) + return _FormulaRecord( + formula_type=element.get("t", "normal"), + text=element.text if element.text not in {None, ""} else None, + reference=element.get("ref"), + attributes=attributes, + ) + + +def _worksheet_sidecar(data: bytes, *, part_name: str) -> tuple[_CellRecord, ...]: + root = _safe_worksheet_xml(data, part_name=part_name) + records: list[_CellRecord] = [] + for element in root.iter(_CELL_TAG): + coordinate = element.get("r") + if coordinate is None: + raise CorruptDocumentError("XLSX cell coordinate is missing") + formula_node = element.find(_FORMULA_TAG) + value_node = element.find(_VALUE_TAG) + records.append( + _CellRecord( + coordinate=coordinate, + cell_type=element.get("t"), + formula=_formula_record(formula_node) if formula_node is not None else None, + cache_present=value_node is not None, + cached_value=value_node.text if value_node is not None else None, + ) + ) + return tuple(records) + + +def _read_sidecars( + path: Path, + preflight: XlsxPreflight, +) -> dict[int, tuple[_CellRecord, ...]]: + records: dict[int, tuple[_CellRecord, ...]] = {} + try: + with ZipFile(path) as archive: + for sheet in preflight.sheets: + if sheet.kind is XlsxSheetKind.CHARTSHEET: + records[sheet.sheet_index] = () + continue + try: + data = archive.read(sheet.part_name) + except KeyError as error: + raise CorruptDocumentError("XLSX worksheet part is missing") from error + records[sheet.sheet_index] = _worksheet_sidecar( + data, + part_name=sheet.part_name, + ) + except BadZipFile as error: + raise CorruptDocumentError("XLSX package is corrupt") from error + except OSError as error: + raise CorruptDocumentError("XLSX package could not be read") from error + return records + + +def _decode_cached_value(record: _CellRecord) -> object: + value = record.cached_value + if value is None: + return None + if record.cell_type == "b": + if value not in {"0", "1"}: + raise CorruptDocumentError("XLSX formula boolean cache is invalid") + return value == "1" + if record.cell_type in {"e", "str", "inlineStr"}: + return value + if record.cell_type == "d": + try: + return datetime.fromisoformat(value.replace("Z", "+00:00")) + except ValueError as error: + raise CorruptDocumentError("XLSX formula date cache is invalid") from error + try: + return Decimal(value) + except InvalidOperation as error: + raise CorruptDocumentError("XLSX formula numeric cache is invalid") from error + + +def _loaded_formula_text(value: object) -> str | None: + if isinstance(value, str): + return value if value.startswith("=") else f"={value}" + text = getattr(value, "text", None) + if isinstance(text, str) and text: + return text if text.startswith("=") else f"={text}" + return None + + +def _literal_formula_text(record: _CellRecord, loaded_value: object) -> str | None: + if record.formula is not None and record.formula.text: + return ( + record.formula.text + if record.formula.text.startswith("=") + else f"={record.formula.text}" + ) + return _loaded_formula_text(loaded_value) + + +def _references_external_workbook(record: _CellRecord, loaded_value: object) -> bool: + formula_text = _literal_formula_text(record, loaded_value) + return formula_text is not None and _EXTERNAL_FORMULA_REFERENCE.search(formula_text) is not None + + +def _record_format_warning( + warning_collector: _WarningCollector, + *, + sheet: XlsxPreflightSheet, + coordinate: str, + warning: str | None, +) -> None: + if warning is None: + return + warning_collector.add( + "xlsx_unsupported_number_format", + sheet=sheet, + coordinate=coordinate, + detail=warning, + ) + + +def _special_formula_text( + record: _CellRecord, + loaded_value: object, + *, + saved_value: str, +) -> str | None: + formula = record.formula + if formula is None: + return None + reference = formula.reference or record.coordinate + if formula.formula_type == "array": + expression = _literal_formula_text(record, loaded_value) or "(expression unavailable)" + rendered = f"Array/spill formula {reference}: {expression}" + if record.cache_present and saved_value: + rendered += f"; saved value: {saved_value}" + return rendered + if formula.formula_type != "dataTable": + return None + parameters = [ + f"{name}={value}" for name, value in formula.attributes if name not in {"t", "ref"} + ] + suffix = f" ({', '.join(parameters)})" if parameters else "" + rendered = f"Data-table formula {reference}{suffix}" + if record.cache_present and saved_value: + rendered += f"; saved value: {saved_value}" + return rendered + + +def _formula_text( + record: _CellRecord, + loaded_value: object, + *, + number_format: str, + epoch: datetime, + conditional_number_format: bool, + sheet: XlsxPreflightSheet, + warnings: _WarningCollector, +) -> str: + if _references_external_workbook(record, loaded_value): + warnings.add( + "xlsx_external_reference", + sheet=sheet, + coordinate=record.coordinate, + detail="external workbook formula was preserved without access", + ) + saved_value = "" + if record.cache_present: + formatted = format_saved_value( + _decode_cached_value(record), + number_format, + epoch=epoch, + conditional_number_format=conditional_number_format, + ) + saved_value = formatted.text + _record_format_warning( + warnings, + sheet=sheet, + coordinate=record.coordinate, + warning=formatted.warning, + ) + + special = _special_formula_text(record, loaded_value, saved_value=saved_value) + if special is not None: + if record.formula is not None and record.formula.formula_type == "dataTable": + warnings.add( + "xlsx_data_table_formula", + sheet=sheet, + coordinate=record.coordinate, + detail="data-table formula has no portable literal expression", + ) + if not record.cache_present: + warnings.add( + "xlsx_formula_cache_missing", + sheet=sheet, + coordinate=record.coordinate, + detail="formula has no saved cache; formula text was preserved", + ) + return special + if record.cache_present: + return saved_value + warnings.add( + "xlsx_formula_cache_missing", + sheet=sheet, + coordinate=record.coordinate, + detail="formula has no saved cache; formula text was preserved", + ) + return _literal_formula_text(record, loaded_value) or "Formula expression unavailable" + + +def _conditional_number_format_coordinates( + worksheet: Any, + records: tuple[_CellRecord, ...], +) -> frozenset[str]: + events: list[tuple[int, int, int, int]] = [] + for conditional in worksheet.conditional_formatting: + rules = worksheet.conditional_formatting[conditional] + changes_number_format = False + for rule in rules: + dxf = getattr(rule, "dxf", None) + if dxf is not None: + changes_number_format |= getattr(dxf, "numFmt", None) is not None + elif getattr(rule, "dxfId", None) is not None: + changes_number_format = True + if changes_number_format: + for cell_range in conditional.sqref.ranges: + events.append( + ( + cell_range.min_row, + 1, + cell_range.min_col, + cell_range.max_col, + ) + ) + events.append( + ( + cell_range.max_row + 1, + -1, + cell_range.min_col, + cell_range.max_col, + ) + ) + if not events or not records: + return frozenset() + + events.sort() + indexed_records = sorted( + (coordinate_to_tuple(record.coordinate), record.coordinate) for record in records + ) + column_deltas = [0] * (_MAX_EXCEL_COLUMN + 2) + + def update(column: int, delta: int) -> None: + while column < len(column_deltas): + column_deltas[column] += delta + column += column & -column + + def active_at(column: int) -> int: + total = 0 + while column > 0: + total += column_deltas[column] + column -= column & -column + return total + + affected: set[str] = set() + event_index = 0 + for (row, column), coordinate in indexed_records: + while event_index < len(events) and events[event_index][0] <= row: + _, delta, minimum_column, maximum_column = events[event_index] + update(minimum_column, delta) + update(maximum_column + 1, -delta) + event_index += 1 + if active_at(column) > 0: + affected.add(coordinate) + return frozenset(affected) + + +def _cell_texts( + worksheet: Any, + records: tuple[_CellRecord, ...], + *, + epoch: datetime, + sheet: XlsxPreflightSheet, + warnings: _WarningCollector, +) -> tuple[dict[_Coordinate, str], set[_Coordinate]]: + texts: dict[_Coordinate, str] = {} + semantic: set[_Coordinate] = set() + conditional_coordinates = _conditional_number_format_coordinates(worksheet, records) + for record in records: + cell = worksheet[record.coordinate] + conditional_number_format = record.coordinate in conditional_coordinates + if record.formula is not None: + text = _formula_text( + record, + cell.value, + number_format=cell.number_format, + epoch=epoch, + conditional_number_format=conditional_number_format, + sheet=sheet, + warnings=warnings, + ) + else: + formatted = format_saved_value( + cell.value, + cell.number_format, + epoch=epoch, + conditional_number_format=conditional_number_format, + ) + text = formatted.text + _record_format_warning( + warnings, + sheet=sheet, + coordinate=record.coordinate, + warning=formatted.warning, + ) + coordinate = coordinate_to_tuple(record.coordinate) + texts[coordinate] = text + if text != "": + semantic.add(coordinate) + return texts, semantic + + +def _area(bounds: _Bounds) -> int: + minimum_column, minimum_row, maximum_column, maximum_row = bounds + return (maximum_column - minimum_column + 1) * (maximum_row - minimum_row + 1) + + +def _coordinates(bounds: _Bounds) -> set[_Coordinate]: + minimum_column, minimum_row, maximum_column, maximum_row = bounds + return { + (row, column) + for row in range(minimum_row, maximum_row + 1) + for column in range(minimum_column, maximum_column + 1) + } + + +def _anchor(bounds: _Bounds) -> str: + minimum_column, minimum_row, maximum_column, maximum_row = bounds + start = f"{get_column_letter(minimum_column)}{minimum_row}" + end = f"{get_column_letter(maximum_column)}{maximum_row}" + return start if start == end else f"{start}:{end}" + + +def _merge_specs(worksheet: Any) -> tuple[_MergeSpec, ...]: + return tuple( + sorted( + ( + _MergeSpec(range_boundaries(str(cell_range))) + for cell_range in worksheet.merged_cells.ranges + ), + key=lambda item: item.bounds, + ) + ) + + +def _table_specs(worksheet: Any, merges: tuple[_MergeSpec, ...]) -> tuple[_RegionSpec, ...]: + tables = sorted(worksheet.tables.values(), key=lambda table: range_boundaries(table.ref)) + specs: list[_RegionSpec] = [] + for table in tables: + bounds = range_boundaries(table.ref) + table_merges = tuple(merge for merge in merges if _bounds_within(merge.bounds, bounds)) + specs.append( + _RegionSpec( + bounds=bounds, + header_rows=1 if (table.headerRowCount or 0) > 0 else 0, + merges=table_merges, + kind="table", + ) + ) + return tuple(specs) + + +def _bounds_within(inner: _Bounds, outer: _Bounds) -> bool: + inner_left, inner_top, inner_right, inner_bottom = inner + outer_left, outer_top, outer_right, outer_bottom = outer + return ( + outer_left <= inner_left <= inner_right <= outer_right + and outer_top <= inner_top <= inner_bottom <= outer_bottom + ) + + +def _component_specs( + semantic: set[_Coordinate], + merges: tuple[_MergeSpec, ...], + occupied: set[_Coordinate], +) -> tuple[_RegionSpec, ...]: + available = set(semantic) - occupied + available_merges: list[_MergeSpec] = [] + for merge in merges: + footprint = _coordinates(merge.bounds) - occupied + if not footprint: + continue + available.update(footprint) + available_merges.append(merge) + + remaining = set(available) + specs: list[_RegionSpec] = [] + for seed in sorted(available): + if seed not in remaining: + continue + stack = [seed] + remaining.remove(seed) + component: set[_Coordinate] = set() + while stack: + row, column = stack.pop() + component.add((row, column)) + for neighbor in ( + (row - 1, column), + (row + 1, column), + (row, column - 1), + (row, column + 1), + ): + if neighbor in remaining: + remaining.remove(neighbor) + stack.append(neighbor) + rows = [row for row, _ in component] + columns = [column for _, column in component] + bounds = (min(columns), min(rows), max(columns), max(rows)) + component_merges = tuple( + merge for merge in available_merges if _bounds_within(merge.bounds, bounds) + ) + specs.append(_RegionSpec(bounds, 0, component_merges, "region")) + return tuple(sorted(specs, key=lambda item: item.bounds)) + + +def _region_specs( + worksheet: Any, + semantic: set[_Coordinate], +) -> tuple[_RegionSpec, ...]: + merges = _merge_specs(worksheet) + table_specs = _table_specs(worksheet, merges) + occupied: set[_Coordinate] = set() + for table in table_specs: + occupied.update(_coordinates(table.bounds)) + component_specs = _component_specs(semantic, merges, occupied) + return tuple(sorted((*table_specs, *component_specs), key=lambda item: item.bounds)) + + +def _region_block( + spec: _RegionSpec, texts: dict[_Coordinate, str] +) -> TableBlock | SpannedTableBlock: + minimum_column, minimum_row, maximum_column, maximum_row = spec.bounds + row_count = maximum_row - minimum_row + 1 + column_count = maximum_column - minimum_column + 1 + if not spec.merges: + return TableBlock( + tuple( + tuple( + texts.get((row, column), "") + for column in range(minimum_column, maximum_column + 1) + ) + for row in range(minimum_row, maximum_row + 1) + ), + spec.header_rows, + ) + + merge_by_origin = {(merge.bounds[1], merge.bounds[0]): merge for merge in spec.merges} + covered: set[_Coordinate] = set() + for merge in spec.merges: + covered.update(_coordinates(merge.bounds)) + cells: list[SpannedTableCell] = [] + for row in range(minimum_row, maximum_row + 1): + for column in range(minimum_column, maximum_column + 1): + merge = merge_by_origin.get((row, column)) + if merge is not None: + left, top, right, bottom = merge.bounds + cells.append( + SpannedTableCell( + row - minimum_row, + column - minimum_column, + bottom - top + 1, + right - left + 1, + texts.get((row, column), ""), + ) + ) + elif (row, column) in covered: + continue + else: + cells.append( + SpannedTableCell( + row - minimum_row, + column - minimum_column, + 1, + 1, + texts.get((row, column), ""), + ) + ) + return SpannedTableBlock(row_count, column_count, tuple(cells), spec.header_rows) + + +def _sheet_prelude(sheet: XlsxPreflightSheet) -> XlsxNativeSlot: + return XlsxNativeSlot( + source_index=0, + anchor="A1", + blocks=( + MarkdownBlock(f""), + HeadingBlock(1, (InlineText(sheet.name),)), + ParagraphBlock((InlineText(f"Sheet state: {sheet.state.value}"),)), + ), + ) + + +def _sheet_slots( + worksheet: Any, + *, + sheet: XlsxPreflightSheet, + budget: _MaterializationBudget, + text_objects: tuple[XlsxTextObject, ...], + texts: dict[_Coordinate, str], + semantic: set[_Coordinate], +) -> tuple[XlsxNativeSlot, ...]: + candidates: list[_SlotCandidate] = [] + for ordinal, spec in enumerate(_region_specs(worksheet, semantic), start=1): + budget.consume(_area(spec.bounds)) + anchor = _anchor(spec.bounds) + candidates.append( + _SlotCandidate( + row=spec.bounds[1], + column=spec.bounds[0], + kind_rank=0, + source_ordinal=ordinal, + anchor=anchor, + region=spec, + ) + ) + candidates.extend( + _SlotCandidate( + row=item.row, + column=item.column, + kind_rank=item.kind_rank, + source_ordinal=item.source_ordinal, + anchor=item.anchor, + text_object=item, + ) + for item in text_objects + ) + slots: list[XlsxNativeSlot] = [_sheet_prelude(sheet)] + for source_index, candidate in enumerate( + sorted( + candidates, + key=lambda item: (item.row, item.column, item.kind_rank, item.source_ordinal), + ), + start=1, + ): + if candidate.region is not None: + spec = candidate.region + comment = ( + f"" + ) + block: Block = _region_block(spec, texts) + blocks = (MarkdownBlock(comment), block) + elif candidate.text_object is not None: + blocks = text_object_blocks( + candidate.text_object, + fallback_label=texts.get((candidate.row, candidate.column), ""), + object_index=source_index, + ) + else: + raise AssertionError("XLSX slot candidate is invalid") + slots.append( + XlsxNativeSlot( + source_index=source_index, + anchor=candidate.anchor, + blocks=blocks, + ) + ) + return tuple(slots) + + +def _non_worksheet_slots( + sheet: XlsxPreflightSheet, + text_objects: tuple[XlsxTextObject, ...], +) -> tuple[XlsxNativeSlot, ...]: + slots = [_sheet_prelude(sheet)] + for source_index, item in enumerate( + sorted( + text_objects, + key=lambda value: ( + value.row, + value.column, + value.kind_rank, + value.source_ordinal, + ), + ), + start=1, + ): + slots.append( + XlsxNativeSlot( + source_index=source_index, + anchor=item.anchor, + blocks=text_object_blocks( + item, + fallback_label="", + object_index=source_index, + ), + ) + ) + return tuple(slots) + + +def _visual_slot( + occurrence: XlsxVisualOccurrence, + *, + source_index: int, +) -> XlsxNativeSlot | XlsxImageSlot | XlsxChartSlot: + if occurrence.artifact_name is None or occurrence.content_sha256 is None: + return XlsxNativeSlot(source_index, occurrence.anchor, occurrence.blocks) + if occurrence.kind == "image": + return XlsxImageSlot( + source_index=source_index, + anchor=occurrence.anchor, + artifact_name=occurrence.artifact_name, + content_sha256=occurrence.content_sha256, + alt_text=occurrence.alt_text, + object_name=occurrence.object_name, + title=occurrence.title, + ) + return XlsxChartSlot( + source_index=source_index, + anchor=occurrence.anchor, + artifact_name=occurrence.artifact_name, + content_sha256=occurrence.content_sha256, + blocks=occurrence.blocks, + alt_text=occurrence.alt_text, + object_name=occurrence.object_name, + title=occurrence.title, + ) + + +def extract_xlsx( + path: Path, + preflight: XlsxPreflight, + *, + artifact_dir: Path | None = None, +) -> XlsxDocument: + if not isinstance(path, Path): + raise TypeError("path must be a Path") + if not isinstance(preflight, XlsxPreflight): + raise TypeError("preflight must be an XlsxPreflight") + if artifact_dir is not None and not isinstance(artifact_dir, Path): + raise TypeError("artifact_dir must be a Path or None") + sidecars = _read_sidecars(path, preflight) + text_objects = read_xlsx_text_objects(path, preflight) + warnings = _WarningCollector() + for warning in text_objects.warnings: + sheet = preflight.sheets[warning.sheet_index - 1] + warnings.add( + warning.code, + sheet=sheet, + coordinate=warning.anchor.split(":", 1)[0], + detail=( + f"sheet={warning.sheet_index} anchor={warning.anchor} " + f"object={warning.object_ordinal}: {warning.detail}" + ), + ) + with path.open("rb") as package: + workbook = openpyxl.load_workbook( + package, + read_only=False, + data_only=False, + rich_text=False, + keep_links=False, + ) + try: + worksheets = {worksheet.title: worksheet for worksheet in workbook.worksheets} + sheet_values: dict[int, tuple[dict[_Coordinate, str], set[_Coordinate]]] = {} + workbook_values: dict[str, dict[str, str]] = {} + for sheet in preflight.sheets: + if sheet.kind is XlsxSheetKind.CHARTSHEET: + workbook_values[sheet.name] = {} + continue + worksheet = worksheets.get(sheet.name) + if worksheet is None: + raise CorruptDocumentError("XLSX worksheet is missing after full-mode load") + texts, semantic = _cell_texts( + worksheet, + sidecars[sheet.sheet_index], + epoch=workbook.epoch, + sheet=sheet, + warnings=warnings, + ) + sheet_values[sheet.sheet_index] = (texts, semantic) + workbook_values[sheet.name] = { + f"{get_column_letter(column)}{row}": text for (row, column), text in texts.items() + } + visual_objects = read_xlsx_visual_objects( + path, + preflight, + workbook_values, + artifact_dir=artifact_dir, + ) + sheets: list[XlsxSheet] = [] + budget = _MaterializationBudget() + for sheet in preflight.sheets: + sheet_text_objects = text_objects.by_sheet[sheet.sheet_index - 1] + if sheet.kind is XlsxSheetKind.CHARTSHEET: + slots = _non_worksheet_slots(sheet, sheet_text_objects) + else: + worksheet = worksheets.get(sheet.name) + if worksheet is None: + raise CorruptDocumentError("XLSX worksheet is missing after full-mode load") + texts, semantic = sheet_values[sheet.sheet_index] + slots = _sheet_slots( + worksheet, + sheet=sheet, + budget=budget, + text_objects=sheet_text_objects, + texts=texts, + semantic=semantic, + ) + slots = ( + *slots, + *( + _visual_slot(occurrence, source_index=len(slots) + offset) + for offset, occurrence in enumerate( + visual_objects.by_sheet[sheet.sheet_index - 1] + ) + ), + ) + sheets.append( + XlsxSheet( + sheet_index=sheet.sheet_index, + name=sheet.name, + kind=sheet.kind, + state=sheet.state, + slots=slots, + ) + ) + finally: + workbook.close() + combined_warnings = (*warnings.freeze(), *visual_objects.warnings) + return XlsxDocument(sheets=tuple(sheets), warnings=combined_warnings) diff --git a/src/opendocs/parsers/xlsx/media.py b/src/opendocs/parsers/xlsx/media.py new file mode 100644 index 0000000..f9bcf25 --- /dev/null +++ b/src/opendocs/parsers/xlsx/media.py @@ -0,0 +1,1003 @@ +from __future__ import annotations + +import hashlib +import io +import re +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Literal +from zipfile import BadZipFile, ZipFile, ZipInfo + +from defusedxml import ElementTree as DefusedET +from defusedxml.common import DefusedXmlException +from openpyxl.utils.cell import get_column_letter, range_boundaries +from PIL import Image, ImageDraw, ImageFont + +from opendocs._models import ( + Block, + HeadingBlock, + InlineText, + ParagraphBlock, + TableBlock, + WarningRecord, +) +from opendocs.errors import CorruptDocumentError, LimitExceededError +from opendocs.parsers.xlsx.models import XlsxChartSlot, XlsxDocument, XlsxImageSlot +from opendocs.parsers.xlsx.preflight import ( + MAX_CHART_CACHE_POINTS, + MAX_DRAWING_OBJECTS, + XlsxPreflight, + XlsxPreflightSheet, + _read_relationships, + _relationships_part, +) +from opendocs.vision.base import VisionRequest, VisionRequestKind +from opendocs.vision.images import PreparedImage, prepare_image + +_SPREADSHEET_NS = "http://schemas.openxmlformats.org/spreadsheetml/2006/main" +_OFFICE_REL_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" +_DRAWING_NS = "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing" +_DRAWING_MAIN_NS = "http://schemas.openxmlformats.org/drawingml/2006/main" +_CHART_NS = "http://schemas.openxmlformats.org/drawingml/2006/chart" +_DRAWING_RELATIONSHIP = f"{_OFFICE_REL_NS}/drawing" +_CHART_RELATIONSHIP = f"{_OFFICE_REL_NS}/chart" +_IMAGE_RELATIONSHIP = f"{_OFFICE_REL_NS}/image" +_RELATIONSHIP_ID = f"{{{_OFFICE_REL_NS}}}id" +_RELATIONSHIP_EMBED = f"{{{_OFFICE_REL_NS}}}embed" + +_CHART_TYPES = { + "lineChart": "line", + "barChart": "bar", + "pieChart": "pie", + "doughnutChart": "doughnut", + "scatterChart": "scatter", +} +_AXIS_TYPES = {"catAx", "valAx", "dateAx", "serAx"} +_LOCAL_RANGE = re.compile( + r"^(?P'(?:[^']|'')+'|[^'!\[\]]+)!" + r"(?P\$?[A-Z]{1,3}\$?[1-9][0-9]{0,6})" + r"(?::(?P\$?[A-Z]{1,3}\$?[1-9][0-9]{0,6}))?$", + re.IGNORECASE, +) +_SAFE_ARTIFACT_SUFFIX = re.compile(r"^\.[A-Za-z0-9]{1,8}$") +_PREVIEW_WIDTH = 1_280 +_PREVIEW_PADDING = 40 +_PREVIEW_LINE_HEIGHT = 26 +_PREVIEW_MAX_LINES = 80 +_PREVIEW_MAX_LINE_CHARS = 180 +_PREVIEW_SERIES_POINTS = 24 + +XLSX_CHART_VISION_PROMPT = ( + "这是由 XLSX 原生图表事实生成的语义卡片, 不是 Excel 外观还原。" + "仅补充可由卡片支持的趋势、关系、标注和含义; 不要改写原生数值, " + "不要猜测缺失数据, 并将结论明确标记为 '视觉解释'。" +) +XLSX_IMAGE_VISION_PROMPT = ( + "仅解释图片中可见的文字、标注、关系和含义; 涉及趋势时只描述可见证据, " + "不要猜测未显示的内容, 并将结论明确标记为 '视觉解释'。" +) + + +@dataclass(frozen=True, slots=True) +class XlsxChartSeriesFacts: + name: str + categories: tuple[str, ...] + values: tuple[str, ...] + x_values: tuple[str, ...] + y_values: tuple[str, ...] + formulas: tuple[str, ...] + unresolved_formulas: tuple[str, ...] + + +@dataclass(frozen=True, slots=True) +class XlsxChartFacts: + chart_type: str + title: str + axis_titles: tuple[str, ...] + axis_labels: tuple[str, ...] + data_labels: tuple[str, ...] + formulas: tuple[str, ...] + unresolved_formulas: tuple[str, ...] + series: tuple[XlsxChartSeriesFacts, ...] + + +@dataclass(frozen=True, slots=True) +class XlsxVisualOccurrence: + kind: Literal["chart", "image"] + anchor: str + blocks: tuple[Block, ...] + artifact_name: str | None + content_sha256: str | None + alt_text: str | None + object_name: str | None + title: str | None + + +@dataclass(frozen=True, slots=True) +class XlsxVisualObjects: + by_sheet: tuple[tuple[XlsxVisualOccurrence, ...], ...] + warnings: tuple[WarningRecord, ...] + + +@dataclass(frozen=True, slots=True) +class XlsxVisualRequest: + digest: str + image_path: Path + prompt: str + source_index: int + kind: VisionRequestKind + + def to_vision_request(self) -> VisionRequest: + return VisionRequest( + image_path=self.image_path, + prompt=self.prompt, + source_index=self.source_index, + kind=self.kind, + ) + + +@dataclass(frozen=True, slots=True) +class _OccurrenceWarning: + code: str + anchor: str + detail: str + + +class _UnsupportedChartType(Exception): + pass + + +def _safe_root(archive: ZipFile, part_name: str, *, message: str) -> Any: + try: + data = archive.read(part_name) + return DefusedET.fromstring( + data, + forbid_dtd=True, + forbid_entities=True, + forbid_external=True, + ) + except (KeyError, OSError, DefusedXmlException, DefusedET.ParseError) as error: + raise CorruptDocumentError(message) from error + + +def _local_name(element: Any) -> str: + return str(element.tag).rsplit("}", 1)[-1] + + +def _relationship_index( + archive: ZipFile, + infos: dict[str, ZipInfo], + part_name: str, +) -> dict[str, Any]: + return _read_relationships( + archive, + infos, + part_name, + required=_relationships_part(part_name) in infos, + ) + + +def _relationship( + relationships: dict[str, Any], + relationship_id: str | None, + *, + expected_type: str, +) -> Any: + relationship = relationships.get(relationship_id or "") + if ( + relationship is None + or relationship.external + or relationship.relationship_type != expected_type + ): + raise CorruptDocumentError("XLSX drawing relationship is invalid") + return relationship + + +def _marker_coordinate(marker: Any) -> tuple[int, int]: + try: + column = int(marker.findtext(f"{{{_DRAWING_NS}}}col", "")) + 1 + row = int(marker.findtext(f"{{{_DRAWING_NS}}}row", "")) + 1 + except ValueError as error: + raise CorruptDocumentError("XLSX drawing anchor is invalid") from error + if not 1 <= column <= 16_384 or not 1 <= row <= 1_048_576: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + return row, column + + +def _drawing_anchor(element: Any) -> str: + kind = _local_name(element) + if kind == "absoluteAnchor": + return "A1" + start = element.find(f"{{{_DRAWING_NS}}}from") + if start is None: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + start_row, start_column = _marker_coordinate(start) + start_text = f"{get_column_letter(start_column)}{start_row}" + if kind == "oneCellAnchor": + return start_text + end = element.find(f"{{{_DRAWING_NS}}}to") + if kind != "twoCellAnchor" or end is None: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + end_row, end_column = _marker_coordinate(end) + if end_row < start_row or end_column < start_column: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + end_text = f"{get_column_letter(end_column)}{end_row}" + return start_text if start_text == end_text else f"{start_text}:{end_text}" + + +def _metadata(anchor: Any) -> tuple[str | None, str | None, str | None]: + node = next(anchor.iter(f"{{{_DRAWING_NS}}}cNvPr"), None) + if node is None: + return None, None, None + return node.get("name"), node.get("descr"), node.get("title") + + +def _plain_text(element: Any | None) -> str: + if element is None: + return "" + rich = "".join(node.text or "" for node in element.iter(f"{{{_DRAWING_MAIN_NS}}}t")) + if rich: + return rich.strip() + values = "".join(node.text or "" for node in element.iter(f"{{{_CHART_NS}}}v")) + return values.strip() + + +def _cache_values(reference: Any) -> tuple[str, ...] | None: + cache = next( + ( + child + for child in reference + if _local_name(child) + in {"strCache", "numCache", "multiLvlStrCache", "strLit", "numLit", "multiLvlStrLit"} + ), + None, + ) + if cache is None: + return None + if _local_name(cache) == "multiLvlStrCache": + levels: list[dict[int, str]] = [] + for level in cache.findall(f"{{{_CHART_NS}}}lvl"): + indexed_level: dict[int, str] = {} + for point in level.findall(f"{{{_CHART_NS}}}pt"): + try: + index = int(point.get("idx", "")) + except ValueError as error: + raise CorruptDocumentError("XLSX chart cache index is invalid") from error + if index < 0 or index in indexed_level: + raise CorruptDocumentError("XLSX chart cache index is invalid") + indexed_level[index] = point.findtext(f"{{{_CHART_NS}}}v", "") + levels.append(indexed_level) + declared_node = cache.find(f"{{{_CHART_NS}}}ptCount") + declared = max( + (max(level, default=-1) + 1 for level in levels), + default=0, + ) + if declared_node is not None: + try: + declared = int(declared_node.get("val", "")) + except ValueError as error: + raise CorruptDocumentError("XLSX chart cache count is invalid") from error + if declared < 0 or declared > MAX_CHART_CACHE_POINTS: + raise LimitExceededError("XLSX exceeds the chart cache point limit") + return tuple( + " / ".join(value for level in levels if (value := level.get(index, ""))) + for index in range(declared) + ) + indexed: dict[int, str] = {} + for point in cache.iter(f"{{{_CHART_NS}}}pt"): + try: + index = int(point.get("idx", "")) + except ValueError as error: + raise CorruptDocumentError("XLSX chart cache index is invalid") from error + if index < 0 or index in indexed: + raise CorruptDocumentError("XLSX chart cache index is invalid") + indexed[index] = point.findtext(f"{{{_CHART_NS}}}v", "") + declared_node = next(cache.iter(f"{{{_CHART_NS}}}ptCount"), None) + declared = max(indexed, default=-1) + 1 + if declared_node is not None: + try: + declared = int(declared_node.get("val", "")) + except ValueError as error: + raise CorruptDocumentError("XLSX chart cache count is invalid") from error + if declared < 0 or declared > MAX_CHART_CACHE_POINTS: + raise LimitExceededError("XLSX exceeds the chart cache point limit") + return tuple(indexed.get(index, "") for index in range(declared)) + + +def _unquote_sheet(value: str) -> str: + if value.startswith("'") and value.endswith("'"): + return value[1:-1].replace("''", "'") + return value + + +def _resolve_local_formula( + formula: str, + workbook_values: dict[str, dict[str, str]], +) -> tuple[str, ...] | None: + local_range = _local_formula_range(formula, workbook_values) + if local_range is None: + return None + values, bounds = local_range + minimum_column, minimum_row, maximum_column, maximum_row = bounds + point_count = (maximum_column - minimum_column + 1) * (maximum_row - minimum_row + 1) + if point_count > MAX_CHART_CACHE_POINTS: + return None + return tuple( + values.get(f"{get_column_letter(column)}{row}", "") + for row in range(minimum_row, maximum_row + 1) + for column in range(minimum_column, maximum_column + 1) + ) + + +def _local_formula_range( + formula: str, + workbook_values: dict[str, dict[str, str]], +) -> tuple[dict[str, str], tuple[int, int, int, int]] | None: + match = _LOCAL_RANGE.fullmatch(formula.removeprefix("=")) + if match is None: + return None + sheet_name = _unquote_sheet(match.group("sheet")) + values = workbook_values.get(sheet_name) + if values is None: + return None + start = match.group("start").replace("$", "").upper() + end = (match.group("end") or start).replace("$", "").upper() + try: + bounds = range_boundaries(f"{start}:{end}") + except ValueError: + return None + return values, bounds + + +def _reference_values( + container: Any | None, + workbook_values: dict[str, dict[str, str]], +) -> tuple[tuple[str, ...], tuple[str, ...], tuple[str, ...]]: + if container is None: + return (), (), () + reference = next( + ( + node + for node in container.iter() + if _local_name(node) + in { + "strRef", + "numRef", + "multiLvlStrRef", + "strLit", + "numLit", + "multiLvlStrLit", + } + ), + None, + ) + if reference is None: + direct = container.find(f"{{{_CHART_NS}}}v") + return ((direct.text or "",), (), ()) if direct is not None else ((), (), ()) + formula = reference.findtext(f"{{{_CHART_NS}}}f", "").strip() + formulas = (formula,) if formula else () + cached = _cache_values(reference) + if cached is not None: + local_range = _local_formula_range(formula, workbook_values) if formula else None + return cached, formulas, () if local_range is not None else formulas + resolved = _resolve_local_formula(formula, workbook_values) if formula else None + if resolved is not None: + return resolved, formulas, () + return (), formulas, formulas + + +def _series_facts( + element: Any, + *, + chart_type: str, + workbook_values: dict[str, dict[str, str]], +) -> XlsxChartSeriesFacts: + name_values, name_formulas, name_unresolved = _reference_values( + element.find(f"{{{_CHART_NS}}}tx"), + workbook_values, + ) + categories, category_formulas, category_unresolved = _reference_values( + element.find(f"{{{_CHART_NS}}}cat"), + workbook_values, + ) + values, value_formulas, value_unresolved = _reference_values( + element.find(f"{{{_CHART_NS}}}val"), + workbook_values, + ) + x_values, x_formulas, x_unresolved = _reference_values( + element.find(f"{{{_CHART_NS}}}xVal"), + workbook_values, + ) + y_values, y_formulas, y_unresolved = _reference_values( + element.find(f"{{{_CHART_NS}}}yVal"), + workbook_values, + ) + if chart_type == "scatter": + categories = () + values = () + formulas = tuple( + dict.fromkeys( + (*name_formulas, *category_formulas, *value_formulas, *x_formulas, *y_formulas) + ) + ) + unresolved = tuple( + dict.fromkeys( + ( + *name_unresolved, + *category_unresolved, + *value_unresolved, + *x_unresolved, + *y_unresolved, + ) + ) + ) + return XlsxChartSeriesFacts( + name=next((value for value in name_values if value), "Series"), + categories=categories, + values=values, + x_values=x_values, + y_values=y_values, + formulas=formulas, + unresolved_formulas=unresolved, + ) + + +def _data_label_facts(chart: Any) -> tuple[str, ...]: + facts: list[str] = [] + labels = chart.find(f"{{{_CHART_NS}}}dLbls") + if labels is None: + return () + flag_labels = { + "showLegendKey": "legend key", + "showVal": "value", + "showCatName": "category name", + "showSerName": "series name", + "showPercent": "percentage", + "showBubbleSize": "bubble size", + } + for name, label in flag_labels.items(): + node = labels.find(f"{{{_CHART_NS}}}{name}") + if node is not None and node.get("val", "1") not in {"0", "false", "False"}: + facts.append(label) + for item in labels.findall(f"{{{_CHART_NS}}}dLbl"): + text = _plain_text(item.find(f"{{{_CHART_NS}}}tx")) + if text: + facts.append(text) + return tuple(dict.fromkeys(facts)) + + +def _chart_facts( + root: Any, + workbook_values: dict[str, dict[str, str]], +) -> XlsxChartFacts: + if root.tag != f"{{{_CHART_NS}}}chartSpace": + raise CorruptDocumentError("XLSX chart namespace is invalid") + chart = root.find(f"{{{_CHART_NS}}}chart") + plot_area = chart.find(f"{{{_CHART_NS}}}plotArea") if chart is not None else None + if chart is None or plot_area is None: + raise CorruptDocumentError("XLSX chart part is corrupt") + chart_nodes = [child for child in plot_area if _local_name(child) in _CHART_TYPES] + if not chart_nodes: + raise _UnsupportedChartType + chart_node = chart_nodes[0] + chart_type = _CHART_TYPES[_local_name(chart_node)] + chart_title = chart.find(f"{{{_CHART_NS}}}title") + title_values, title_formulas, title_unresolved = _reference_values( + chart_title.find(f"{{{_CHART_NS}}}tx") if chart_title is not None else None, + workbook_values, + ) + title = next((value for value in title_values if value), _plain_text(chart_title)) + axis_titles: list[str] = [] + axis_labels: list[str] = [] + axis_formulas: list[str] = [] + axis_unresolved: list[str] = [] + for axis in plot_area: + if _local_name(axis) not in _AXIS_TYPES: + continue + axis_title_node = axis.find(f"{{{_CHART_NS}}}title") + axis_title_values, formulas, unresolved = _reference_values( + axis_title_node.find(f"{{{_CHART_NS}}}tx") if axis_title_node is not None else None, + workbook_values, + ) + axis_title = next( + (value for value in axis_title_values if value), + _plain_text(axis_title_node), + ) + if axis_title: + axis_titles.append(axis_title) + axis_formulas.extend(formulas) + axis_unresolved.extend(unresolved) + label_position = axis.find(f"{{{_CHART_NS}}}tickLblPos") + if label_position is not None and label_position.get("val"): + axis_labels.append(label_position.get("val", "")) + return XlsxChartFacts( + chart_type=chart_type, + title=title, + axis_titles=tuple(axis_titles), + axis_labels=tuple(axis_labels), + data_labels=_data_label_facts(chart_node), + formulas=tuple(dict.fromkeys((*title_formulas, *axis_formulas))), + unresolved_formulas=tuple(dict.fromkeys((*title_unresolved, *axis_unresolved))), + series=tuple( + _series_facts( + series, + chart_type=_CHART_TYPES[_local_name(node)], + workbook_values=workbook_values, + ) + for node in chart_nodes + for series in node.findall(f"{{{_CHART_NS}}}ser") + ), + ) + + +def _paragraph(text: str) -> ParagraphBlock: + return ParagraphBlock((InlineText(text),)) + + +def chart_fact_blocks(facts: XlsxChartFacts) -> tuple[Block, ...]: + blocks: list[Block] = [ + HeadingBlock(2, (InlineText(facts.title or f"{facts.chart_type.title()} chart"),)), + _paragraph(f"Chart type: {facts.chart_type}"), + ] + if facts.axis_titles: + blocks.append(_paragraph(f"Axis titles: {'; '.join(facts.axis_titles)}")) + if facts.axis_labels: + blocks.append(_paragraph(f"Axis label positions: {'; '.join(facts.axis_labels)}")) + if facts.data_labels: + blocks.append(_paragraph(f"Data labels: {'; '.join(facts.data_labels)}")) + series_names = tuple(dict.fromkeys(series.name for series in facts.series)) + if series_names: + blocks.append(_paragraph(f"Series names: {'; '.join(series_names)}")) + formulas = tuple( + dict.fromkeys( + (*facts.formulas, *(formula for item in facts.series for formula in item.formulas)) + ) + ) + if formulas: + blocks.append(_paragraph(f"Local/formula references: {'; '.join(formulas)}")) + unresolved = tuple( + dict.fromkeys( + ( + *facts.unresolved_formulas, + *(formula for item in facts.series for formula in item.unresolved_formulas), + ) + ) + ) + if unresolved: + blocks.append( + _paragraph(f"References preserved without evaluation: {'; '.join(unresolved)}") + ) + rows: list[tuple[str, ...]] = [] + for series in facts.series: + if facts.chart_type == "scatter": + width = max(len(series.x_values), len(series.y_values)) + rows.extend( + ( + "Series", + series.name, + "X", + series.x_values[index] if index < len(series.x_values) else "", + "Y", + series.y_values[index] if index < len(series.y_values) else "", + ) + for index in range(width) + ) + else: + width = max(len(series.categories), len(series.values)) + rows.extend( + ( + "Series", + series.name, + "Category", + series.categories[index] if index < len(series.categories) else "", + "Value", + series.values[index] if index < len(series.values) else "", + ) + for index in range(width) + ) + if rows: + blocks.append(TableBlock(tuple(rows), header_rows=0)) + elif facts.series: + blocks.append(_paragraph("Chart series have no saved cache or resolvable local values.")) + else: + blocks.append(_paragraph("Chart has no readable series.")) + return tuple(blocks) + + +def _sample_pairs( + left: tuple[str, ...], + right: tuple[str, ...], +) -> tuple[tuple[str, str], ...]: + count = max(len(left), len(right)) + if count == 0: + return () + if count <= _PREVIEW_SERIES_POINTS: + indexes = range(count) + else: + indexes = tuple( + round(index * (count - 1) / (_PREVIEW_SERIES_POINTS - 1)) + for index in range(_PREVIEW_SERIES_POINTS) + ) + return tuple( + ( + left[index] if index < len(left) else "", + right[index] if index < len(right) else "", + ) + for index in indexes + ) + + +def _preview_lines(facts: XlsxChartFacts) -> tuple[str, ...]: + lines = [ + "视觉解释输入 / 非 Excel 外观还原", + f"Chart type: {facts.chart_type}", + f"Title: {facts.title or '(none)'}", + ] + if facts.axis_titles: + lines.append(f"Axis titles: {'; '.join(facts.axis_titles)}") + if facts.data_labels: + lines.append(f"Data labels: {'; '.join(facts.data_labels)}") + if facts.formulas: + lines.append(f"Chart references: {'; '.join(facts.formulas)}") + for series in facts.series: + lines.append(f"Series: {series.name}") + if facts.chart_type == "scatter": + count = max(len(series.x_values), len(series.y_values)) + pairs = _sample_pairs(series.x_values, series.y_values) + lines.append( + f"Points (sampled across {count}): " + "; ".join(f"({x}, {y})" for x, y in pairs) + ) + else: + count = max(len(series.categories), len(series.values)) + pairs = _sample_pairs(series.categories, series.values) + lines.append( + f"Values (sampled across {count}): " + + "; ".join(f"{category}={value}" for category, value in pairs) + ) + if series.formulas: + lines.append(f"References: {'; '.join(series.formulas)}") + normalized = [line[:_PREVIEW_MAX_LINE_CHARS] for line in lines[:_PREVIEW_MAX_LINES]] + if len(lines) > _PREVIEW_MAX_LINES: + normalized.append("(additional native facts omitted from preview only)") + return tuple(normalized) + + +def render_chart_semantic_preview(facts: XlsxChartFacts) -> bytes: + lines = _preview_lines(facts) + height = _PREVIEW_PADDING * 2 + _PREVIEW_LINE_HEIGHT * len(lines) + image = Image.new("RGB", (_PREVIEW_WIDTH, max(160, height)), "white") + try: + draw = ImageDraw.Draw(image) + font = ImageFont.load_default(size=18) + for index, line in enumerate(lines): + draw.text( + (_PREVIEW_PADDING, _PREVIEW_PADDING + index * _PREVIEW_LINE_HEIGHT), + line, + fill="black", + font=font, + ) + output = io.BytesIO() + image.save(output, format="PNG", optimize=False) + return output.getvalue() + finally: + image.close() + + +def _write_artifact(directory: Path, name: str, data: bytes) -> None: + directory.mkdir(parents=True, exist_ok=True) + root = directory.resolve() + path = directory / name + if path.resolve().parent != root: + raise ValueError("XLSX visual artifact escapes the artifact directory") + if path.exists(): + if path.read_bytes() != data: + raise CorruptDocumentError("XLSX visual artifact digest collision") + return + path.write_bytes(data) + + +def _chart_occurrence( + archive: ZipFile, + relationship: Any, + *, + anchor: str, + metadata: tuple[str | None, str | None, str | None], + workbook_values: dict[str, dict[str, str]], + artifact_dir: Path | None, + facts_cache: dict[str, XlsxChartFacts], +) -> tuple[XlsxVisualOccurrence, tuple[str, ...]]: + facts = facts_cache.get(relationship.target) + if facts is None: + root = _safe_root(archive, relationship.target, message="XLSX chart part is corrupt") + facts = _chart_facts(root, workbook_values) + facts_cache[relationship.target] = facts + blocks = chart_fact_blocks(facts) + artifact_name: str | None = None + digest: str | None = None + if artifact_dir is not None: + preview = render_chart_semantic_preview(facts) + digest = hashlib.sha256(preview).hexdigest() + artifact_name = f"xlsx-chart-{digest}.png" + _write_artifact(artifact_dir, artifact_name, preview) + object_name, alt_text, title = metadata + warnings = tuple( + dict.fromkeys( + ( + *facts.unresolved_formulas, + *(formula for series in facts.series for formula in series.unresolved_formulas), + ) + ) + ) + return ( + XlsxVisualOccurrence( + kind="chart", + anchor=anchor, + blocks=blocks, + artifact_name=artifact_name, + content_sha256=digest, + alt_text=alt_text, + object_name=object_name, + title=title, + ), + warnings, + ) + + +def _image_occurrence( + archive: ZipFile, + relationship: Any, + *, + anchor: str, + metadata: tuple[str | None, str | None, str | None], + artifact_dir: Path | None, +) -> XlsxVisualOccurrence: + try: + data = archive.read(relationship.target) + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX embedded image part is corrupt") from error + digest = hashlib.sha256(data).hexdigest() + suffix = Path(relationship.target).suffix.lower() + if _SAFE_ARTIFACT_SUFFIX.fullmatch(suffix) is None: + suffix = ".bin" + artifact_name = f"xlsx-media-{digest}{suffix}" + if artifact_dir is not None: + _write_artifact(artifact_dir, artifact_name, data) + object_name, alt_text, title = metadata + description = tuple( + f"{label}: {value}" + for label, value in ( + ("Image name", object_name), + ("Image description", alt_text), + ("Image title", title), + ) + if value + ) + blocks: tuple[Block, ...] = ( + HeadingBlock(3, (InlineText("Embedded image"),)), + *(_paragraph(value) for value in description), + ) + return XlsxVisualOccurrence( + kind="image", + anchor=anchor, + blocks=blocks, + artifact_name=artifact_name if artifact_dir is not None else None, + content_sha256=digest if artifact_dir is not None else None, + alt_text=alt_text, + object_name=object_name, + title=title, + ) + + +def _sheet_drawings( + archive: ZipFile, + infos: dict[str, ZipInfo], + sheet: XlsxPreflightSheet, +) -> tuple[str, ...]: + root = _safe_root(archive, sheet.part_name, message="XLSX sheet part is corrupt") + relationships = _relationship_index(archive, infos, sheet.part_name) + targets: list[str] = [] + for node in root.iter(f"{{{_SPREADSHEET_NS}}}drawing"): + relationship = _relationship( + relationships, + node.get(_RELATIONSHIP_ID), + expected_type=_DRAWING_RELATIONSHIP, + ) + targets.append(relationship.target) + return tuple(targets) + + +def _drawing_occurrences( + archive: ZipFile, + infos: dict[str, ZipInfo], + drawing_part: str, + *, + workbook_values: dict[str, dict[str, str]], + artifact_dir: Path | None, + chart_facts_cache: dict[str, XlsxChartFacts], +) -> tuple[tuple[XlsxVisualOccurrence, ...], tuple[_OccurrenceWarning, ...]]: + root = _safe_root(archive, drawing_part, message="XLSX drawing part is corrupt") + if root.tag != f"{{{_DRAWING_NS}}}wsDr": + raise CorruptDocumentError("XLSX drawing namespace is invalid") + relationships = _relationship_index(archive, infos, drawing_part) + occurrences: list[XlsxVisualOccurrence] = [] + warnings: list[_OccurrenceWarning] = [] + for anchor_node in root: + if _local_name(anchor_node) not in {"oneCellAnchor", "twoCellAnchor", "absoluteAnchor"}: + continue + anchor = _drawing_anchor(anchor_node) + metadata = _metadata(anchor_node) + chart_node = next(anchor_node.iter(f"{{{_CHART_NS}}}chart"), None) + image_node = next(anchor_node.iter(f"{{{_DRAWING_MAIN_NS}}}blip"), None) + if chart_node is not None: + relationship = _relationship( + relationships, + chart_node.get(_RELATIONSHIP_ID), + expected_type=_CHART_RELATIONSHIP, + ) + try: + occurrence, formulas = _chart_occurrence( + archive, + relationship, + anchor=anchor, + metadata=metadata, + workbook_values=workbook_values, + artifact_dir=artifact_dir, + facts_cache=chart_facts_cache, + ) + except _UnsupportedChartType: + warnings.append( + _OccurrenceWarning( + "xlsx_unsupported_object", + anchor, + "unsupported chart type was skipped without guessing", + ) + ) + continue + occurrences.append(occurrence) + warnings.extend( + _OccurrenceWarning( + "xlsx_external_reference", + anchor, + f"chart reference was preserved without access: {formula}", + ) + for formula in formulas + ) + elif image_node is not None: + relationship = _relationship( + relationships, + image_node.get(_RELATIONSHIP_EMBED), + expected_type=_IMAGE_RELATIONSHIP, + ) + occurrences.append( + _image_occurrence( + archive, + relationship, + anchor=anchor, + metadata=metadata, + artifact_dir=artifact_dir, + ) + ) + return tuple(occurrences), tuple(warnings) + + +def _visual_warning( + code: str, + sheet: XlsxPreflightSheet, + anchor: str, + detail: str, +) -> WarningRecord: + return WarningRecord( + code=code, + message=f"{sheet.name}!{anchor}: {detail}", + ) + + +def read_xlsx_visual_objects( + path: Path, + preflight: XlsxPreflight, + workbook_values: dict[str, dict[str, str]], + *, + artifact_dir: Path | None, +) -> XlsxVisualObjects: + if preflight.usage.drawing_objects > MAX_DRAWING_OBJECTS: + raise LimitExceededError("XLSX exceeds the drawing object limit") + if preflight.usage.chart_cache_points > MAX_CHART_CACHE_POINTS: + raise LimitExceededError("XLSX exceeds the chart cache point limit") + by_sheet: list[tuple[XlsxVisualOccurrence, ...]] = [] + warnings: list[WarningRecord] = [] + try: + with ZipFile(path) as archive: + infos = {info.filename: info for info in archive.infolist()} + chart_facts_cache: dict[str, XlsxChartFacts] = {} + for sheet in preflight.sheets: + occurrences: list[XlsxVisualOccurrence] = [] + for drawing_part in _sheet_drawings(archive, infos, sheet): + drawing_items, drawing_warnings = _drawing_occurrences( + archive, + infos, + drawing_part, + workbook_values=workbook_values, + artifact_dir=artifact_dir, + chart_facts_cache=chart_facts_cache, + ) + occurrences.extend(drawing_items) + for warning in drawing_warnings: + warnings.append( + _visual_warning( + warning.code, + sheet, + warning.anchor, + warning.detail, + ) + ) + if artifact_dir is None: + warnings.extend( + _visual_warning( + "xlsx_visual_artifact_unavailable", + sheet, + occurrence.anchor, + f"{occurrence.kind} visual artifact requires an explicit directory", + ) + for occurrence in occurrences + ) + by_sheet.append(tuple(occurrences)) + except BadZipFile as error: + raise CorruptDocumentError("XLSX package is corrupt") from error + except OSError as error: + raise CorruptDocumentError("XLSX package could not be read") from error + return XlsxVisualObjects(tuple(by_sheet), tuple(warnings)) + + +def _artifact_path(artifact_dir: Path, artifact_name: str) -> Path: + root = artifact_dir.resolve() + path = artifact_dir / artifact_name + if path.resolve().parent != root: + raise ValueError("XLSX visual artifact escapes the artifact directory") + return path + + +def build_xlsx_visual_requests( + document: XlsxDocument, + artifact_dir: Path, +) -> tuple[XlsxVisualRequest, ...]: + requests: list[XlsxVisualRequest] = [] + seen: set[tuple[str, str]] = set() + for sheet in document.sheets: + for slot in sorted(sheet.slots, key=lambda item: item.source_index): + if not isinstance(slot, XlsxImageSlot | XlsxChartSlot): + continue + slot_kind = "chart" if isinstance(slot, XlsxChartSlot) else "image" + key = (slot_kind, slot.content_sha256) + if key in seen: + continue + seen.add(key) + requests.append( + XlsxVisualRequest( + digest=slot.content_sha256, + image_path=_artifact_path(artifact_dir, slot.artifact_name), + prompt=( + XLSX_CHART_VISION_PROMPT + if isinstance(slot, XlsxChartSlot) + else XLSX_IMAGE_VISION_PROMPT + ), + source_index=len(requests), + kind=VisionRequestKind.PROSE, + ) + ) + return tuple(requests) + + +def prepare_xlsx_visual_artifact( + slot: XlsxImageSlot | XlsxChartSlot, + artifact_dir: Path, + output_directory: Path, + output_stem: str, +) -> PreparedImage: + return prepare_image( + _artifact_path(artifact_dir, slot.artifact_name), + output_directory, + output_stem, + slot.artifact_name, + "embedded", + None, + ) diff --git a/src/opendocs/parsers/xlsx/merge.py b/src/opendocs/parsers/xlsx/merge.py new file mode 100644 index 0000000..8800ba6 --- /dev/null +++ b/src/opendocs/parsers/xlsx/merge.py @@ -0,0 +1,204 @@ +from __future__ import annotations + +import re +from collections.abc import Mapping +from dataclasses import dataclass +from itertools import groupby + +from opendocs._models import ( + Block, + DocumentType, + HeadingBlock, + InlineText, + MarkdownBlock, + ParagraphBlock, + ParsedDocument, + TableBlock, + TextBlock, + WarningRecord, +) +from opendocs.parsers.xlsx.models import ( + XlsxChartSlot, + XlsxDocument, + XlsxImageSlot, + XlsxNativeSlot, + XlsxSheet, + XlsxSheetState, + XlsxSlot, +) +from opendocs.vision.base import VisionResult, VisionTableElement, VisionTextElement + +_A1_START = re.compile(r"^([A-Z]{1,3})([1-9][0-9]{0,6})") +_STATE_LABELS = { + XlsxSheetState.VISIBLE: "Visible", + XlsxSheetState.HIDDEN: "Hidden", + XlsxSheetState.VERY_HIDDEN: "Very Hidden", +} + + +@dataclass(frozen=True, slots=True) +class XlsxVisualOutcome: + result: VisionResult | None + warning_code: str | None = None + + def __post_init__(self) -> None: + if self.result is not None and not isinstance(self.result, VisionResult): + raise TypeError("result must be a VisionResult or None") + if self.warning_code is not None and not isinstance(self.warning_code, str): + raise TypeError("warning_code must be a str or None") + if self.warning_code == "": + raise ValueError("warning_code must not be empty") + + +def _column_number(label: str) -> int: + number = 0 + for character in label: + number = number * 26 + ord(character) - ord("A") + 1 + return number + + +def _anchor_position(anchor: str) -> tuple[int, int]: + match = _A1_START.match(anchor) + if match is None: + raise ValueError("XLSX slot anchor is invalid") + return int(match.group(2)), _column_number(match.group(1)) + + +def _slot_kind_rank(slot: XlsxSlot) -> int: + if isinstance(slot, XlsxNativeSlot): + return 0 + if isinstance(slot, XlsxChartSlot): + return 1 + return 2 + + +def _slot_sort_key(slot: XlsxSlot) -> tuple[int, int, int, int]: + row, column = _anchor_position(slot.anchor) + return row, column, _slot_kind_rank(slot), slot.source_index + + +def _is_extractor_prelude(slot: XlsxSlot, sheet_index: int) -> bool: + if not isinstance(slot, XlsxNativeSlot) or slot.source_index != 0: + return False + expected = f"" + return any( + isinstance(block, MarkdownBlock) and block.markdown == expected for block in slot.blocks + ) + + +def _sheet_prelude(sheet: XlsxSheet) -> tuple[Block, ...]: + return ( + MarkdownBlock(f""), + HeadingBlock(1, (InlineText(f"{sheet.name} ({_STATE_LABELS[sheet.state]})"),)), + ) + + +def _object_anchor(sheet: XlsxSheet, slot: XlsxImageSlot | XlsxChartSlot) -> MarkdownBlock: + return MarkdownBlock( + f"" + ) + + +def _metadata_blocks( + kind: str, + slot: XlsxImageSlot | XlsxChartSlot, +) -> tuple[Block, ...]: + labels = ( + (f"{kind} name", slot.object_name), + (f"{kind} description", slot.alt_text), + (f"{kind} title", slot.title), + ) + return ( + *((HeadingBlock(3, (InlineText("Embedded image"),)),) if kind == "Image" else ()), + *(ParagraphBlock((InlineText(f"{label}: {value}"),)) for label, value in labels if value), + ) + + +def _native_blocks(sheet: XlsxSheet, slot: XlsxSlot) -> tuple[Block, ...]: + if isinstance(slot, XlsxNativeSlot): + return slot.blocks + if isinstance(slot, XlsxChartSlot): + return ( + _object_anchor(sheet, slot), + *_metadata_blocks("Chart", slot), + *slot.blocks, + ) + return (_object_anchor(sheet, slot), *_metadata_blocks("Image", slot)) + + +def _vision_blocks(result: VisionResult) -> tuple[Block, ...]: + blocks: list[Block] = [HeadingBlock(3, (InlineText("Visual interpretation"),))] + for element in sorted(result.elements, key=lambda item: item.source_index): + if isinstance(element, VisionTextElement): + if element.text.strip(): + blocks.append(TextBlock(element.text.strip())) + elif isinstance(element, VisionTableElement): + blocks.append(TableBlock(element.grid, element.header_rows)) + return tuple(blocks) + + +def _visual_blocks( + slot: XlsxSlot, + visual_outcomes: Mapping[str, XlsxVisualOutcome], +) -> tuple[Block, ...]: + if not isinstance(slot, XlsxImageSlot | XlsxChartSlot): + return () + outcome = visual_outcomes.get(slot.content_sha256) + if outcome is None or outcome.result is None: + return () + return _vision_blocks(outcome.result) + + +def _visual_warning( + sheet: XlsxSheet, + slot: XlsxImageSlot | XlsxChartSlot, + code: str, +) -> WarningRecord: + kind = "chart" if isinstance(slot, XlsxChartSlot) else "image" + return WarningRecord( + code=code, + message=f"{sheet.name}!{slot.anchor}: {kind} visual interpretation was not completed", + ) + + +def merge_xlsx_document( + document: XlsxDocument, + visual_outcomes: Mapping[str, XlsxVisualOutcome], +) -> ParsedDocument: + if not isinstance(document, XlsxDocument): + raise TypeError("document must be an XlsxDocument") + for digest, outcome in visual_outcomes.items(): + if not isinstance(digest, str): + raise TypeError("visual outcome keys must be strings") + if not isinstance(outcome, XlsxVisualOutcome): + raise TypeError("visual outcomes must contain XlsxVisualOutcome values") + + blocks: list[Block] = [] + warnings = list(document.warnings) + for sheet in document.sheets: + blocks.extend(_sheet_prelude(sheet)) + slots = tuple( + sorted( + ( + slot + for slot in sheet.slots + if not _is_extractor_prelude(slot, sheet.sheet_index) + ), + key=_slot_sort_key, + ) + ) + for _position, positioned_slots in groupby( + slots, + key=lambda item: _anchor_position(item.anchor), + ): + group = tuple(positioned_slots) + for slot in group: + blocks.extend(_native_blocks(sheet, slot)) + for slot in group: + blocks.extend(_visual_blocks(slot, visual_outcomes)) + if isinstance(slot, XlsxImageSlot | XlsxChartSlot): + outcome = visual_outcomes.get(slot.content_sha256) + if outcome is not None and outcome.warning_code is not None: + warnings.append(_visual_warning(sheet, slot, outcome.warning_code)) + return ParsedDocument(DocumentType.XLSX, tuple(blocks), tuple(warnings)) diff --git a/src/opendocs/parsers/xlsx/models.py b/src/opendocs/parsers/xlsx/models.py new file mode 100644 index 0000000..d61c5ce --- /dev/null +++ b/src/opendocs/parsers/xlsx/models.py @@ -0,0 +1,589 @@ +from __future__ import annotations + +import re +from dataclasses import dataclass, fields, is_dataclass +from enum import Enum, StrEnum +from functools import cache +from pathlib import Path +from typing import Any, TypeAlias, cast, get_type_hints + +import opendocs._models as core_models +from opendocs._models import Block, WarningRecord +from opendocs.errors import LimitExceededError + +MAX_NATIVE_WIRE_ESTIMATE = 8 * 1024 * 1024 + +_SHA256_RE = re.compile(r"^[0-9a-f]{64}$") +_A1_RE = re.compile(r"^([A-Z]{1,3})([1-9][0-9]{0,6})(?::([A-Z]{1,3})([1-9][0-9]{0,6}))?$") +_DATACLASS_TYPE = "__xlsx_dataclass__" +_XLSX_MAX_COLUMN = 16_384 +_XLSX_MAX_ROW = 1_048_576 +_WIRE_NODE_OVERHEAD = 32 + + +def _require_int(name: str, value: object) -> int: + if isinstance(value, bool) or not isinstance(value, int): + raise TypeError(f"{name} must be an int") + return value + + +def _require_string(name: str, value: object) -> str: + if not isinstance(value, str): + raise TypeError(f"{name} must be a str") + return value + + +def _require_non_empty_string(name: str, value: object) -> str: + normalized = _require_string(name, value) + if not normalized: + raise ValueError(f"{name} must not be empty") + if any(ord(character) < 32 or ord(character) == 127 for character in normalized): + raise ValueError(f"{name} must not contain control characters") + return normalized + + +def _require_sheet_name(value: object) -> str: + name = _require_non_empty_string("name", value) + if len(name) > 31 or any(character in "[]:*?/\\" for character in name): + raise ValueError("name must be a valid XLSX sheet name") + return name + + +def _require_optional_string(name: str, value: object) -> str | None: + if value is None: + return None + return _require_string(name, value) + + +def _require_tuple(name: str, value: object) -> tuple[object, ...]: + if not isinstance(value, tuple): + raise TypeError(f"{name} must be a tuple") + return value + + +def _column_number(label: str) -> int: + number = 0 + for character in label: + number = number * 26 + ord(character) - ord("A") + 1 + return number + + +def _require_anchor(name: str, value: object) -> str: + anchor = _require_string(name, value) + match = _A1_RE.fullmatch(anchor) + if match is None: + raise ValueError(f"{name} must be a canonical A1 anchor or range") + start_column = _column_number(match.group(1)) + start_row = int(match.group(2)) + end_column = _column_number(match.group(3) or match.group(1)) + end_row = int(match.group(4) or match.group(2)) + if ( + start_column > _XLSX_MAX_COLUMN + or end_column > _XLSX_MAX_COLUMN + or start_row > _XLSX_MAX_ROW + or end_row > _XLSX_MAX_ROW + or end_column < start_column + or end_row < start_row + ): + raise ValueError(f"{name} must be a canonical A1 anchor or range") + return anchor + + +def _require_basename(name: str, value: object) -> str: + artifact_name = _require_string(name, value) + candidate = Path(artifact_name) + windows_stem = artifact_name.split(".", 1)[0].rstrip(" ").upper() + windows_reserved = windows_stem in { + "CON", + "PRN", + "AUX", + "NUL", + "CONIN$", + "CONOUT$", + } or ( + len(windows_stem) == 4 + and windows_stem[:3] in {"COM", "LPT"} + and windows_stem[3] in "123456789¹²³" + ) + forbidden = '<>:"/\\|?*' + if ( + not artifact_name + or artifact_name[-1] in {" ", "."} + or any( + ord(character) < 32 or ord(character) == 127 or character in forbidden + for character in artifact_name + ) + or candidate.is_absolute() + or candidate.name != artifact_name + or artifact_name in {".", ".."} + or windows_reserved + ): + raise ValueError(f"{name} must be a portable non-empty basename") + return artifact_name + + +def _require_sha256(name: str, value: object) -> str: + digest = _require_string(name, value) + if not _SHA256_RE.fullmatch(digest): + raise ValueError(f"{name} must be a lowercase 64-character SHA-256 hex digest") + return digest + + +def _require_source_index(name: str, value: object) -> int: + source_index = _require_int(name, value) + if source_index < 0: + raise ValueError(f"{name} must be greater than or equal to zero") + return source_index + + +def _block_class_names() -> tuple[str, ...]: + return ( + "TextBlock", + "MarkdownBlock", + "TableBlock", + "InlineText", + "InlineLink", + "ParagraphBlock", + "HeadingBlock", + "ListItemBlock", + "SpannedTableCell", + "SpannedTableBlock", + ) + + +_MODEL_REGISTRY: dict[str, type[Any]] = {} +for _name in (*_block_class_names(), "WarningRecord"): + _class = getattr(core_models, _name, None) + if isinstance(_class, type) and is_dataclass(_class): + _MODEL_REGISTRY[_name] = _class + +_KNOWN_BLOCK_TYPES = tuple( + value for name, value in _MODEL_REGISTRY.items() if name != "WarningRecord" +) + + +def _require_blocks(value: object) -> tuple[Block, ...]: + blocks = _require_tuple("blocks", value) + if not blocks: + raise ValueError("blocks must contain at least one block") + for index, block in enumerate(blocks): + if not isinstance(block, _KNOWN_BLOCK_TYPES): + raise TypeError(f"blocks[{index}] is not a supported block type") + return cast(tuple[Block, ...], blocks) + + +class XlsxSheetKind(StrEnum): + WORKSHEET = "worksheet" + CHARTSHEET = "chartsheet" + + +class XlsxSheetState(StrEnum): + VISIBLE = "visible" + HIDDEN = "hidden" + VERY_HIDDEN = "veryHidden" + + +@dataclass(frozen=True, slots=True) +class XlsxNativeSlot: + source_index: int + anchor: str + blocks: tuple[Block, ...] + + def __post_init__(self) -> None: + object.__setattr__( + self, "source_index", _require_source_index("source_index", self.source_index) + ) + object.__setattr__(self, "anchor", _require_anchor("anchor", self.anchor)) + object.__setattr__(self, "blocks", _require_blocks(self.blocks)) + + +@dataclass(frozen=True, slots=True) +class XlsxImageSlot: + source_index: int + anchor: str + artifact_name: str + content_sha256: str + alt_text: str | None = None + object_name: str | None = None + title: str | None = None + + def __post_init__(self) -> None: + object.__setattr__( + self, "source_index", _require_source_index("source_index", self.source_index) + ) + object.__setattr__(self, "anchor", _require_anchor("anchor", self.anchor)) + object.__setattr__( + self, "artifact_name", _require_basename("artifact_name", self.artifact_name) + ) + object.__setattr__( + self, + "content_sha256", + _require_sha256("content_sha256", self.content_sha256), + ) + object.__setattr__(self, "alt_text", _require_optional_string("alt_text", self.alt_text)) + object.__setattr__( + self, + "object_name", + _require_optional_string("object_name", self.object_name), + ) + object.__setattr__(self, "title", _require_optional_string("title", self.title)) + + +@dataclass(frozen=True, slots=True) +class XlsxChartSlot: + source_index: int + anchor: str + artifact_name: str + content_sha256: str + blocks: tuple[Block, ...] + alt_text: str | None = None + object_name: str | None = None + title: str | None = None + + def __post_init__(self) -> None: + object.__setattr__( + self, "source_index", _require_source_index("source_index", self.source_index) + ) + object.__setattr__(self, "anchor", _require_anchor("anchor", self.anchor)) + object.__setattr__( + self, "artifact_name", _require_basename("artifact_name", self.artifact_name) + ) + object.__setattr__( + self, + "content_sha256", + _require_sha256("content_sha256", self.content_sha256), + ) + object.__setattr__(self, "blocks", _require_blocks(self.blocks)) + object.__setattr__(self, "alt_text", _require_optional_string("alt_text", self.alt_text)) + object.__setattr__( + self, + "object_name", + _require_optional_string("object_name", self.object_name), + ) + object.__setattr__(self, "title", _require_optional_string("title", self.title)) + + +XlsxSlot: TypeAlias = XlsxNativeSlot | XlsxImageSlot | XlsxChartSlot + + +@dataclass(frozen=True, slots=True) +class XlsxSheet: + sheet_index: int + name: str + kind: XlsxSheetKind + state: XlsxSheetState + slots: tuple[XlsxSlot, ...] + + def __post_init__(self) -> None: + sheet_index = _require_int("sheet_index", self.sheet_index) + if sheet_index <= 0: + raise ValueError("sheet_index must be greater than zero") + if not isinstance(self.kind, XlsxSheetKind): + raise TypeError("kind must be an XlsxSheetKind") + if not isinstance(self.state, XlsxSheetState): + raise TypeError("state must be an XlsxSheetState") + slots = _require_tuple("slots", self.slots) + seen: set[int] = set() + for index, slot in enumerate(slots): + if not isinstance(slot, XlsxNativeSlot | XlsxImageSlot | XlsxChartSlot): + raise TypeError(f"slots[{index}] must be an XLSX slot") + if slot.source_index in seen: + raise ValueError("XLSX sheet source indexes must be unique") + seen.add(slot.source_index) + object.__setattr__(self, "sheet_index", sheet_index) + object.__setattr__(self, "name", _require_sheet_name(self.name)) + object.__setattr__(self, "slots", cast(tuple[XlsxSlot, ...], slots)) + + +@dataclass(frozen=True, slots=True) +class XlsxDocument: + sheets: tuple[XlsxSheet, ...] + warnings: tuple[WarningRecord, ...] = () + + def __post_init__(self) -> None: + sheets = _require_tuple("sheets", self.sheets) + seen: set[int] = set() + for position, sheet in enumerate(sheets, start=1): + if not isinstance(sheet, XlsxSheet): + raise TypeError(f"sheets[{position - 1}] must be an XlsxSheet") + if sheet.sheet_index in seen or sheet.sheet_index != position: + raise ValueError("XLSX sheet indexes must be unique and preserve source order") + seen.add(sheet.sheet_index) + warnings = _require_tuple("warnings", self.warnings) + for index, warning in enumerate(warnings): + if not isinstance(warning, WarningRecord): + raise TypeError(f"warnings[{index}] must be a WarningRecord") + object.__setattr__(self, "sheets", cast(tuple[XlsxSheet, ...], sheets)) + object.__setattr__(self, "warnings", cast(tuple[WarningRecord, ...], warnings)) + + +def _dataclass_to_wire(value: object) -> dict[str, object]: + return { + _DATACLASS_TYPE: type(value).__name__, + "fields": { + field.name: _value_to_wire(getattr(value, field.name)) + for field in fields(cast(Any, value)) + }, + } + + +def _value_to_wire(value: object) -> object: + if value is None or isinstance(value, bool | int | float | str): + return value + if isinstance(value, Enum): + return value.value + if isinstance(value, tuple): + return tuple(_value_to_wire(item) for item in value) + if is_dataclass(value) and type(value).__name__ in _MODEL_REGISTRY: + return _dataclass_to_wire(value) + raise TypeError(f"XLSX wire value type is not supported: {type(value).__name__}") + + +@cache +def _resolved_field_types(cls: type[Any]) -> dict[str, Any]: + return get_type_hints(cls) + + +def _restore_enum_field(value: object, field_type: object) -> object: + if not isinstance(field_type, type) or not issubclass(field_type, Enum): + return value + if isinstance(value, field_type): + return value + try: + return field_type(value) + except (TypeError, ValueError): + return value + + +def _decode_dataclass(value: dict[str, object]) -> object: + if set(value) != {_DATACLASS_TYPE, "fields"}: + raise ValueError("XLSX dataclass wire is invalid") + class_name = value[_DATACLASS_TYPE] + fields_value = value["fields"] + if not isinstance(class_name, str) or not isinstance(fields_value, dict): + raise ValueError("XLSX dataclass wire is invalid") + cls = _MODEL_REGISTRY.get(class_name) + if cls is None: + raise ValueError("XLSX dataclass type is invalid") + field_names = {field.name for field in fields(cast(Any, cls))} + typed_fields = cast(dict[str, object], fields_value) + if set(typed_fields) != field_names: + raise ValueError("XLSX dataclass wire is invalid") + field_types = _resolved_field_types(cls) + kwargs: dict[str, Any] = { + name: _restore_enum_field(_value_from_wire(item), field_types.get(name)) + for name, item in typed_fields.items() + } + return cls(**kwargs) + + +def _value_from_wire(value: object) -> object: + if value is None or isinstance(value, bool | int | float | str): + return value + if isinstance(value, tuple): + return tuple(_value_from_wire(item) for item in value) + if isinstance(value, dict) and _DATACLASS_TYPE in value: + return _decode_dataclass(cast(dict[str, object], value)) + raise ValueError("XLSX wire value is invalid") + + +def _slot_to_wire(slot: XlsxSlot) -> dict[str, object]: + common: dict[str, object] = { + "source_index": slot.source_index, + "anchor": slot.anchor, + } + if isinstance(slot, XlsxNativeSlot): + return { + "type": "xlsx_native_slot", + **common, + "blocks": tuple(_value_to_wire(block) for block in slot.blocks), + } + if isinstance(slot, XlsxImageSlot): + return { + "type": "xlsx_image_slot", + **common, + "artifact_name": slot.artifact_name, + "content_sha256": slot.content_sha256, + "alt_text": slot.alt_text, + "object_name": slot.object_name, + "title": slot.title, + } + return { + "type": "xlsx_chart_slot", + **common, + "artifact_name": slot.artifact_name, + "content_sha256": slot.content_sha256, + "blocks": tuple(_value_to_wire(block) for block in slot.blocks), + "alt_text": slot.alt_text, + "object_name": slot.object_name, + "title": slot.title, + } + + +def _slot_from_wire(value: object) -> XlsxSlot: + if not isinstance(value, dict) or not isinstance(value.get("type"), str): + raise ValueError("XLSX slot wire is invalid") + payload = cast(dict[str, object], value) + kind = payload["type"] + if kind == "xlsx_native_slot": + if set(payload) != {"type", "source_index", "anchor", "blocks"}: + raise ValueError("XLSX native slot wire is invalid") + blocks_value = payload["blocks"] + if not isinstance(blocks_value, tuple): + raise ValueError("XLSX native slot blocks are invalid") + return XlsxNativeSlot( + source_index=_require_source_index("source_index", payload["source_index"]), + anchor=_require_anchor("anchor", payload["anchor"]), + blocks=tuple(cast(Block, _value_from_wire(block)) for block in blocks_value), + ) + if kind == "xlsx_image_slot": + if set(payload) != { + "type", + "source_index", + "anchor", + "artifact_name", + "content_sha256", + "alt_text", + "object_name", + "title", + }: + raise ValueError("XLSX image slot wire is invalid") + return XlsxImageSlot( + source_index=_require_source_index("source_index", payload["source_index"]), + anchor=_require_anchor("anchor", payload["anchor"]), + artifact_name=_require_basename("artifact_name", payload["artifact_name"]), + content_sha256=_require_sha256("content_sha256", payload["content_sha256"]), + alt_text=_require_optional_string("alt_text", payload["alt_text"]), + object_name=_require_optional_string("object_name", payload["object_name"]), + title=_require_optional_string("title", payload["title"]), + ) + if kind == "xlsx_chart_slot": + if set(payload) != { + "type", + "source_index", + "anchor", + "artifact_name", + "content_sha256", + "blocks", + "alt_text", + "object_name", + "title", + }: + raise ValueError("XLSX chart slot wire is invalid") + blocks_value = payload["blocks"] + if not isinstance(blocks_value, tuple): + raise ValueError("XLSX chart slot blocks are invalid") + return XlsxChartSlot( + source_index=_require_source_index("source_index", payload["source_index"]), + anchor=_require_anchor("anchor", payload["anchor"]), + artifact_name=_require_basename("artifact_name", payload["artifact_name"]), + content_sha256=_require_sha256("content_sha256", payload["content_sha256"]), + blocks=tuple(cast(Block, _value_from_wire(block)) for block in blocks_value), + alt_text=_require_optional_string("alt_text", payload["alt_text"]), + object_name=_require_optional_string("object_name", payload["object_name"]), + title=_require_optional_string("title", payload["title"]), + ) + raise ValueError("XLSX slot type is invalid") + + +def _primitive_wire_estimate(value: object) -> int: + if value is None or isinstance(value, bool | int | float): + return _WIRE_NODE_OVERHEAD + if isinstance(value, str): + return _WIRE_NODE_OVERHEAD + len(value) * 4 + if isinstance(value, Enum): + return _primitive_wire_estimate(value.value) + if isinstance(value, tuple): + return _WIRE_NODE_OVERHEAD + sum(_primitive_wire_estimate(item) for item in value) + if is_dataclass(value) and type(value).__name__ in _MODEL_REGISTRY: + estimate = _WIRE_NODE_OVERHEAD + estimate += _primitive_wire_estimate(_DATACLASS_TYPE) + estimate += _primitive_wire_estimate(type(value).__name__) + estimate += _primitive_wire_estimate("fields") + _WIRE_NODE_OVERHEAD + for field in fields(cast(Any, value)): + estimate += _primitive_wire_estimate(field.name) + estimate += _primitive_wire_estimate(getattr(value, field.name)) + return estimate + raise TypeError(f"XLSX wire value type is not supported: {type(value).__name__}") + + +def _wire_estimate(document: XlsxDocument) -> int: + estimate = 512 + for sheet in document.sheets: + estimate += 512 + len(sheet.name) * 4 + for slot in sheet.slots: + estimate += 512 + len(slot.anchor) * 4 + if isinstance(slot, XlsxNativeSlot | XlsxChartSlot): + estimate += sum(_primitive_wire_estimate(block) for block in slot.blocks) + if isinstance(slot, XlsxImageSlot | XlsxChartSlot): + estimate += 512 + len(slot.artifact_name) * 4 + len(slot.content_sha256) * 4 + if isinstance(slot, XlsxImageSlot | XlsxChartSlot): + estimate += sum( + len(value) * 4 + for value in (slot.alt_text, slot.object_name, slot.title) + if value is not None + ) + estimate += sum(_primitive_wire_estimate(warning) for warning in document.warnings) + return estimate + + +def document_to_wire(document: XlsxDocument) -> dict[str, object]: + if not isinstance(document, XlsxDocument): + raise TypeError("document must be an XlsxDocument") + if _wire_estimate(document) > MAX_NATIVE_WIRE_ESTIMATE: + raise LimitExceededError("XLSX native document exceeds the inline result budget") + return { + "type": "xlsx_document", + "sheets": tuple( + { + "type": "xlsx_sheet", + "sheet_index": sheet.sheet_index, + "name": sheet.name, + "kind": sheet.kind.value, + "state": sheet.state.value, + "slots": tuple(_slot_to_wire(slot) for slot in sheet.slots), + } + for sheet in document.sheets + ), + "warnings": tuple(_value_to_wire(warning) for warning in document.warnings), + } + + +def document_from_wire(value: object) -> XlsxDocument: + if not isinstance(value, dict) or value.get("type") != "xlsx_document": + raise ValueError("XLSX document wire is invalid") + payload = cast(dict[str, object], value) + if set(payload) != {"type", "sheets", "warnings"}: + raise ValueError("XLSX document wire is invalid") + sheets_value = payload["sheets"] + if not isinstance(sheets_value, tuple): + raise ValueError("XLSX document sheets are invalid") + sheets: list[XlsxSheet] = [] + for sheet_value in sheets_value: + if not isinstance(sheet_value, dict) or sheet_value.get("type") != "xlsx_sheet": + raise ValueError("XLSX sheet wire is invalid") + sheet_payload = cast(dict[str, object], sheet_value) + if set(sheet_payload) != {"type", "sheet_index", "name", "kind", "state", "slots"}: + raise ValueError("XLSX sheet wire is invalid") + slots_value = sheet_payload["slots"] + if not isinstance(slots_value, tuple): + raise ValueError("XLSX sheet slots are invalid") + try: + kind = XlsxSheetKind(sheet_payload["kind"]) + state = XlsxSheetState(sheet_payload["state"]) + except (TypeError, ValueError) as error: + raise ValueError("XLSX sheet kind or state is invalid") from error + sheets.append( + XlsxSheet( + sheet_index=_require_int("sheet_index", sheet_payload["sheet_index"]), + name=_require_sheet_name(sheet_payload["name"]), + kind=kind, + state=state, + slots=tuple(_slot_from_wire(slot) for slot in slots_value), + ) + ) + warnings_value = payload["warnings"] + if not isinstance(warnings_value, tuple): + raise ValueError("XLSX document warnings are invalid") + warnings = tuple(cast(WarningRecord, _value_from_wire(item)) for item in warnings_value) + return XlsxDocument(sheets=tuple(sheets), warnings=warnings) diff --git a/src/opendocs/parsers/xlsx/parser.py b/src/opendocs/parsers/xlsx/parser.py new file mode 100644 index 0000000..f1b3973 --- /dev/null +++ b/src/opendocs/parsers/xlsx/parser.py @@ -0,0 +1,311 @@ +from __future__ import annotations + +import asyncio +from contextlib import suppress +from dataclasses import dataclass +from pathlib import Path +from typing import cast + +from opendocs._models import ParsedDocument +from opendocs._runtime import ParserRuntime +from opendocs.errors import DocumentTimeoutError, RuntimeDependencyError +from opendocs.options import ParseOptions, VisionConfig +from opendocs.parsers.xlsx.extract import extract_xlsx +from opendocs.parsers.xlsx.media import ( + XlsxVisualRequest, + build_xlsx_visual_requests, + prepare_xlsx_visual_artifact, +) +from opendocs.parsers.xlsx.merge import XlsxVisualOutcome, merge_xlsx_document +from opendocs.parsers.xlsx.models import ( + XlsxChartSlot, + XlsxDocument, + XlsxImageSlot, + document_from_wire, + document_to_wire, +) +from opendocs.parsers.xlsx.preflight import preflight_xlsx +from opendocs.source import ResolvedSource +from opendocs.vision.base import VisionClient, VisionRequest, VisionResult, VisionTextElement +from opendocs.vision.images import ( + PreparedImage, + merge_tiled_results, + prepared_paths, + tile_prompt, +) + + +def _cleanup_worker_artifacts(workspace_path: Path) -> None: + for pattern in ("xlsx-media-*", "xlsx-chart-*", "xlsx-prepared-*"): + for artifact in workspace_path.glob(pattern): + with suppress(OSError): + artifact.unlink(missing_ok=True) + + +def _extract_xlsx_to_wire(path: Path, workspace_path: Path) -> dict[str, object]: + preflight = preflight_xlsx(path) + try: + document = extract_xlsx(path, preflight, artifact_dir=workspace_path) + return document_to_wire(document) + except BaseException: + _cleanup_worker_artifacts(workspace_path) + raise + + +def _artifact_slot_to_wire(slot: XlsxImageSlot | XlsxChartSlot) -> dict[str, object]: + return { + "type": "xlsx_visual_artifact", + "source_index": slot.source_index, + "anchor": slot.anchor, + "artifact_name": slot.artifact_name, + "content_sha256": slot.content_sha256, + "alt_text": slot.alt_text, + "object_name": slot.object_name, + "title": slot.title, + } + + +def _artifact_slot_from_wire(value: object) -> XlsxImageSlot: + if not isinstance(value, dict) or set(value) != { + "type", + "source_index", + "anchor", + "artifact_name", + "content_sha256", + "alt_text", + "object_name", + "title", + }: + raise ValueError("XLSX visual artifact wire is invalid") + if value.get("type") != "xlsx_visual_artifact": + raise ValueError("XLSX visual artifact wire is invalid") + payload = cast(dict[str, object], value) + return XlsxImageSlot( + source_index=cast(int, payload["source_index"]), + anchor=cast(str, payload["anchor"]), + artifact_name=cast(str, payload["artifact_name"]), + content_sha256=cast(str, payload["content_sha256"]), + alt_text=cast(str | None, payload["alt_text"]), + object_name=cast(str | None, payload["object_name"]), + title=cast(str | None, payload["title"]), + ) + + +def _prepare_xlsx_visual_to_wire( + slot_wire: dict[str, object], + artifact_dir: Path, + output_directory: Path, + output_stem: str, +) -> PreparedImage: + return prepare_xlsx_visual_artifact( + _artifact_slot_from_wire(slot_wire), + artifact_dir, + output_directory, + output_stem, + ) + + +def _visual_slots(document: XlsxDocument) -> tuple[XlsxImageSlot | XlsxChartSlot, ...]: + return tuple( + slot + for sheet in document.sheets + for slot in sheet.slots + if isinstance(slot, XlsxImageSlot | XlsxChartSlot) + ) + + +def _unique_requests( + document: XlsxDocument, + artifact_dir: Path, +) -> tuple[XlsxVisualRequest, ...]: + unique: list[XlsxVisualRequest] = [] + seen: set[str] = set() + for request in build_xlsx_visual_requests(document, artifact_dir): + if request.digest in seen: + continue + seen.add(request.digest) + unique.append(request) + return tuple(unique) + + +def _has_visual_content(result: VisionResult) -> bool: + return any( + not isinstance(element, VisionTextElement) or bool(element.text.strip()) + for element in result.elements + ) + + +@dataclass(frozen=True, slots=True) +class _PreparedVisual: + request: XlsxVisualRequest + prepared: PreparedImage + paths: tuple[Path, ...] + + +class XlsxParser: + def __init__( + self, + runtime: ParserRuntime, + vision: VisionClient | None, + vision_config: VisionConfig | None, + *, + deadline: float | None = None, + ) -> None: + if not isinstance(runtime, ParserRuntime): + raise TypeError("runtime must be a ParserRuntime") + if vision_config is not None and not isinstance(vision_config, VisionConfig): + raise TypeError("vision_config must be a VisionConfig or None") + self._runtime = runtime + self._vision = vision + self._vision_config = vision_config + self._deadline = deadline + + async def parse( + self, + source: ResolvedSource, + *, + options: ParseOptions, + ) -> ParsedDocument: + loop = asyncio.get_running_loop() + deadline = loop.time() + float(options.timeout) + if self._deadline is not None: + deadline = min(deadline, self._deadline) + try: + async with asyncio.timeout_at(deadline): + document = await self._extract(source) + visual_outcomes = await self._visual_outcomes(document) + except asyncio.CancelledError: + raise + except TimeoutError: + raise DocumentTimeoutError("XLSX parsing exceeded the document deadline") from None + return merge_xlsx_document(document, visual_outcomes) + + async def _extract(self, source: ResolvedSource) -> XlsxDocument: + wire = await self._runtime.run_native( + _extract_xlsx_to_wire, + source.path, + self._runtime.workspace.path, + ) + try: + return document_from_wire(wire) + except (TypeError, ValueError) as error: + _cleanup_worker_artifacts(self._runtime.workspace.path) + raise RuntimeDependencyError("native XLSX worker returned invalid data") from error + + async def _visual_outcomes( + self, + document: XlsxDocument, + ) -> dict[str, XlsxVisualOutcome]: + slots = _visual_slots(document) + if not slots: + return {} + representatives: dict[str, XlsxImageSlot | XlsxChartSlot] = {} + for slot in slots: + representatives.setdefault(slot.content_sha256, slot) + artifact_dir = self._runtime.workspace.path + raw_paths = {artifact_dir / slot.artifact_name for slot in slots} + prepared_items: list[_PreparedVisual] = [] + outcomes: dict[str, XlsxVisualOutcome] = {} + try: + if self._vision is None or self._vision_config is None: + return { + digest: XlsxVisualOutcome(None, "xlsx_vision_unavailable") + for digest in representatives + } + for request in _unique_requests(document, artifact_dir): + slot = representatives[request.digest] + try: + prepared = await self._runtime.run_native( + _prepare_xlsx_visual_to_wire, + _artifact_slot_to_wire(slot), + artifact_dir, + artifact_dir, + f"xlsx-prepared-{request.source_index}", + ) + paths = prepared_paths(prepared, artifact_dir) + if bool(prepared.get("skipped")) or not paths: + outcomes[request.digest] = XlsxVisualOutcome( + None, + "xlsx_vision_failed", + ) + for path in paths: + with suppress(OSError): + path.unlink(missing_ok=True) + continue + prepared_items.append(_PreparedVisual(request, prepared, paths)) + except asyncio.CancelledError: + raise + except BaseException: + outcomes[request.digest] = XlsxVisualOutcome(None, "xlsx_vision_failed") + + analyzed = await asyncio.gather( + *(self._analyze_prepared(item) for item in prepared_items), + return_exceptions=True, + ) + for item, outcome in zip(prepared_items, analyzed, strict=True): + if isinstance(outcome, asyncio.CancelledError): + raise outcome + if isinstance(outcome, BaseException): + outcomes[item.request.digest] = XlsxVisualOutcome( + None, + "xlsx_vision_failed", + ) + else: + outcomes[item.request.digest] = outcome + return outcomes + finally: + for item in prepared_items: + for path in item.paths: + with suppress(OSError): + path.unlink(missing_ok=True) + for path in raw_paths: + with suppress(OSError): + path.unlink(missing_ok=True) + _cleanup_worker_artifacts(artifact_dir) + + async def _analyze_prepared(self, item: _PreparedVisual) -> XlsxVisualOutcome: + vision = self._vision + config = self._vision_config + if vision is None or config is None: + return XlsxVisualOutcome(None, "xlsx_vision_unavailable") + + async def analyze_tile(request: VisionRequest) -> VisionResult | BaseException | object: + try: + return await vision.analyze(request) + except asyncio.CancelledError: + raise + except BaseException as error: + return error + + requests = tuple( + VisionRequest( + path, + tile_prompt(item.request.prompt, tile_index, len(item.paths)), + item.request.source_index * 10_000 + tile_index, + item.request.kind, + ) + for tile_index, path in enumerate(item.paths) + ) + try: + async with asyncio.timeout(float(config.timeout)): + results = await asyncio.gather(*(analyze_tile(request) for request in requests)) + except asyncio.CancelledError: + raise + except TimeoutError: + return XlsxVisualOutcome(None, "xlsx_vision_timeout") + if any(isinstance(result, TimeoutError) for result in results): + return XlsxVisualOutcome(None, "xlsx_vision_timeout") + if any(not isinstance(result, VisionResult) for result in results): + return XlsxVisualOutcome(None, "xlsx_vision_failed") + typed_results = cast(tuple[VisionResult, ...], results) + if any(not _has_visual_content(result) for result in typed_results): + return XlsxVisualOutcome(None, "xlsx_vision_failed") + try: + merged = merge_tiled_results(item.prepared, typed_results) + except asyncio.CancelledError: + raise + except BaseException: + return XlsxVisualOutcome(None, "xlsx_vision_failed") + if not _has_visual_content(merged): + return XlsxVisualOutcome(None, "xlsx_vision_failed") + return XlsxVisualOutcome(merged) diff --git a/src/opendocs/parsers/xlsx/preflight.py b/src/opendocs/parsers/xlsx/preflight.py new file mode 100644 index 0000000..5f30496 --- /dev/null +++ b/src/opendocs/parsers/xlsx/preflight.py @@ -0,0 +1,2018 @@ +from __future__ import annotations + +import posixpath +import re +from collections.abc import Iterator +from dataclasses import dataclass +from pathlib import Path, PurePosixPath +from typing import IO, Any +from zipfile import BadZipFile, ZipFile, ZipInfo + +from defusedxml import ElementTree as DefusedET +from defusedxml.common import DefusedXmlException + +from opendocs._models import DocumentType +from opendocs.errors import CorruptDocumentError, LimitExceededError +from opendocs.parsers.office.package import validate_office_package +from opendocs.parsers.xlsx.models import XlsxSheetKind, XlsxSheetState + +MAX_SHEETS = 128 +MAX_DECLARED_CELLS = 2_000_000 +MAX_SERIALIZED_CELLS = 200_000 +MAX_NON_EMPTY_CELLS = 50_000 +MAX_MATERIALIZED_GRID_CELLS = 200_000 +MAX_MERGE_RANGES = 10_000 +MAX_MERGE_FOOTPRINT = 50_000 +MAX_SHARED_STRINGS = 100_000 +MAX_SHARED_STRING_CHARS = 1_000_000 +MAX_TABLES = 1_024 +MAX_TABLE_FOOTPRINT = 200_000 +MAX_TABLE_COLUMNS = 10_000 +MAX_HYPERLINKS_AND_COMMENTS = 20_000 +MAX_HYPERLINK_FOOTPRINT = 50_000 +MAX_DRAWING_OBJECTS = 256 +MAX_CHART_CACHE_POINTS = 200_000 +MAX_NATIVE_TEXT_CHARS = 1_000_000 +MAX_STYLE_RECORDS = 50_000 +MAX_NUMBER_FORMATS = 10_000 +MAX_FONTS = 10_000 +MAX_FILLS = 10_000 +MAX_BORDERS = 10_000 +MAX_CELL_STYLE_XFS = 10_000 +MAX_CELL_XFS = 10_000 +MAX_NAMED_CELL_STYLES = 1_000 +MAX_DXFS = 10_000 +MAX_TABLE_STYLES = 1_000 +MAX_CONDITIONAL_FORMATTING_RULES = 20_000 +MAX_CONDITIONAL_FORMATTING_RANGES = 10_000 +MAX_DATA_VALIDATIONS = 10_000 +MAX_DEFINED_NAMES = 10_000 +MAX_PIVOT_CACHES = 128 +MAX_PIVOT_TABLES = 128 +MAX_PIVOT_CACHE_RECORDS = 50_000 +MAX_PIVOT_ITEMS = 200_000 +MAX_CUSTOM_PROPERTIES = 1_000 +MAX_ROW_DIMENSIONS = 50_000 +MAX_COLUMN_DIMENSIONS = 10_000 +MAX_PAGE_BREAKS = 10_000 +MAX_SCENARIOS = 1_000 +MAX_SHEET_VIEWS = 256 +MAX_FILTER_ITEMS = 10_000 +MAX_WORKBOOK_VIEWS = 256 +MAX_EXTERNAL_REFERENCES = 128 +MAX_COMMENT_AUTHORS = 10_000 +MAX_XML_ELEMENTS_PER_PART = 200_000 +MAX_TOTAL_XML_ELEMENTS = 1_000_000 +MAX_PROJECTED_WIRE_BYTES = 8 * 1024 * 1024 +MAX_PROJECTED_WORKBOOK_BYTES = 96 * 1024 * 1024 + +_SPREADSHEET_NS = "http://schemas.openxmlformats.org/spreadsheetml/2006/main" +_OFFICE_REL_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" +_PACKAGE_REL_NS = "http://schemas.openxmlformats.org/package/2006/relationships" +_WORKSHEET_RELATIONSHIP = f"{_OFFICE_REL_NS}/worksheet" +_CHARTSHEET_RELATIONSHIP = f"{_OFFICE_REL_NS}/chartsheet" +_SHARED_STRINGS_RELATIONSHIP = f"{_OFFICE_REL_NS}/sharedStrings" +_STYLES_RELATIONSHIP = f"{_OFFICE_REL_NS}/styles" +_DRAWING_RELATIONSHIP = f"{_OFFICE_REL_NS}/drawing" +_CHART_RELATIONSHIP = f"{_OFFICE_REL_NS}/chart" +_IMAGE_RELATIONSHIP = f"{_OFFICE_REL_NS}/image" +_COMMENTS_RELATIONSHIP = f"{_OFFICE_REL_NS}/comments" +_TABLE_RELATIONSHIP = f"{_OFFICE_REL_NS}/table" +_HYPERLINK_RELATIONSHIP = f"{_OFFICE_REL_NS}/hyperlink" +_PIVOT_TABLE_RELATIONSHIP = f"{_OFFICE_REL_NS}/pivotTable" +_PIVOT_CACHE_DEFINITION_RELATIONSHIP = f"{_OFFICE_REL_NS}/pivotCacheDefinition" +_PIVOT_CACHE_RECORDS_RELATIONSHIP = f"{_OFFICE_REL_NS}/pivotCacheRecords" +_EXTERNAL_LINK_RELATIONSHIP = f"{_OFFICE_REL_NS}/externalLink" +_EXTERNAL_LINK_PATH_RELATIONSHIP = f"{_OFFICE_REL_NS}/externalLinkPath" +_CONNECTIONS_RELATIONSHIP = f"{_OFFICE_REL_NS}/connections" +_THREADED_REL_NS = "http://schemas.microsoft.com/office/2017/10/relationships" +_THREADED_COMMENTS_RELATIONSHIP = f"{_THREADED_REL_NS}/threadedComment" +_PERSON_RELATIONSHIP = f"{_THREADED_REL_NS}/person" +_RELATIONSHIP_ID = f"{{{_OFFICE_REL_NS}}}id" +_RELATIONSHIP_TAG = f"{{{_PACKAGE_REL_NS}}}Relationship" +_A1_RANGE_RE = re.compile(r"^([A-Z]{1,3})([1-9][0-9]{0,6})(?::([A-Z]{1,3})([1-9][0-9]{0,6}))?$") +_MAX_COLUMN = 16_384 +_MAX_ROW = 1_048_576 +_DRAWING_NS = "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing" +_DRAWING_MAIN_NS = "http://schemas.openxmlformats.org/drawingml/2006/main" +_CHART_NS = "http://schemas.openxmlformats.org/drawingml/2006/chart" +_CUSTOM_PROPERTIES_NS = "http://schemas.openxmlformats.org/officeDocument/2006/custom-properties" +_THREADED_COMMENTS_NS = "http://schemas.microsoft.com/office/spreadsheetml/2018/threadedcomments" + + +@dataclass(frozen=True, slots=True) +class XlsxUnsupportedObjectRef: + sheet_index: int + source_index: int + kind: str + relationship_id: str + target: str + + +@dataclass(frozen=True, slots=True) +class XlsxPreflightSheet: + sheet_index: int + name: str + kind: XlsxSheetKind + state: XlsxSheetState + part_name: str + declared_cells: int + serialized_cells: int + non_empty_cells: int + unsupported_objects: tuple[XlsxUnsupportedObjectRef, ...] + + +@dataclass(frozen=True, slots=True) +class XlsxPreflight: + sheets: tuple[XlsxPreflightSheet, ...] + date_1904: bool + serialized_cells: int + non_empty_cells: int + native_text_chars: int + projected_wire_bytes: int + projected_workbook_bytes: int + usage: XlsxResourceUsage + + +@dataclass(frozen=True, slots=True) +class XlsxResourceUsage: + serialized_cells: int = 0 + non_empty_cells: int = 0 + materialized_grid_cells: int = 0 + merge_ranges: int = 0 + merge_footprint: int = 0 + shared_strings: int = 0 + shared_string_chars: int = 0 + tables: int = 0 + table_footprint: int = 0 + table_columns: int = 0 + hyperlinks_and_comments: int = 0 + hyperlink_footprint: int = 0 + drawing_objects: int = 0 + chart_cache_points: int = 0 + native_text_chars: int = 0 + style_records: int = 0 + number_formats: int = 0 + fonts: int = 0 + fills: int = 0 + borders: int = 0 + cell_style_xfs: int = 0 + cell_xfs: int = 0 + named_cell_styles: int = 0 + dxfs: int = 0 + table_styles: int = 0 + conditional_formatting_rules: int = 0 + conditional_formatting_ranges: int = 0 + data_validations: int = 0 + defined_names: int = 0 + pivot_caches: int = 0 + pivot_tables: int = 0 + pivot_cache_records: int = 0 + pivot_items: int = 0 + custom_properties: int = 0 + row_dimensions: int = 0 + column_dimensions: int = 0 + page_breaks: int = 0 + scenarios: int = 0 + sheet_views: int = 0 + filter_items: int = 0 + workbook_views: int = 0 + external_references: int = 0 + comment_authors: int = 0 + xml_elements: int = 0 + + +@dataclass(slots=True) +class _ResourceUsage: + serialized_cells: int = 0 + non_empty_cells: int = 0 + materialized_grid_cells: int = 0 + merge_ranges: int = 0 + merge_footprint: int = 0 + shared_strings: int = 0 + shared_string_chars: int = 0 + tables: int = 0 + table_footprint: int = 0 + table_columns: int = 0 + hyperlinks_and_comments: int = 0 + hyperlink_footprint: int = 0 + drawing_objects: int = 0 + chart_cache_points: int = 0 + native_text_chars: int = 0 + style_records: int = 0 + number_formats: int = 0 + fonts: int = 0 + fills: int = 0 + borders: int = 0 + cell_style_xfs: int = 0 + cell_xfs: int = 0 + named_cell_styles: int = 0 + dxfs: int = 0 + table_styles: int = 0 + conditional_formatting_rules: int = 0 + conditional_formatting_ranges: int = 0 + data_validations: int = 0 + defined_names: int = 0 + pivot_caches: int = 0 + pivot_tables: int = 0 + pivot_cache_records: int = 0 + pivot_items: int = 0 + custom_properties: int = 0 + row_dimensions: int = 0 + column_dimensions: int = 0 + page_breaks: int = 0 + scenarios: int = 0 + sheet_views: int = 0 + filter_items: int = 0 + workbook_views: int = 0 + external_references: int = 0 + comment_authors: int = 0 + xml_elements: int = 0 + + def freeze(self) -> XlsxResourceUsage: + return XlsxResourceUsage( + **{name: getattr(self, name) for name in XlsxResourceUsage.__dataclass_fields__} + ) + + +@dataclass(frozen=True, slots=True) +class _Relationship: + relationship_type: str + target: str + external: bool + + +@dataclass(frozen=True, slots=True) +class _WorksheetCounts: + declared_cells: int + serialized_cells: int + non_empty_cells: int + native_text_chars: int + + +def _increment( + usage: _ResourceUsage, + field_name: str, + amount: int, + *, + limit: int, + message: str, +) -> None: + if amount < 0: + raise ValueError("XLSX resource increments must be non-negative") + value = getattr(usage, field_name) + amount + if value > limit: + raise LimitExceededError(message) + setattr(usage, field_name, value) + + +def _add_native_text(usage: _ResourceUsage, characters: int) -> None: + usage.native_text_chars += characters + if usage.native_text_chars > MAX_NATIVE_TEXT_CHARS: + raise LimitExceededError("XLSX exceeds the native text limit") + + +def _relationship_for( + relationships: dict[str, _Relationship], + relationship_id: str | None, + *, + expected_type: str, +) -> _Relationship: + relationship = relationships.get(relationship_id or "") + if relationship is None or relationship.relationship_type != expected_type: + raise CorruptDocumentError("XLSX object relationship is invalid") + return relationship + + +def _record_unsupported_object( + unsupported_objects: list[XlsxUnsupportedObjectRef], + *, + sheet_index: int, + relationship_id: str, + relationship: _Relationship, +) -> None: + unsupported_objects.append( + XlsxUnsupportedObjectRef( + sheet_index=sheet_index, + source_index=len(unsupported_objects), + kind=relationship.relationship_type.rsplit("/", 1)[-1] or "unknown", + relationship_id=relationship_id, + target=relationship.target, + ) + ) + + +def _column_number(label: str) -> int: + number = 0 + for character in label: + number = number * 26 + ord(character) - ord("A") + 1 + return number + + +def _parse_a1_range(value: str, *, message: str) -> tuple[int, int, int, int]: + match = _A1_RANGE_RE.fullmatch(value) + if match is None: + raise CorruptDocumentError(message) + start_column = _column_number(match.group(1)) + start_row = int(match.group(2)) + end_column = _column_number(match.group(3) or match.group(1)) + end_row = int(match.group(4) or match.group(2)) + if ( + start_column > _MAX_COLUMN + or end_column > _MAX_COLUMN + or start_row > _MAX_ROW + or end_row > _MAX_ROW + or end_column < start_column + or end_row < start_row + ): + raise CorruptDocumentError(message) + return start_column, start_row, end_column, end_row + + +def _area(bounds: tuple[int, int, int, int]) -> int: + start_column, start_row, end_column, end_row = bounds + return (end_column - start_column + 1) * (end_row - start_row + 1) + + +def _xml_events(stream: IO[bytes], *, message: str) -> Iterator[tuple[str, Any]]: + try: + yield from DefusedET.iterparse( + stream, + events=("start", "end"), + forbid_dtd=True, + forbid_entities=True, + forbid_external=True, + ) + except (DefusedXmlException, DefusedET.ParseError) as error: + raise CorruptDocumentError(message) from error + + +def _preflight_all_xml_parts( + archive: ZipFile, + infos: dict[str, ZipInfo], + usage: _ResourceUsage, +) -> None: + for part_name in sorted(infos): + if not ( + part_name.endswith(".xml") + or part_name.endswith(".rels") + or part_name == "[Content_Types].xml" + ): + continue + part_elements = 0 + try: + with archive.open(part_name) as stream: + for event, element in _xml_events( + stream, + message=f"XLSX XML part is corrupt: {part_name}", + ): + if event != "end": + continue + part_elements += 1 + if part_elements > MAX_XML_ELEMENTS_PER_PART: + raise LimitExceededError("XLSX XML part exceeds the element limit") + usage.xml_elements += 1 + if usage.xml_elements > MAX_TOTAL_XML_ELEMENTS: + raise LimitExceededError("XLSX exceeds the aggregate XML element limit") + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError(f"XLSX XML part is corrupt: {part_name}") from error + + +def _safe_relationship_target(source_part: str, target: str) -> str: + if not target or target.startswith(("//", "\\")) or "\\" in target: + raise CorruptDocumentError("XLSX relationship target is invalid") + package_absolute = target.startswith("/") + candidate = target[1:] if package_absolute else target + if not candidate: + raise CorruptDocumentError("XLSX relationship target is invalid") + first_parts = PurePosixPath(candidate).parts[:1] + if first_parts and ":" in first_parts[0]: + raise CorruptDocumentError("XLSX relationship target is invalid") + base = PurePosixPath(source_part).parent.as_posix() + normalized = ( + posixpath.normpath(candidate) + if package_absolute + else posixpath.normpath(posixpath.join(base, candidate)) + ) + if normalized in {"", ".", ".."} or normalized.startswith(("../", "/")): + raise CorruptDocumentError("XLSX relationship target is invalid") + return normalized + + +def _relationships_part(source_part: str) -> str: + path = PurePosixPath(source_part) + return (path.parent / "_rels" / f"{path.name}.rels").as_posix() + + +def _read_relationships( + archive: ZipFile, + infos: dict[str, ZipInfo], + source_part: str, + *, + required: bool, +) -> dict[str, _Relationship]: + relationships_part = _relationships_part(source_part) + if relationships_part not in infos: + if required: + raise CorruptDocumentError("XLSX relationships part is missing") + return {} + relationships: dict[str, _Relationship] = {} + root_seen = False + try: + with archive.open(relationships_part) as stream: + for event, element in _xml_events(stream, message="XLSX relationships part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_PACKAGE_REL_NS}}}Relationships": + raise CorruptDocumentError("XLSX relationships namespace is invalid") + if ( + event == "end" + and isinstance(element.tag, str) + and element.tag.rsplit("}", 1)[-1] == "Relationship" + and element.tag != _RELATIONSHIP_TAG + ): + raise CorruptDocumentError("XLSX relationships namespace is invalid") + if event != "end" or element.tag != _RELATIONSHIP_TAG: + continue + relationship_id = element.get("Id") + relationship_type = element.get("Type") + target = element.get("Target") + target_mode = element.get("TargetMode") + if ( + not relationship_id + or not relationship_type + or not target + or target_mode not in {None, "External"} + ): + raise CorruptDocumentError("XLSX relationship is malformed") + if relationship_id in relationships: + raise CorruptDocumentError("XLSX relationship identifiers must be unique") + external = target_mode == "External" + normalized_target = ( + target if external else _safe_relationship_target(source_part, target) + ) + if not external and normalized_target not in infos: + raise CorruptDocumentError("XLSX relationship target is missing") + relationships[relationship_id] = _Relationship( + relationship_type=relationship_type, + target=normalized_target, + external=external, + ) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX relationships part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX relationships part is corrupt") + return relationships + + +def _preflight_persons( + archive: ZipFile, + part_name: str, + usage: _ResourceUsage, +) -> None: + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX persons part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_THREADED_COMMENTS_NS}}}personList": + raise CorruptDocumentError("XLSX persons namespace is invalid") + if event == "end" and element.tag == f"{{{_THREADED_COMMENTS_NS}}}person": + if not element.get("id") or not element.get("displayName"): + raise CorruptDocumentError("XLSX person entry is invalid") + _increment( + usage, + "comment_authors", + 1, + limit=MAX_COMMENT_AUTHORS, + message="XLSX exceeds the comment author limit", + ) + _add_native_text( + usage, + sum( + len(element.get(attribute, "")) + for attribute in ("displayName", "userId", "providerId") + ), + ) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX persons part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX persons part is corrupt") + + +def _preflight_connections( + archive: ZipFile, + part_name: str, + usage: _ResourceUsage, +) -> None: + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX connections part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}connections": + raise CorruptDocumentError("XLSX connections namespace is invalid") + if event != "end" or element.tag != f"{{{_SPREADSHEET_NS}}}connection": + continue + references = tuple( + element.get(attribute, "") + for attribute in ("sourceFile", "odcFile", "connectionFile") + if element.get(attribute) + ) + for reference in references: + _increment( + usage, + "external_references", + 1, + limit=MAX_EXTERNAL_REFERENCES, + message="XLSX exceeds the external reference limit", + ) + _add_native_text(usage, len(reference)) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX connections part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX connections part is corrupt") + + +def _preflight_workbook_text_relationships( + archive: ZipFile, + infos: dict[str, ZipInfo], + relationships: dict[str, _Relationship], + usage: _ResourceUsage, + referenced_external_ids: set[str], +) -> None: + person_targets = tuple( + relationship.target + for relationship in relationships.values() + if relationship.relationship_type == _PERSON_RELATIONSHIP and not relationship.external + ) + if ( + any( + relationship.external + for relationship in relationships.values() + if relationship.relationship_type == _PERSON_RELATIONSHIP + ) + or len(person_targets) > 1 + ): + raise CorruptDocumentError("XLSX persons relationship is invalid") + for target in person_targets: + _preflight_persons(archive, target, usage) + + for relationship_id, relationship in relationships.items(): + if relationship.relationship_type == _CONNECTIONS_RELATIONSHIP: + if relationship.external: + raise CorruptDocumentError("XLSX connections relationship is invalid") + _preflight_connections(archive, relationship.target, usage) + elif relationship.relationship_type == _EXTERNAL_LINK_RELATIONSHIP: + if relationship.external: + if relationship_id not in referenced_external_ids: + _increment( + usage, + "external_references", + 1, + limit=MAX_EXTERNAL_REFERENCES, + message="XLSX exceeds the external reference limit", + ) + _add_native_text(usage, len(relationship.target)) + continue + nested = _read_relationships( + archive, + infos, + relationship.target, + required=_relationships_part(relationship.target) in infos, + ) + for nested_relationship in nested.values(): + if nested_relationship.relationship_type != _EXTERNAL_LINK_PATH_RELATIONSHIP: + continue + if relationship_id not in referenced_external_ids: + _increment( + usage, + "external_references", + 1, + limit=MAX_EXTERNAL_REFERENCES, + message="XLSX exceeds the external reference limit", + ) + _add_native_text(usage, len(nested_relationship.target)) + + +def _parse_workbook( + archive: ZipFile, + infos: dict[str, ZipInfo], + usage: _ResourceUsage, +) -> tuple[ + tuple[tuple[str, XlsxSheetKind, XlsxSheetState, str], ...], + bool, + tuple[str, ...], + str | None, + str | None, +]: + relationships = _read_relationships( + archive, + infos, + "xl/workbook.xml", + required="xl/_rels/workbook.xml.rels" in infos, + ) + sheet_entries: list[tuple[str, XlsxSheetKind, XlsxSheetState, str]] = [] + sheet_ids: set[int] = set() + names: set[str] = set() + pivot_cache_targets: list[str] = [] + referenced_external_ids: set[str] = set() + date_1904 = False + root_seen = False + try: + with archive.open("xl/workbook.xml") as stream: + for event, element in _xml_events(stream, message="XLSX workbook part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}workbook": + raise CorruptDocumentError("XLSX workbook namespace is invalid") + if event != "end": + continue + if element.tag == f"{{{_SPREADSHEET_NS}}}workbookPr": + raw_date = element.get("date1904") + if raw_date not in {None, "0", "1", "false", "true"}: + raise CorruptDocumentError("XLSX workbook date system is invalid") + date_1904 = raw_date in {"1", "true"} + elif element.tag == f"{{{_SPREADSHEET_NS}}}definedName": + _increment( + usage, + "defined_names", + 1, + limit=MAX_DEFINED_NAMES, + message="XLSX exceeds the defined name limit", + ) + _add_native_text(usage, len(element.text or "")) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}workbookView": + _increment( + usage, + "workbook_views", + 1, + limit=MAX_WORKBOOK_VIEWS, + message="XLSX exceeds the workbook view limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}externalReference": + _increment( + usage, + "external_references", + 1, + limit=MAX_EXTERNAL_REFERENCES, + message="XLSX exceeds the external reference limit", + ) + relationship_id = element.get(_RELATIONSHIP_ID) + relationship = relationships.get(relationship_id or "") + if ( + relationship_id is None + or relationship is None + or relationship.relationship_type != _EXTERNAL_LINK_RELATIONSHIP + ): + raise CorruptDocumentError("XLSX external reference is invalid") + referenced_external_ids.add(relationship_id) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}pivotCache": + _increment( + usage, + "pivot_caches", + 1, + limit=MAX_PIVOT_CACHES, + message="XLSX exceeds the pivot cache limit", + ) + relationship = _relationship_for( + relationships, + element.get(_RELATIONSHIP_ID), + expected_type=_PIVOT_CACHE_DEFINITION_RELATIONSHIP, + ) + if relationship.external: + raise CorruptDocumentError("XLSX pivot cache relationship is invalid") + pivot_cache_targets.append(relationship.target) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}sheet": + if len(sheet_entries) >= MAX_SHEETS: + raise LimitExceededError("XLSX exceeds the sheet count limit") + name = element.get("name") + relationship_id = element.get(_RELATIONSHIP_ID) + raw_sheet_id = element.get("sheetId") + raw_state = element.get("state", XlsxSheetState.VISIBLE.value) + if not name or not relationship_id or not raw_sheet_id: + raise CorruptDocumentError("XLSX workbook sheet entry is malformed") + if ( + len(name) > 31 + or any(ord(character) < 32 for character in name) + or any(character in "[]:*?/\\" for character in name) + ): + raise CorruptDocumentError("XLSX workbook sheet name is invalid") + try: + sheet_id = int(raw_sheet_id) + state = XlsxSheetState(raw_state) + except ValueError as error: + raise CorruptDocumentError( + "XLSX workbook sheet entry is malformed" + ) from error + if sheet_id <= 0 or sheet_id in sheet_ids or name.casefold() in names: + raise CorruptDocumentError("XLSX workbook sheet entries must be unique") + relationship = relationships.get(relationship_id) + if relationship is None or relationship.external: + raise CorruptDocumentError("XLSX workbook sheet relationship is invalid") + if relationship.relationship_type == _WORKSHEET_RELATIONSHIP: + kind = XlsxSheetKind.WORKSHEET + elif relationship.relationship_type == _CHARTSHEET_RELATIONSHIP: + kind = XlsxSheetKind.CHARTSHEET + else: + raise CorruptDocumentError( + "XLSX workbook sheet relationship type is invalid" + ) + sheet_ids.add(sheet_id) + names.add(name.casefold()) + sheet_entries.append((name, kind, state, relationship.target)) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX workbook part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX workbook part is corrupt") + if sheet_entries and not relationships: + raise CorruptDocumentError("XLSX relationships part is missing") + if any( + relationship.external + for relationship in relationships.values() + if relationship.relationship_type + in { + _SHARED_STRINGS_RELATIONSHIP, + _STYLES_RELATIONSHIP, + } + ): + raise CorruptDocumentError("XLSX workbook singleton relationship is external") + shared_string_targets = tuple( + relationship.target + for relationship in relationships.values() + if relationship.relationship_type == _SHARED_STRINGS_RELATIONSHIP + and not relationship.external + ) + style_targets = tuple( + relationship.target + for relationship in relationships.values() + if relationship.relationship_type == _STYLES_RELATIONSHIP and not relationship.external + ) + if len(shared_string_targets) > 1 or len(style_targets) > 1: + raise CorruptDocumentError("XLSX workbook singleton relationships are duplicated") + _preflight_workbook_text_relationships( + archive, + infos, + relationships, + usage, + referenced_external_ids, + ) + shared_strings_part = ( + shared_string_targets[0] + if shared_string_targets + else "xl/sharedStrings.xml" + if "xl/sharedStrings.xml" in infos + else None + ) + styles_part = ( + style_targets[0] if style_targets else "xl/styles.xml" if "xl/styles.xml" in infos else None + ) + return ( + tuple(sheet_entries), + date_1904, + tuple(pivot_cache_targets), + shared_strings_part, + styles_part, + ) + + +def _preflight_shared_strings( + archive: ZipFile, + part_name: str | None, + usage: _ResourceUsage, +) -> None: + if part_name is None: + return + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events( + stream, message="XLSX shared strings part is corrupt" + ): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}sst": + raise CorruptDocumentError("XLSX shared strings namespace is invalid") + if event == "end" and element.tag == f"{{{_SPREADSHEET_NS}}}si": + _increment( + usage, + "shared_strings", + 1, + limit=MAX_SHARED_STRINGS, + message="XLSX exceeds the shared string item limit", + ) + characters = sum( + len(node.text or "") for node in element.iter(f"{{{_SPREADSHEET_NS}}}t") + ) + _increment( + usage, + "shared_string_chars", + characters, + limit=MAX_SHARED_STRING_CHARS, + message="XLSX exceeds the shared string text limit", + ) + _add_native_text(usage, characters) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX shared strings part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX shared strings part is corrupt") + + +def _preflight_styles( + archive: ZipFile, + part_name: str | None, + usage: _ResourceUsage, +) -> None: + if part_name is None: + return + containers = { + "numFmts": ("number_formats", MAX_NUMBER_FORMATS, "number format"), + "fonts": ("fonts", MAX_FONTS, "font"), + "fills": ("fills", MAX_FILLS, "fill"), + "borders": ("borders", MAX_BORDERS, "border"), + "cellStyleXfs": ("cell_style_xfs", MAX_CELL_STYLE_XFS, "cellStyleXfs"), + "cellXfs": ("cell_xfs", MAX_CELL_XFS, "cellXfs"), + "cellStyles": ("named_cell_styles", MAX_NAMED_CELL_STYLES, "named style"), + "dxfs": ("dxfs", MAX_DXFS, "dxf"), + "tableStyles": ("table_styles", MAX_TABLE_STYLES, "table style"), + } + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX stylesheet part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}styleSheet": + raise CorruptDocumentError("XLSX stylesheet namespace is invalid") + if event != "end" or not isinstance(element.tag, str): + continue + local_name = element.tag.rsplit("}", 1)[-1] + details = containers.get(local_name) + if details is None or not element.tag.startswith(f"{{{_SPREADSHEET_NS}}}"): + continue + field_name, limit, label = details + count = len(element) + _increment( + usage, + field_name, + count, + limit=limit, + message=f"XLSX exceeds the {label} limit", + ) + _increment( + usage, + "style_records", + count, + limit=MAX_STYLE_RECORDS, + message="XLSX exceeds the aggregate style record limit", + ) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX stylesheet part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX stylesheet part is corrupt") + + +def _preflight_custom_properties( + archive: ZipFile, + infos: dict[str, ZipInfo], + usage: _ResourceUsage, +) -> None: + part_name = "docProps/custom.xml" + if part_name not in infos: + return + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX custom properties are corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_CUSTOM_PROPERTIES_NS}}}Properties": + raise CorruptDocumentError("XLSX custom properties namespace is invalid") + if event == "end" and element.tag == f"{{{_CUSTOM_PROPERTIES_NS}}}property": + _increment( + usage, + "custom_properties", + 1, + limit=MAX_CUSTOM_PROPERTIES, + message="XLSX exceeds the custom property limit", + ) + characters = sum(len(node.text or "") for node in element.iter()) + _add_native_text(usage, characters) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX custom properties are corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX custom properties are corrupt") + + +def _preflight_table( + archive: ZipFile, + part_name: str, + usage: _ResourceUsage, +) -> int: + root_seen = False + table_reference: str | None = None + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX table part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}table": + raise CorruptDocumentError("XLSX table namespace is invalid") + table_reference = element.get("ref") + if event == "end" and element.tag == f"{{{_SPREADSHEET_NS}}}tableColumn": + _increment( + usage, + "table_columns", + 1, + limit=MAX_TABLE_COLUMNS, + message="XLSX exceeds the table column limit", + ) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX table part is corrupt") from error + if not root_seen or table_reference is None: + raise CorruptDocumentError("XLSX table part is corrupt") + footprint = _area(_parse_a1_range(table_reference, message="XLSX table range is invalid")) + _increment( + usage, + "table_footprint", + footprint, + limit=MAX_TABLE_FOOTPRINT, + message="XLSX exceeds the table footprint limit", + ) + return footprint + + +def _preflight_comments( + archive: ZipFile, + part_name: str, + usage: _ResourceUsage, +) -> None: + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX comments part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}comments": + raise CorruptDocumentError("XLSX comments namespace is invalid") + if event == "end" and element.tag == f"{{{_SPREADSHEET_NS}}}comment": + reference = element.get("ref") + if reference is None: + raise CorruptDocumentError("XLSX comment anchor is invalid") + bounds = _parse_a1_range(reference, message="XLSX comment anchor is invalid") + if _area(bounds) != 1: + raise CorruptDocumentError("XLSX comment anchor is invalid") + _increment( + usage, + "hyperlinks_and_comments", + 1, + limit=MAX_HYPERLINKS_AND_COMMENTS, + message="XLSX exceeds the hyperlink and comment limit", + ) + characters = sum( + len(node.text or "") for node in element.iter(f"{{{_SPREADSHEET_NS}}}t") + ) + _add_native_text(usage, characters) + element.clear() + elif event == "end" and element.tag == f"{{{_SPREADSHEET_NS}}}author": + _increment( + usage, + "comment_authors", + 1, + limit=MAX_COMMENT_AUTHORS, + message="XLSX exceeds the comment author limit", + ) + _add_native_text(usage, len(element.text or "")) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX comments part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX comments part is corrupt") + + +def _preflight_threaded_comments( + archive: ZipFile, + part_name: str, + usage: _ResourceUsage, +) -> None: + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events( + stream, + message="XLSX threaded comments part is corrupt", + ): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_THREADED_COMMENTS_NS}}}ThreadedComments": + raise CorruptDocumentError("XLSX threaded comments namespace is invalid") + if event == "end" and element.tag == f"{{{_THREADED_COMMENTS_NS}}}threadedComment": + reference = element.get("ref") + if reference is None or not element.get("personId"): + raise CorruptDocumentError("XLSX threaded comment is invalid") + bounds = _parse_a1_range( + reference, + message="XLSX threaded comment anchor is invalid", + ) + if _area(bounds) != 1: + raise CorruptDocumentError("XLSX threaded comment anchor is invalid") + _increment( + usage, + "hyperlinks_and_comments", + 1, + limit=MAX_HYPERLINKS_AND_COMMENTS, + message="XLSX exceeds the hyperlink and comment limit", + ) + _add_native_text( + usage, + sum( + len(node.text or "") + for node in element.iter(f"{{{_THREADED_COMMENTS_NS}}}text") + ), + ) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX threaded comments part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX threaded comments part is corrupt") + + +def _preflight_chart( + archive: ZipFile, + part_name: str, + usage: _ResourceUsage, +) -> None: + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX chart part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_CHART_NS}}}chartSpace": + raise CorruptDocumentError("XLSX chart namespace is invalid") + if event != "end": + continue + if element.tag == f"{{{_CHART_NS}}}pt": + _increment( + usage, + "chart_cache_points", + 1, + limit=MAX_CHART_CACHE_POINTS, + message="XLSX exceeds the chart cache point limit", + ) + elif element.tag == f"{{{_CHART_NS}}}ptCount": + raw_count = element.get("val") + try: + declared_count = int(raw_count or "") + except ValueError as error: + raise CorruptDocumentError("XLSX chart cache count is invalid") from error + if declared_count < 0 or declared_count > MAX_CHART_CACHE_POINTS: + raise LimitExceededError("XLSX exceeds the chart cache point limit") + elif element.tag in { + f"{{{_CHART_NS}}}v", + f"{{{_CHART_NS}}}f", + f"{{{_DRAWING_MAIN_NS}}}t", + }: + _add_native_text(usage, len(element.text or "")) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX chart part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX chart part is corrupt") + + +def _require_anchor_index(value: str | None, *, maximum: int) -> int: + try: + index = int(value or "") + except ValueError as error: + raise CorruptDocumentError("XLSX drawing anchor is invalid") from error + if index < 0 or index >= maximum: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + return index + + +def _validate_drawing_anchor(anchor: Any, *, allow_zero_absolute_extent: bool) -> None: + local_name = anchor.tag.rsplit("}", 1)[-1] + if local_name in {"oneCellAnchor", "twoCellAnchor"}: + starts = anchor.findall(f"{{{_DRAWING_NS}}}from") + if len(starts) != 1: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + markers = starts + if local_name == "twoCellAnchor": + ends = anchor.findall(f"{{{_DRAWING_NS}}}to") + if len(ends) != 1: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + markers = [*markers, *ends] + for marker in markers: + _require_anchor_index( + marker.findtext(f"{{{_DRAWING_NS}}}col"), + maximum=_MAX_COLUMN, + ) + _require_anchor_index( + marker.findtext(f"{{{_DRAWING_NS}}}row"), + maximum=_MAX_ROW, + ) + if local_name == "oneCellAnchor": + extents = anchor.findall(f"{{{_DRAWING_NS}}}ext") + if len(extents) != 1: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + try: + cx = int(extents[0].get("cx", "")) + cy = int(extents[0].get("cy", "")) + except ValueError as error: + raise CorruptDocumentError("XLSX drawing anchor is invalid") from error + if cx <= 0 or cy <= 0: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + elif local_name == "absoluteAnchor": + positions = anchor.findall(f"{{{_DRAWING_NS}}}pos") + extents = anchor.findall(f"{{{_DRAWING_NS}}}ext") + if len(positions) != 1 or len(extents) != 1: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + try: + values = tuple( + int(value) + for value in ( + positions[0].get("x", ""), + positions[0].get("y", ""), + extents[0].get("cx", ""), + extents[0].get("cy", ""), + ) + ) + except ValueError as error: + raise CorruptDocumentError("XLSX drawing anchor is invalid") from error + invalid_extent = ( + values[2] < 0 or values[3] < 0 + if allow_zero_absolute_extent + else values[2] <= 0 or values[3] <= 0 + ) + if values[0] < 0 or values[1] < 0 or invalid_extent: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + else: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + + +def _preflight_drawing( + archive: ZipFile, + infos: dict[str, ZipInfo], + part_name: str, + usage: _ResourceUsage, + visited_charts: set[str], + sheet_index: int, + unsupported_objects: list[XlsxUnsupportedObjectRef], + *, + allow_zero_absolute_extent: bool = False, +) -> None: + relationships = _read_relationships( + archive, + infos, + part_name, + required=_relationships_part(part_name) in infos, + ) + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX drawing part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_DRAWING_NS}}}wsDr": + raise CorruptDocumentError("XLSX drawing namespace is invalid") + if event != "end" or element.tag not in { + f"{{{_DRAWING_NS}}}oneCellAnchor", + f"{{{_DRAWING_NS}}}twoCellAnchor", + f"{{{_DRAWING_NS}}}absoluteAnchor", + }: + continue + _validate_drawing_anchor( + element, + allow_zero_absolute_extent=allow_zero_absolute_extent, + ) + _increment( + usage, + "drawing_objects", + 1, + limit=MAX_DRAWING_OBJECTS, + message="XLSX exceeds the drawing object limit", + ) + _add_native_text( + usage, + sum(len(node.text or "") for node in element.iter(f"{{{_DRAWING_MAIN_NS}}}t")), + ) + _add_native_text( + usage, + sum( + len(node.get(attribute, "")) + for node in element.iter(f"{{{_DRAWING_NS}}}cNvPr") + for attribute in ("name", "descr", "title") + ), + ) + for node in element.iter(): + for attribute_name, relationship_id in node.attrib.items(): + if not attribute_name.startswith(f"{{{_OFFICE_REL_NS}}}"): + continue + relationship = relationships.get(relationship_id) + if relationship is None or relationship.external: + raise CorruptDocumentError("XLSX drawing relationship is invalid") + local_attribute = attribute_name.rsplit("}", 1)[-1] + local_tag = node.tag.rsplit("}", 1)[-1] + expected_type = None + if local_tag == "chart" and local_attribute == "id": + expected_type = _CHART_RELATIONSHIP + elif local_tag == "blip" and local_attribute in {"embed", "link"}: + expected_type = _IMAGE_RELATIONSHIP + if ( + expected_type is not None + and relationship.relationship_type != expected_type + ): + raise CorruptDocumentError("XLSX drawing relationship type is invalid") + if ( + expected_type == _CHART_RELATIONSHIP + and relationship.target not in visited_charts + ): + visited_charts.add(relationship.target) + _preflight_chart(archive, relationship.target, usage) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX drawing part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX drawing part is corrupt") + known_relationship_types = { + _CHART_RELATIONSHIP, + _IMAGE_RELATIONSHIP, + _HYPERLINK_RELATIONSHIP, + } + for relationship_id, relationship in relationships.items(): + if relationship.relationship_type not in known_relationship_types: + _add_native_text(usage, len(relationship.target)) + _record_unsupported_object( + unsupported_objects, + sheet_index=sheet_index, + relationship_id=relationship_id, + relationship=relationship, + ) + + +def _preflight_pivot_cache( + archive: ZipFile, + infos: dict[str, ZipInfo], + part_name: str, + usage: _ResourceUsage, +) -> None: + relationships = _read_relationships( + archive, + infos, + part_name, + required=False, + ) + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX pivot cache part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}pivotCacheDefinition": + raise CorruptDocumentError("XLSX pivot cache namespace is invalid") + if event == "end" and element.tag in { + f"{{{_SPREADSHEET_NS}}}cacheField", + f"{{{_SPREADSHEET_NS}}}s", + f"{{{_SPREADSHEET_NS}}}n", + f"{{{_SPREADSHEET_NS}}}d", + f"{{{_SPREADSHEET_NS}}}b", + f"{{{_SPREADSHEET_NS}}}e", + f"{{{_SPREADSHEET_NS}}}m", + }: + _increment( + usage, + "pivot_items", + 1, + limit=MAX_PIVOT_ITEMS, + message="XLSX exceeds the pivot item limit", + ) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX pivot cache part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX pivot cache part is corrupt") + for relationship in relationships.values(): + if relationship.relationship_type != _PIVOT_CACHE_RECORDS_RELATIONSHIP: + continue + _preflight_pivot_records(archive, relationship.target, usage) + + +def _preflight_pivot_records( + archive: ZipFile, + part_name: str, + usage: _ResourceUsage, +) -> None: + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX pivot records are corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}pivotCacheRecords": + raise CorruptDocumentError("XLSX pivot records namespace is invalid") + if event == "end" and element.tag == f"{{{_SPREADSHEET_NS}}}r": + _increment( + usage, + "pivot_cache_records", + 1, + limit=MAX_PIVOT_CACHE_RECORDS, + message="XLSX exceeds the pivot cache record limit", + ) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX pivot records are corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX pivot records are corrupt") + + +def _worksheet_counts( + archive: ZipFile, + infos: dict[str, ZipInfo], + part_name: str, + usage: _ResourceUsage, + visited_drawings: set[str], + visited_charts: set[str], + sheet_index: int, + unsupported_objects: list[XlsxUnsupportedObjectRef], +) -> _WorksheetCounts: + relationships = _read_relationships( + archive, + infos, + part_name, + required=False, + ) + declared_bounds: tuple[int, int, int, int] | None = None + serialized_cells = 0 + non_empty_cells = 0 + native_text_chars = 0 + actual_columns: list[int] = [] + actual_rows: list[int] = [] + seen_cells: set[tuple[int, int]] = set() + drawing_targets: list[str] = [] + table_targets: list[str] = [] + pivot_targets: list[str] = [] + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX worksheet part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}worksheet": + raise CorruptDocumentError("XLSX worksheet namespace is invalid") + if event != "end": + continue + if element.tag == f"{{{_SPREADSHEET_NS}}}dimension": + reference = element.get("ref") + if reference is None: + raise CorruptDocumentError("XLSX worksheet dimension is invalid") + declared_bounds = _parse_a1_range( + reference, + message="XLSX worksheet dimension is invalid", + ) + if _area(declared_bounds) > MAX_DECLARED_CELLS: + raise LimitExceededError( + "XLSX worksheet exceeds the declared dimension limit" + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}c": + serialized_cells += 1 + if serialized_cells > MAX_SERIALIZED_CELLS: + raise LimitExceededError("XLSX exceeds the serialized cell limit") + reference = element.get("r") + if reference is None: + raise CorruptDocumentError("XLSX cell coordinate is missing") + column, row, end_column, end_row = _parse_a1_range( + reference, + message="XLSX cell coordinate is invalid", + ) + if column != end_column or row != end_row: + raise CorruptDocumentError("XLSX cell coordinate is invalid") + coordinate = (row, column) + if coordinate in seen_cells: + raise CorruptDocumentError("XLSX cell coordinates must be unique") + seen_cells.add(coordinate) + actual_columns.append(column) + actual_rows.append(row) + value_nodes = tuple( + child + for child in element.iter() + if child is not element + and child.tag + in { + f"{{{_SPREADSHEET_NS}}}v", + f"{{{_SPREADSHEET_NS}}}f", + f"{{{_SPREADSHEET_NS}}}t", + } + ) + if any((node.text or "") != "" for node in value_nodes): + non_empty_cells += 1 + if non_empty_cells > MAX_NON_EMPTY_CELLS: + raise LimitExceededError("XLSX exceeds the non-empty cell limit") + native_text_chars += sum(len(node.text or "") for node in value_nodes) + if native_text_chars > MAX_NATIVE_TEXT_CHARS: + raise LimitExceededError("XLSX exceeds the native text limit") + if element.get("t") == "s": + value_node = element.find(f"{{{_SPREADSHEET_NS}}}v") + try: + shared_string_index = int( + value_node.text if value_node is not None else "" + ) + except (TypeError, ValueError) as error: + raise CorruptDocumentError( + "XLSX shared string index is invalid" + ) from error + if not 0 <= shared_string_index < usage.shared_strings: + raise CorruptDocumentError("XLSX shared string index is invalid") + raw_style = element.get("s") + if raw_style is not None: + try: + style_index = int(raw_style) + except ValueError as error: + raise CorruptDocumentError( + "XLSX cell style index is invalid" + ) from error + if style_index < 0 or style_index >= usage.cell_xfs: + raise CorruptDocumentError("XLSX cell style index is invalid") + _increment( + usage, + "materialized_grid_cells", + 1, + limit=MAX_MATERIALIZED_GRID_CELLS, + message="XLSX exceeds the materialized grid limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}mergeCell": + reference = element.get("ref") + if reference is None: + raise CorruptDocumentError("XLSX merge range is invalid") + footprint = _area( + _parse_a1_range(reference, message="XLSX merge range is invalid") + ) + _increment( + usage, + "merge_ranges", + 1, + limit=MAX_MERGE_RANGES, + message="XLSX exceeds the merge range limit", + ) + _increment( + usage, + "merge_footprint", + footprint, + limit=MAX_MERGE_FOOTPRINT, + message="XLSX exceeds the merge footprint limit", + ) + _increment( + usage, + "materialized_grid_cells", + footprint, + limit=MAX_MATERIALIZED_GRID_CELLS, + message="XLSX exceeds the materialized grid limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}cfRule": + _increment( + usage, + "conditional_formatting_rules", + 1, + limit=MAX_CONDITIONAL_FORMATTING_RULES, + message="XLSX exceeds the conditional formatting rule limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}conditionalFormatting": + _increment( + usage, + "conditional_formatting_ranges", + 1, + limit=MAX_CONDITIONAL_FORMATTING_RANGES, + message="XLSX exceeds the conditional formatting range limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}dataValidation": + _increment( + usage, + "data_validations", + 1, + limit=MAX_DATA_VALIDATIONS, + message="XLSX exceeds the data validation limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}row": + raw_row = element.get("r") + if raw_row is not None: + try: + row_number = int(raw_row) + except ValueError as error: + raise CorruptDocumentError("XLSX row index is invalid") from error + if not 1 <= row_number <= _MAX_ROW: + raise CorruptDocumentError("XLSX row index is invalid") + if set(element.attrib) - {"r", "spans"}: + _increment( + usage, + "row_dimensions", + 1, + limit=MAX_ROW_DIMENSIONS, + message="XLSX exceeds the row dimension limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}col": + try: + minimum_column = int(element.get("min", "")) + maximum_column = int(element.get("max", "")) + except ValueError as error: + raise CorruptDocumentError("XLSX column range is invalid") from error + if ( + not 1 <= minimum_column <= _MAX_COLUMN + or not minimum_column <= maximum_column <= _MAX_COLUMN + ): + raise CorruptDocumentError("XLSX column range is invalid") + _increment( + usage, + "column_dimensions", + 1, + limit=MAX_COLUMN_DIMENSIONS, + message="XLSX exceeds the column dimension limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}brk": + _increment( + usage, + "page_breaks", + 1, + limit=MAX_PAGE_BREAKS, + message="XLSX exceeds the page break limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}scenario": + _increment( + usage, + "scenarios", + 1, + limit=MAX_SCENARIOS, + message="XLSX exceeds the scenario limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}sheetView": + _increment( + usage, + "sheet_views", + 1, + limit=MAX_SHEET_VIEWS, + message="XLSX exceeds the sheet view limit", + ) + element.clear() + elif element.tag in { + f"{{{_SPREADSHEET_NS}}}filterColumn", + f"{{{_SPREADSHEET_NS}}}customFilter", + f"{{{_SPREADSHEET_NS}}}filter", + }: + _increment( + usage, + "filter_items", + 1, + limit=MAX_FILTER_ITEMS, + message="XLSX exceeds the filter item limit", + ) + element.clear() + elif element.tag in { + f"{{{_SPREADSHEET_NS}}}oddHeader", + f"{{{_SPREADSHEET_NS}}}oddFooter", + f"{{{_SPREADSHEET_NS}}}evenHeader", + f"{{{_SPREADSHEET_NS}}}evenFooter", + f"{{{_SPREADSHEET_NS}}}firstHeader", + f"{{{_SPREADSHEET_NS}}}firstFooter", + }: + _add_native_text(usage, len(element.text or "")) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}hyperlink": + reference = element.get("ref") + if reference is None: + raise CorruptDocumentError("XLSX hyperlink anchor is invalid") + footprint = _area( + _parse_a1_range(reference, message="XLSX hyperlink anchor is invalid") + ) + _increment( + usage, + "hyperlinks_and_comments", + 1, + limit=MAX_HYPERLINKS_AND_COMMENTS, + message="XLSX exceeds the hyperlink and comment limit", + ) + _increment( + usage, + "hyperlink_footprint", + footprint, + limit=MAX_HYPERLINK_FOOTPRINT, + message="XLSX exceeds the hyperlink footprint limit", + ) + _increment( + usage, + "materialized_grid_cells", + footprint, + limit=MAX_MATERIALIZED_GRID_CELLS, + message="XLSX exceeds the materialized grid limit", + ) + relationship_id = element.get(_RELATIONSHIP_ID) + if relationship_id is not None: + relationship = _relationship_for( + relationships, + relationship_id, + expected_type=_HYPERLINK_RELATIONSHIP, + ) + _add_native_text(usage, len(relationship.target)) + _add_native_text( + usage, + sum( + len(element.get(attribute, "")) + for attribute in ("display", "location", "tooltip") + ), + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}drawing": + relationship = _relationship_for( + relationships, + element.get(_RELATIONSHIP_ID), + expected_type=_DRAWING_RELATIONSHIP, + ) + if relationship.external: + raise CorruptDocumentError("XLSX drawing relationship is invalid") + drawing_targets.append(relationship.target) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}tablePart": + relationship = _relationship_for( + relationships, + element.get(_RELATIONSHIP_ID), + expected_type=_TABLE_RELATIONSHIP, + ) + if relationship.external: + raise CorruptDocumentError("XLSX table relationship is invalid") + _increment( + usage, + "tables", + 1, + limit=MAX_TABLES, + message="XLSX exceeds the table count limit", + ) + table_targets.append(relationship.target) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX worksheet part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX worksheet part is corrupt") + for relationship in relationships.values(): + if relationship.relationship_type == _COMMENTS_RELATIONSHIP: + if relationship.external: + raise CorruptDocumentError("XLSX comments relationship is invalid") + _preflight_comments(archive, relationship.target, usage) + elif relationship.relationship_type == _THREADED_COMMENTS_RELATIONSHIP: + if relationship.external: + raise CorruptDocumentError("XLSX threaded comments relationship is invalid") + _preflight_threaded_comments(archive, relationship.target, usage) + elif relationship.relationship_type == _PIVOT_TABLE_RELATIONSHIP: + if relationship.external: + raise CorruptDocumentError("XLSX pivot table relationship is invalid") + pivot_targets.append(relationship.target) + for target in table_targets: + footprint = _preflight_table(archive, target, usage) + _increment( + usage, + "materialized_grid_cells", + footprint, + limit=MAX_MATERIALIZED_GRID_CELLS, + message="XLSX exceeds the materialized grid limit", + ) + for target in drawing_targets: + if target not in visited_drawings: + visited_drawings.add(target) + _preflight_drawing( + archive, + infos, + target, + usage, + visited_charts, + sheet_index, + unsupported_objects, + ) + for target in pivot_targets: + _increment( + usage, + "pivot_tables", + 1, + limit=MAX_PIVOT_TABLES, + message="XLSX exceeds the pivot table limit", + ) + _preflight_xml_root( + archive, + target, + expected_tag=f"{{{_SPREADSHEET_NS}}}pivotTableDefinition", + message="XLSX pivot table part is corrupt", + ) + known_relationship_types = { + _DRAWING_RELATIONSHIP, + _COMMENTS_RELATIONSHIP, + _THREADED_COMMENTS_RELATIONSHIP, + _TABLE_RELATIONSHIP, + _HYPERLINK_RELATIONSHIP, + _PIVOT_TABLE_RELATIONSHIP, + } + for relationship_id, relationship in relationships.items(): + if relationship.relationship_type in known_relationship_types: + continue + _increment( + usage, + "drawing_objects", + 1, + limit=MAX_DRAWING_OBJECTS, + message="XLSX exceeds the drawing object limit", + ) + _add_native_text(usage, len(relationship.target)) + _record_unsupported_object( + unsupported_objects, + sheet_index=sheet_index, + relationship_id=relationship_id, + relationship=relationship, + ) + if actual_columns: + actual_bounds = ( + min(actual_columns), + min(actual_rows), + max(actual_columns), + max(actual_rows), + ) + if _area(actual_bounds) > MAX_DECLARED_CELLS: + raise LimitExceededError("XLSX worksheet actual coordinates exceed the resource budget") + if declared_bounds is not None: + left, top, right, bottom = declared_bounds + actual_left, actual_top, actual_right, actual_bottom = actual_bounds + if ( + actual_left < left + or actual_top < top + or actual_right > right + or actual_bottom > bottom + ): + raise CorruptDocumentError("XLSX cells fall outside the declared dimension") + return _WorksheetCounts( + declared_cells=_area(declared_bounds) if declared_bounds is not None else 0, + serialized_cells=serialized_cells, + non_empty_cells=non_empty_cells, + native_text_chars=native_text_chars, + ) + + +def _preflight_xml_root( + archive: ZipFile, + part_name: str, + *, + expected_tag: str, + message: str, +) -> None: + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message=message): + if not root_seen and event == "start": + root_seen = True + if element.tag != expected_tag: + raise CorruptDocumentError(message) + if event == "end": + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError(message) from error + if not root_seen: + raise CorruptDocumentError(message) + + +def _chartsheet_preflight( + archive: ZipFile, + infos: dict[str, ZipInfo], + part_name: str, + usage: _ResourceUsage, + visited_drawings: set[str], + visited_charts: set[str], + sheet_index: int, + unsupported_objects: list[XlsxUnsupportedObjectRef], +) -> None: + relationships = _read_relationships(archive, infos, part_name, required=False) + drawing_ids: list[str] = [] + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX chartsheet part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}chartsheet": + raise CorruptDocumentError("XLSX chartsheet namespace is invalid") + if event == "end": + if element.tag == f"{{{_SPREADSHEET_NS}}}drawing": + relationship_id = element.get(_RELATIONSHIP_ID) + if relationship_id is None: + raise CorruptDocumentError("XLSX chartsheet drawing is invalid") + drawing_ids.append(relationship_id) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX chartsheet part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX chartsheet part is corrupt") + for relationship_id in drawing_ids: + relationship = _relationship_for( + relationships, + relationship_id, + expected_type=_DRAWING_RELATIONSHIP, + ) + if relationship.external: + raise CorruptDocumentError("XLSX chartsheet drawing is invalid") + if relationship.target not in visited_drawings: + visited_drawings.add(relationship.target) + _preflight_drawing( + archive, + infos, + relationship.target, + usage, + visited_charts, + sheet_index, + unsupported_objects, + allow_zero_absolute_extent=True, + ) + for relationship_id, relationship in relationships.items(): + if relationship.relationship_type == _DRAWING_RELATIONSHIP: + continue + _increment( + usage, + "drawing_objects", + 1, + limit=MAX_DRAWING_OBJECTS, + message="XLSX exceeds the drawing object limit", + ) + _add_native_text(usage, len(relationship.target)) + _record_unsupported_object( + unsupported_objects, + sheet_index=sheet_index, + relationship_id=relationship_id, + relationship=relationship, + ) + + +def _projected_budgets( + *, + sheet_count: int, + usage: _ResourceUsage, +) -> tuple[int, int]: + projected_wire = ( + sheet_count * 1_024 + + usage.serialized_cells * 256 + + usage.non_empty_cells * 1_024 + + usage.materialized_grid_cells * 128 + + usage.drawing_objects * 2_048 + + usage.chart_cache_points * 512 + + usage.hyperlinks_and_comments * 1_024 + + usage.native_text_chars * 8 + ) + projected_workbook = ( + sheet_count * 16_384 + + usage.serialized_cells * 512 + + usage.materialized_grid_cells * 256 + + usage.shared_strings * 128 + + usage.style_records * 1_024 + + usage.table_columns * 512 + + usage.conditional_formatting_rules * 1_024 + + usage.conditional_formatting_ranges * 512 + + usage.data_validations * 1_024 + + usage.defined_names * 512 + + usage.pivot_caches * 16_384 + + usage.pivot_tables * 16_384 + + usage.pivot_cache_records * 512 + + usage.pivot_items * 256 + + usage.custom_properties * 1_024 + + usage.row_dimensions * 512 + + usage.column_dimensions * 512 + + usage.page_breaks * 256 + + usage.scenarios * 2_048 + + usage.sheet_views * 2_048 + + usage.filter_items * 512 + + usage.workbook_views * 2_048 + + usage.external_references * 2_048 + + usage.comment_authors * 256 + + usage.drawing_objects * 32_768 + + usage.chart_cache_points * 512 + + usage.native_text_chars * 4 + + usage.xml_elements * 128 + ) + if projected_wire > MAX_PROJECTED_WIRE_BYTES: + raise LimitExceededError("XLSX projected native wire exceeds the inline result budget") + if projected_workbook > MAX_PROJECTED_WORKBOOK_BYTES: + raise LimitExceededError("XLSX projected workbook exceeds the resource budget") + return projected_wire, projected_workbook + + +def preflight_xlsx(path: Path) -> XlsxPreflight: + validate_office_package(path, document_type=DocumentType.XLSX) + try: + with ZipFile(path) as archive: + infos = {info.filename: info for info in archive.infolist()} + usage = _ResourceUsage() + _preflight_all_xml_parts(archive, infos, usage) + ( + sheet_entries, + date_1904, + pivot_cache_targets, + shared_strings_part, + styles_part, + ) = _parse_workbook( + archive, + infos, + usage, + ) + _preflight_shared_strings(archive, shared_strings_part, usage) + _preflight_styles(archive, styles_part, usage) + _preflight_custom_properties(archive, infos, usage) + for target in pivot_cache_targets: + _preflight_pivot_cache(archive, infos, target, usage) + sheets: list[XlsxPreflightSheet] = [] + serialized_cells = 0 + non_empty_cells = 0 + visited_drawings: set[str] = set() + visited_charts: set[str] = set() + for sheet_index, (name, kind, state, part_name) in enumerate( + sheet_entries, + start=1, + ): + unsupported_objects: list[XlsxUnsupportedObjectRef] = [] + if kind is XlsxSheetKind.WORKSHEET: + counts = _worksheet_counts( + archive, + infos, + part_name, + usage, + visited_drawings, + visited_charts, + sheet_index, + unsupported_objects, + ) + else: + _chartsheet_preflight( + archive, + infos, + part_name, + usage, + visited_drawings, + visited_charts, + sheet_index, + unsupported_objects, + ) + counts = _WorksheetCounts(0, 0, 0, 0) + serialized_cells += counts.serialized_cells + non_empty_cells += counts.non_empty_cells + if serialized_cells > MAX_SERIALIZED_CELLS: + raise LimitExceededError("XLSX exceeds the serialized cell limit") + if non_empty_cells > MAX_NON_EMPTY_CELLS: + raise LimitExceededError("XLSX exceeds the non-empty cell limit") + usage.serialized_cells = serialized_cells + usage.non_empty_cells = non_empty_cells + _add_native_text(usage, counts.native_text_chars) + sheets.append( + XlsxPreflightSheet( + sheet_index=sheet_index, + name=name, + kind=kind, + state=state, + part_name=part_name, + declared_cells=counts.declared_cells, + serialized_cells=counts.serialized_cells, + non_empty_cells=counts.non_empty_cells, + unsupported_objects=tuple(unsupported_objects), + ) + ) + projected_wire, projected_workbook = _projected_budgets( + sheet_count=len(sheets), + usage=usage, + ) + except BadZipFile as error: + raise CorruptDocumentError("XLSX package is corrupt") from error + except OSError as error: + raise CorruptDocumentError("XLSX package could not be read") from error + return XlsxPreflight( + sheets=tuple(sheets), + date_1904=date_1904, + serialized_cells=serialized_cells, + non_empty_cells=non_empty_cells, + native_text_chars=usage.native_text_chars, + projected_wire_bytes=projected_wire, + projected_workbook_bytes=projected_workbook, + usage=usage.freeze(), + ) diff --git a/src/opendocs/parsers/xlsx/text_objects.py b/src/opendocs/parsers/xlsx/text_objects.py new file mode 100644 index 0000000..00abbf7 --- /dev/null +++ b/src/opendocs/parsers/xlsx/text_objects.py @@ -0,0 +1,694 @@ +from __future__ import annotations + +import re +from dataclasses import dataclass +from pathlib import Path +from typing import Any +from urllib.parse import quote, urlsplit +from zipfile import BadZipFile, ZipFile, ZipInfo + +from defusedxml import ElementTree as DefusedET +from defusedxml.common import DefusedXmlException +from openpyxl.utils.cell import get_column_letter, range_boundaries + +from opendocs._models import Block, InlineLink, InlineText, MarkdownBlock, ParagraphBlock +from opendocs.errors import CorruptDocumentError +from opendocs.parsers.xlsx.preflight import ( + XlsxPreflight, + XlsxPreflightSheet, + _read_relationships, + _relationships_part, +) + +_SPREADSHEET_NS = "http://schemas.openxmlformats.org/spreadsheetml/2006/main" +_OFFICE_REL_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" +_DRAWING_NS = "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing" +_DRAWING_MAIN_NS = "http://schemas.openxmlformats.org/drawingml/2006/main" +_THREADED_REL_NS = "http://schemas.microsoft.com/office/2017/10/relationships" +_THREADED_COMMENTS_NS = "http://schemas.microsoft.com/office/spreadsheetml/2018/threadedcomments" + +_COMMENTS_RELATIONSHIP = f"{_OFFICE_REL_NS}/comments" +_DRAWING_RELATIONSHIP = f"{_OFFICE_REL_NS}/drawing" +_HYPERLINK_RELATIONSHIP = f"{_OFFICE_REL_NS}/hyperlink" +_EXTERNAL_LINK_RELATIONSHIP = f"{_OFFICE_REL_NS}/externalLink" +_EXTERNAL_LINK_PATH_RELATIONSHIP = f"{_OFFICE_REL_NS}/externalLinkPath" +_CONNECTIONS_RELATIONSHIP = f"{_OFFICE_REL_NS}/connections" +_THREADED_COMMENTS_RELATIONSHIP = f"{_THREADED_REL_NS}/threadedComment" +_PERSON_RELATIONSHIP = f"{_THREADED_REL_NS}/person" +_RELATIONSHIP_ID = f"{{{_OFFICE_REL_NS}}}id" + +_KIND_HYPERLINK = 10 +_KIND_COMMENT = 20 +_KIND_TEXT_BOX = 30 +_KIND_HEADER_FOOTER = 40 +_KIND_EXTERNAL_REFERENCE = 50 +_SAFE_LINK_SCHEMES = {"http", "https", "mailto"} +_UNSUPPORTED_KIND = re.compile(r"[^A-Za-z0-9_.-]+") +_HEADER_TAGS = ( + ("oddHeader", "Odd header"), + ("oddFooter", "Odd footer"), + ("evenHeader", "Even header"), + ("evenFooter", "Even footer"), + ("firstHeader", "First header"), + ("firstFooter", "First footer"), +) +_HEADER_SECTIONS = (("L", "left"), ("C", "center"), ("R", "right")) +_HEADER_FIELDS = { + "P": "{page}", + "N": "{pages}", + "D": "{date}", + "T": "{time}", + "F": "{file}", + "Z": "{path}", + "A": "{sheet}", +} +_HEADER_NAMED_FIELDS = { + "Page": "{page}", + "Pages": "{pages}", + "Date": "{date}", + "Time": "{time}", + "File": "{file}", + "Path": "{path}", + "Tab": "{sheet}", +} +_HEADER_FORMAT_CODES = {"B", "I", "U", "E", "S", "X", "Y", "O", "H"} + + +@dataclass(frozen=True, slots=True) +class XlsxTextObject: + sheet_index: int + anchor: str + row: int + column: int + kind_rank: int + source_ordinal: int + paragraphs: tuple[str, ...] = () + link_label: str | None = None + link_target: str | None = None + safe_link: bool = False + + +@dataclass(frozen=True, slots=True) +class XlsxTextWarning: + code: str + sheet_index: int + anchor: str + object_ordinal: int + detail: str + + +@dataclass(frozen=True, slots=True) +class XlsxTextObjects: + by_sheet: tuple[tuple[XlsxTextObject, ...], ...] + warnings: tuple[XlsxTextWarning, ...] + + +@dataclass(slots=True) +class _SheetCollector: + sheet: XlsxPreflightSheet + objects: list[XlsxTextObject] + warnings: list[XlsxTextWarning] + next_ordinal: int = 1 + + def add_object( + self, + *, + anchor: str, + kind_rank: int, + paragraphs: tuple[str, ...] = (), + link_label: str | None = None, + link_target: str | None = None, + safe_link: bool = False, + ) -> int: + row, column = _top_left(anchor) + ordinal = self.next_ordinal + self.next_ordinal += 1 + self.objects.append( + XlsxTextObject( + sheet_index=self.sheet.sheet_index, + anchor=anchor, + row=row, + column=column, + kind_rank=kind_rank, + source_ordinal=ordinal, + paragraphs=paragraphs, + link_label=link_label, + link_target=link_target, + safe_link=safe_link, + ) + ) + return ordinal + + def add_warning( + self, + code: str, + *, + anchor: str, + detail: str, + object_ordinal: int | None = None, + ) -> None: + ordinal = object_ordinal + if ordinal is None: + ordinal = self.next_ordinal + self.next_ordinal += 1 + else: + self.next_ordinal = max(self.next_ordinal, ordinal + 1) + self.warnings.append( + XlsxTextWarning( + code=code, + sheet_index=self.sheet.sheet_index, + anchor=anchor, + object_ordinal=ordinal, + detail=detail, + ) + ) + + +def _safe_root(archive: ZipFile, part_name: str, *, message: str) -> Any: + try: + data = archive.read(part_name) + return DefusedET.fromstring( + data, + forbid_dtd=True, + forbid_entities=True, + forbid_external=True, + ) + except (KeyError, OSError, DefusedXmlException, DefusedET.ParseError) as error: + raise CorruptDocumentError(message) from error + + +def _top_left(anchor: str) -> tuple[int, int]: + try: + minimum_column, minimum_row, _, _ = range_boundaries(anchor) + except ValueError as error: + raise CorruptDocumentError("XLSX object anchor is invalid") from error + return minimum_row, minimum_column + + +def _relationship_index( + archive: ZipFile, + infos: dict[str, ZipInfo], + part_name: str, +) -> dict[str, Any]: + return _read_relationships( + archive, + infos, + part_name, + required=_relationships_part(part_name) in infos, + ) + + +def _relationship( + relationships: dict[str, Any], + relationship_id: str | None, + *, + expected_type: str, +) -> Any: + relationship = relationships.get(relationship_id or "") + if relationship is None or relationship.relationship_type != expected_type: + raise CorruptDocumentError("XLSX object relationship is invalid") + return relationship + + +def _person_names( + archive: ZipFile, + infos: dict[str, ZipInfo], +) -> dict[str, str]: + relationships = _relationship_index(archive, infos, "xl/workbook.xml") + person_targets = [ + relationship.target + for relationship in relationships.values() + if relationship.relationship_type == _PERSON_RELATIONSHIP and not relationship.external + ] + if ( + any( + relationship.external + for relationship in relationships.values() + if relationship.relationship_type == _PERSON_RELATIONSHIP + ) + or len(person_targets) > 1 + ): + raise CorruptDocumentError("XLSX persons relationship is invalid") + if not person_targets: + return {} + root = _safe_root(archive, person_targets[0], message="XLSX persons part is corrupt") + if root.tag != f"{{{_THREADED_COMMENTS_NS}}}personList": + raise CorruptDocumentError("XLSX persons namespace is invalid") + people: dict[str, str] = {} + for person in root.findall(f"{{{_THREADED_COMMENTS_NS}}}person"): + person_id = person.get("id") + display_name = person.get("displayName") + if not person_id or not display_name or person_id in people: + raise CorruptDocumentError("XLSX person entry is invalid") + people[person_id] = display_name + return people + + +def _classic_comments( + archive: ZipFile, + relationship: Any, + collector: _SheetCollector, +) -> None: + if relationship.external: + raise CorruptDocumentError("XLSX comments relationship is invalid") + root = _safe_root(archive, relationship.target, message="XLSX comments part is corrupt") + if root.tag != f"{{{_SPREADSHEET_NS}}}comments": + raise CorruptDocumentError("XLSX comments namespace is invalid") + authors_node = root.find(f"{{{_SPREADSHEET_NS}}}authors") + authors = ( + tuple(node.text or "" for node in authors_node.findall(f"{{{_SPREADSHEET_NS}}}author")) + if authors_node is not None + else () + ) + comment_list = root.find(f"{{{_SPREADSHEET_NS}}}commentList") + if comment_list is None: + raise CorruptDocumentError("XLSX comments part is corrupt") + for comment in comment_list.findall(f"{{{_SPREADSHEET_NS}}}comment"): + reference = comment.get("ref") + try: + author_index = int(comment.get("authorId", "")) + except ValueError as error: + raise CorruptDocumentError("XLSX comment author is invalid") from error + if reference is None or not 0 <= author_index < len(authors): + raise CorruptDocumentError("XLSX comment author or anchor is invalid") + _top_left(reference) + text = "".join(node.text or "" for node in comment.iter(f"{{{_SPREADSHEET_NS}}}t")) + author = authors[author_index] + prefix = f"Comment by {author}: " if author else "Comment: " + collector.add_object( + anchor=reference, + kind_rank=_KIND_COMMENT, + paragraphs=(f"{prefix}{text}",), + ) + + +def _threaded_comments( + archive: ZipFile, + relationship: Any, + people: dict[str, str], + collector: _SheetCollector, +) -> None: + if relationship.external: + raise CorruptDocumentError("XLSX threaded comments relationship is invalid") + root = _safe_root( + archive, + relationship.target, + message="XLSX threaded comments part is corrupt", + ) + if root.tag != f"{{{_THREADED_COMMENTS_NS}}}ThreadedComments": + raise CorruptDocumentError("XLSX threaded comments namespace is invalid") + for comment in root.findall(f"{{{_THREADED_COMMENTS_NS}}}threadedComment"): + reference = comment.get("ref") + person_id = comment.get("personId") + if reference is None or not person_id or person_id not in people: + raise CorruptDocumentError("XLSX threaded comment person or anchor is invalid") + _top_left(reference) + text = "".join(node.text or "" for node in comment.iter(f"{{{_THREADED_COMMENTS_NS}}}text")) + collector.add_object( + anchor=reference, + kind_rank=_KIND_COMMENT, + paragraphs=(f"Threaded comment by {people[person_id]}: {text}",), + ) + + +def _safe_link_target(target: str) -> bool: + if target.startswith("#"): + return not any(character.isspace() for character in target) + if any(character.isspace() or ord(character) < 32 for character in target): + return False + try: + return urlsplit(target).scheme.lower() in _SAFE_LINK_SCHEMES + except ValueError: + return False + + +def _internal_target(location: str) -> str: + return f"#{_quoted_location(location)}" + + +def _quoted_location(location: str) -> str: + return quote(location, safe="!$&'()*+,-./:;=@_~?") + + +def _hyperlinks( + worksheet_root: Any, + relationships: dict[str, Any], + collector: _SheetCollector, +) -> None: + for hyperlink in worksheet_root.iter(f"{{{_SPREADSHEET_NS}}}hyperlink"): + reference = hyperlink.get("ref") + if reference is None: + raise CorruptDocumentError("XLSX hyperlink anchor is invalid") + _top_left(reference) + relationship_id = hyperlink.get(_RELATIONSHIP_ID) + location = hyperlink.get("location") or "" + target = "" + if relationship_id is not None: + target = _relationship( + relationships, + relationship_id, + expected_type=_HYPERLINK_RELATIONSHIP, + ).target + if target and location: + target = f"{target}#{_quoted_location(location)}" + elif location: + target = location if location.startswith("[") else _internal_target(location) + if not target: + raise CorruptDocumentError("XLSX hyperlink target is missing") + safe = _safe_link_target(target) + ordinal = collector.add_object( + anchor=reference, + kind_rank=_KIND_HYPERLINK, + link_label=hyperlink.get("display"), + link_target=target, + safe_link=safe, + ) + if not safe: + collector.add_warning( + "xlsx_external_reference", + anchor=reference, + object_ordinal=ordinal, + detail="hyperlink target was preserved as plain text without access", + ) + + +def _drawing_anchor(anchor: Any) -> str: + local_name = anchor.tag.rsplit("}", 1)[-1] + if local_name == "absoluteAnchor": + return "A1" + start = anchor.find(f"{{{_DRAWING_NS}}}from") + if start is None: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + + def coordinate(marker: Any) -> tuple[int, int]: + try: + column = int(marker.findtext(f"{{{_DRAWING_NS}}}col", "")) + 1 + row = int(marker.findtext(f"{{{_DRAWING_NS}}}row", "")) + 1 + except ValueError as error: + raise CorruptDocumentError("XLSX drawing anchor is invalid") from error + if not 1 <= column <= 16_384 or not 1 <= row <= 1_048_576: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + return row, column + + start_row, start_column = coordinate(start) + start_address = f"{get_column_letter(start_column)}{start_row}" + if local_name == "oneCellAnchor": + return start_address + end = anchor.find(f"{{{_DRAWING_NS}}}to") + if end is None: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + end_row, end_column = coordinate(end) + if end_row < start_row or end_column < start_column: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + end_address = f"{get_column_letter(end_column)}{end_row}" + return start_address if start_address == end_address else f"{start_address}:{end_address}" + + +def _drawing_text_boxes( + archive: ZipFile, + relationship: Any, + collector: _SheetCollector, +) -> None: + if relationship.external: + raise CorruptDocumentError("XLSX drawing relationship is invalid") + root = _safe_root(archive, relationship.target, message="XLSX drawing part is corrupt") + if root.tag != f"{{{_DRAWING_NS}}}wsDr": + raise CorruptDocumentError("XLSX drawing namespace is invalid") + for anchor in root: + if anchor.tag not in { + f"{{{_DRAWING_NS}}}oneCellAnchor", + f"{{{_DRAWING_NS}}}twoCellAnchor", + f"{{{_DRAWING_NS}}}absoluteAnchor", + }: + continue + address = _drawing_anchor(anchor) + for shape in anchor.iter(f"{{{_DRAWING_NS}}}sp"): + paragraphs: list[str] = [] + text_paragraphs = [ + "".join(node.text or "" for node in paragraph.iter(f"{{{_DRAWING_MAIN_NS}}}t")) + for paragraph in shape.iter(f"{{{_DRAWING_MAIN_NS}}}p") + ] + text = "\n".join(value for value in text_paragraphs if value) + if text: + paragraphs.append(f"Text box: {text}") + metadata = shape.find(f".//{{{_DRAWING_NS}}}cNvPr") + if metadata is not None: + for attribute, label in ( + ("name", "Shape name"), + ("descr", "Alt text"), + ("title", "Alt title"), + ): + value = metadata.get(attribute) + if value: + paragraphs.append(f"{label}: {value}") + if paragraphs: + collector.add_object( + anchor=address, + kind_rank=_KIND_TEXT_BOX, + paragraphs=tuple(paragraphs), + ) + + +def _split_header_footer(raw: str) -> tuple[dict[str, str], set[str]]: + sections = {"L": [], "C": [], "R": []} + image_sections: set[str] = set() + current = "C" + index = 0 + while index < len(raw): + if raw[index] != "&": + sections[current].append(raw[index]) + index += 1 + continue + if index + 1 >= len(raw): + sections[current].append("&") + break + code = raw[index + 1] + index += 2 + if code == "&": + sections[current].append("&") + elif code in sections: + current = code + elif code == '"': + closing = raw.find('"', index) + index = len(raw) if closing < 0 else closing + 1 + elif code == "K": + index = min(index + 6, len(raw)) + elif code == "[": + closing = raw.find("]", index) + if closing < 0: + sections[current].append("&[") + continue + field = raw[index:closing] + index = closing + 1 + if field == "Picture": + image_sections.add(current) + elif field in _HEADER_NAMED_FIELDS: + sections[current].append(_HEADER_NAMED_FIELDS[field]) + else: + sections[current].append(f"&[{field}]") + elif code.isdigit(): + while index < len(raw) and raw[index].isdigit(): + index += 1 + elif code in _HEADER_FIELDS: + sections[current].append(_HEADER_FIELDS[code]) + elif code == "G": + image_sections.add(current) + elif code in _HEADER_FORMAT_CODES: + continue + else: + sections[current].append(f"&{code}") + return ( + {section: "".join(value).strip() for section, value in sections.items()}, + image_sections, + ) + + +def _headers_and_footers(worksheet_root: Any, collector: _SheetCollector) -> None: + for tag, label in _HEADER_TAGS: + element = worksheet_root.find(f".//{{{_SPREADSHEET_NS}}}{tag}") + if element is None or element.text is None: + continue + sections, image_sections = _split_header_footer(element.text) + for section, section_label in _HEADER_SECTIONS: + value = sections[section] + ordinal: int | None = None + if value: + ordinal = collector.add_object( + anchor="A1", + kind_rank=_KIND_HEADER_FOOTER, + paragraphs=(f"{label} {section_label}: {value}",), + ) + if section in image_sections: + collector.add_warning( + "xlsx_unsupported_object", + anchor="A1", + object_ordinal=ordinal, + detail="header/footer image field was skipped", + ) + + +def _visible_vml_text(archive: ZipFile, part_name: str) -> bool: + root = _safe_root(archive, part_name, message="XLSX VML drawing part is corrupt") + return any( + "".join(node.itertext()).strip() + for node in root.iter() + if isinstance(node.tag, str) and node.tag.rsplit("}", 1)[-1] == "textbox" + ) + + +def _unsupported_objects( + archive: ZipFile, + worksheet_root: Any, + collector: _SheetCollector, +) -> None: + for reference in collector.sheet.unsupported_objects: + if reference.kind == "vmlDrawing" and not _visible_vml_text(archive, reference.target): + continue + kind = _UNSUPPORTED_KIND.sub("_", reference.kind).strip("_") or "unknown" + collector.add_warning( + "xlsx_unsupported_object", + anchor="A1", + object_ordinal=reference.source_index + 1, + detail=f"unsupported {kind} object was skipped", + ) + for _extension in worksheet_root.iter(f"{{{_SPREADSHEET_NS}}}ext"): + collector.add_warning( + "xlsx_unsupported_object", + anchor="A1", + detail="unsupported vendor extension was skipped", + ) + + +def _external_reference( + collector: _SheetCollector, + target: str, + *, + detail: str, +) -> None: + ordinal = collector.add_object( + anchor="A1", + kind_rank=_KIND_EXTERNAL_REFERENCE, + paragraphs=(f"External reference: {target}",), + ) + collector.add_warning( + "xlsx_external_reference", + anchor="A1", + object_ordinal=ordinal, + detail=detail, + ) + + +def _workbook_external_references( + archive: ZipFile, + infos: dict[str, ZipInfo], + collector: _SheetCollector, +) -> None: + relationships = _relationship_index(archive, infos, "xl/workbook.xml") + for relationship in relationships.values(): + if relationship.relationship_type == _EXTERNAL_LINK_RELATIONSHIP: + if relationship.external: + _external_reference( + collector, + relationship.target, + detail="external workbook reference was preserved without access", + ) + continue + nested = _relationship_index(archive, infos, relationship.target) + for nested_relationship in nested.values(): + if nested_relationship.relationship_type == _EXTERNAL_LINK_PATH_RELATIONSHIP: + _external_reference( + collector, + nested_relationship.target, + detail="external workbook reference was preserved without access", + ) + elif relationship.relationship_type == _CONNECTIONS_RELATIONSHIP: + if relationship.external: + raise CorruptDocumentError("XLSX connections relationship is invalid") + root = _safe_root( + archive, + relationship.target, + message="XLSX connections part is corrupt", + ) + if root.tag != f"{{{_SPREADSHEET_NS}}}connections": + raise CorruptDocumentError("XLSX connections namespace is invalid") + for connection in root.findall(f"{{{_SPREADSHEET_NS}}}connection"): + for attribute in ("sourceFile", "odcFile", "connectionFile"): + target = connection.get(attribute) + if target: + _external_reference( + collector, + target, + detail="external data reference was preserved without access", + ) + + +def read_xlsx_text_objects(path: Path, preflight: XlsxPreflight) -> XlsxTextObjects: + collectors = { + sheet.sheet_index: _SheetCollector(sheet=sheet, objects=[], warnings=[]) + for sheet in preflight.sheets + } + try: + with ZipFile(path) as archive: + infos = {info.filename: info for info in archive.infolist()} + people = _person_names(archive, infos) + for sheet in preflight.sheets: + collector = collectors[sheet.sheet_index] + worksheet_root = _safe_root( + archive, + sheet.part_name, + message="XLSX worksheet part is corrupt", + ) + if sheet.kind.value != "worksheet": + _unsupported_objects(archive, worksheet_root, collector) + continue + if worksheet_root.tag != f"{{{_SPREADSHEET_NS}}}worksheet": + raise CorruptDocumentError("XLSX worksheet namespace is invalid") + relationships = _relationship_index(archive, infos, sheet.part_name) + _hyperlinks(worksheet_root, relationships, collector) + for relationship in relationships.values(): + if relationship.relationship_type == _COMMENTS_RELATIONSHIP: + _classic_comments(archive, relationship, collector) + elif relationship.relationship_type == _THREADED_COMMENTS_RELATIONSHIP: + _threaded_comments(archive, relationship, people, collector) + for drawing in worksheet_root.iter(f"{{{_SPREADSHEET_NS}}}drawing"): + relationship = _relationship( + relationships, + drawing.get(_RELATIONSHIP_ID), + expected_type=_DRAWING_RELATIONSHIP, + ) + _drawing_text_boxes(archive, relationship, collector) + _headers_and_footers(worksheet_root, collector) + _unsupported_objects(archive, worksheet_root, collector) + if preflight.sheets: + _workbook_external_references(archive, infos, collectors[1]) + except BadZipFile as error: + raise CorruptDocumentError("XLSX package is corrupt") from error + except OSError as error: + raise CorruptDocumentError("XLSX package could not be read") from error + warnings = tuple( + warning for sheet in preflight.sheets for warning in collectors[sheet.sheet_index].warnings + ) + return XlsxTextObjects( + by_sheet=tuple(tuple(collectors[sheet.sheet_index].objects) for sheet in preflight.sheets), + warnings=warnings, + ) + + +def text_object_blocks( + item: XlsxTextObject, + *, + fallback_label: str, + object_index: int, +) -> tuple[Block, ...]: + marker = MarkdownBlock( + f"" + ) + if item.link_target is not None: + label = item.link_label or fallback_label or item.link_target + inline = ( + InlineLink(label, item.link_target) + if item.safe_link + else InlineText(label if label == item.link_target else f"{label} ({item.link_target})") + ) + return marker, ParagraphBlock((inline,)) + return (marker, *(ParagraphBlock((InlineText(text),)) for text in item.paragraphs)) diff --git a/src/opendocs/parsers/xlsx/values.py b/src/opendocs/parsers/xlsx/values.py new file mode 100644 index 0000000..09ea0bd --- /dev/null +++ b/src/opendocs/parsers/xlsx/values.py @@ -0,0 +1,374 @@ +from __future__ import annotations + +import math +import re +from dataclasses import dataclass +from datetime import date, datetime, time, timedelta +from decimal import ROUND_HALF_UP, Decimal, InvalidOperation + +from openpyxl.styles.numbers import is_date_format, is_timedelta_format +from openpyxl.utils.datetime import from_excel + +_CURRENCY_SYMBOLS = frozenset({"$", "€", "£", "¥"}) +_BRACKET_TOKEN = re.compile(r"\[([^]]+)]") +_SCIENTIFIC_FORMAT = re.compile(r"[0#?](?:\.[0#?]+)?[Ee][+-]?[0#?]+") +_FRACTION_FORMAT = re.compile(r"(?:^|[^A-Za-z])[0#?]+(?:\s+[0#?]+)?/[0#?]+") +_LOCALE_CURRENCY = re.compile(r"^\$([^\]-]*)-[0-9A-Fa-f]+$") +_QUOTED_LITERAL = re.compile(r'"([^"]*)"') + + +@dataclass(frozen=True, slots=True) +class FormattedSavedValue: + text: str + warning: str | None = None + + +def _stable_decimal(value: Decimal) -> str: + if not value.is_finite(): + return str(value) + rendered = format(value, "f") + if "." in rendered: + rendered = rendered.rstrip("0").rstrip(".") + return rendered or "0" + + +def stable_raw_value(value: object) -> str: + if value is None: + return "" + if isinstance(value, bool): + return "TRUE" if value else "FALSE" + if isinstance(value, datetime): + return value.isoformat(sep=" ", timespec="seconds") + if isinstance(value, date): + return value.isoformat() + if isinstance(value, time): + return value.isoformat(timespec="seconds") + if isinstance(value, timedelta): + return _stable_decimal(Decimal(str(value.total_seconds()))) + if isinstance(value, Decimal): + return _stable_decimal(value) + if isinstance(value, int): + return str(value) + if isinstance(value, float): + if not math.isfinite(value): + return str(value) + return _stable_decimal(Decimal(str(value))) + return str(value) + + +def _split_sections(number_format: str) -> tuple[str, ...]: + sections: list[str] = [] + current: list[str] = [] + quoted = False + bracket_depth = 0 + escaped = False + for character in number_format: + if escaped: + current.append(character) + escaped = False + elif character == "\\": + current.append(character) + escaped = True + elif character == '"': + current.append(character) + quoted = not quoted + elif not quoted and character == "[": + bracket_depth += 1 + current.append(character) + elif not quoted and character == "]": + bracket_depth = max(0, bracket_depth - 1) + current.append(character) + elif not quoted and bracket_depth == 0 and character == ";": + sections.append("".join(current)) + current = [] + else: + current.append(character) + sections.append("".join(current)) + return tuple(sections) + + +def _has_unsupported_token(number_format: str) -> bool: + if _SCIENTIFIC_FORMAT.search(number_format) or _FRACTION_FORMAT.search(number_format): + return True + for literal in _QUOTED_LITERAL.findall(number_format): + if any(character not in {*_CURRENCY_SYMBOLS, " ", "-", "(", ")"} for character in literal): + return True + for match in _BRACKET_TOKEN.finditer(number_format): + token = match.group(1) + if token.casefold() in {"h", "hh", "m", "mm", "s", "ss"}: + continue + currency = _LOCALE_CURRENCY.fullmatch(token) + if currency is not None and currency.group(1) in _CURRENCY_SYMBOLS: + continue + return True + remaining = _QUOTED_LITERAL.sub("", number_format) + remaining = _BRACKET_TOKEN.sub("", remaining) + remaining = remaining.casefold().replace("am/pm", "") + index = 0 + while index < len(remaining): + character = remaining[index] + if character in {"_", "*"}: + index += 2 + continue + if character == "\\" and index + 1 < len(remaining): + if remaining[index + 1].isalnum(): + return True + index += 2 + continue + if character.isalpha() and character not in "ymdhs": + return True + index += 1 + return False + + +def _clean_section(section: str) -> str: + cleaned: list[str] = [] + index = 0 + while index < len(section): + character = section[index] + if character in {"_", "*"}: + index += 2 + continue + if character == "\\" and index + 1 < len(section): + cleaned.append(section[index + 1]) + index += 2 + continue + if character == '"': + end = section.find('"', index + 1) + if end == -1: + return section + cleaned.append(section[index + 1 : end]) + index = end + 1 + continue + if character == "[": + end = section.find("]", index + 1) + if end == -1: + return section + token = section[index + 1 : end] + currency = _LOCALE_CURRENCY.fullmatch(token) + cleaned.append(currency.group(1) if currency is not None else f"[{token}]") + index = end + 1 + continue + cleaned.append(character) + index += 1 + return "".join(cleaned).strip() + + +def _as_decimal(value: object) -> Decimal | None: + if isinstance(value, bool): + return None + if isinstance(value, Decimal): + return value + if isinstance(value, int): + return Decimal(value) + if isinstance(value, float) and math.isfinite(value): + return Decimal(str(value)) + return None + + +def _selected_numeric_section(sections: tuple[str, ...], value: Decimal) -> tuple[str, bool]: + if value < 0: + return (sections[1] if len(sections) > 1 else sections[0], True) + if value == 0 and len(sections) > 2: + return sections[2], False + return sections[0], False + + +def _decimal_places(pattern: str) -> tuple[int, int]: + if "." not in pattern: + return 0, 0 + suffix = pattern.split(".", 1)[1] + placeholders: list[str] = [] + for character in suffix: + if character in "0#?": + placeholders.append(character) + elif placeholders: + break + return placeholders.count("0"), len(placeholders) + + +def _format_number(value: Decimal, number_format: str) -> str | None: + sections = _split_sections(number_format) + section, negative = _selected_numeric_section(sections, value) + pattern = _clean_section(section) + if value == 0 and "-" in pattern: + symbol = next((item for item in _CURRENCY_SYMBOLS if item in pattern), "") + if symbol: + return f"{symbol}-" + if not any(character in "0#?" for character in pattern): + return None + + percent_count = pattern.count("%") + magnitude = abs(value) * (Decimal(100) ** percent_count) + integer_pattern = pattern.split(".", 1)[0] + scale_commas = len(integer_pattern) - len(integer_pattern.rstrip(",")) + if scale_commas: + magnitude /= Decimal(1000) ** scale_commas + integer_pattern = integer_pattern.rstrip(",") + minimum_decimals, maximum_decimals = _decimal_places(pattern) + if maximum_decimals: + quantum = Decimal(1).scaleb(-maximum_decimals) + magnitude = magnitude.quantize(quantum, rounding=ROUND_HALF_UP) + rendered = f"{magnitude:.{maximum_decimals}f}" + integer, fraction = rendered.split(".", 1) + if maximum_decimals > minimum_decimals: + fraction = fraction.rstrip("0") + fraction += "0" * max(0, minimum_decimals - len(fraction)) + rendered = integer + (f".{fraction}" if fraction else "") + else: + rendered = str(magnitude.quantize(Decimal(1), rounding=ROUND_HALF_UP)) + + integer, separator, fraction = rendered.partition(".") + if "," in integer_pattern: + integer = f"{int(integer):,}" + rendered = integer + (separator + fraction if separator else "") + + currency = next((symbol for symbol in _CURRENCY_SYMBOLS if symbol in pattern), "") + first_placeholder = min( + (pattern.find(character) for character in "0#?" if character in pattern), + default=0, + ) + if currency: + rendered = ( + f"{currency}{rendered}" + if pattern.find(currency) <= first_placeholder + else f"{rendered}{currency}" + ) + if percent_count: + rendered += "%" * percent_count + if negative: + rendered = f"({rendered})" if "(" in pattern and ")" in pattern else f"-{rendered}" + return rendered + + +def _time_fraction_digits(number_format: str) -> int: + match = re.search(r"s{1,2}\.([0#]+)", number_format, flags=re.IGNORECASE) + return len(match.group(1)) if match is not None else 0 + + +def _render_clock(value: time, number_format: str) -> str: + lowered = number_format.casefold() + include_seconds = "s" in lowered + twelve_hour = "am/pm" in lowered + raw_hour = value.hour % 12 or 12 if twelve_hour else value.hour + hour_token = re.search(r"(? str: + total_seconds = Decimal(str(value.total_seconds())) + digits = _time_fraction_digits(number_format) + quantum = Decimal(1).scaleb(-digits) if digits else Decimal(1) + total_seconds = total_seconds.quantize(quantum, rounding=ROUND_HALF_UP) + negative = total_seconds < 0 + total_seconds = abs(total_seconds) + whole_seconds = int(total_seconds) + fraction = total_seconds - whole_seconds + lowered = number_format.casefold() + if "[h]" in lowered or "[hh]" in lowered: + total_hours = whole_seconds // 3600 + rendered_hours = f"{total_hours:02d}" if "[hh]" in lowered else str(total_hours) + rendered = f"{rendered_hours}:{whole_seconds % 3600 // 60:02d}" + if "s" in lowered: + rendered += f":{whole_seconds % 60:02d}" + elif "[m]" in lowered or "[mm]" in lowered: + total_minutes = whole_seconds // 60 + rendered = f"{total_minutes:02d}" if "[mm]" in lowered else str(total_minutes) + if "s" in lowered: + rendered += f":{whole_seconds % 60:02d}" + else: + rendered = str(whole_seconds) + if digits: + rendered += f".{str(fraction)[2:]:0<{digits}}"[: digits + 1] + return f"-{rendered}" if negative else rendered + + +def _format_date_value(value: object, number_format: str, epoch: datetime) -> str | None: + converted = value + decimal = _as_decimal(value) + try: + if decimal is not None: + converted = from_excel( + float(decimal), + epoch, + timedelta=is_timedelta_format(number_format), + ) + except (OverflowError, ValueError): + return None + if isinstance(converted, timedelta): + return _render_elapsed(converted, number_format) + if isinstance(converted, datetime): + lowered = number_format.casefold() + has_date = "y" in lowered or "d" in lowered + has_time = "h" in lowered or "s" in lowered or not has_date + if has_date and has_time: + return ( + f"{converted.date().isoformat()} {_render_clock(converted.time(), number_format)}" + ) + if has_date: + return converted.date().isoformat() + return _render_clock(converted.time(), number_format) + if isinstance(converted, date): + return converted.isoformat() + if isinstance(converted, time): + return _render_clock(converted, number_format) + return None + + +def format_saved_value( + value: object, + number_format: str, + *, + epoch: datetime, + conditional_number_format: bool = False, +) -> FormattedSavedValue: + raw = stable_raw_value(value) + if value is None or isinstance(value, bool | str): + return FormattedSavedValue(raw) + if conditional_number_format: + return FormattedSavedValue(raw, "unsupported number format") + if number_format.casefold() == "general": + return FormattedSavedValue(raw) + if _has_unsupported_token(number_format): + return FormattedSavedValue(raw, "unsupported number format") + if is_date_format(number_format): + rendered_date = _format_date_value(value, number_format, epoch) + if rendered_date is not None: + return FormattedSavedValue(rendered_date) + return FormattedSavedValue(raw, "unsupported number format") + decimal = _as_decimal(value) + if decimal is None: + return FormattedSavedValue(raw) + try: + rendered_number = _format_number(decimal, number_format) + except (InvalidOperation, ValueError): + rendered_number = None + if rendered_number is None: + return FormattedSavedValue(raw, "unsupported number format") + return FormattedSavedValue(rendered_number) diff --git a/tests/native_worker_helpers.py b/tests/native_worker_helpers.py index 868e543..a97890a 100644 --- a/tests/native_worker_helpers.py +++ b/tests/native_worker_helpers.py @@ -24,6 +24,10 @@ def sleep_and_echo(delay: float, value: object) -> object: return value +def hard_exit(status: int) -> None: + os._exit(status) + + def raise_corrupt(message: str) -> None: raise CorruptDocumentError(message) @@ -52,3 +56,35 @@ def noisy_echo(value: object) -> object: def make_bytes(size: int) -> bytes: return b"x" * size + + +def make_xlsx_wire(block_count: int) -> dict[str, object]: + from opendocs._models import TextBlock + from opendocs.parsers.xlsx.models import ( + XlsxDocument, + XlsxNativeSlot, + XlsxSheet, + XlsxSheetKind, + XlsxSheetState, + document_to_wire, + ) + + return document_to_wire( + XlsxDocument( + sheets=( + XlsxSheet( + sheet_index=1, + name="Sheet", + kind=XlsxSheetKind.WORKSHEET, + state=XlsxSheetState.VISIBLE, + slots=( + XlsxNativeSlot( + source_index=0, + anchor="A1", + blocks=tuple(TextBlock("x" * 200) for _ in range(block_count)), + ), + ), + ), + ) + ) + ) diff --git a/tests/test_acceptance_corpus.py b/tests/test_acceptance_corpus.py index 30cdbd3..a696ff6 100644 --- a/tests/test_acceptance_corpus.py +++ b/tests/test_acceptance_corpus.py @@ -165,3 +165,14 @@ def test_private_corpus_matches_manifest( path = corpus_dir / entry["name"] assert path.is_file(), f"missing acceptance file: {path}" assert _sha256(path) == entry["sha256"], f"hash mismatch: {path}" + + +def test_private_xlsx_gate_is_not_run_until_a_maintainer_provides_a_real_workbook() -> None: + entries = _entries() + release_plan = ( + Path(__file__).resolve().parents[1] / "docs/plans/2026-08-07-v0.2.0-release-plan.md" + ).read_text(encoding="utf-8") + + assert not any(entry["name"].lower().endswith(".xlsx") for entry in entries) + assert "XLSX 私有探索门状态: `not_run`" in release_plan + assert "不得提交真实工作簿、hash、模型输出或完成检查表" in release_plan diff --git a/tests/test_api_xlsx.py b/tests/test_api_xlsx.py new file mode 100644 index 0000000..ba95e01 --- /dev/null +++ b/tests/test_api_xlsx.py @@ -0,0 +1,440 @@ +from __future__ import annotations + +import asyncio +import io +import warnings +from collections.abc import Callable +from pathlib import Path +from typing import Any, cast +from zipfile import ZipFile + +import pytest +from openpyxl import Workbook +from openpyxl.drawing.image import Image as SpreadsheetImage +from PIL import Image + +import opendocs.api as api_module +import opendocs.source as source_module +from opendocs import ( + DocumentTimeoutError, + LimitExceededError, + OpenDocsWarning, + ParseOptions, + RuntimeDependencyError, + VisionConfig, + aparse, + parse, +) +from opendocs._runtime import ParserRuntime +from opendocs.source import Source +from tests.native_worker_helpers import hard_exit, sleep_and_echo +from tests.xlsx_fixtures import rewrite_xlsx, write_public_contract_xlsx + + +class NamedBytesIO(io.BytesIO): + def __init__(self, data: bytes, name: str) -> None: + super().__init__(data) + self.name = name + + +def _source_factory(kind: str, path: Path, content: bytes) -> Callable[[], Source]: + if kind == "path": + return lambda: path + if kind == "bytes": + return lambda: content + if kind == "named_stream": + return lambda: NamedBytesIO(content, str(path)) + if kind == "unnamed_stream": + return lambda: io.BytesIO(content) + raise AssertionError(f"unknown source kind: {kind}") + + +def _warning_codes(captured: list[warnings.WarningMessage]) -> tuple[str, ...]: + return tuple( + item.message.code for item in captured if isinstance(item.message, OpenDocsWarning) + ) + + +def _invoke( + api_kind: str, + source: Source, + *, + options: ParseOptions | None = None, + vision: VisionConfig | None = None, +) -> tuple[str, tuple[str, ...], tuple[str, ...], tuple[str, ...]]: + with warnings.catch_warnings(record=True) as captured: + warnings.simplefilter("always", OpenDocsWarning) + result = ( + parse(source, options=options, vision=vision) + if api_kind == "parse" + else asyncio.run(aparse(source, options=options, vision=vision)) + ) + public_warnings = tuple(item for item in captured if isinstance(item.message, OpenDocsWarning)) + return ( + result, + _warning_codes(captured), + tuple(str(item.message) for item in public_warnings), + tuple(item.filename for item in public_warnings), + ) + + +def test_xlsx_public_api_has_eight_equivalent_input_and_api_combinations( + tmp_path: Path, +) -> None: + path = tmp_path / "contract.xlsx" + write_public_contract_xlsx(path) + content = path.read_bytes() + outcomes: list[tuple[str, tuple[str, ...], tuple[str, ...], tuple[str, ...]]] = [] + streams: list[Source] = [] + for source_kind in ("path", "bytes", "named_stream", "unnamed_stream"): + source_factory = _source_factory(source_kind, path, content) + for api_kind in ("parse", "aparse"): + source = source_factory() + streams.append(source) + outcomes.append(_invoke(api_kind, source)) + + assert len(outcomes) == 8 + assert [outcome[:3] for outcome in outcomes] == [outcomes[0][:3]] * 8 + result, warning_codes, _messages, filenames = outcomes[0] + assert warning_codes == ( + "xlsx_formula_cache_missing", + "xlsx_unsupported_number_format", + ) + assert filenames == (__file__, __file__) + assert "# Ledger (Visible)" in result + assert "$1,234.50" in result + assert "2026-08-14" in result + assert "=B2*2" in result + assert "# Hidden (Hidden)" in result + assert "# Very Hidden (Very Hidden)" in result + assert "# Empty (Visible)" in result + assert all(not source.closed for source in streams if hasattr(source, "closed")) + + +@pytest.mark.asyncio +async def test_xlsx_async_warning_emission_points_to_the_public_api_caller( + tmp_path: Path, +) -> None: + path = tmp_path / "warning-location.xlsx" + write_public_contract_xlsx(path) + + with warnings.catch_warnings(record=True) as captured: + warnings.simplefilter("always", OpenDocsWarning) + await aparse(path) + + public_warnings = tuple(item for item in captured if isinstance(item.message, OpenDocsWarning)) + assert tuple(item.filename for item in public_warnings) == (__file__, __file__) + + +def test_xlsx_public_output_is_repeatable_and_max_pages_does_not_limit_sheets( + tmp_path: Path, +) -> None: + path = tmp_path / "repeatable.xlsx" + write_public_contract_xlsx(path) + + outcomes = [_invoke("parse", path) for _ in range(3)] + assert outcomes == [outcomes[0]] * 3 + + limited = _invoke("parse", path, options=ParseOptions(max_pages=1)) + assert limited == outcomes[0] + assert limited[0].count("" + assert sheet.name not in prelude.blocks[0].markdown + assert isinstance(prelude.blocks[1], HeadingBlock) + assert isinstance(prelude.blocks[2], ParagraphBlock) + assert len(_native_slots(document.sheets[1])) == 1 + assert len(_native_slots(document.sheets[2])) == 1 + assert len(_native_slots(document.sheets[3])) == 2 + + +def test_extract_builds_tables_regions_merges_and_ignores_style_only_cells( + tmp_path: Path, +) -> None: + path = tmp_path / "regions.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet.title = "Regions" + sheet.append(("Name", "Amount")) + sheet.append(("A", 1)) + sheet.append(("B", 2)) + table = Table(displayName="Ledger", ref="A1:B3") + table.headerRowCount = 1 + sheet.add_table(table) + sheet["D1"] = "Merged" + sheet.merge_cells("D1:E1") + sheet["D2"] = "left" + sheet["E2"] = "right" + sheet["H1"] = "first" + sheet["H3"] = "second" + sheet["J1"].font = Font(bold=True, color="FF0000") + workbook.save(path) + + document = extract_xlsx(path, preflight_xlsx(path)) + + slots = _native_slots(document.sheets[0]) + assert [slot.anchor for slot in slots] == ["A1", "A1:B3", "D1:E2", "H1", "H3"] + assert [slot.source_index for slot in slots] == list(range(5)) + assert isinstance(slots[1].blocks[1], TableBlock) + assert slots[1].blocks[1].header_rows == 1 + assert slots[1].blocks[1].grid == (("Name", "Amount"), ("A", "1"), ("B", "2")) + assert isinstance(slots[2].blocks[1], SpannedTableBlock) + assert slots[2].blocks[1].header_rows == 0 + assert [(cell.row_span, cell.column_span, cell.text) for cell in slots[2].blocks[1].cells] == [ + (1, 2, "Merged"), + (1, 1, "left"), + (1, 1, "right"), + ] + assert isinstance(slots[3].blocks[1], TableBlock) + assert slots[3].blocks[1].header_rows == 0 + assert all( + "Regions" not in block.markdown + for slot in slots + for block in slot.blocks + if isinstance(block, MarkdownBlock) + ) + assert all(slot.anchor != "J1" for slot in slots) + + +def test_excel_table_header_row_count_zero_is_not_promoted_to_header(tmp_path: Path) -> None: + path = tmp_path / "headerless.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet["A1"] = "one" + sheet["A2"] = "two" + table = Table(displayName="Headerless", ref="A1:A2") + table.headerRowCount = 0 + sheet.add_table(table) + workbook.save(path) + + document = extract_xlsx(path, preflight_xlsx(path)) + + table_block = _native_slots(document.sheets[0])[1].blocks[1] + assert isinstance(table_block, TableBlock) + assert table_block.header_rows == 0 + + +def test_extracted_native_blocks_render_to_stable_markdown(tmp_path: Path) -> None: + path = tmp_path / "golden.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet.title = "Ledger" + sheet.append(("Item", "Amount")) + sheet.append(("Book", 12.5)) + sheet["B2"].number_format = "$0.00" + workbook.save(path) + + document = extract_xlsx(path, preflight_xlsx(path)) + blocks = tuple( + block + for extracted_sheet in document.sheets + for slot in extracted_sheet.slots + if isinstance(slot, XlsxNativeSlot) + for block in slot.blocks + ) + rendered = render_markdown( + ParsedDocument(DocumentType.XLSX, blocks, document.warnings), + max_output_chars=10_000, + ) + + assert rendered.markdown == ( + "\n\n" + "# Ledger\n\n" + "Sheet state: visible\n\n" + "\n\n" + "\n" + "\n" + "\n" + "\n" + "\n" + "
ItemAmount
Book$12.50
\n" + ) + + +def test_extract_rechecks_component_bounding_box_before_materializing( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "bounded.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet["A1"] = "a" + sheet["B1"] = "b" + sheet["B2"] = "c" + workbook.save(path) + index = preflight_xlsx(path) + monkeypatch.setattr(extract_module, "MAX_MATERIALIZED_GRID_CELLS", 3) + + with pytest.raises(LimitExceededError, match="materialized grid"): + extract_xlsx(path, index) + + +def test_extract_rechecks_materialized_grid_across_all_sheets( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "bounded-workbook.xlsx" + workbook = Workbook() + first = workbook.active + first.append(("a", "b")) + second = workbook.create_sheet("Second") + second.append(("c", "d")) + workbook.save(path) + index = preflight_xlsx(path) + monkeypatch.setattr(extract_module, "MAX_MATERIALIZED_GRID_CELLS", 3) + + with pytest.raises(LimitExceededError, match="materialized grid"): + extract_xlsx(path, index) + + +def test_extract_aggregates_number_format_warnings_after_twenty_coordinates( + tmp_path: Path, +) -> None: + path = tmp_path / "warnings.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet.title = "Warnings" + for row in range(1, 23): + cell = sheet.cell(row, 1, row) + cell.number_format = "0.00E+00" + workbook.save(path) + + document = extract_xlsx(path, preflight_xlsx(path)) + + warnings = [ + warning for warning in document.warnings if warning.code == "xlsx_unsupported_number_format" + ] + assert len(warnings) == 21 + assert warnings[0].message.startswith("Warnings!A1:") + assert warnings[19].message.startswith("Warnings!A20:") + assert warnings[20].message == "2 additional xlsx_unsupported_number_format warnings suppressed" + + +def test_extract_falls_back_when_conditional_rule_can_change_number_format( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "conditional-number-format.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet["A1"] = 1234.5 + sheet.conditional_formatting.add( + "A1", + Rule( + type="expression", + formula=("1",), + dxf=DifferentialStyle(numFmt=NumberFormat(numFmtId=164, formatCode="$0.00")), + ), + ) + workbook.save(path) + + def forbidden_linear_scan(coordinate: str, ranges: tuple[Any, ...]) -> bool: + del coordinate, ranges + raise AssertionError("conditional formats must not scan every range per cell") + + monkeypatch.setattr( + extract_module, + "_has_conditional_number_format", + forbidden_linear_scan, + raising=False, + ) + + document = extract_xlsx(path, preflight_xlsx(path)) + + region = _native_slots(document.sheets[0])[1] + assert isinstance(region.blocks[1], TableBlock) + assert region.blocks[1].grid == (("1234.5",),) + assert [warning.code for warning in document.warnings] == ["xlsx_unsupported_number_format"] + + +def test_extract_loads_full_mode_openpyxl_once_with_links_disabled( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "once.xlsx" + workbook = Workbook() + workbook.active["A1"] = "value" + workbook.save(path) + index = preflight_xlsx(path) + calls: list[tuple[object, dict[str, object], bool]] = [] + original = extract_module.openpyxl.load_workbook + + def recording_loader(filename: object, **kwargs: object) -> object: + calls.append((filename, kwargs, bool(getattr(filename, "closed", True)))) + return original(filename, **kwargs) + + monkeypatch.setattr(extract_module.openpyxl, "load_workbook", recording_loader) + + extract_xlsx(path, index) + + assert len(calls) == 1 + filename, kwargs, was_closed = calls[0] + assert Path(cast(Any, filename).name) == path + assert was_closed is False + assert kwargs == { + "read_only": False, + "data_only": False, + "rich_text": False, + "keep_links": False, + } + + +def test_formula_sidecar_prefers_cache_and_distinguishes_missing_empty_and_special_formulas( + tmp_path: Path, +) -> None: + path = tmp_path / "formulas.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet.title = "Formulas" + for row in range(1, 9): + sheet.cell(row, 1, row) + sheet.cell(row, 2, f"=A{row}*2") + sheet["B1"].number_format = "$#,##0.00" + workbook.save(path) + with ZipFile(path) as archive: + xml = archive.read("xl/worksheets/sheet1.xml") + replacements = { + b'A1*2': b'A1*22', + b'A2*2': b'A2*2', + b'A3*2': b'A3*2', + b'A4*2': b'[1]Sheet1!A1', + b'A5*2': ( + b'SUM(A5:A6)10' + ), + b'A6*2': ( + b'' + ), + b'A7*2': ( + b'A7*2' + ), + b'A8*2': b'', + } + for old, new in replacements.items(): + assert old in xml + xml = xml.replace(old, new) + rewrite_xlsx(path, {"xl/worksheets/sheet1.xml": xml}) + + document = extract_xlsx(path, preflight_xlsx(path)) + + text_by_anchor = { + slot.anchor: "\n".join(cell for row in slot.blocks[1].grid for cell in row) + for slot in _native_slots(document.sheets[0])[1:] + if isinstance(slot.blocks[1], TableBlock) + } + all_text = "\n".join(text_by_anchor.values()) + assert "$2.00" in all_text + assert "=A2*2" in all_text + assert "=A3*2" not in all_text + assert "=[1]Sheet1!A1" in all_text + assert "Array/spill formula B5:C5: =SUM(A5:A6); saved value: 10" in all_text + assert "Data-table formula B6:C7 (dt2D=1, r1=A1, r2=A2)" in all_text + assert "=A7*2" in all_text + assert "=A8*2" in all_text + codes = [warning.code for warning in document.warnings] + assert codes.count("xlsx_formula_cache_missing") == 4 + assert codes.count("xlsx_data_table_formula") == 1 + assert codes.count("xlsx_external_reference") == 1 + + +def test_repeated_extraction_is_deterministic(tmp_path: Path) -> None: + path = tmp_path / "repeat.xlsx" + _save_ordered_workbook(path) + index = preflight_xlsx(path) + + assert extract_xlsx(path, index) == extract_xlsx(path, index) + + +def test_extracts_comments_threaded_text_boxes_links_and_headers_in_stable_order( + tmp_path: Path, +) -> None: + path = tmp_path / "text-objects.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet.title = "Objects" + sheet.append(("HTTP [docs]", "HTTPS", "Mail", "Jump")) + workbook.save(path) + with ZipFile(path) as archive: + worksheet_xml = _with_relationship_namespace(archive.read("xl/worksheets/sheet1.xml")) + workbook_rels = archive.read("xl/_rels/workbook.xml.rels") + worksheet_xml = _append_before( + worksheet_xml, + b"", + ( + '' + '' + '' + '' + '' + "" + "&LOdd header left && kept &P &N" + "&C&D &T&R&"Arial"&KFF0000&12&B" + "Odd header right &F &Z &A &G" + "&LOdd footer left&COdd footer center&ROdd footer right" + "" + "&LEven header left&CEven header center&REven header right" + "" + "&LEven footer left&CEven footer center&REven footer right" + "" + "&LFirst header left&CFirst header center&RFirst header right" + "" + "&LFirst footer left&CFirst footer center&RFirst footer right" + "" + "" + ), + ) + workbook_rels = _append_before( + workbook_rels, + b"", + ( + f'' + ), + ) + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": worksheet_xml, + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ("rIdHttp", f"{OFFICE_REL_NS}/hyperlink", "http://example.test/a(b)", True), + ("rIdHttps", f"{OFFICE_REL_NS}/hyperlink", "https://example.test/two", True), + ("rIdMail", f"{OFFICE_REL_NS}/hyperlink", "mailto:team@example.test", True), + ("rIdComment", f"{OFFICE_REL_NS}/comments", "../comments1.xml", False), + ( + "rIdThreaded", + f"{THREADED_REL_NS}/threadedComment", + "../threadedComments/threadedComment1.xml", + False, + ), + ("rIdDrawing", f"{OFFICE_REL_NS}/drawing", "../drawings/drawing1.xml", False), + ), + "xl/comments1.xml": ( + f'Alice' + 'Classic note' + "" + ).encode(), + "xl/threadedComments/threadedComment1.xml": ( + f'Threaded reply' + "" + ).encode(), + "xl/persons/person.xml": ( + f'' + "" + ).encode(), + "xl/_rels/workbook.xml.rels": workbook_rels, + "xl/drawings/drawing1.xml": ( + f'' + "41" + '' + '' + "Box & text" + "" + "" + ).encode(), + }, + ) + + document = extract_xlsx(path, preflight_xlsx(path)) + + slots = _native_slots(document.sheets[0]) + assert [slot.source_index for slot in slots] == list(range(len(slots))) + assert slots[0].blocks[0] == MarkdownBlock("") + links = [ + inline + for slot in slots + for block in slot.blocks + if isinstance(block, ParagraphBlock) + for inline in block.inlines + if isinstance(inline, InlineLink) + ] + assert [(link.label, link.target) for link in links] == [ + ("HTTP [docs]", "http://example.test/a(b)"), + ("HTTPS", "https://example.test/two"), + ("Mail", "mailto:team@example.test"), + ("Jump", "#Objects!A1"), + ] + all_text = [_paragraph_text(slot) for slot in slots] + assert "Comment by Alice: Classic note" in all_text + assert "Threaded comment by Bob: Threaded reply" in all_text + assert any( + text == "Text box: Box & text\nShape name: Callout 1\nAlt text: Shape details\n" + "Alt title: Shape title" + for text in all_text + ) + header_text = [text for text in all_text if text.startswith(("Odd ", "Even ", "First "))] + assert header_text == [ + "Odd header left: Odd header left & kept {page} {pages}", + "Odd header center: {date} {time}", + "Odd header right: Odd header right {file} {path} {sheet}", + "Odd footer left: Odd footer left", + "Odd footer center: Odd footer center", + "Odd footer right: Odd footer right", + "Even header left: Even header left", + "Even header center: Even header center", + "Even header right: Even header right", + "Even footer left: Even footer left", + "Even footer center: Even footer center", + "Even footer right: Even footer right", + "First header left: First header left", + "First header center: First header center", + "First header right: First header right", + "First footer left: First footer left", + "First footer center: First footer center", + "First footer right: First footer right", + ] + assert all("Arial" not in text and "FF0000" not in text for text in header_text) + assert [warning.code for warning in document.warnings] == ["xlsx_unsupported_object"] + assert "sheet=1 anchor=A1" in document.warnings[0].message + assert "header/footer image field" in document.warnings[0].message + assert all( + "Objects" not in block.markdown + for slot in slots[1:] + for block in slot.blocks + if isinstance(block, MarkdownBlock) + ) + rendered = render_markdown( + ParsedDocument( + DocumentType.XLSX, + tuple(block for slot in slots for block in slot.blocks), + document.warnings, + ), + max_output_chars=100_000, + ) + assert "[HTTP \\[docs\\]](http://example.test/a\\(b\\))" in rendered.markdown + + +def test_standard_note_vml_presentation_is_not_reported_as_lost_text(tmp_path: Path) -> None: + path = tmp_path / "standard-note.xlsx" + workbook = Workbook() + workbook.active["B2"].comment = Comment("A standard note", "Reviewer") + workbook.save(path) + + document = extract_xlsx(path, preflight_xlsx(path)) + + paragraphs = [_paragraph_text(slot) for slot in _native_slots(document.sheets[0])] + assert "Comment by Reviewer: A standard note" in paragraphs + assert not any( + warning.code == "xlsx_unsupported_object" and "vmlDrawing" in warning.message + for warning in document.warnings + ) + + +def test_unsafe_hyperlinks_and_remote_data_are_plain_text_and_never_accessed( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "external.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet.append(("Script", "File", "Book")) + workbook.save(path) + with ZipFile(path) as archive: + worksheet_xml = _with_relationship_namespace(archive.read("xl/worksheets/sheet1.xml")) + workbook_rels = archive.read("xl/_rels/workbook.xml.rels") + worksheet_xml = _append_before( + worksheet_xml, + b"", + ( + '' + '' + '' + ), + ) + workbook_rels = _append_before( + workbook_rels, + b"", + ( + f'' + f'' + ), + ) + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": worksheet_xml, + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ("rIdScript", f"{OFFICE_REL_NS}/hyperlink", "javascript:alert(1)", True), + ("rIdFile", f"{OFFICE_REL_NS}/hyperlink", "file:///tmp/private.xlsx", True), + ("rIdBook", f"{OFFICE_REL_NS}/hyperlink", "../other.xlsx", True), + ), + "xl/_rels/workbook.xml.rels": workbook_rels, + "xl/externalLinks/externalLink1.xml": (f'').encode(), + "xl/externalLinks/_rels/externalLink1.xml.rels": _relationships( + ( + "rIdPath", + f"{OFFICE_REL_NS}/externalLinkPath", + "https://remote.example.test/book.xlsx", + True, + ), + ), + "xl/connections.xml": ( + f'' + "" + ).encode(), + }, + ) + + def deny_network(*args: object, **kwargs: object) -> None: + raise AssertionError(f"network access attempted: {args!r} {kwargs!r}") + + monkeypatch.setattr(socket, "create_connection", deny_network) + monkeypatch.setattr(urllib.request, "urlopen", deny_network) + for module_name in ("httpx", "requests"): + module = pytest.importorskip(module_name) + monkeypatch.setattr(module, "get", deny_network) + + document = extract_xlsx(path, preflight_xlsx(path)) + + paragraphs = [ + block + for slot in _native_slots(document.sheets[0]) + for block in slot.blocks + if isinstance(block, ParagraphBlock) + ] + assert not any( + isinstance(inline, InlineLink) for paragraph in paragraphs for inline in paragraph.inlines + ) + plain = "\n".join( + inline.text + for paragraph in paragraphs + for inline in paragraph.inlines + if isinstance(inline, InlineText) + ) + assert "Script (javascript:alert(1))" in plain + assert "File (file:///tmp/private.xlsx)" in plain + assert "Book (../other.xlsx)" in plain + assert "https://remote.example.test/book.xlsx" in plain + assert "https://remote.example.test/data.csv" in plain + external_warnings = [ + warning for warning in document.warnings if warning.code == "xlsx_external_reference" + ] + assert len(external_warnings) == 5 + assert all("sheet=1" in warning.message for warning in external_warnings) + + +def test_malformed_hyperlink_target_is_plain_text() -> None: + assert text_objects_module._safe_link_target("http://[::1") is False + + +def test_unsupported_xlsx_objects_are_locatable_and_aggregated(tmp_path: Path) -> None: + path = tmp_path / "unsupported.xlsx" + workbook = Workbook() + workbook.active["A1"] = "kept" + workbook.save(path) + with ZipFile(path) as archive: + worksheet_xml = _with_relationship_namespace(archive.read("xl/worksheets/sheet1.xml")) + worksheet_xml = _append_before( + worksheet_xml, + b"", + '', + ) + unsupported = ( + ("rIdSmartArt", f"{OFFICE_REL_NS}/diagramData", "../diagrams/data1.xml", False), + ("rIdOle", f"{OFFICE_REL_NS}/oleObject", "../embeddings/ole1.bin", False), + ("rIdActiveX", f"{OFFICE_REL_NS}/activeXControl", "../activeX/activeX1.xml", False), + ("rIdControl", f"{OFFICE_REL_NS}/control", "../controls/control1.xml", False), + ("rIdVml", f"{OFFICE_REL_NS}/vmlDrawing", "../drawings/vmlDrawing1.vml", False), + ) + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": worksheet_xml, + "xl/worksheets/_rels/sheet1.xml.rels": _relationships(*unsupported), + "xl/diagrams/data1.xml": b'', + "xl/embeddings/ole1.bin": b"ole", + "xl/activeX/activeX1.xml": b'', + "xl/controls/control1.xml": b'', + "xl/drawings/vmlDrawing1.vml": ( + b'' + b"VML text" + ), + }, + ) + + document = extract_xlsx(path, preflight_xlsx(path)) + + warnings = [ + warning for warning in document.warnings if warning.code == "xlsx_unsupported_object" + ] + assert len(warnings) == 6 + assert all("sheet=1" in warning.message for warning in warnings) + assert all("object=" in warning.message for warning in warnings) + assert any("diagramData" in warning.message for warning in warnings) + assert any("oleObject" in warning.message for warning in warnings) + assert any("activeXControl" in warning.message for warning in warnings) + assert any("control" in warning.message for warning in warnings) + assert any("vmlDrawing" in warning.message for warning in warnings) + assert any("vendor extension" in warning.message for warning in warnings) + + +def test_unsupported_object_warnings_keep_twenty_then_summarize(tmp_path: Path) -> None: + path = tmp_path / "many-unsupported.xlsx" + workbook = Workbook() + workbook.active["A1"] = "kept" + workbook.save(path) + with ZipFile(path) as archive: + worksheet_xml = archive.read("xl/worksheets/sheet1.xml") + extensions = "".join( + f'' for index in range(22) + ) + worksheet_xml = _append_before( + worksheet_xml, + b"", + f"{extensions}", + ) + rewrite_xlsx(path, {"xl/worksheets/sheet1.xml": worksheet_xml}) + + document = extract_xlsx(path, preflight_xlsx(path)) + + warnings = [ + warning for warning in document.warnings if warning.code == "xlsx_unsupported_object" + ] + assert len(warnings) == 21 + assert warnings[-1].message == "2 additional xlsx_unsupported_object warnings suppressed" + + +@pytest.mark.parametrize("relationship_id", ["missing", "wrong-type"]) +def test_hyperlink_relationship_must_exist_with_the_expected_type( + tmp_path: Path, + relationship_id: str, +) -> None: + path = tmp_path / f"bad-link-{relationship_id}.xlsx" + workbook = Workbook() + workbook.active["A1"] = "link" + workbook.save(path) + with ZipFile(path) as archive: + worksheet_xml = _with_relationship_namespace(archive.read("xl/worksheets/sheet1.xml")) + worksheet_xml = _append_before( + worksheet_xml, + b"", + f'', + ) + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": worksheet_xml, + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ( + "wrong-type", + f"{OFFICE_REL_NS}/comments", + "../comments1.xml", + False, + ), + ), + "xl/comments1.xml": ( + f'' + ).encode(), + }, + ) + + with pytest.raises(CorruptDocumentError, match="relationship"): + preflight_xlsx(path) diff --git a/tests/test_xlsx_merge.py b/tests/test_xlsx_merge.py new file mode 100644 index 0000000..105e720 --- /dev/null +++ b/tests/test_xlsx_merge.py @@ -0,0 +1,212 @@ +from __future__ import annotations + +from opendocs._models import ( + HeadingBlock, + InlineText, + MarkdownBlock, + ParagraphBlock, + TableBlock, + TextBlock, + WarningRecord, +) +from opendocs.parsers.xlsx.merge import XlsxVisualOutcome, merge_xlsx_document +from opendocs.parsers.xlsx.models import ( + XlsxChartSlot, + XlsxDocument, + XlsxImageSlot, + XlsxNativeSlot, + XlsxSheet, + XlsxSheetKind, + XlsxSheetState, +) +from opendocs.vision.base import VisionResult, VisionTextElement + + +def _prelude(sheet_index: int, name: str) -> XlsxNativeSlot: + return XlsxNativeSlot( + 0, + "A1", + ( + MarkdownBlock(f""), + HeadingBlock(1, (InlineText(name),)), + ParagraphBlock((InlineText("legacy state"),)), + ), + ) + + +def test_merge_orders_by_anchor_and_keeps_all_native_facts_before_same_anchor_vision() -> None: + chart_digest = "a" * 64 + image_digest = "b" * 64 + document = XlsxDocument( + ( + XlsxSheet( + 1, + "Data", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.HIDDEN, + ( + _prelude(1, "Data"), + XlsxNativeSlot(1, "A10", (TextBlock("late"),)), + XlsxChartSlot( + 3, + "B2", + "chart.png", + chart_digest, + (TextBlock("native chart facts"),), + ), + XlsxImageSlot( + 4, + "B2", + "image.png", + image_digest, + alt_text="Quarterly dashboard", + object_name="Picture 1", + ), + XlsxNativeSlot(2, "A2", (TextBlock("early"),)), + ), + ), + XlsxSheet( + 2, + "Empty", + XlsxSheetKind.CHARTSHEET, + XlsxSheetState.VERY_HIDDEN, + (_prelude(2, "Empty"),), + ), + ), + (WarningRecord("xlsx_unsupported_object", "Data!C4: unsupported control"),), + ) + outcomes = { + chart_digest: XlsxVisualOutcome( + VisionResult((VisionTextElement("chart interpretation", 0),)) + ), + image_digest: XlsxVisualOutcome( + VisionResult((VisionTextElement("image interpretation", 0),)) + ), + } + + merged = merge_xlsx_document(document, outcomes) + + assert merged.blocks[0:2] == ( + MarkdownBlock(""), + HeadingBlock(1, (InlineText("Data (Hidden)"),)), + ) + assert merged.blocks[-2:] == ( + MarkdownBlock(""), + HeadingBlock(1, (InlineText("Empty (Very Hidden)"),)), + ) + rendered_text = [block.text for block in merged.blocks if isinstance(block, TextBlock)] + assert rendered_text == ( + [ + "early", + "native chart facts", + "chart interpretation", + "image interpretation", + "late", + ] + ) + chart_native = merged.blocks.index(TextBlock("native chart facts")) + image_metadata = merged.blocks.index( + ParagraphBlock((InlineText("Image description: Quarterly dashboard"),)) + ) + chart_visual = merged.blocks.index(TextBlock("chart interpretation")) + image_visual = merged.blocks.index(TextBlock("image interpretation")) + assert chart_native < chart_visual + assert image_metadata < chart_visual + assert chart_visual < image_visual + assert merged.warnings == document.warnings + + +def test_merge_replays_digest_failure_for_every_occurrence_without_losing_metadata() -> None: + digest = "c" * 64 + document = XlsxDocument( + ( + XlsxSheet( + 1, + "One", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.VISIBLE, + ( + _prelude(1, "One"), + XlsxImageSlot(1, "C3", "same.png", digest, title="Logo"), + ), + ), + XlsxSheet( + 2, + "Two", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.VISIBLE, + ( + _prelude(2, "Two"), + XlsxImageSlot(1, "D4", "same.png", digest, alt_text="Brand"), + ), + ), + ) + ) + + merged = merge_xlsx_document(document, {digest: XlsxVisualOutcome(None, "xlsx_vision_failed")}) + + assert ParagraphBlock((InlineText("Image title: Logo"),)) in merged.blocks + assert ParagraphBlock((InlineText("Image description: Brand"),)) in merged.blocks + assert [warning.code for warning in merged.warnings] == [ + "xlsx_vision_failed", + "xlsx_vision_failed", + ] + assert "One!C3" in merged.warnings[0].message + assert "Two!D4" in merged.warnings[1].message + + +def test_merge_keeps_valid_vision_tables_after_native_chart_data() -> None: + digest = "d" * 64 + document = XlsxDocument( + ( + XlsxSheet( + 1, + "Chart", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.VISIBLE, + ( + _prelude(1, "Chart"), + XlsxChartSlot( + 1, + "A1", + "chart.png", + digest, + (TableBlock((("native", "1"),), 0),), + ), + ), + ), + ) + ) + outcome = XlsxVisualOutcome( + VisionResult((VisionTextElement("visual trend", 0),)), + ) + + merged = merge_xlsx_document(document, {digest: outcome}) + + assert merged.blocks.index(TableBlock((("native", "1"),), 0)) < merged.blocks.index( + TextBlock("visual trend") + ) + + +def test_merge_returns_headings_and_warnings_for_unsupported_only_workbook() -> None: + warning = WarningRecord("xlsx_unsupported_object", "Only!A1: unsupported object") + document = XlsxDocument( + ( + XlsxSheet( + 1, + "Only", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.VISIBLE, + (_prelude(1, "Only"),), + ), + ), + (warning,), + ) + + merged = merge_xlsx_document(document, {}) + + assert merged.blocks == ( + MarkdownBlock(""), + HeadingBlock(1, (InlineText("Only (Visible)"),)), + ) + assert merged.warnings == (warning,) diff --git a/tests/test_xlsx_models.py b/tests/test_xlsx_models.py new file mode 100644 index 0000000..5934d79 --- /dev/null +++ b/tests/test_xlsx_models.py @@ -0,0 +1,218 @@ +from __future__ import annotations + +from dataclasses import FrozenInstanceError +from typing import Any, cast + +import pytest + +from opendocs._models import ( + InlineLink, + PageBreakBlock, + ParagraphBlock, + TableBlock, + TextBlock, + WarningRecord, +) +from opendocs._native_protocol import MAX_FRAME_BYTES, encode_message +from opendocs.errors import LimitExceededError +from opendocs.parsers.xlsx.models import ( + XlsxChartSlot, + XlsxDocument, + XlsxImageSlot, + XlsxNativeSlot, + XlsxSheet, + XlsxSheetKind, + XlsxSheetState, + document_from_wire, + document_to_wire, +) + + +def _document() -> XlsxDocument: + return XlsxDocument( + sheets=( + XlsxSheet( + sheet_index=1, + name="Visible", + kind=XlsxSheetKind.WORKSHEET, + state=XlsxSheetState.VISIBLE, + slots=( + XlsxNativeSlot( + source_index=0, + anchor="A1:B2", + blocks=( + TextBlock("alpha"), + ParagraphBlock( + (InlineLink("docs [safe]", "https://example.test/a(b)"),) + ), + TableBlock((("head", None), ("value", "tail")), header_rows=0), + ), + ), + XlsxImageSlot( + source_index=1, + anchor="D4", + artifact_name="xlsx-image-1.png", + content_sha256="a" * 64, + alt_text="diagram", + object_name="Picture 1", + title="Architecture", + ), + XlsxChartSlot( + source_index=2, + anchor="F5:J20", + artifact_name="xlsx-chart-1.png", + content_sha256="b" * 64, + blocks=(TextBlock("Series: 1, 2, 3"),), + alt_text="trend chart", + object_name="Chart 1", + title="Trend", + ), + ), + ), + XlsxSheet( + sheet_index=2, + name="Chart", + kind=XlsxSheetKind.CHARTSHEET, + state=XlsxSheetState.VERY_HIDDEN, + slots=(), + ), + ), + warnings=(WarningRecord(code="kept", message="warning"),), + ) + + +def test_xlsx_document_wire_round_trip_is_sheet_oriented_and_strict() -> None: + document = _document() + + wire = document_to_wire(document) + restored = document_from_wire(wire) + + assert restored == document + assert "pages" not in repr(wire) + assert "page_number" not in repr(wire) + assert isinstance(restored.sheets, tuple) + assert isinstance(restored.sheets[0].slots, tuple) + + +def test_xlsx_models_are_frozen_and_require_tuple_collections() -> None: + document = _document() + + with pytest.raises(FrozenInstanceError): + document.sheets[0].__setattr__("slots", ()) + with pytest.raises(TypeError, match="sheets"): + XlsxDocument(sheets=cast(Any, [])) + with pytest.raises(TypeError, match="slots"): + XlsxSheet( + sheet_index=1, + name="Sheet", + kind=XlsxSheetKind.WORKSHEET, + state=XlsxSheetState.VISIBLE, + slots=cast(Any, []), + ) + with pytest.raises(TypeError, match="supported block"): + XlsxNativeSlot(source_index=0, anchor="A1", blocks=(cast(Any, PageBreakBlock(1)),)) + + +@pytest.mark.parametrize("anchor", ["a1", "$A$1", "A0", "XFE1", "A1048577", "B2:A1"]) +def test_xlsx_slots_reject_noncanonical_or_out_of_range_a1_anchors(anchor: str) -> None: + with pytest.raises(ValueError, match="anchor"): + XlsxNativeSlot(source_index=0, anchor=anchor, blocks=(TextBlock("value"),)) + + +def test_xlsx_models_reject_duplicate_indexes_and_unsafe_artifacts() -> None: + sheet = XlsxSheet( + sheet_index=1, + name="Sheet", + kind=XlsxSheetKind.WORKSHEET, + state=XlsxSheetState.VISIBLE, + slots=( + XlsxNativeSlot(source_index=0, anchor="A1", blocks=(TextBlock("one"),)), + XlsxNativeSlot(source_index=1, anchor="A2", blocks=(TextBlock("two"),)), + ), + ) + with pytest.raises(ValueError, match="sheet indexes"): + XlsxDocument(sheets=(sheet, sheet)) + with pytest.raises(ValueError, match="sheet name"): + XlsxSheet( + sheet_index=1, + name="Bad/Name", + kind=XlsxSheetKind.WORKSHEET, + state=XlsxSheetState.VISIBLE, + slots=(), + ) + with pytest.raises(ValueError, match="source indexes"): + XlsxSheet( + sheet_index=1, + name="Sheet", + kind=XlsxSheetKind.WORKSHEET, + state=XlsxSheetState.VISIBLE, + slots=(sheet.slots[0], sheet.slots[0]), + ) + with pytest.raises(ValueError, match="artifact_name"): + XlsxImageSlot( + source_index=0, + anchor="A1", + artifact_name="../escape.png", + content_sha256="a" * 64, + ) + with pytest.raises(ValueError, match="content_sha256"): + XlsxImageSlot( + source_index=0, + anchor="A1", + artifact_name="image.png", + content_sha256="short", + ) + + +def test_xlsx_wire_rejects_unknown_fields_and_non_tuple_collections() -> None: + wire = document_to_wire(_document()) + wire["page"] = 1 + with pytest.raises(ValueError, match="XLSX document wire"): + document_from_wire(wire) + + wire = document_to_wire(_document()) + wire["sheets"] = list(cast(tuple[object, ...], wire["sheets"])) + with pytest.raises(ValueError, match="sheets"): + document_from_wire(wire) + + +def test_xlsx_wire_estimator_rejects_result_before_protocol_encoding() -> None: + blocks = tuple(TextBlock("x" * 200) for _ in range(8_000)) + document = XlsxDocument( + sheets=( + XlsxSheet( + sheet_index=1, + name="Sheet", + kind=XlsxSheetKind.WORKSHEET, + state=XlsxSheetState.VISIBLE, + slots=(XlsxNativeSlot(source_index=0, anchor="A1", blocks=blocks),), + ), + ) + ) + + with pytest.raises(LimitExceededError, match="inline result budget"): + document_to_wire(document) + + +def test_xlsx_wire_estimator_keeps_accepted_payload_below_frame_limit() -> None: + document = XlsxDocument( + sheets=( + XlsxSheet( + sheet_index=1, + name="Sheet", + kind=XlsxSheetKind.WORKSHEET, + state=XlsxSheetState.VISIBLE, + slots=( + XlsxNativeSlot( + source_index=0, + anchor="A1", + blocks=tuple(TextBlock("x" * 200) for _ in range(7_156)), + ), + ), + ), + ) + ) + + encoded = encode_message({"version": 1, "value": document_to_wire(document)}) + + assert len(encoded) - 8 < MAX_FRAME_BYTES diff --git a/tests/test_xlsx_parser.py b/tests/test_xlsx_parser.py new file mode 100644 index 0000000..3af78d9 --- /dev/null +++ b/tests/test_xlsx_parser.py @@ -0,0 +1,468 @@ +from __future__ import annotations + +import asyncio +from pathlib import Path +from typing import Any, cast + +import pytest +from PIL import Image + +import opendocs.parsers.xlsx.parser as parser_module +from opendocs._models import ( + DocumentType, + HeadingBlock, + InlineText, + MarkdownBlock, + ParagraphBlock, + TextBlock, +) +from opendocs._runtime import ParserRuntime +from opendocs.errors import ( + DocumentTimeoutError, + ModelAuthenticationError, + ModelInvalidRequestError, + ModelInvalidResponseError, + ModelPermissionError, + ModelUnavailableError, + RuntimeDependencyError, +) +from opendocs.options import ParseOptions, VisionConfig +from opendocs.parsers.xlsx.media import build_xlsx_visual_requests +from opendocs.parsers.xlsx.models import ( + XlsxChartSlot, + XlsxDocument, + XlsxImageSlot, + XlsxNativeSlot, + XlsxSheet, + XlsxSheetKind, + XlsxSheetState, + document_to_wire, +) +from opendocs.parsers.xlsx.parser import XlsxParser, _extract_xlsx_to_wire +from opendocs.parsers.xlsx.preflight import XlsxPreflight +from opendocs.source import ParseWorkspace, ResolvedSource +from opendocs.vision.base import VisionRequest, VisionResult, VisionTextElement +from tests.xlsx_fixtures import write_structured_xlsx + + +class RecordingVision: + def __init__(self, result: object | BaseException | None = None) -> None: + self.requests: list[VisionRequest] = [] + self.result = result + + async def analyze(self, request: VisionRequest) -> VisionResult: + self.requests.append(request) + if isinstance(self.result, BaseException): + raise self.result + if self.result is not None: + return cast(VisionResult, self.result) + return VisionResult((VisionTextElement("视觉解释: 收入总体上升", request.source_index),)) + + +class FatalVisionFailure(BaseException): + pass + + +def _document_with_images(*, duplicate: bool = False) -> XlsxDocument: + first = XlsxImageSlot(1, "B2", "same.png", "b" * 64, alt_text="First") + second = XlsxImageSlot(1, "C3", "same.png", "b" * 64, alt_text="Second") + sheets = [ + XlsxSheet( + 1, + "One", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.VISIBLE, + ( + XlsxNativeSlot( + 0, + "A1", + ( + MarkdownBlock(""), + HeadingBlock(1, (InlineText("One"),)), + ), + ), + first, + ), + ) + ] + if duplicate: + sheets.append( + XlsxSheet( + 2, + "Two", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.VISIBLE, + ( + XlsxNativeSlot( + 0, + "A1", + ( + MarkdownBlock(""), + HeadingBlock(1, (InlineText("Two"),)), + ), + ), + second, + ), + ) + ) + return XlsxDocument(tuple(sheets)) + + +def _runtime_with_document( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + document: XlsxDocument | object, +) -> tuple[ParserRuntime, list[str]]: + workspace = tmp_path / "workspace" + workspace.mkdir() + artifact_names = ( + { + slot.artifact_name + for sheet in document.sheets + for slot in sheet.slots + if isinstance(slot, XlsxImageSlot | XlsxChartSlot) + } + if isinstance(document, XlsxDocument) + else set() + ) + for artifact_name in artifact_names or {"same.png"}: + (workspace / artifact_name).write_bytes(b"raw") + runtime = ParserRuntime(ParseWorkspace(workspace)) + calls: list[str] = [] + + async def run_native(function: Any, *args: object, **kwargs: object) -> object: + del kwargs + calls.append(function.__name__) + if function.__name__ == "_extract_xlsx_to_wire": + return document_to_wire(document) if isinstance(document, XlsxDocument) else document + if function.__name__ == "_prepare_xlsx_visual_to_wire": + output_directory = args[-2] + output_stem = args[-1] + assert isinstance(output_directory, Path) + assert isinstance(output_stem, str) + name = f"{output_stem}-0.png" + (output_directory / name).write_bytes(b"sanitized") + return { + "skipped": False, + "reason": None, + "width": 20, + "height": 10, + "parts": [ + { + "name": name, + "top": 0.0, + "bottom": 1.0, + "core_top": 0.0, + "core_bottom": 1.0, + "width": 20, + "height": 10, + } + ], + "facts": { + "alpha_coverage": 1.0, + "components": 1, + "edge_density": 0.5, + "color_count": 8, + "nearly_blank": False, + }, + } + raise AssertionError(f"unexpected native function: {function.__name__}") + + monkeypatch.setattr(runtime, "run_native", run_native) + return runtime, calls + + +async def _parse(parser: XlsxParser, tmp_path: Path, options: ParseOptions | None = None): + source = tmp_path / "source.xlsx" + source.write_bytes(b"source") + return await parser.parse( + ResolvedSource(source, source.name, False), + options=options or ParseOptions(), + ) + + +@pytest.mark.asyncio +async def test_xlsx_visual_request_seam_uses_fake_client_and_bounded_chart_prompt( + tmp_path: Path, +) -> None: + artifact_name = "chart.png" + image = Image.new("RGB", (32, 16), "white") + try: + image.save(tmp_path / artifact_name, "PNG") + finally: + image.close() + document = XlsxDocument( + ( + XlsxSheet( + 1, + "Data", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.VISIBLE, + ( + XlsxChartSlot( + 1, + "D2", + artifact_name, + "a" * 64, + (HeadingBlock(2, (InlineText("Revenue"),)),), + ), + ), + ), + ) + ) + specs = build_xlsx_visual_requests(document, tmp_path) + vision = RecordingVision() + + result = await vision.analyze(specs[0].to_vision_request()) + + assert result.elements == (VisionTextElement("视觉解释: 收入总体上升", 0),) + assert len(vision.requests) == 1 + assert vision.requests[0].image_path == tmp_path / artifact_name + assert "视觉解释" in vision.requests[0].prompt + assert "Excel 外观还原" in vision.requests[0].prompt + + +def test_native_worker_preflights_before_extract_and_strictly_serializes( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + source = tmp_path / "source.xlsx" + write_structured_xlsx( + source, + sheets=(("Data", "worksheet", "visible", "A1", ("A1",)),), + ) + events: list[str] = [] + real_preflight = parser_module.preflight_xlsx + real_extract = parser_module.extract_xlsx + + def record_preflight(path: Path): + events.append("preflight") + return real_preflight(path) + + def record_extract(path: Path, preflight: XlsxPreflight, *, artifact_dir: Path): + events.append("extract") + return real_extract(path, preflight, artifact_dir=artifact_dir) + + monkeypatch.setattr(parser_module, "preflight_xlsx", record_preflight) + monkeypatch.setattr(parser_module, "extract_xlsx", record_extract) + + wire = _extract_xlsx_to_wire(source, tmp_path / "artifacts") + + assert wire["type"] == "xlsx_document" + assert events == ["preflight", "extract"] + + +@pytest.mark.asyncio +async def test_parser_deduplicates_visual_work_and_replays_success_per_occurrence( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + document = _document_with_images(duplicate=True) + runtime, calls = _runtime_with_document(monkeypatch, tmp_path, document) + vision = RecordingVision(VisionResult((VisionTextElement("visual", 0),))) + try: + result = await _parse( + XlsxParser(runtime, vision, VisionConfig("model")), + tmp_path, + ) + finally: + await runtime.aclose() + + assert [block for block in result.blocks if block == TextBlock("visual")] == [ + TextBlock("visual"), + TextBlock("visual"), + ] + assert len(vision.requests) == 1 + assert calls == ["_extract_xlsx_to_wire", "_prepare_xlsx_visual_to_wire"] + assert not (tmp_path / "workspace" / "same.png").exists() + assert not tuple((tmp_path / "workspace").glob("xlsx-prepared-*.png")) + + +@pytest.mark.asyncio +async def test_parser_without_vision_returns_native_and_occurrence_warning( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + runtime, calls = _runtime_with_document(monkeypatch, tmp_path, _document_with_images()) + try: + result = await _parse(XlsxParser(runtime, None, None), tmp_path) + finally: + await runtime.aclose() + + assert result.document_type is DocumentType.XLSX + assert [warning.code for warning in result.warnings] == ["xlsx_vision_unavailable"] + assert "One!B2" in result.warnings[0].message + assert calls == ["_extract_xlsx_to_wire"] + assert not (tmp_path / "workspace" / "same.png").exists() + + +@pytest.mark.asyncio +async def test_parser_partial_failure_is_replayed_in_anchor_order_not_completion_order( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + slow_digest = "d" * 64 + failed_digest = "e" * 64 + document = XlsxDocument( + ( + XlsxSheet( + 1, + "Mixed", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.VISIBLE, + ( + XlsxNativeSlot( + 0, + "A1", + ( + MarkdownBlock(""), + HeadingBlock(1, (InlineText("Mixed"),)), + ), + ), + XlsxImageSlot(1, "A5", "slow.png", slow_digest, alt_text="Slow"), + XlsxImageSlot(2, "A2", "failed.png", failed_digest, alt_text="Failed"), + ), + ), + ) + ) + + class ReverseCompletionVision: + async def analyze(self, request: VisionRequest) -> VisionResult: + if request.source_index == 0: + await asyncio.sleep(0.02) + return VisionResult((VisionTextElement("slow success", 0),)) + raise ModelUnavailableError("fast failure") + + runtime, calls = _runtime_with_document(monkeypatch, tmp_path, document) + try: + result = await _parse( + XlsxParser(runtime, ReverseCompletionVision(), VisionConfig("model")), + tmp_path, + ) + finally: + await runtime.aclose() + + failed_metadata = ParagraphBlock((InlineText("Image description: Failed"),)) + slow_metadata = ParagraphBlock((InlineText("Image description: Slow"),)) + assert result.blocks.index(failed_metadata) < result.blocks.index(slow_metadata) + assert TextBlock("slow success") in result.blocks + assert [warning.code for warning in result.warnings] == ["xlsx_vision_failed"] + assert "Mixed!A2" in result.warnings[0].message + assert calls == [ + "_extract_xlsx_to_wire", + "_prepare_xlsx_visual_to_wire", + "_prepare_xlsx_visual_to_wire", + ] + assert not tuple((tmp_path / "workspace").glob("*.png")) + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + "failure", + [ + ModelAuthenticationError("auth"), + ModelPermissionError("permission"), + ModelInvalidRequestError("invalid"), + ModelUnavailableError("provider"), + ModelInvalidResponseError("invalid response"), + RuntimeDependencyError("runtime"), + ValueError("plain provider error"), + FatalVisionFailure("base exception"), + object(), + VisionResult(()), + ], +) +async def test_parser_fails_open_for_every_non_timeout_visual_failure( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + failure: object, +) -> None: + runtime, _ = _runtime_with_document(monkeypatch, tmp_path, _document_with_images()) + vision = RecordingVision(failure) + try: + result = await _parse( + XlsxParser(runtime, vision, VisionConfig("model")), + tmp_path, + ) + finally: + await runtime.aclose() + + assert [warning.code for warning in result.warnings] == ["xlsx_vision_failed"] + assert HeadingBlock(1, (InlineText("One (Visible)"),)) in result.blocks + + +@pytest.mark.asyncio +async def test_parser_classifies_per_object_timeout_but_document_deadline_is_fatal( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + class SlowVision: + async def analyze(self, request: VisionRequest) -> VisionResult: + del request + await asyncio.sleep(1) + raise AssertionError("unreachable") + + runtime, _ = _runtime_with_document(monkeypatch, tmp_path, _document_with_images()) + try: + result = await _parse( + XlsxParser(runtime, SlowVision(), VisionConfig("model", timeout=0.01)), + tmp_path, + ParseOptions(timeout=1), + ) + assert [warning.code for warning in result.warnings] == ["xlsx_vision_timeout"] + finally: + await runtime.aclose() + + second_path = tmp_path / "deadline" + second_path.mkdir() + runtime, _ = _runtime_with_document(monkeypatch, second_path, _document_with_images()) + + async def slow_native(function: Any, *args: object, **kwargs: object) -> object: + del function, args, kwargs + await asyncio.sleep(1) + raise AssertionError("unreachable") + + monkeypatch.setattr(runtime, "run_native", slow_native) + try: + with pytest.raises(DocumentTimeoutError): + await _parse( + XlsxParser(runtime, None, None), + second_path, + ParseOptions(timeout=0.01), + ) + finally: + await runtime.aclose() + + +@pytest.mark.asyncio +async def test_parser_propagates_caller_cancellation( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + runtime, _ = _runtime_with_document(monkeypatch, tmp_path, _document_with_images()) + + async def cancelled(function: Any, *args: object, **kwargs: object) -> object: + del function, args, kwargs + raise asyncio.CancelledError + + monkeypatch.setattr(runtime, "run_native", cancelled) + try: + with pytest.raises(asyncio.CancelledError): + await _parse(XlsxParser(runtime, None, None), tmp_path) + finally: + await runtime.aclose() + + +@pytest.mark.asyncio +@pytest.mark.parametrize("wire", [{"type": "wrong"}, object()]) +async def test_parser_maps_invalid_native_wire_to_runtime_dependency( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + wire: object, +) -> None: + runtime, _ = _runtime_with_document(monkeypatch, tmp_path, wire) + try: + with pytest.raises(RuntimeDependencyError, match="invalid data"): + await _parse(XlsxParser(runtime, None, None), tmp_path) + finally: + await runtime.aclose() diff --git a/tests/test_xlsx_preflight.py b/tests/test_xlsx_preflight.py new file mode 100644 index 0000000..47a550a --- /dev/null +++ b/tests/test_xlsx_preflight.py @@ -0,0 +1,1009 @@ +from __future__ import annotations + +from pathlib import Path +from zipfile import ZIP_DEFLATED, ZipFile + +import pytest +from openpyxl import Workbook +from openpyxl.chart import BarChart, Reference + +import opendocs.parsers.xlsx.parser as parser_module +import opendocs.parsers.xlsx.preflight as preflight_module +from opendocs._models import HeadingBlock, InlineText +from opendocs._runtime import ParserRuntime +from opendocs.errors import CorruptDocumentError, LimitExceededError +from opendocs.options import ParseOptions +from opendocs.parsers.xlsx import XlsxParser +from opendocs.parsers.xlsx.models import XlsxSheetKind, XlsxSheetState +from opendocs.parsers.xlsx.parser import _extract_xlsx_to_wire +from opendocs.parsers.xlsx.preflight import MAX_SHEETS, preflight_xlsx +from opendocs.source import ParseWorkspace, ResolvedSource +from tests.xlsx_fixtures import rewrite_xlsx, write_structured_xlsx + +SHEET_NS = "http://schemas.openxmlformats.org/spreadsheetml/2006/main" +OFFICE_REL_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" +PACKAGE_REL_NS = "http://schemas.openxmlformats.org/package/2006/relationships" + + +def _one_sheet(path: Path, *, cells: tuple[str, ...] = ()) -> None: + write_structured_xlsx( + path, + sheets=(("Sheet", "worksheet", "visible", "A1:B2", cells),), + ) + + +def _worksheet(body: str) -> bytes: + return (f'{body}').encode() + + +def _relationships(*items: tuple[str, str, str]) -> bytes: + values = "".join( + f'' + for relationship_id, relationship_type, target in items + ) + return f'{values}'.encode() + + +def test_preflight_preserves_worksheet_chartsheet_state_and_empty_sheet_order( + tmp_path: Path, +) -> None: + path = tmp_path / "ordered.xlsx" + write_structured_xlsx( + path, + sheets=( + ("Visible", "worksheet", "visible", "A1", ("A1",)), + ("Hidden", "worksheet", "hidden", None, ()), + ("Very Hidden", "worksheet", "veryHidden", "C3", ("C3",)), + ("Chart", "chartsheet", "visible", None, ()), + ), + ) + + result = preflight_xlsx(path) + + assert [(sheet.sheet_index, sheet.name) for sheet in result.sheets] == [ + (1, "Visible"), + (2, "Hidden"), + (3, "Very Hidden"), + (4, "Chart"), + ] + assert [sheet.kind for sheet in result.sheets] == [ + XlsxSheetKind.WORKSHEET, + XlsxSheetKind.WORKSHEET, + XlsxSheetKind.WORKSHEET, + XlsxSheetKind.CHARTSHEET, + ] + assert [sheet.state for sheet in result.sheets] == [ + XlsxSheetState.VISIBLE, + XlsxSheetState.HIDDEN, + XlsxSheetState.VERY_HIDDEN, + XlsxSheetState.VISIBLE, + ] + assert result.serialized_cells == 2 + + +def test_preflight_accepts_openpyxl_package_absolute_relationship_targets( + tmp_path: Path, +) -> None: + path = tmp_path / "openpyxl.xlsx" + workbook = Workbook() + workbook.active["A1"] = "value" + chart = BarChart() + chart.add_data(Reference(workbook.active, min_col=1, min_row=1, max_row=1)) + chart_sheet = workbook.create_chartsheet("Chart") + chart_sheet.add_chart(chart) + workbook.save(path) + + result = preflight_xlsx(path) + + assert [(sheet.name, sheet.kind) for sheet in result.sheets] == [ + ("Sheet", XlsxSheetKind.WORKSHEET), + ("Chart", XlsxSheetKind.CHARTSHEET), + ] + + +def test_preflight_accepts_128_sheets_and_rejects_129(tmp_path: Path) -> None: + accepted = tmp_path / "accepted.xlsx" + rejected = tmp_path / "rejected.xlsx" + sheets = tuple( + (f"Sheet {index}", "worksheet", "visible", None, ()) for index in range(1, MAX_SHEETS + 1) + ) + write_structured_xlsx(accepted, sheets=sheets) + write_structured_xlsx( + rejected, + sheets=(*sheets, ("One too many", "worksheet", "visible", None, ())), + ) + + assert len(preflight_xlsx(accepted).sheets) == MAX_SHEETS + with pytest.raises(LimitExceededError, match="sheet count"): + preflight_xlsx(rejected) + + +def test_sparse_full_grid_dimension_fails_before_any_loader(tmp_path: Path) -> None: + path = tmp_path / "sparse.xlsx" + write_structured_xlsx( + path, + sheets=(("Sparse", "worksheet", "visible", "A1:XFD1048576", ("A1",)),), + ) + + with pytest.raises(LimitExceededError, match="declared dimension"): + preflight_xlsx(path) + + +@pytest.mark.asyncio +async def test_parser_seam_preflights_limits_and_does_not_map_max_pages_to_sheets( + tmp_path: Path, +) -> None: + accepted = tmp_path / "two-sheets.xlsx" + rejected = tmp_path / "too-many-sheets.xlsx" + workbook = Workbook() + workbook.active.title = "One" + workbook.create_sheet("Two") + workbook.save(accepted) + workbook.close() + write_structured_xlsx( + rejected, + sheets=tuple( + (f"S{index}", "worksheet", "visible", None, ()) for index in range(MAX_SHEETS + 1) + ), + ) + + workspace = tmp_path / "workspace" + workspace.mkdir() + runtime = ParserRuntime(ParseWorkspace(workspace)) + parser = XlsxParser(runtime, None, None) + try: + result = await parser.parse( + ResolvedSource(accepted, "two-sheets.xlsx", False), + options=ParseOptions(max_pages=1), + ) + assert [block for block in result.blocks if isinstance(block, HeadingBlock)] == [ + HeadingBlock(1, (InlineText("One (Visible)"),)), + HeadingBlock(1, (InlineText("Two (Visible)"),)), + ] + with pytest.raises(LimitExceededError, match="sheet count"): + await parser.parse( + ResolvedSource(rejected, "too-many-sheets.xlsx", False), + options=ParseOptions(max_pages=1), + ) + finally: + await runtime.aclose() + + +@pytest.mark.parametrize( + ("limit_name", "body", "message"), + [ + ( + "MAX_SERIALIZED_CELLS", + '', + "serialized cell", + ), + ( + "MAX_NON_EMPTY_CELLS", + '' + '12', + "non-empty cell", + ), + ( + "MAX_MERGE_RANGES", + '' + '', + "merge range", + ), + ( + "MAX_MERGE_FOOTPRINT", + '', + "merge footprint", + ), + ( + "MAX_CONDITIONAL_FORMATTING_RULES", + '' + '' + '', + "conditional formatting", + ), + ( + "MAX_DATA_VALIDATIONS", + '' + '' + '', + "data validation", + ), + ( + "MAX_ROW_DIMENSIONS", + '', + "row dimension", + ), + ( + "MAX_COLUMN_DIMENSIONS", + '' + '', + "column dimension", + ), + ( + "MAX_PAGE_BREAKS", + '' + '', + "page break", + ), + ( + "MAX_SCENARIOS", + '' + '', + "scenario", + ), + ], +) +def test_worksheet_loader_collections_are_bounded( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + limit_name: str, + body: str, + message: str, +) -> None: + path = tmp_path / f"{limit_name}.xlsx" + _one_sheet(path) + monkeypatch.setattr(preflight_module, limit_name, 1) + rewrite_xlsx(path, {"xl/worksheets/sheet1.xml": _worksheet(body)}) + + with pytest.raises(LimitExceededError, match=message): + preflight_xlsx(path) + + +def test_loader_collection_boundary_values_are_accepted(tmp_path: Path) -> None: + path = tmp_path / "loader-boundaries.xlsx" + _one_sheet(path) + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": _worksheet( + '' + '' + '' + '' + '' + '' + '' + '' + ) + }, + ) + + usage = preflight_xlsx(path).usage + + assert usage.serialized_cells == 1 + assert usage.non_empty_cells == 1 + assert usage.merge_ranges == 1 + assert usage.merge_footprint == 2 + assert usage.conditional_formatting_rules == 1 + assert usage.data_validations == 1 + assert usage.row_dimensions == 1 + assert usage.column_dimensions == 1 + assert usage.page_breaks == 1 + assert usage.scenarios == 1 + + +def test_shared_string_item_and_text_budgets_have_boundary_and_overflow( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "strings.xlsx" + _one_sheet(path) + monkeypatch.setattr(preflight_module, "MAX_SHARED_STRINGS", 2) + monkeypatch.setattr(preflight_module, "MAX_SHARED_STRING_CHARS", 3) + rewrite_xlsx( + path, + { + "xl/sharedStrings.xml": ( + f'abc' + ).encode() + }, + ) + assert preflight_xlsx(path).usage.shared_strings == 2 + + rewrite_xlsx( + path, + {"xl/sharedStrings.xml": (f'abcd').encode()}, + ) + with pytest.raises(LimitExceededError, match="shared string text"): + preflight_xlsx(path) + + +@pytest.mark.parametrize( + ("element", "limit_name", "message"), + [ + ("font", "MAX_FONTS", "font"), + ("fill", "MAX_FILLS", "fill"), + ("border", "MAX_BORDERS", "border"), + ("xf", "MAX_CELL_XFS", "cellXfs"), + ("dxf", "MAX_DXFS", "dxf"), + ], +) +def test_stylesheet_loader_collections_are_bounded( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + element: str, + limit_name: str, + message: str, +) -> None: + path = tmp_path / f"{element}.xlsx" + _one_sheet(path) + monkeypatch.setattr(preflight_module, limit_name, 1) + container = { + "font": "fonts", + "fill": "fills", + "border": "borders", + "xf": "cellXfs", + "dxf": "dxfs", + }[element] + rewrite_xlsx( + path, + { + "xl/styles.xml": ( + f'<{container} count="2">' + f"<{element}/><{element}/>" + ).encode() + }, + ) + + with pytest.raises(LimitExceededError, match=message): + preflight_xlsx(path) + + +def test_stylesheet_boundary_values_are_counted(tmp_path: Path) -> None: + path = tmp_path / "styles-boundary.xlsx" + _one_sheet(path) + rewrite_xlsx( + path, + { + "xl/styles.xml": ( + f'' + "" + "" + "" + ).encode() + }, + ) + + usage = preflight_xlsx(path).usage + + assert usage.style_records == 8 + assert usage.cell_xfs == 1 + assert usage.dxfs == 1 + + +def test_defined_names_and_custom_properties_are_bounded( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "names.xlsx" + _one_sheet(path) + monkeypatch.setattr(preflight_module, "MAX_DEFINED_NAMES", 1) + one_name = ( + f'' + '' + 'Sheet!$A$1' + ).encode() + rewrite_xlsx(path, {"xl/workbook.xml": one_name}) + assert preflight_xlsx(path).usage.defined_names == 1 + + rewrite_xlsx( + path, + { + "xl/workbook.xml": ( + f'' + '' + 'Sheet!$A$1' + 'Sheet!$A$2' + "" + ).encode() + }, + ) + with pytest.raises(LimitExceededError, match="defined name"): + preflight_xlsx(path) + + monkeypatch.setattr(preflight_module, "MAX_DEFINED_NAMES", 10) + monkeypatch.setattr(preflight_module, "MAX_CUSTOM_PROPERTIES", 1) + rewrite_xlsx( + path, + { + "docProps/custom.xml": ( + b'' + ) + }, + ) + assert preflight_xlsx(path).usage.custom_properties == 1 + + rewrite_xlsx( + path, + { + "docProps/custom.xml": ( + b'' + ) + }, + ) + with pytest.raises(LimitExceededError, match="custom propert"): + preflight_xlsx(path) + + +def test_table_count_and_footprint_are_bounded_before_loader( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "table.xlsx" + _one_sheet(path) + monkeypatch.setattr(preflight_module, "MAX_TABLES", 1) + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": _worksheet( + '' + '' + ), + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ("rIdT1", f"{OFFICE_REL_NS}/table", "../tables/table1.xml"), + ("rIdT2", f"{OFFICE_REL_NS}/table", "../tables/table2.xml"), + ), + "xl/tables/table1.xml": f''.encode(), + "xl/tables/table2.xml": f'
'.encode(), + }, + ) + with pytest.raises(LimitExceededError, match="table count"): + preflight_xlsx(path) + + monkeypatch.setattr(preflight_module, "MAX_TABLES", 2) + monkeypatch.setattr(preflight_module, "MAX_TABLE_FOOTPRINT", 8) + assert preflight_xlsx(path).usage.tables == 2 + + monkeypatch.setattr(preflight_module, "MAX_TABLE_FOOTPRINT", 7) + with pytest.raises(LimitExceededError, match="table footprint"): + preflight_xlsx(path) + + +def test_drawing_object_chart_cache_and_anchor_are_preflighted( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "drawing.xlsx" + _one_sheet(path) + drawing_rel = f"{OFFICE_REL_NS}/drawing" + chart_rel = f"{OFFICE_REL_NS}/chart" + drawing_ns = "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing" + chart_ns = "http://schemas.openxmlformats.org/drawingml/2006/chart" + drawing = ( + f'' + "00" + '' + "" + ).encode() + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": _worksheet( + '' + ), + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ("rIdD1", drawing_rel, "../drawings/drawing1.xml"), + ), + "xl/drawings/drawing1.xml": drawing, + "xl/drawings/_rels/drawing1.xml.rels": _relationships( + ("rIdC1", chart_rel, "../charts/chart1.xml"), + ), + "xl/charts/chart1.xml": ( + f'1' + '2' + ).encode(), + }, + ) + monkeypatch.setattr(preflight_module, "MAX_CHART_CACHE_POINTS", 1) + with pytest.raises(LimitExceededError, match="chart cache"): + preflight_xlsx(path) + + monkeypatch.setattr(preflight_module, "MAX_CHART_CACHE_POINTS", 2) + assert preflight_xlsx(path).usage.drawing_objects == 1 + + monkeypatch.setattr(preflight_module, "MAX_DRAWING_OBJECTS", 0) + with pytest.raises(LimitExceededError, match="drawing object"): + preflight_xlsx(path) + monkeypatch.setattr(preflight_module, "MAX_DRAWING_OBJECTS", 256) + + invalid = drawing.replace(b"0", b"16384") + rewrite_xlsx(path, {"xl/drawings/drawing1.xml": invalid}) + with pytest.raises(CorruptDocumentError, match="drawing anchor"): + preflight_xlsx(path) + + +def test_hyperlink_comment_and_relationship_references_are_bounded_and_resolved( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "links.xlsx" + _one_sheet(path) + monkeypatch.setattr(preflight_module, "MAX_HYPERLINKS_AND_COMMENTS", 1) + one_link = _worksheet( + '' + '' + ) + one_link_relationship = ( + f'' + f'' + ).encode() + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": one_link, + "xl/worksheets/_rels/sheet1.xml.rels": one_link_relationship, + }, + ) + assert preflight_xlsx(path).usage.hyperlinks_and_comments == 1 + + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": _worksheet( + '' + '' + "" + ), + "xl/worksheets/_rels/sheet1.xml.rels": ( + f'' + f'' + f'' + ).encode(), + }, + ) + with pytest.raises(LimitExceededError, match="hyperlink and comment"): + preflight_xlsx(path) + + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": _worksheet( + '' + ), + }, + ) + with pytest.raises(CorruptDocumentError, match="relationship"): + preflight_xlsx(path) + + +def test_pivot_cache_collections_are_bounded( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + path = tmp_path / "pivot.xlsx" + _one_sheet(path) + monkeypatch.setattr(preflight_module, "MAX_PIVOT_CACHES", 1) + one_cache_workbook = ( + f'' + '' + '' + ).encode() + one_cache_relationships = _relationships( + ("rId1", f"{OFFICE_REL_NS}/worksheet", "worksheets/sheet1.xml"), + ("rIdP1", f"{OFFICE_REL_NS}/pivotCacheDefinition", "pivotCache/a.xml"), + ) + rewrite_xlsx( + path, + { + "xl/workbook.xml": one_cache_workbook, + "xl/_rels/workbook.xml.rels": one_cache_relationships, + "xl/pivotCache/a.xml": (f'').encode(), + }, + ) + assert preflight_xlsx(path).usage.pivot_caches == 1 + + rewrite_xlsx( + path, + { + "xl/workbook.xml": ( + f'' + '' + '' + "" + ).encode(), + "xl/_rels/workbook.xml.rels": _relationships( + ("rId1", f"{OFFICE_REL_NS}/worksheet", "worksheets/sheet1.xml"), + ("rIdP1", f"{OFFICE_REL_NS}/pivotCacheDefinition", "pivotCache/a.xml"), + ("rIdP2", f"{OFFICE_REL_NS}/pivotCacheDefinition", "pivotCache/b.xml"), + ), + "xl/pivotCache/a.xml": b"", + "xl/pivotCache/b.xml": b"", + }, + ) + with pytest.raises(LimitExceededError, match="pivot cache"): + preflight_xlsx(path) + + +def test_comment_and_pivot_record_limits_apply_before_loader( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "comments-pivot.xlsx" + _one_sheet(path) + monkeypatch.setattr(preflight_module, "MAX_HYPERLINKS_AND_COMMENTS", 1) + rewrite_xlsx( + path, + { + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ("rIdC1", f"{OFFICE_REL_NS}/comments", "../comments1.xml"), + ), + "xl/comments1.xml": ( + f'A' + 'one' + "" + ).encode(), + }, + ) + assert preflight_xlsx(path).usage.hyperlinks_and_comments == 1 + + rewrite_xlsx( + path, + { + "xl/comments1.xml": ( + f'A' + 'one' + 'two' + ).encode() + }, + ) + with pytest.raises(LimitExceededError, match="hyperlink and comment"): + preflight_xlsx(path) + + monkeypatch.setattr(preflight_module, "MAX_HYPERLINKS_AND_COMMENTS", 20_000) + monkeypatch.setattr(preflight_module, "MAX_PIVOT_CACHE_RECORDS", 1) + rewrite_xlsx( + path, + { + "xl/comments1.xml": None, + "xl/worksheets/_rels/sheet1.xml.rels": None, + "xl/pivotCache/cache.xml": f''.encode(), + "xl/pivotCache/_rels/cache.xml.rels": _relationships( + ( + "rIdR1", + f"{OFFICE_REL_NS}/pivotCacheRecords", + "records.xml", + ), + ), + "xl/pivotCache/records.xml": ( + f'' + ).encode(), + "xl/workbook.xml": ( + f'' + '' + '' + ).encode(), + "xl/_rels/workbook.xml.rels": _relationships( + ("rId1", f"{OFFICE_REL_NS}/worksheet", "worksheets/sheet1.xml"), + ( + "rIdP1", + f"{OFFICE_REL_NS}/pivotCacheDefinition", + "pivotCache/cache.xml", + ), + ), + }, + ) + with pytest.raises(LimitExceededError, match="pivot cache record"): + preflight_xlsx(path) + + +def test_unknown_relationship_objects_are_deterministically_locatable(tmp_path: Path) -> None: + path = tmp_path / "unknown.xlsx" + _one_sheet(path) + rewrite_xlsx( + path, + { + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ("rIdOle", f"{OFFICE_REL_NS}/oleObject", "../embeddings/ole1.bin"), + ), + "xl/embeddings/ole1.bin": b"opaque", + }, + ) + + unsupported = preflight_xlsx(path).sheets[0].unsupported_objects + + assert [ + (item.source_index, item.kind, item.relationship_id, item.target) for item in unsupported + ] == [(0, "oleObject", "rIdOle", "xl/embeddings/ole1.bin")] + + +def test_threaded_comments_and_person_text_are_preflighted_before_extraction( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "threaded-comments.xlsx" + _one_sheet(path) + threaded_relationship = ( + "http://schemas.microsoft.com/office/2017/10/relationships/threadedComment" + ) + person_relationship = "http://schemas.microsoft.com/office/2017/10/relationships/person" + threaded_namespace = "http://schemas.microsoft.com/office/spreadsheetml/2018/threadedcomments" + with ZipFile(path) as archive: + workbook_relationships = archive.read("xl/_rels/workbook.xml.rels") + workbook_relationships = workbook_relationships.replace( + b"", + ( + f'' + ).encode(), + ) + rewrite_xlsx( + path, + { + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ( + "rIdThreaded", + threaded_relationship, + "../threadedComments/threadedComment1.xml", + ), + ), + "xl/threadedComments/threadedComment1.xml": ( + f'' + 'hello' + "" + ).encode(), + "xl/persons/person.xml": ( + f'' + ).encode(), + "xl/_rels/workbook.xml.rels": workbook_relationships, + }, + ) + + result = preflight_xlsx(path) + + assert result.usage.hyperlinks_and_comments == 1 + assert result.native_text_chars >= len("helloReviewer") + assert result.sheets[0].unsupported_objects == () + + monkeypatch.setattr(preflight_module, "MAX_HYPERLINKS_AND_COMMENTS", 0) + with pytest.raises(LimitExceededError, match="hyperlink and comment"): + preflight_xlsx(path) + + +def test_drawing_shape_metadata_is_included_in_native_text_budget( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "drawing-text-budget.xlsx" + _one_sheet(path) + drawing_ns = "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing" + drawing_main_ns = "http://schemas.openxmlformats.org/drawingml/2006/main" + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": _worksheet( + '' + ), + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ("rIdDrawing", f"{OFFICE_REL_NS}/drawing", "../drawings/drawing1.xml"), + ), + "xl/drawings/drawing1.xml": ( + f'' + "00" + '' + '' + "Text" + "" + "" + ).encode(), + }, + ) + + result = preflight_xlsx(path) + expected = len("NamedDescriptionTitleText") + assert result.native_text_chars >= expected + + monkeypatch.setattr(preflight_module, "MAX_NATIVE_TEXT_CHARS", expected - 1) + with pytest.raises(LimitExceededError, match="native text"): + preflight_xlsx(path) + + +@pytest.mark.parametrize( + "xml", + [ + b']>', + b'', + b'', + ], +) +def test_worksheet_dtd_entity_namespace_and_malformed_xml_are_typed_corruption( + tmp_path: Path, + xml: bytes, +) -> None: + path = tmp_path / "corrupt.xlsx" + _one_sheet(path) + rewrite_xlsx(path, {"xl/worksheets/sheet1.xml": xml}) + + with pytest.raises(CorruptDocumentError): + preflight_xlsx(path) + + +def test_zip_slip_is_rejected_by_preflight(tmp_path: Path) -> None: + path = tmp_path / "zip-slip.xlsx" + _one_sheet(path) + with ZipFile(path, "a", ZIP_DEFLATED) as archive: + archive.writestr("../escape.xml", b"") + + with pytest.raises(CorruptDocumentError, match="unsafe member"): + preflight_xlsx(path) + + +def test_projected_wire_budget_is_a_stricter_success_boundary_than_grid_limit( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "wire.xlsx" + _one_sheet(path, cells=("A1", "B1")) + assert preflight_module.MAX_MATERIALIZED_GRID_CELLS == 200_000 + monkeypatch.setattr(preflight_module, "MAX_PROJECTED_WIRE_BYTES", 2_000) + + with pytest.raises(LimitExceededError, match="inline result budget"): + preflight_xlsx(path) + + +def test_projected_workbook_budget_bounds_full_mode_loader_work( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "projected-workbook.xlsx" + _one_sheet(path, cells=("A1",)) + monkeypatch.setattr(preflight_module, "MAX_PROJECTED_WORKBOOK_BYTES", 1) + + with pytest.raises(LimitExceededError, match="projected workbook"): + preflight_xlsx(path) + + +def test_materialized_grid_and_native_text_have_independent_outer_limits( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "grid-text.xlsx" + _one_sheet(path, cells=("A1", "B1")) + monkeypatch.setattr(preflight_module, "MAX_MATERIALIZED_GRID_CELLS", 1) + with pytest.raises(LimitExceededError, match="materialized grid"): + preflight_xlsx(path) + + monkeypatch.setattr(preflight_module, "MAX_MATERIALIZED_GRID_CELLS", 200_000) + monkeypatch.setattr(preflight_module, "MAX_NATIVE_TEXT_CHARS", 3) + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": _worksheet( + '1234' + ) + }, + ) + with pytest.raises(LimitExceededError, match="native text"): + preflight_xlsx(path) + + +def test_representative_adversarial_limits_stop_before_any_expensive_xlsx_stage( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + def forbidden_extract(*args: object, **kwargs: object) -> object: + del args, kwargs + raise AssertionError("XLSX extraction, materialization, and vision must not start") + + monkeypatch.setattr(parser_module, "extract_xlsx", forbidden_extract) + cases: list[tuple[Path, type[BaseException]]] = [] + + zip_bomb = tmp_path / "zip-bomb.xlsx" + _one_sheet(zip_bomb) + with ZipFile(zip_bomb, "a", ZIP_DEFLATED) as archive: + archive.writestr("xl/filler.bin", b"x" * 200_000) + cases.append((zip_bomb, LimitExceededError)) + + entity = tmp_path / "entity.xlsx" + _one_sheet(entity) + rewrite_xlsx( + entity, + { + "xl/worksheets/sheet1.xml": ( + b']>&x;' + ) + }, + ) + cases.append((entity, CorruptDocumentError)) + + dimension = tmp_path / "dimension.xlsx" + write_structured_xlsx( + dimension, + sheets=(("Sparse", "worksheet", "visible", "A1:XFD1048576", ("A1",)),), + ) + cases.append((dimension, LimitExceededError)) + + merge = tmp_path / "merge.xlsx" + _one_sheet(merge) + monkeypatch.setattr(preflight_module, "MAX_MERGE_RANGES", 1) + rewrite_xlsx( + merge, + { + "xl/worksheets/sheet1.xml": _worksheet( + '' + '' + ) + }, + ) + cases.append((merge, LimitExceededError)) + + shared_strings = tmp_path / "shared-strings.xlsx" + _one_sheet(shared_strings) + monkeypatch.setattr(preflight_module, "MAX_SHARED_STRINGS", 1) + rewrite_xlsx( + shared_strings, + { + "xl/_rels/workbook.xml.rels": _relationships( + ("rId1", f"{OFFICE_REL_NS}/worksheet", "worksheets/sheet1.xml"), + ("rIdS", f"{OFFICE_REL_NS}/sharedStrings", "sharedStrings.xml"), + ), + "xl/sharedStrings.xml": ( + f'onetwo' + ).encode(), + }, + ) + cases.append((shared_strings, LimitExceededError)) + + objects = tmp_path / "objects.xlsx" + _one_sheet(objects) + monkeypatch.setattr(preflight_module, "MAX_DRAWING_OBJECTS", 1) + rewrite_xlsx( + objects, + { + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ("rIdO1", f"{OFFICE_REL_NS}/oleObject", "../embeddings/one.bin"), + ("rIdO2", f"{OFFICE_REL_NS}/oleObject", "../embeddings/two.bin"), + ), + "xl/embeddings/one.bin": b"one", + "xl/embeddings/two.bin": b"two", + }, + ) + cases.append((objects, LimitExceededError)) + + chart_cache = tmp_path / "chart-cache.xlsx" + _one_sheet(chart_cache) + monkeypatch.setattr(preflight_module, "MAX_CHART_CACHE_POINTS", 1) + drawing_ns = "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing" + chart_ns = "http://schemas.openxmlformats.org/drawingml/2006/chart" + rewrite_xlsx( + chart_cache, + { + "xl/worksheets/sheet1.xml": _worksheet( + '' + ), + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ("rIdD", f"{OFFICE_REL_NS}/drawing", "../drawings/drawing1.xml"), + ), + "xl/drawings/drawing1.xml": ( + f'' + "00" + '' + "" + ).encode(), + "xl/drawings/_rels/drawing1.xml.rels": _relationships( + ("rIdC", f"{OFFICE_REL_NS}/chart", "../charts/chart1.xml"), + ), + "xl/charts/chart1.xml": ( + f'' + '12' + "" + ).encode(), + }, + ) + cases.append((chart_cache, LimitExceededError)) + + for path, error_type in cases: + artifact_dir = tmp_path / f"{path.stem}-artifacts" + with pytest.raises(error_type): + _extract_xlsx_to_wire(path, artifact_dir) + assert not artifact_dir.exists() diff --git a/tests/test_xlsx_values.py b/tests/test_xlsx_values.py new file mode 100644 index 0000000..98655eb --- /dev/null +++ b/tests/test_xlsx_values.py @@ -0,0 +1,113 @@ +from __future__ import annotations + +from datetime import datetime, time +from decimal import Decimal + +import pytest +from openpyxl.utils.datetime import MAC_EPOCH, WINDOWS_EPOCH + +from opendocs.parsers.xlsx.values import format_saved_value + + +@pytest.mark.parametrize( + ("value", "number_format", "expected"), + [ + (True, "General", "TRUE"), + ("#DIV/0!", "General", "#DIV/0!"), + (Decimal("1234"), "0", "1234"), + (Decimal("1234.5"), "0.00", "1234.50"), + (Decimal("1234.5"), "#,##0.00", "1,234.50"), + (Decimal("0.125"), "0.00%", "12.50%"), + (Decimal("1234.5"), "$#,##0.00", "$1,234.50"), + (Decimal("1234"), "¥#,##0", "¥1,234"), + (Decimal("-1234.5"), "#,##0.00;(#,##0.00)", "(1,234.50)"), + ( + Decimal("-1234.5"), + '_(€* #,##0.00_);_(€* (#,##0.00);_(€* "-"??_);_(@_)', + "(€1,234.50)", + ), + ( + Decimal("0"), + '_(€* #,##0.00_);_(€* (#,##0.00);_(€* "-"??_);_(@_)', + "€-", + ), + ], +) +def test_format_saved_value_supports_core_and_accounting_formats( + value: object, + number_format: str, + expected: str, +) -> None: + result = format_saved_value(value, number_format, epoch=WINDOWS_EPOCH) + + assert result.text == expected + assert result.warning is None + + +def test_format_saved_value_supports_both_date_systems_and_elapsed_time() -> None: + windows = format_saved_value(43831, "yyyy-mm-dd", epoch=WINDOWS_EPOCH) + mac = format_saved_value(42369, "yyyy-mm-dd", epoch=MAC_EPOCH) + timestamp = format_saved_value( + datetime(2020, 1, 2, 3, 4, 5), + "yyyy-mm-dd hh:mm:ss", + epoch=WINDOWS_EPOCH, + ) + elapsed = format_saved_value(1.5, "[h]:mm:ss", epoch=WINDOWS_EPOCH) + + assert windows.text == "2020-01-01" + assert mac.text == "2020-01-01" + assert timestamp.text == "2020-01-02 03:04:05" + assert elapsed.text == "36:00:00" + assert not any(item.warning for item in (windows, mac, timestamp, elapsed)) + + +def test_format_saved_value_supports_common_twelve_hour_time() -> None: + result = format_saved_value( + time(15, 4, 5), + "h:mm:ss AM/PM", + epoch=WINDOWS_EPOCH, + ) + + assert result.text == "3:04:05 PM" + assert result.warning is None + + +@pytest.mark.parametrize( + ("value", "number_format", "expected"), + [ + (Decimal("1234567"), "#,##0,", "1,235"), + (Decimal("1234567"), "#,##0,,", "1"), + (time(1, 2, 3), "mm:ss", "02:03"), + (time(1, 2, 3), "ss", "03"), + ], +) +def test_format_saved_value_handles_scaled_and_partial_time_formats( + value: object, + number_format: str, + expected: str, +) -> None: + result = format_saved_value(value, number_format, epoch=WINDOWS_EPOCH) + + assert result.text == expected + assert result.warning is None + + +@pytest.mark.parametrize( + "number_format", + [ + "0.00E+00", + "# ?/?", + "[Red]0.00", + "[$-409]d-mmm-yy", + "[>100]0;0", + '0.00" kg"', + "0.00\\k", + ], +) +def test_format_saved_value_falls_back_for_unsupported_number_formats( + number_format: str, +) -> None: + result = format_saved_value(1234.5, number_format, epoch=WINDOWS_EPOCH) + + assert result.text == "1234.5" + assert result.warning == "unsupported number format" diff --git a/tests/xlsx_fixtures.py b/tests/xlsx_fixtures.py new file mode 100644 index 0000000..1cfecd7 --- /dev/null +++ b/tests/xlsx_fixtures.py @@ -0,0 +1,233 @@ +from __future__ import annotations + +import io +from datetime import date +from pathlib import Path +from typing import Literal +from zipfile import ZIP_DEFLATED, ZipFile + +from openpyxl import Workbook + +XLSX_CONTENT_TYPE = "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml" +OFFICE_DOCUMENT_RELATIONSHIP = ( + "http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" +) + + +def minimal_xlsx_entries( + *, + include_content_types: bool = True, + include_root_relationships: bool = True, + include_workbook: bool = True, + workbook_content_type: str = XLSX_CONTENT_TYPE, + root_target: str = "xl/workbook.xml", + root_target_mode: str | None = None, +) -> list[tuple[str, bytes]]: + entries: list[tuple[str, bytes]] = [] + if include_content_types: + entries.append( + ( + "[Content_Types].xml", + f""" + + + +""".encode(), + ) + ) + if include_root_relationships: + target_mode = f' TargetMode="{root_target_mode}"' if root_target_mode else "" + entries.append( + ( + "_rels/.rels", + f""" + + + +""".encode(), + ) + ) + if include_workbook: + entries.append( + ( + "xl/workbook.xml", + b"", + ) + ) + return entries + + +def xlsx_bytes( + *, + include_content_types: bool = True, + include_root_relationships: bool = True, + include_workbook: bool = True, + workbook_content_type: str = XLSX_CONTENT_TYPE, + root_target: str = "xl/workbook.xml", + root_target_mode: str | None = None, +) -> bytes: + output = io.BytesIO() + with ZipFile(output, "w", ZIP_DEFLATED) as archive: + for name, data in minimal_xlsx_entries( + include_content_types=include_content_types, + include_root_relationships=include_root_relationships, + include_workbook=include_workbook, + workbook_content_type=workbook_content_type, + root_target=root_target, + root_target_mode=root_target_mode, + ): + archive.writestr(name, data) + return output.getvalue() + + +def write_xlsx( + path: Path, + *, + include_content_types: bool = True, + include_root_relationships: bool = True, + include_workbook: bool = True, + workbook_content_type: str = XLSX_CONTENT_TYPE, + root_target: str = "xl/workbook.xml", + root_target_mode: str | None = None, +) -> None: + path.write_bytes( + xlsx_bytes( + include_content_types=include_content_types, + include_root_relationships=include_root_relationships, + include_workbook=include_workbook, + workbook_content_type=workbook_content_type, + root_target=root_target, + root_target_mode=root_target_mode, + ) + ) + + +def write_structured_xlsx( + path: Path, + *, + sheets: tuple[ + tuple[ + str, + Literal["worksheet", "chartsheet"], + Literal["visible", "hidden", "veryHidden"], + str | None, + tuple[str, ...], + ], + ..., + ], +) -> None: + workbook_sheets: list[str] = [] + relationships: list[str] = [] + entries = minimal_xlsx_entries(include_workbook=False) + for index, (name, kind, state, dimension, cells) in enumerate(sheets, start=1): + relationship_id = f"rId{index}" + workbook_sheets.append( + f'' + ) + folder = "worksheets" if kind == "worksheet" else "chartsheets" + relationship_type = ( + f"http://schemas.openxmlformats.org/officeDocument/2006/relationships/{kind}" + ) + relationships.append( + f'' + ) + if kind == "worksheet": + dimension_xml = f'' if dimension else "" + cell_xml = "".join(f'1' for cell in cells) + entries.append( + ( + f"xl/{folder}/sheet{index}.xml", + ( + '' + '' + f"{dimension_xml}{cell_xml}" + "" + ).encode(), + ) + ) + else: + entries.append( + ( + f"xl/{folder}/sheet{index}.xml", + b'', + ) + ) + entries.extend( + [ + ( + "xl/workbook.xml", + ( + '' + '' + f"{''.join(workbook_sheets)}" + "" + ).encode(), + ), + ( + "xl/_rels/workbook.xml.rels", + ( + '' + '{"".join(relationships)}' + ).encode(), + ), + ] + ) + with ZipFile(path, "w", ZIP_DEFLATED) as archive: + for name, data in entries: + archive.writestr(name, data) + + +def rewrite_xlsx( + path: Path, + replacements: dict[str, bytes | None], +) -> None: + with ZipFile(path) as archive: + entries = {info.filename: archive.read(info) for info in archive.infolist()} + for name, data in replacements.items(): + if data is None: + entries.pop(name, None) + else: + entries[name] = data + with ZipFile(path, "w", ZIP_DEFLATED) as archive: + for name, data in entries.items(): + archive.writestr(name, data) + + +def write_public_contract_xlsx(path: Path) -> None: + workbook = Workbook() + ledger = workbook.active + ledger.title = "Ledger" + ledger.append(("Item", "Amount", "Date", "Formula", "Scientific")) + ledger.append(("Book", 1234.5, date(2026, 8, 14), "=B2*2", 1200)) + ledger["B2"].number_format = "$#,##0.00" + ledger["C2"].number_format = "yyyy-mm-dd" + ledger["E2"].number_format = "0.00E+00" + ledger["A4"] = "Merged note" + ledger.merge_cells("A4:B4") + + hidden = workbook.create_sheet("Hidden") + hidden.sheet_state = "hidden" + hidden["A1"] = "hidden value" + very_hidden = workbook.create_sheet("Very Hidden") + very_hidden.sheet_state = "veryHidden" + very_hidden["A1"] = "very hidden value" + workbook.create_sheet("Empty") + workbook.save(path) + workbook.close() + + with ZipFile(path) as archive: + worksheet = archive.read("xl/worksheets/sheet1.xml") + formula = b'B2*2' + assert formula in worksheet + rewrite_xlsx( + path, + {"xl/worksheets/sheet1.xml": worksheet.replace(formula, formula.replace(b"", b""))}, + ) diff --git a/uv.lock b/uv.lock index dabe0e5..c0261fc 100644 --- a/uv.lock +++ b/uv.lock @@ -431,6 +431,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/aa/50/a9caea39ad19c431c1a3f8a31114df65b260cdfe67786b6c7e7c040c4c44/cryptography-49.0.0-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:be9fcb48a55f023493482827d4f459bd263cc20efde64f204b97c123201850c6", size = 3783731, upload-time = "2026-06-12T20:02:43.319Z" }, ] +[[package]] +name = "defusedxml" +version = "0.7.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/0f/d5/c66da9b79e5bdb124974bfe172b4daf3c984ebd9c2a06e2b8a4dc7331c72/defusedxml-0.7.1.tar.gz", hash = "sha256:1bb3032db185915b62d7c6209c5a8792be6a32ab2fedacc84e01b52c51aa3e69", size = 75520, upload-time = "2021-03-08T10:59:26.269Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/07/6c/aa3f2f849e01cb6a001cd8554a88d4c77c5c1a31c95bdf1cf9301e6d9ef4/defusedxml-0.7.1-py2.py3-none-any.whl", hash = "sha256:a352e7e428770286cc899e2542b6cdaedb2b4953ff269a210103ec58f6198a61", size = 25604, upload-time = "2021-03-08T10:59:24.45Z" }, +] + [[package]] name = "distro" version = "1.9.0" @@ -440,6 +449,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/12/b3/231ffd4ab1fc9d679809f356cebee130ac7daa00d6d6f3206dd4fd137e9e/distro-1.9.0-py3-none-any.whl", hash = "sha256:7bffd925d65168f85027d8da9af6bddab658135b840670a223589bc0c8ef02b2", size = 20277, upload-time = "2023-12-24T09:54:30.421Z" }, ] +[[package]] +name = "et-xmlfile" +version = "2.0.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/d3/38/af70d7ab1ae9d4da450eeec1fa3918940a5fafb9055e934af8d6eb0c2313/et_xmlfile-2.0.0.tar.gz", hash = "sha256:dab3f4764309081ce75662649be815c4c9081e88f0837825f90fd28317d4da54", size = 17234, upload-time = "2024-10-25T17:25:40.039Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/c1/8b/5fe2cc11fee489817272089c4203e679c63b570a5aaeb18d852ae3cbba6a/et_xmlfile-2.0.0-py3-none-any.whl", hash = "sha256:7a91720bc756843502c3b7504c77b8fe44217c85c537d85037f0f536151b2caa", size = 18059, upload-time = "2024-10-25T17:25:39.051Z" }, +] + [[package]] name = "fastuuid" version = "0.14.0" @@ -1198,7 +1216,9 @@ name = "opendocs-sdk" version = "0.1.0" source = { editable = "." } dependencies = [ + { name = "defusedxml" }, { name = "litellm" }, + { name = "openpyxl" }, { name = "pdfplumber" }, { name = "pillow" }, { name = "python-docx" }, @@ -1215,7 +1235,9 @@ dev = [ [package.metadata] requires-dist = [ + { name = "defusedxml", specifier = ">=0.7.1,<1" }, { name = "litellm", specifier = ">=1.93,<2" }, + { name = "openpyxl", specifier = ">=3.1.5,<3.2" }, { name = "pdfplumber", specifier = ">=0.11.10,<0.12" }, { name = "pillow", specifier = ">=12.3,<13" }, { name = "python-docx", specifier = ">=1.1.2,<2" }, @@ -1230,6 +1252,18 @@ dev = [ { name = "ty", specifier = ">=0.0.63" }, ] +[[package]] +name = "openpyxl" +version = "3.1.5" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "et-xmlfile" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/3d/f9/88d94a75de065ea32619465d2f77b29a0469500e99012523b91cc4141cd1/openpyxl-3.1.5.tar.gz", hash = "sha256:cf0e3cf56142039133628b5acffe8ef0c12bc902d2aadd3e0fe5878dc08d1050", size = 186464, upload-time = "2024-06-28T14:03:44.161Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/c0/da/977ded879c29cbd04de313843e76868e6e13408a94ed6b987245dc7c8506/openpyxl-3.1.5-py2.py3-none-any.whl", hash = "sha256:5282c12b107bffeef825f4617dc029afaf41d0ea60823bbb665ef3079dc79de2", size = 250910, upload-time = "2024-06-28T14:03:41.161Z" }, +] + [[package]] name = "packaging" version = "26.2"