Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,15 @@

本项目的发布说明遵循语义化版本,并记录用户可见的兼容性边界。

## 未发布

### 修复

- 限制单页 PDF 视觉候选数量,避免异常重叠对象触发高复杂度区域合并。
- 异步取消不再依赖仍在运行的事件循环清理 SDK 自有临时源文件。
- PPTX 在完整提取媒体和幻灯片内容前执行 `max_pages` 校验。
- 空 TXT、Markdown 和无名文本字节返回更明确的 `NoUsableContentError` 消息。

## 0.1.0 - Alpha

OpenDocs 的首个公开 Alpha 将本地文档转换为 Markdown。
Expand Down
27 changes: 27 additions & 0 deletions src/opendocs/parsers/office/package.py
Original file line number Diff line number Diff line change
Expand Up @@ -224,6 +224,33 @@ def validate_office_package(path: Path, *, document_type: DocumentType) -> Offic
)


def _local_name(tag: str) -> str:
return tag.rsplit("}", 1)[-1]


def enforce_pptx_page_limit(path: Path, *, max_pages: int) -> None:
if isinstance(max_pages, bool) or not isinstance(max_pages, int):
raise TypeError("max_pages must be an int")
if max_pages <= 0:
raise ValueError("max_pages must be greater than zero")

layout = validate_office_package(path, document_type=DocumentType.PPTX)
try:
with ZipFile(path) as archive:
root = ET.fromstring(archive.read(layout.main_part_name))
except (BadZipFile, KeyError, OSError, ET.ParseError) as error:
raise CorruptDocumentError("PPTX presentation part is corrupt") from error

page_count = sum(
_local_name(slide.tag) == "sldId"
for child in root
if _local_name(child.tag) == "sldIdLst"
for slide in child
)
if page_count > max_pages:
raise LimitExceededError(f"PPTX exceeds the configured {max_pages} page limit")


def open_validated_office_document(
path: Path,
*,
Expand Down
8 changes: 6 additions & 2 deletions src/opendocs/parsers/office/parser.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@
document_from_wire,
document_to_wire,
)
from opendocs.parsers.office.package import enforce_pptx_page_limit
from opendocs.source import ParseWorkspace, ResolvedSource
from opendocs.vision.base import VisionClient, VisionRequest, VisionRequestKind, VisionResult
from opendocs.vision.images import (
Expand Down Expand Up @@ -58,6 +59,7 @@ def _extract_office_to_wire(
document_type_value: str,
path: Path,
workspace_path: Path,
max_pages: int,
) -> dict[str, object]:
document_type = DocumentType(document_type_value)
workspace = ParseWorkspace(workspace_path)
Expand All @@ -68,6 +70,7 @@ def _extract_office_to_wire(
elif document_type is DocumentType.PPTX:
from opendocs.parsers.office.pptx import extract_pptx

enforce_pptx_page_limit(path, max_pages=max_pages)
document = extract_pptx(path, workspace)
else:
raise ValueError("native Office extraction requires DOCX or PPTX")
Expand Down Expand Up @@ -143,7 +146,7 @@ async def parse(self, source: ResolvedSource, *, options: ParseOptions) -> Parse
deadline = min(deadline, self._deadline)
try:
async with asyncio.timeout_at(deadline):
document = await self._extract(source)
document = await self._extract(source, max_pages=options.max_pages)
if (
self._document_type is DocumentType.PPTX
and len(document.pages) > options.max_pages
Expand Down Expand Up @@ -178,12 +181,13 @@ async def parse(self, source: ResolvedSource, *, options: ParseOptions) -> Parse
f"{self._document_type.value.upper()} produced no usable content"
)

async def _extract(self, source: ResolvedSource) -> OfficeDocument:
async def _extract(self, source: ResolvedSource, *, max_pages: int) -> OfficeDocument:
wire = await self._runtime.run_native(
_extract_office_to_wire,
self._document_type.value,
source.path,
self._runtime.workspace.path,
max_pages,
)
try:
document = document_from_wire(wire)
Expand Down
5 changes: 5 additions & 0 deletions src/opendocs/parsers/pdf/analyze.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,7 @@
page_to_wire,
)
from opendocs.parsers.pdf.routing import (
MAX_VISUAL_CANDIDATES_PER_PAGE,
VECTOR_OBJECT_COUNT_MIN,
build_visual_regions,
significant_image,
Expand Down Expand Up @@ -165,6 +166,10 @@ def _reading_order_ambiguity(words: Sequence[PdfWord]) -> tuple[bool, list[BBox]
left_area = (left.bbox.right - left.bbox.left) * (left.bbox.bottom - left.bbox.top)
right_area = (right.bbox.right - right.bbox.left) * (right.bbox.bottom - right.bbox.top)
if horizontal * vertical / min(left_area, right_area) >= 0.20:
if len(overlapping) >= MAX_VISUAL_CANDIDATES_PER_PAGE:
raise LimitExceededError(
"PDF visual region candidates exceed the resource budget"
)
overlapping.append(
BBox(
min(left.bbox.left, right.bbox.left),
Expand Down
4 changes: 4 additions & 0 deletions src/opendocs/parsers/pdf/routing.py
Original file line number Diff line number Diff line change
@@ -1,13 +1,15 @@
from __future__ import annotations

from opendocs._models import BBox
from opendocs.errors import LimitExceededError
from opendocs.parsers.pdf.extract import bbox_area, intersection_area, union_area
from opendocs.parsers.pdf.models import PageFacts, PageRoute, PageRouteDecision, VisualRegion

SIGNIFICANT_REGION_AREA_MIN = 0.05
FULL_PAGE_IMAGE_AREA_MIN = 0.85
FULL_VISION_UNION_AREA_MIN = 0.60
MAX_REGIONS_PER_PAGE = 4
MAX_VISUAL_CANDIDATES_PER_PAGE = 1_024
REGION_PADDING = 0.015
REGION_MERGE_GAP = 0.02
VECTOR_OBJECT_COUNT_MIN = 30
Expand All @@ -33,6 +35,8 @@ def _should_merge(left: BBox, right: BBox) -> bool:
def build_visual_regions(
candidates: list[tuple[BBox, str]],
) -> tuple[VisualRegion, ...]:
if len(candidates) > MAX_VISUAL_CANDIDATES_PER_PAGE:
raise LimitExceededError("PDF visual region candidates exceed the resource budget")
merged = [(_expand(bbox), {reason}, index) for index, (bbox, reason) in enumerate(candidates)]
while True:
for left_index, (left, reasons, source_index) in enumerate(merged):
Expand Down
4 changes: 3 additions & 1 deletion src/opendocs/parsers/text.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@
import re

from opendocs._models import DocumentType, MarkdownBlock, ParsedDocument, TextBlock
from opendocs.errors import CorruptDocumentError, LimitExceededError
from opendocs.errors import CorruptDocumentError, LimitExceededError, NoUsableContentError
from opendocs.options import ParseOptions
from opendocs.source import ResolvedSource

Expand Down Expand Up @@ -47,6 +47,8 @@ async def parse(
) -> ParsedDocument:
del options
value = await asyncio.to_thread(_read_utf8, source)
if not value.strip():
raise NoUsableContentError(f"{self._document_type.value} document is empty")

if self._document_type is DocumentType.MARKDOWN:
blocks = (MarkdownBlock(markdown=value),)
Expand Down
47 changes: 32 additions & 15 deletions src/opendocs/source.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@
import shutil
import sys
import tempfile
import threading
import warnings
from collections.abc import AsyncIterator
from contextlib import asynccontextmanager, suppress
Expand Down Expand Up @@ -145,10 +146,21 @@ def _write_temporary(path: Path, data: bytes) -> None:
handle.write(data)


def _cleanup_finished_write(task: asyncio.Task[None], path: Path) -> None:
def _consume_finished_write(task: asyncio.Task[None]) -> None:
if not task.cancelled():
task.exception()
_schedule_background_cleanup(path)


def _write_temporary_with_cancellation_cleanup(
path: Path,
data: bytes,
cleanup_requested: threading.Event,
) -> None:
try:
_write_temporary(path, data)
finally:
if cleanup_requested.is_set():
_cleanup_owned_path_now(path)


async def _unlink_if_exists(path: Path) -> None:
Expand All @@ -169,21 +181,16 @@ def _warn_cleanup_failure(path: Path, error: BaseException) -> None:
)


def _consume_task_exception(task: asyncio.Task[object], path: Path) -> None:
if not task.cancelled():
exception = task.exception()
if exception is not None:
_warn_cleanup_failure(path, exception)


def _schedule_background_cleanup(path: Path) -> None:
cleanup_task = asyncio.create_task(_cleanup_owned_path(path))
cleanup_task.add_done_callback(lambda task: _consume_task_exception(task, path))
def _cleanup_owned_path_now(path: Path) -> None:
try:
path.unlink(missing_ok=True)
except OSError as error:
_warn_cleanup_failure(path, error)


async def _cleanup_after_cancellation(path: Path, *, wait: bool) -> None:
if not wait:
_schedule_background_cleanup(path)
_cleanup_owned_path_now(path)
return

try:
Expand All @@ -196,7 +203,15 @@ async def _cleanup_after_cancellation(path: Path, *, wait: bool) -> None:

async def _write_owned(data: bytes, *, wait_for_cleanup_on_cancel: bool) -> Path:
path = _create_temporary_path(data)
write_task = asyncio.create_task(asyncio.to_thread(_write_temporary, path, data))
cleanup_requested = threading.Event()
write_task = asyncio.create_task(
asyncio.to_thread(
_write_temporary_with_cancellation_cleanup,
path,
data,
cleanup_requested,
)
)
try:
await asyncio.shield(write_task)
except asyncio.CancelledError:
Expand All @@ -205,7 +220,9 @@ async def _write_owned(data: bytes, *, wait_for_cleanup_on_cancel: bool) -> Path
await write_task
await _cleanup_after_cancellation(path, wait=True)
else:
write_task.add_done_callback(lambda completed: _cleanup_finished_write(completed, path))
cleanup_requested.set()
await _cleanup_after_cancellation(path, wait=False)
write_task.add_done_callback(_consume_finished_write)
raise
except OSError:
await _unlink_if_exists(path)
Expand Down
Loading
Loading