Test markitdown-ocr in CI against the sibling markitdown (#2378)

* Test markitdown-ocr in CI against the sibling markitdown
* Run the full matrix.
This commit is contained in:
afourney
2026-09-03 20:28:53 -07:00
committed by GitHub
parent 2dffd8bb6e
commit bc90c5d7a5
4 changed files with 56 additions and 11 deletions
+31 -4
View File
@@ -1,18 +1,45 @@
name: tests
on: [pull_request]
permissions:
contents: read
jobs:
tests:
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix:
python-version: ["3.10", "3.11", "3.12", "3.13"]
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: |
3.10
3.11
3.12
python-version: ${{ matrix.python-version }}
- name: Install Hatch
run: pipx install hatch
- name: Run tests
run: cd packages/markitdown; hatch test
ocr-tests:
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix:
python-version: ["3.10", "3.11", "3.12", "3.13"]
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: ${{ matrix.python-version }}
# Install markitdown from the local checkout rather than PyPI, so the
# plugin is tested against the sibling package it ships alongside.
- name: Install packages
run: pip install ./packages/markitdown ./packages/markitdown-ocr pytest
- name: Run tests
working-directory: packages/markitdown-ocr
run: pytest
@@ -144,17 +144,30 @@ def test_pdf_complex_layout(svc: MockOCRService) -> None:
# ---------------------------------------------------------------------------
# pdf_multipage.pdf — pdfplumber/pdfminer fail (EOF); PyMuPDF fallback used
# pdf_multipage.pdf
# ---------------------------------------------------------------------------
def test_pdf_multipage(svc: MockOCRService) -> None:
# pdfplumber cannot open this file (Unexpected EOF), so _ocr_full_pages
# falls back to PyMuPDF for page rendering. Each page becomes one OCR block.
expected = (
f"## Page 1\n\n\n{_OCR_BLOCK}\n\n\n"
f"## Page 2\n\n\n{_OCR_BLOCK}\n\n\n"
f"## Page 3\n\n\n{_OCR_BLOCK}"
"## Page 1\n\n\n"
"Page 1 - Content before image\n\n"
"This is important text that appears BEFORE the image.\n\n\n\n"
"*[Image OCR]\nMOCK_OCR_TEXT_12345\n[End OCR]*\n\n\n"
"This text appears AFTER the image on page 1.\n\n"
"More content follows here.\n\n\n"
"## Page 2\n\n\n"
"Page 2 - Content with image at end\n\n"
"Main content of page 2 starts here.\n\n"
"This is paragraph 1.\n\n"
"This is paragraph 2.\n\n"
"Final paragraph before image.\n\n\n\n"
"*[Image OCR]\nMOCK_OCR_TEXT_12345\n[End OCR]*\n\n\n\n"
"## Page 3\n\n\n"
"Page 3 - Image at top\n\n\n\n"
"*[Image OCR]\nMOCK_OCR_TEXT_12345\n[End OCR]*\n\n\n"
"Content that follows the image.\n\n"
"This text is AFTER the image."
)
assert _convert("pdf_multipage.pdf", svc) == expected
@@ -219,7 +219,7 @@ def main():
elif args.use_cu:
if args.cu_endpoint is None:
_exit_with_error(
"Content Understanding Endpoint (--cu-endpoint) is required when using --use-cu."
"Content Understanding Endpoint (--cu-endpoint) is required when using --use-cu. "
"Pass --cu-endpoint or set MARKITDOWN_CU_ENDPOINT."
)
@@ -114,6 +114,11 @@ def _pre_process_strike(content: bytes) -> bytes:
Returns:
bytes: The processed content with "dstrike" elements renamed to "strike", encoded as bytes.
"""
# Double strikethrough is rare, and parsing/reserializing the XML is expensive
# on large documents, so skip the round-trip when there is nothing to rename.
if b"dstrike" not in content:
return content
soup = BeautifulSoup(content.decode(), features="xml")
for tag in soup.find_all("dstrike"):
tag.name = "strike"