mirror of
https://github.com/microsoft/markitdown.git
synced 2026-10-02 04:14:35 +08:00
fix: fall back to plain text when RSS item content triggers RecursionError (#2333)
* fix: fall back to plain text when RSS item content triggers RecursionError Deeply nested HTML inside an RSS item's description or content:encoded previously hit the broad except in _parse_content and returned the raw unconverted HTML, silently embedding HTML tags in the markdown output. Catch RecursionError specifically and fall back to BeautifulSoup's iterative get_text(), mirroring the HTML converter fix from #1644. Includes a deterministic regression test. * Honor strict=true, similar to HTML
This commit is contained in:
@@ -1,3 +1,5 @@
|
||||
import warnings
|
||||
|
||||
from defusedxml import minidom
|
||||
from xml.dom.minidom import Document, Element
|
||||
from typing import BinaryIO, Any, Union
|
||||
@@ -87,18 +89,23 @@ class RssConverter(DocumentConverter):
|
||||
stream_info: StreamInfo,
|
||||
**kwargs: Any, # Options to pass to the converter
|
||||
) -> DocumentConverterResult:
|
||||
# Pop our own keyword before forwarding the rest to markdownify.
|
||||
# strict=True raises RecursionError instead of falling back to plain text.
|
||||
strict: bool = kwargs.pop("strict", False)
|
||||
self._kwargs = kwargs
|
||||
doc = minidom.parse(file_stream)
|
||||
feed_type = self._feed_type(doc)
|
||||
|
||||
if feed_type == "rss":
|
||||
return self._parse_rss_type(doc)
|
||||
return self._parse_rss_type(doc, strict=strict)
|
||||
elif feed_type == "atom":
|
||||
return self._parse_atom_type(doc)
|
||||
return self._parse_atom_type(doc, strict=strict)
|
||||
else:
|
||||
raise ValueError("Unknown feed type")
|
||||
|
||||
def _parse_atom_type(self, doc: Document) -> DocumentConverterResult:
|
||||
def _parse_atom_type(
|
||||
self, doc: Document, *, strict: bool = False
|
||||
) -> DocumentConverterResult:
|
||||
"""Parse the type of an Atom feed.
|
||||
|
||||
Returns None if the feed type is not recognized or something goes wrong.
|
||||
@@ -121,16 +128,18 @@ class RssConverter(DocumentConverter):
|
||||
if entry_updated:
|
||||
md_text += f"Updated on: {entry_updated}\n"
|
||||
if entry_summary:
|
||||
md_text += self._parse_content(entry_summary)
|
||||
md_text += self._parse_content(entry_summary, strict=strict)
|
||||
if entry_content:
|
||||
md_text += self._parse_content(entry_content)
|
||||
md_text += self._parse_content(entry_content, strict=strict)
|
||||
|
||||
return DocumentConverterResult(
|
||||
markdown=md_text,
|
||||
title=title,
|
||||
)
|
||||
|
||||
def _parse_rss_type(self, doc: Document) -> DocumentConverterResult:
|
||||
def _parse_rss_type(
|
||||
self, doc: Document, *, strict: bool = False
|
||||
) -> DocumentConverterResult:
|
||||
"""Parse the type of an RSS feed.
|
||||
|
||||
Returns None if the feed type is not recognized or something goes wrong.
|
||||
@@ -158,21 +167,34 @@ class RssConverter(DocumentConverter):
|
||||
if pubDate:
|
||||
md_text += f"Published on: {pubDate}\n"
|
||||
if description:
|
||||
md_text += self._parse_content(description)
|
||||
md_text += self._parse_content(description, strict=strict)
|
||||
if content:
|
||||
md_text += self._parse_content(content)
|
||||
md_text += self._parse_content(content, strict=strict)
|
||||
|
||||
return DocumentConverterResult(
|
||||
markdown=md_text,
|
||||
title=channel_title,
|
||||
)
|
||||
|
||||
def _parse_content(self, content: str) -> str:
|
||||
def _parse_content(self, content: str, *, strict: bool = False) -> str:
|
||||
"""Parse the content of an RSS feed item"""
|
||||
try:
|
||||
# using bs4 because many RSS feeds have HTML-styled content
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
return _CustomMarkdownify(**self._kwargs).convert_soup(soup)
|
||||
except RecursionError:
|
||||
if strict:
|
||||
raise
|
||||
# Deeply nested item content can exceed Python's recursion limit
|
||||
# during markdownify's recursive DOM traversal. Fall back to
|
||||
# BeautifulSoup's iterative get_text() so the caller still gets
|
||||
# usable plain-text content instead of raw HTML.
|
||||
warnings.warn(
|
||||
"RSS item content is too deeply nested for markdown conversion "
|
||||
"(RecursionError). Falling back to plain-text extraction.",
|
||||
stacklevel=2,
|
||||
)
|
||||
return BeautifulSoup(content, "html.parser").get_text("\n", strip=True)
|
||||
except BaseException as _:
|
||||
return content
|
||||
|
||||
|
||||
@@ -9,6 +9,7 @@ from unittest.mock import MagicMock
|
||||
|
||||
from markitdown._uri_utils import parse_data_uri, file_uri_to_path
|
||||
from markitdown._markitdown import _get_content_disposition_filename
|
||||
from markitdown.converters import RssConverter
|
||||
|
||||
from markitdown import (
|
||||
MarkItDown,
|
||||
@@ -434,6 +435,7 @@ def test_deeply_nested_html_fallback() -> None:
|
||||
# Should have emitted a warning about the fallback
|
||||
recursion_warnings = [x for x in w if "deeply nested" in str(x.message)]
|
||||
assert len(recursion_warnings) > 0
|
||||
|
||||
finally:
|
||||
sys.setrecursionlimit(original_limit)
|
||||
|
||||
@@ -444,6 +446,96 @@ def test_deeply_nested_html_fallback() -> None:
|
||||
assert "<p>" not in result.markdown
|
||||
|
||||
|
||||
def test_deeply_nested_rss_item_fallback() -> None:
|
||||
"""Deeply nested HTML inside an RSS item should fall back to plain-text
|
||||
extraction instead of silently embedding raw unconverted HTML in the
|
||||
markdown output (same failure class as the HTML converter fix in #1644).
|
||||
|
||||
Note: This test uses sys.setrecursionlimit to guarantee a RecursionError
|
||||
regardless of the host environment's default limit, making it deterministic
|
||||
across different platforms and CI configurations.
|
||||
"""
|
||||
import sys
|
||||
import warnings
|
||||
|
||||
markitdown = MarkItDown()
|
||||
|
||||
# Use a small recursion limit so the test is environment-independent.
|
||||
# We restore the original limit in a finally block to avoid side-effects.
|
||||
original_limit = sys.getrecursionlimit()
|
||||
low_limit = 200 # well below markdownify's traversal depth for depth=500
|
||||
|
||||
# Build an RSS item whose content is deeply nested HTML
|
||||
depth = 500
|
||||
item_html = ""
|
||||
for _ in range(depth):
|
||||
item_html += '<div style="margin-left:10px">'
|
||||
item_html += "<p>Deep feed content with <b>bold text</b></p>"
|
||||
for _ in range(depth):
|
||||
item_html += "</div>"
|
||||
|
||||
rss = (
|
||||
'<?xml version="1.0" encoding="UTF-8"?>'
|
||||
'<rss version="2.0" '
|
||||
'xmlns:content="http://purl.org/rss/1.0/modules/content/">'
|
||||
"<channel>"
|
||||
"<title>Test Feed</title>"
|
||||
"<description>A test feed</description>"
|
||||
"<item>"
|
||||
"<title>Deep Item</title>"
|
||||
f"<content:encoded><![CDATA[{item_html}]]></content:encoded>"
|
||||
"</item>"
|
||||
"</channel>"
|
||||
"</rss>"
|
||||
)
|
||||
atom = (
|
||||
'<?xml version="1.0" encoding="UTF-8"?>'
|
||||
'<feed xmlns="http://www.w3.org/2005/Atom">'
|
||||
"<title>Test Feed</title>"
|
||||
"<entry>"
|
||||
"<title>Deep Entry</title>"
|
||||
f'<content type="html"><![CDATA[{item_html}]]></content>'
|
||||
"</entry>"
|
||||
"</feed>"
|
||||
)
|
||||
|
||||
try:
|
||||
sys.setrecursionlimit(low_limit)
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
warnings.simplefilter("always")
|
||||
result = markitdown.convert_stream(
|
||||
io.BytesIO(rss.encode("utf-8")),
|
||||
file_extension=".rss",
|
||||
)
|
||||
|
||||
# Should have emitted a warning about the fallback
|
||||
recursion_warnings = [x for x in w if "deeply nested" in str(x.message)]
|
||||
assert len(recursion_warnings) > 0
|
||||
|
||||
# strict=True should expose the conversion failure rather than applying
|
||||
# the plain-text fallback.
|
||||
with pytest.raises(RecursionError):
|
||||
RssConverter().convert(
|
||||
io.BytesIO(rss.encode("utf-8")),
|
||||
StreamInfo(extension=".rss"),
|
||||
strict=True,
|
||||
)
|
||||
with pytest.raises(RecursionError):
|
||||
RssConverter().convert(
|
||||
io.BytesIO(atom.encode("utf-8")),
|
||||
StreamInfo(extension=".atom"),
|
||||
strict=True,
|
||||
)
|
||||
finally:
|
||||
sys.setrecursionlimit(original_limit)
|
||||
|
||||
# The output should contain the text content, not raw HTML
|
||||
assert "Deep feed content" in result.markdown
|
||||
assert "bold text" in result.markdown
|
||||
assert "<div" not in result.markdown
|
||||
assert "<p>" not in result.markdown
|
||||
|
||||
|
||||
def test_doc_rlink() -> None:
|
||||
# Test for: CVE-2025-11849
|
||||
markitdown = MarkItDown()
|
||||
@@ -644,7 +736,7 @@ def test_json_with_late_non_ascii_character(tmp_path) -> None:
|
||||
"title": "Example record",
|
||||
"abstract": "This is sample test. " * 500,
|
||||
},
|
||||
"notes": "non-ASCII character: è",
|
||||
"notes": "non-ASCII character: è",
|
||||
}
|
||||
json_path = tmp_path / "input.json"
|
||||
json_path.write_text(
|
||||
@@ -653,7 +745,7 @@ def test_json_with_late_non_ascii_character(tmp_path) -> None:
|
||||
|
||||
result = MarkItDown().convert(str(json_path))
|
||||
|
||||
assert "non-ASCII character: è" in result.text_content
|
||||
assert "non-ASCII character: è" in result.text_content
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
Reference in New Issue
Block a user