fix(rss): drop layout whitespace from feed and entry titles (#2428)

* fix(rss): drop layout whitespace from feed and entry titles
* Refactor return value handling in RSS converter
* Do not flatten descriptions.
This commit is contained in:
kevin
2026-09-09 11:23:30 -07:00
committed by GitHub
parent cb785cb2f7
commit a04fd8bd19
2 changed files with 172 additions and 8 deletions
@@ -145,16 +145,17 @@ class RssConverter(DocumentConverter):
Returns None if the feed type is not recognized or something goes wrong.
"""
root = doc.documentElement
title = self._get_data_by_tag_name(root, "title")
subtitle = self._get_data_by_tag_name(root, "subtitle")
title = self._get_flattened_text(root, "title")
subtitle = self._get_flattened_text(root, "subtitle")
entries = root.getElementsByTagNameNS(root.namespaceURI, "entry")
md_text = f"# {title}\n"
md_text = f"# {title}\n" if title else ""
if subtitle:
md_text += f"{subtitle}\n"
for entry in entries:
entry_title = self._get_data_by_tag_name(entry, "title")
entry_title = self._get_flattened_text(entry, "title")
entry_summary, summary_is_markup = self._get_atom_content(entry, "summary")
entry_updated = self._get_data_by_tag_name(entry, "updated")
entry_updated = self._get_flattened_text(entry, "updated")
entry_content, content_is_markup = self._get_atom_content(entry, "content")
if entry_title:
@@ -244,6 +245,21 @@ class RssConverter(DocumentConverter):
self._localize_xhtml_names(child)
return node
def _get_flattened_text(self, element: Element, tag_name: str) -> Union[str, None]:
"""Get a value that is rendered as a heading or a metadata line.
Feeds are routinely pretty-printed, so a value written as
``<title>\\n Example feed\\n</title>`` carries the indentation of the
element it sits in. Emitted verbatim it ends the Markdown heading
before the text begins, leaving a bare ``#`` and the title as body
text, so the layout whitespace is dropped here. A value that is
nothing but whitespace is reported as absent.
"""
value = self._get_data_by_tag_name(element, tag_name)
if value is None:
return None
return " ".join(part.strip() for part in value.splitlines()).strip() or None
def _parse_rss_type(
self, doc: Document, *, strict: bool = False
) -> DocumentConverterResult:
@@ -256,7 +272,7 @@ class RssConverter(DocumentConverter):
if not channel_list:
raise ValueError("No channel found in RSS feed")
channel = channel_list[0]
channel_title = self._get_data_by_tag_name(channel, "title")
channel_title = self._get_flattened_text(channel, "title")
channel_description = self._get_data_by_tag_name(channel, "description")
items = channel.getElementsByTagName("item")
md_text = ""
@@ -265,9 +281,9 @@ class RssConverter(DocumentConverter):
if channel_description:
md_text += f"{channel_description}\n"
for item in items:
title = self._get_data_by_tag_name(item, "title")
title = self._get_flattened_text(item, "title")
description = self._get_data_by_tag_name(item, "description")
pubDate = self._get_data_by_tag_name(item, "pubDate")
pubDate = self._get_flattened_text(item, "pubDate")
content = self._get_data_by_tag_name(item, "content:encoded")
if title:
@@ -0,0 +1,148 @@
#!/usr/bin/env python3 -m pytest
"""Feed and entry titles must survive a pretty-printed feed."""
import io
from markitdown import StreamInfo
from markitdown.converters import RssConverter
def test_rss_pretty_printed_titles_still_make_headings() -> None:
"""A title written on its own line must not end the heading before its text."""
feed = b"""<?xml version="1.0" encoding="utf-8"?>
<rss version="2.0"><channel>
<title>
Example feed
</title>
<description>
Example feed description
</description>
<item>
<title>
A story about things
</title>
<pubDate>
Mon, 01 Jan 2024 00:00:00 GMT
</pubDate>
<description>Body text.</description>
</item>
</channel></rss>
"""
result = RssConverter().convert(io.BytesIO(feed), StreamInfo(extension=".rss"))
assert result.title == "Example feed"
# The channel description retains its original whitespace.
assert result.markdown.splitlines() == [
"# Example feed",
"",
" Example feed description",
" ",
"",
"## A story about things",
"Published on: Mon, 01 Jan 2024 00:00:00 GMT",
"Body text.",
]
def test_atom_pretty_printed_titles_still_make_headings() -> None:
feed = b"""<?xml version="1.0" encoding="utf-8"?>
<feed xmlns="http://www.w3.org/2005/Atom">
<title>
Example feed
</title>
<subtitle>
Example subtitle
</subtitle>
<entry>
<title>
Example entry
</title>
<updated>
2024-01-01T00:00:00Z
</updated>
<content type="text">Body text.</content>
</entry>
</feed>
"""
result = RssConverter().convert(
io.BytesIO(feed), StreamInfo(mimetype="application/atom+xml")
)
assert result.title == "Example feed"
assert result.markdown.splitlines() == [
"# Example feed",
"Example subtitle",
"",
"## Example entry",
"Updated on: 2024-01-01T00:00:00Z",
"Body text.",
]
def test_rss_whitespace_only_title_makes_no_heading() -> None:
"""An empty title must be absent, not an empty '#' line."""
feed = b"""<?xml version="1.0" encoding="utf-8"?>
<rss version="2.0"><channel>
<title> </title>
<description>Example feed description</description>
<item>
<title>Example item</title>
<description>Body text.</description>
</item>
</channel></rss>
"""
result = RssConverter().convert(io.BytesIO(feed), StreamInfo(extension=".rss"))
assert result.title is None
assert not result.markdown.startswith("#\n")
assert result.markdown.splitlines()[0] == "Example feed description"
def test_atom_feed_without_any_title_makes_no_heading() -> None:
"""A missing title must not render the word 'None' as the document heading."""
feed = b"""<?xml version="1.0" encoding="utf-8"?>
<feed xmlns="http://www.w3.org/2005/Atom">
<id>urn:example:feed</id>
<entry>
<id>urn:example:entry</id>
<content type="text">Body text.</content>
</entry>
</feed>
"""
result = RssConverter().convert(
io.BytesIO(feed), StreamInfo(mimetype="application/atom+xml")
)
assert "None" not in result.markdown
assert "Body text." in result.markdown
def test_rss_single_line_titles_are_unchanged() -> None:
"""The ordinary layout must produce exactly the same markdown as before."""
feed = b"""<?xml version="1.0" encoding="utf-8"?>
<rss version="2.0"><channel>
<title>Example feed</title>
<description>Example feed description</description>
<item>
<title>Example item</title>
<pubDate>Mon, 01 Jan 2024 00:00:00 GMT</pubDate>
<description>Body text.</description>
</item>
</channel></rss>
"""
result = RssConverter().convert(io.BytesIO(feed), StreamInfo(extension=".rss"))
assert result.title == "Example feed"
assert result.markdown == (
"# Example feed\n"
"Example feed description\n"
"\n"
"## Example item\n"
"Published on: Mon, 01 Jan 2024 00:00:00 GMT\n"
"Body text."
)