mirror of
https://github.com/microsoft/markitdown.git
synced 2026-10-02 04:14:35 +08:00
fix(rss): drop layout whitespace from feed and entry titles (#2428)
* fix(rss): drop layout whitespace from feed and entry titles * Refactor return value handling in RSS converter * Do not flatten descriptions.
This commit is contained in:
@@ -145,16 +145,17 @@ class RssConverter(DocumentConverter):
|
||||
Returns None if the feed type is not recognized or something goes wrong.
|
||||
"""
|
||||
root = doc.documentElement
|
||||
title = self._get_data_by_tag_name(root, "title")
|
||||
subtitle = self._get_data_by_tag_name(root, "subtitle")
|
||||
title = self._get_flattened_text(root, "title")
|
||||
subtitle = self._get_flattened_text(root, "subtitle")
|
||||
entries = root.getElementsByTagNameNS(root.namespaceURI, "entry")
|
||||
md_text = f"# {title}\n"
|
||||
md_text = f"# {title}\n" if title else ""
|
||||
|
||||
if subtitle:
|
||||
md_text += f"{subtitle}\n"
|
||||
for entry in entries:
|
||||
entry_title = self._get_data_by_tag_name(entry, "title")
|
||||
entry_title = self._get_flattened_text(entry, "title")
|
||||
entry_summary, summary_is_markup = self._get_atom_content(entry, "summary")
|
||||
entry_updated = self._get_data_by_tag_name(entry, "updated")
|
||||
entry_updated = self._get_flattened_text(entry, "updated")
|
||||
entry_content, content_is_markup = self._get_atom_content(entry, "content")
|
||||
|
||||
if entry_title:
|
||||
@@ -244,6 +245,21 @@ class RssConverter(DocumentConverter):
|
||||
self._localize_xhtml_names(child)
|
||||
return node
|
||||
|
||||
def _get_flattened_text(self, element: Element, tag_name: str) -> Union[str, None]:
|
||||
"""Get a value that is rendered as a heading or a metadata line.
|
||||
|
||||
Feeds are routinely pretty-printed, so a value written as
|
||||
``<title>\\n Example feed\\n</title>`` carries the indentation of the
|
||||
element it sits in. Emitted verbatim it ends the Markdown heading
|
||||
before the text begins, leaving a bare ``#`` and the title as body
|
||||
text, so the layout whitespace is dropped here. A value that is
|
||||
nothing but whitespace is reported as absent.
|
||||
"""
|
||||
value = self._get_data_by_tag_name(element, tag_name)
|
||||
if value is None:
|
||||
return None
|
||||
return " ".join(part.strip() for part in value.splitlines()).strip() or None
|
||||
|
||||
def _parse_rss_type(
|
||||
self, doc: Document, *, strict: bool = False
|
||||
) -> DocumentConverterResult:
|
||||
@@ -256,7 +272,7 @@ class RssConverter(DocumentConverter):
|
||||
if not channel_list:
|
||||
raise ValueError("No channel found in RSS feed")
|
||||
channel = channel_list[0]
|
||||
channel_title = self._get_data_by_tag_name(channel, "title")
|
||||
channel_title = self._get_flattened_text(channel, "title")
|
||||
channel_description = self._get_data_by_tag_name(channel, "description")
|
||||
items = channel.getElementsByTagName("item")
|
||||
md_text = ""
|
||||
@@ -265,9 +281,9 @@ class RssConverter(DocumentConverter):
|
||||
if channel_description:
|
||||
md_text += f"{channel_description}\n"
|
||||
for item in items:
|
||||
title = self._get_data_by_tag_name(item, "title")
|
||||
title = self._get_flattened_text(item, "title")
|
||||
description = self._get_data_by_tag_name(item, "description")
|
||||
pubDate = self._get_data_by_tag_name(item, "pubDate")
|
||||
pubDate = self._get_flattened_text(item, "pubDate")
|
||||
content = self._get_data_by_tag_name(item, "content:encoded")
|
||||
|
||||
if title:
|
||||
|
||||
@@ -0,0 +1,148 @@
|
||||
#!/usr/bin/env python3 -m pytest
|
||||
"""Feed and entry titles must survive a pretty-printed feed."""
|
||||
|
||||
import io
|
||||
|
||||
from markitdown import StreamInfo
|
||||
from markitdown.converters import RssConverter
|
||||
|
||||
|
||||
def test_rss_pretty_printed_titles_still_make_headings() -> None:
|
||||
"""A title written on its own line must not end the heading before its text."""
|
||||
feed = b"""<?xml version="1.0" encoding="utf-8"?>
|
||||
<rss version="2.0"><channel>
|
||||
<title>
|
||||
Example feed
|
||||
</title>
|
||||
<description>
|
||||
Example feed description
|
||||
</description>
|
||||
<item>
|
||||
<title>
|
||||
A story about things
|
||||
</title>
|
||||
<pubDate>
|
||||
Mon, 01 Jan 2024 00:00:00 GMT
|
||||
</pubDate>
|
||||
<description>Body text.</description>
|
||||
</item>
|
||||
</channel></rss>
|
||||
"""
|
||||
|
||||
result = RssConverter().convert(io.BytesIO(feed), StreamInfo(extension=".rss"))
|
||||
|
||||
assert result.title == "Example feed"
|
||||
# The channel description retains its original whitespace.
|
||||
assert result.markdown.splitlines() == [
|
||||
"# Example feed",
|
||||
"",
|
||||
" Example feed description",
|
||||
" ",
|
||||
"",
|
||||
"## A story about things",
|
||||
"Published on: Mon, 01 Jan 2024 00:00:00 GMT",
|
||||
"Body text.",
|
||||
]
|
||||
|
||||
|
||||
def test_atom_pretty_printed_titles_still_make_headings() -> None:
|
||||
feed = b"""<?xml version="1.0" encoding="utf-8"?>
|
||||
<feed xmlns="http://www.w3.org/2005/Atom">
|
||||
<title>
|
||||
Example feed
|
||||
</title>
|
||||
<subtitle>
|
||||
Example subtitle
|
||||
</subtitle>
|
||||
<entry>
|
||||
<title>
|
||||
Example entry
|
||||
</title>
|
||||
<updated>
|
||||
2024-01-01T00:00:00Z
|
||||
</updated>
|
||||
<content type="text">Body text.</content>
|
||||
</entry>
|
||||
</feed>
|
||||
"""
|
||||
|
||||
result = RssConverter().convert(
|
||||
io.BytesIO(feed), StreamInfo(mimetype="application/atom+xml")
|
||||
)
|
||||
|
||||
assert result.title == "Example feed"
|
||||
assert result.markdown.splitlines() == [
|
||||
"# Example feed",
|
||||
"Example subtitle",
|
||||
"",
|
||||
"## Example entry",
|
||||
"Updated on: 2024-01-01T00:00:00Z",
|
||||
"Body text.",
|
||||
]
|
||||
|
||||
|
||||
def test_rss_whitespace_only_title_makes_no_heading() -> None:
|
||||
"""An empty title must be absent, not an empty '#' line."""
|
||||
feed = b"""<?xml version="1.0" encoding="utf-8"?>
|
||||
<rss version="2.0"><channel>
|
||||
<title> </title>
|
||||
<description>Example feed description</description>
|
||||
<item>
|
||||
<title>Example item</title>
|
||||
<description>Body text.</description>
|
||||
</item>
|
||||
</channel></rss>
|
||||
"""
|
||||
|
||||
result = RssConverter().convert(io.BytesIO(feed), StreamInfo(extension=".rss"))
|
||||
|
||||
assert result.title is None
|
||||
assert not result.markdown.startswith("#\n")
|
||||
assert result.markdown.splitlines()[0] == "Example feed description"
|
||||
|
||||
|
||||
def test_atom_feed_without_any_title_makes_no_heading() -> None:
|
||||
"""A missing title must not render the word 'None' as the document heading."""
|
||||
feed = b"""<?xml version="1.0" encoding="utf-8"?>
|
||||
<feed xmlns="http://www.w3.org/2005/Atom">
|
||||
<id>urn:example:feed</id>
|
||||
<entry>
|
||||
<id>urn:example:entry</id>
|
||||
<content type="text">Body text.</content>
|
||||
</entry>
|
||||
</feed>
|
||||
"""
|
||||
|
||||
result = RssConverter().convert(
|
||||
io.BytesIO(feed), StreamInfo(mimetype="application/atom+xml")
|
||||
)
|
||||
|
||||
assert "None" not in result.markdown
|
||||
assert "Body text." in result.markdown
|
||||
|
||||
|
||||
def test_rss_single_line_titles_are_unchanged() -> None:
|
||||
"""The ordinary layout must produce exactly the same markdown as before."""
|
||||
feed = b"""<?xml version="1.0" encoding="utf-8"?>
|
||||
<rss version="2.0"><channel>
|
||||
<title>Example feed</title>
|
||||
<description>Example feed description</description>
|
||||
<item>
|
||||
<title>Example item</title>
|
||||
<pubDate>Mon, 01 Jan 2024 00:00:00 GMT</pubDate>
|
||||
<description>Body text.</description>
|
||||
</item>
|
||||
</channel></rss>
|
||||
"""
|
||||
|
||||
result = RssConverter().convert(io.BytesIO(feed), StreamInfo(extension=".rss"))
|
||||
|
||||
assert result.title == "Example feed"
|
||||
assert result.markdown == (
|
||||
"# Example feed\n"
|
||||
"Example feed description\n"
|
||||
"\n"
|
||||
"## Example item\n"
|
||||
"Published on: Mon, 01 Jan 2024 00:00:00 GMT\n"
|
||||
"Body text."
|
||||
)
|
||||
Reference in New Issue
Block a user