← All ivy-nodes
IVYXSTUDIO · IVY NODE
R
RSS Fetch
ivy.node.rss-fetch · v0.1.0
ivyx✓
Fetches an RSS or Atom feed and returns its newest items as title, link, summary and date, with the feed's own title. Summaries come back as plain text.
#rss#atom#feed#news#web
Inputs
| Field | Type | Description |
|---|---|---|
| urlrequired | string | The feed's address. |
| limit | integer | At most this many items. |
Outputs
| Field | Type | Description |
|---|---|---|
| itemsrequired | array | The items as {title, link, summary, published}, in feed order. |
| feed_titlerequired | string | The feed's title. |
| countrequired | integer | How many items came back. |
Source
python
inp = __ivy_ctx__["nodes"][__ivy_node_id__]["input"]
import re, html
import xml.etree.ElementTree as ET
import requests
url = inp["url"]
limit = int(inp.get("limit", 10))
response = requests.get(url, timeout=20, headers={"User-Agent": "IVY node rss-fetch"})
response.raise_for_status()
root = ET.fromstring(response.content)
def plain(value):
return re.sub(r"\s+", " ", re.sub(r"<[^>]+>", " ", html.unescape(value or ""))).strip()
def local(tag):
return tag.rsplit("}", 1)[-1]
def child(node, name):
for c in node:
if local(c.tag) == name:
return c
return None
items, feed_title = [], ""
if local(root.tag) == "rss":
channel = child(root, "channel")
feed_title = (child(channel, "title").text or "").strip() if channel is not None and child(channel, "title") is not None else ""
for it in [c for c in channel if local(c.tag) == "item"] if channel is not None else []:
get = lambda n: (child(it, n).text if child(it, n) is not None else "") or ""
items.append({"title": get("title").strip(), "link": get("link").strip(), "summary": plain(get("description")), "published": get("pubDate").strip()})
elif local(root.tag) == "feed":
feed_title = (child(root, "title").text or "").strip() if child(root, "title") is not None else ""
for entry in [c for c in root if local(c.tag) == "entry"]:
get = lambda n: (child(entry, n).text if child(entry, n) is not None else "") or ""
link = child(entry, "link")
items.append({"title": get("title").strip(), "link": (link.get("href") if link is not None else "") or "",
"summary": plain(get("summary") or get("content")), "published": (get("updated") or get("published")).strip()})
else:
raise ValueError(f"{url} is not an RSS or Atom feed.")
out = __ivy_ctx__["nodes"][__ivy_node_id__]["output"]
out["items"] = items[:limit]
out["feed_title"] = feed_title
out["count"] = len(items[:limit])Tests
Requires: python:3.9, requests
- rss
An RSS feed's items come back with their summaries as plain text.
- atom
An Atom feed reads the same way.
- not-a-feed
A page that is not a feed is refused.