← All ivy-nodes
IVYXSTUDIO · IVY NODE
R

RSS Fetch

ivy.node.rss-fetch · v0.1.0

ivyx✓

Fetches an RSS or Atom feed and returns its newest items as title, link, summary and date, with the feed's own title. Summaries come back as plain text.

#rss#atom#feed#news#web

Inputs

FieldTypeDescription
urlrequiredstringThe feed's address.
limitintegerAt most this many items.

Outputs

FieldTypeDescription
itemsrequiredarrayThe items as {title, link, summary, published}, in feed order.
feed_titlerequiredstringThe feed's title.
countrequiredintegerHow many items came back.

Source

python

inp = __ivy_ctx__["nodes"][__ivy_node_id__]["input"]

import re, html
import xml.etree.ElementTree as ET
import requests

url = inp["url"]
limit = int(inp.get("limit", 10))
response = requests.get(url, timeout=20, headers={"User-Agent": "IVY node rss-fetch"})
response.raise_for_status()
root = ET.fromstring(response.content)


def plain(value):
    return re.sub(r"\s+", " ", re.sub(r"<[^>]+>", " ", html.unescape(value or ""))).strip()


def local(tag):
    return tag.rsplit("}", 1)[-1]


def child(node, name):
    for c in node:
        if local(c.tag) == name:
            return c
    return None


items, feed_title = [], ""
if local(root.tag) == "rss":
    channel = child(root, "channel")
    feed_title = (child(channel, "title").text or "").strip() if channel is not None and child(channel, "title") is not None else ""
    for it in [c for c in channel if local(c.tag) == "item"] if channel is not None else []:
        get = lambda n: (child(it, n).text if child(it, n) is not None else "") or ""
        items.append({"title": get("title").strip(), "link": get("link").strip(), "summary": plain(get("description")), "published": get("pubDate").strip()})
elif local(root.tag) == "feed":
    feed_title = (child(root, "title").text or "").strip() if child(root, "title") is not None else ""
    for entry in [c for c in root if local(c.tag) == "entry"]:
        get = lambda n: (child(entry, n).text if child(entry, n) is not None else "") or ""
        link = child(entry, "link")
        items.append({"title": get("title").strip(), "link": (link.get("href") if link is not None else "") or "",
                      "summary": plain(get("summary") or get("content")), "published": (get("updated") or get("published")).strip()})
else:
    raise ValueError(f"{url} is not an RSS or Atom feed.")

out = __ivy_ctx__["nodes"][__ivy_node_id__]["output"]
out["items"] = items[:limit]
out["feed_title"] = feed_title
out["count"] = len(items[:limit])

Tests

Requires: python:3.9, requests

  • rss

    An RSS feed's items come back with their summaries as plain text.

  • atom

    An Atom feed reads the same way.

  • not-a-feed

    A page that is not a feed is refused.