Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
__pycache__/
*.pyc
116 changes: 116 additions & 0 deletions news_fetcher.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,116 @@
"""Fetch today's news headlines from public RSS feeds."""

import urllib.request
import xml.etree.ElementTree as ET
from datetime import datetime, timezone


# Public RSS feed sources
RSS_FEEDS = {
"BBC News": "https://feeds.bbci.co.uk/news/rss.xml",
"CNN": "http://rss.cnn.com/rss/edition.rss",
"Reuters": "https://feeds.reuters.com/reuters/topNews",
}

DEFAULT_TIMEOUT = 10 # seconds


def fetch_rss(url, timeout=DEFAULT_TIMEOUT):
"""Fetch and parse an RSS feed from the given URL.

Args:
url: The RSS feed URL.
timeout: Request timeout in seconds.

Returns:
A list of dicts with keys 'title', 'link', and 'published'.
"""
req = urllib.request.Request(
url,
headers={"User-Agent": "NewsFetcher/1.0"},
)
with urllib.request.urlopen(req, timeout=timeout) as response:
data = response.read()

return parse_rss_xml(data)


def parse_rss_xml(xml_bytes):
"""Parse RSS XML bytes into a list of article dicts.

Args:
xml_bytes: Raw XML content as bytes.

Returns:
A list of dicts with keys 'title', 'link', and 'published'.
"""
root = ET.fromstring(xml_bytes)
items = root.findall(".//item")

articles = []
for item in items:
title_el = item.find("title")
link_el = item.find("link")
pub_date_el = item.find("pubDate")

articles.append({
"title": title_el.text.strip() if title_el is not None and title_el.text else "",
"link": link_el.text.strip() if link_el is not None and link_el.text else "",
"published": pub_date_el.text.strip() if pub_date_el is not None and pub_date_el.text else "",
})

return articles


def fetch_news(sources=None, max_items=5):
"""Fetch top news headlines from multiple RSS sources.

Args:
sources: A dict mapping source names to RSS URLs.
Defaults to RSS_FEEDS.
max_items: Maximum number of headlines per source.

Returns:
A dict mapping source names to lists of article dicts.
"""
if sources is None:
sources = RSS_FEEDS

all_news = {}
for name, url in sources.items():
try:
articles = fetch_rss(url)
all_news[name] = articles[:max_items]
except Exception as exc:
all_news[name] = [{"title": f"Error fetching feed: {exc}", "link": "", "published": ""}]

return all_news


def display_news(news):
"""Print news headlines to the console.

Args:
news: A dict mapping source names to lists of article dicts.
"""
today = datetime.now(timezone.utc).strftime("%Y-%m-%d")
print(f"\n{'='*60}")
print(f" Today's News Headlines — {today}")
print(f"{'='*60}\n")

for source, articles in news.items():
print(f"📰 {source}")
print(f"{'-'*40}")
for i, article in enumerate(articles, 1):
print(f" {i}. {article['title']}")
if article["link"]:
print(f" 🔗 {article['link']}")
if article["published"]:
print(f" 📅 {article['published']}")
print()


if __name__ == "__main__":
print("Fetching today's news...")
news = fetch_news()
display_news(news)
82 changes: 82 additions & 0 deletions test_news_fetcher.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,82 @@
"""Tests for news_fetcher module."""

import unittest
from news_fetcher import parse_rss_xml, display_news


SAMPLE_RSS = b"""\
<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0">
<channel>
<title>Sample News</title>
<item>
<title>First headline</title>
<link>https://example.com/1</link>
<pubDate>Mon, 10 Feb 2026 00:00:00 GMT</pubDate>
</item>
<item>
<title>Second headline</title>
<link>https://example.com/2</link>
<pubDate>Mon, 10 Feb 2026 01:00:00 GMT</pubDate>
</item>
</channel>
</rss>
"""


class TestParseRssXml(unittest.TestCase):
"""Tests for the parse_rss_xml helper."""

def test_parses_titles(self):
articles = parse_rss_xml(SAMPLE_RSS)
self.assertEqual(len(articles), 2)
self.assertEqual(articles[0]["title"], "First headline")
self.assertEqual(articles[1]["title"], "Second headline")

def test_parses_links(self):
articles = parse_rss_xml(SAMPLE_RSS)
self.assertEqual(articles[0]["link"], "https://example.com/1")

def test_parses_pub_date(self):
articles = parse_rss_xml(SAMPLE_RSS)
self.assertEqual(articles[0]["published"], "Mon, 10 Feb 2026 00:00:00 GMT")

def test_empty_feed(self):
empty_rss = b"""\
<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0"><channel><title>Empty</title></channel></rss>
"""
articles = parse_rss_xml(empty_rss)
self.assertEqual(articles, [])

def test_missing_optional_elements(self):
rss_no_link = b"""\
<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0">
<channel>
<item><title>Only title</title></item>
</channel>
</rss>
"""
articles = parse_rss_xml(rss_no_link)
self.assertEqual(len(articles), 1)
self.assertEqual(articles[0]["title"], "Only title")
self.assertEqual(articles[0]["link"], "")
self.assertEqual(articles[0]["published"], "")


class TestDisplayNews(unittest.TestCase):
"""Tests for the display_news function (smoke test)."""

def test_display_does_not_raise(self):
news = {
"Test Source": [
{"title": "Headline", "link": "https://example.com", "published": "today"},
]
}
# Should not raise
display_news(news)


if __name__ == "__main__":
unittest.main()