News Synthesizer Setup
2025-01-302 turns16,915 charsgpt-4o
Summary
User is building a news synthesizer app using Streamlit and Python libraries for fetching and displaying news content.
Messages
from __future__ import annotations
from typing import List, Dict, Tuple, Optional, Any
import feedparser
import asyncio
import re
import aiohttp
from playwright.async_api import async_playwright
import datetime
import json
import streamlit as st
from streamlit_extras.colored_header import colored_header
from streamlit_extras.add_vertical_space import add_vertical_space
from streamlit_lottie import st_lottie
import logging
import hashlib
import sqlite3
import requests
from datetime import datetime
class NewsSynthesizer:
def __init__(self):
self.rss_feeds = [
('BBC News', 'http://feeds.bbci.co.uk/news/rss.xml'),
('Reuters', 'https://www.reutersagency.com/feed/?taxonomy=best-topics&post_type=best'), # Updated Reuters feed
('AP News', 'https://apnews.com/hub/ap-top-news.rss'),
('NY Times', 'https://rss.nytimes.com/services/xml/rss/nyt/HomePage.xml'),
('Al Jazeera', 'https://www.aljazeera.com/xml/rss/all.xml')
]
self.user_agent = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
self.scrape_timeout = 30
self.logger = logging.getLogger(__name__)
self.setup_database()
def setup_database(self):
"""Initialize SQLite database with necessary tables"""
try:
with sqlite3.connect('news_synthesis.db') as conn:
cursor = conn.cursor()
# Create articles table
cursor.execute('''
CREATE TABLE IF NOT EXISTS articles (
id INTEGER PRIMARY KEY AUTOINCREMENT,
title TEXT NOT NULL,
source TEXT NOT NULL,
url TEXT UNIQUE NOT NULL,
content TEXT,
published_date TEXT,
hash TEXT UNIQUE,
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
)
''')
# Create reports table
cursor.execute('''
CREATE TABLE IF NOT EXISTS reports (
id INTEGER PRIMARY KEY AUTOINCREMENT,
headline TEXT NOT NULL,
sources TEXT NOT NULL,
urls TEXT NOT NULL,
published_dates TEXT NOT NULL,
report_content TEXT,
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
)
''')
conn.commit()
self.logger.info("Database setup completed successfully")
except sqlite3.Error as e:
self.logger.error(f"Database setup error: {str(e)}")
raise Exception(f"Failed to setup database: {str(e)}")
async def _analyze_cluster(self, cluster: List[Dict]) -> Optional[str]:
"""
Analyze a cluster of related articles and generate a synthesized report.
Args:
cluster: List of article dictionaries containing content from different sources
Returns:
Optional[str]: A synthesized report or None if analysis fails
"""
try:
# Extract titles and content
titles = [article['title'] for article in cluster]
contents = [article['content'] for article in cluster]
# Combine all content for analysis
combined_content = " ".join(contents)
# Basic report structure
report_sections = []
# Add overview from titles
report_sections.append("## Overview\n")
report_sections.append(f"This story is being reported by {len(cluster)} news sources. ")
report_sections.append(f"The primary headline from {cluster[0]['source']} reads: '{titles[0]}'\n")
# Extract key information
# Find common locations mentioned
location_pattern = r'in ([A-Z][a-zA-Z]+(?: [A-Z][a-zA-Z]+)*)'
locations = set(re.findall(location_pattern, combined_content))
if locations:
report_sections.append("\n## Location\n")
report_sections.append(f"This story takes place in {', '.join(list(locations)[:3])}.\n")
# Find dates and times
date_pattern = r'(?:January|February|March|April|May|June|July|August|September|October|November|December) \d{1,2}(?:st|nd|rd|th)?(?:,? \d{4})?'
dates = set(re.findall(date_pattern, combined_content))
if dates:
report_sections.append("\n## Timeline\n")
report_sections.append(f"Key dates mentioned: {', '.join(list(dates)[:3])}.\n")
# Extract quotes
quote_pattern = r'"([^"]*)"'
quotes = re.findall(quote_pattern, combined_content)
if quotes:
report_sections.append("\n## Key Quotes\n")
for quote in quotes[:2]: # Limit to 2 most relevant quotes
if len(quote) > 20: # Filter out short quotes
report_sections.append(f'- "{quote}"\n')
# Compare perspectives
report_sections.append("\n## Source Comparison\n")
for article in cluster:
report_sections.append(f"- {article['source']}'s coverage focuses on: {article['title']}\n")
# Combine sections into final report
final_report = "\n".join(report_sections)
return final_report
except Exception as e:
self.logger.error(f"Cluster analysis error: {str(e)}")
return None
def _store_article(self, article: Dict) -> bool:
"""Store article in database if it doesn't exist"""
try:
content_hash = hashlib.sha256(
f"{article['title']}{article['url']}".encode()
).hexdigest()
with sqlite3.connect('news_synthesis.db') as conn:
cursor = conn.cursor()
cursor.execute('''
INSERT OR IGNORE INTO articles
(title, source, url, content, published_date, hash)
VALUES (?, ?, ?, ?, ?, ?)
''', (
article['title'],
article['source'],
article['url'],
article.get('content', ''),
article.get('published', 'N/A'),
content_hash
))
return cursor.rowcount > 0
except sqlite3.Error as e:
self.logger.error(f"Article storage error: {str(e)}")
return False
def _store_report(self, report: Dict) -> bool:
"""Store generated report in database"""
try:
with sqlite3.connect('news_synthesis.db') as conn:
cursor = conn.cursor()
cursor.execute('''
INSERT INTO reports
(headline, sources, urls, published_dates, report_content)
VALUES (?, ?, ?, ?, ?)
''', (
report['headline'],
json.dumps(report['sources']),
json.dumps(report['urls']),
json.dumps(report['published_dates']),
report.get('report', '')
))
return cursor.rowcount > 0
except sqlite3.Error as e:
self.logger.error(f"Report storage error: {str(e)}")
return False
async def _parse_feed_with_retry(self, session: aiohttp.ClientSession, feed_url: str, retries=3) -> Optional[Dict]:
"""Parse RSS feed with improved error handling and retry logic"""
for attempt in range(retries):
try:
async with session.get(feed_url, timeout=self.scrape_timeout) as response:
if response.status == 200:
content = await response.text()
feed_data = feedparser.parse(content)
if feed_data.entries:
return feed_data
await asyncio.sleep(2 ** attempt)
except Exception as e:
self.logger.error(f"Feed parsing error (attempt {attempt + 1}): {str(e)}")
if attempt < retries - 1:
await asyncio.sleep(2 ** attempt)
return None
async def _scrape_article(self, url: str) -> Optional[str]:
try:
async with async_playwright() as p:
browser = await p.chromium.launch(headless=True)
context = await browser.new_context(
user_agent=self.user_agent,
viewport={'width': 1920, 'height': 1080}
)
page = await context.new_page()
await page.route("**/*", lambda route: route.continue_() if route.request.resource_type == "document" else route.abort())
try:
await page.goto(url, timeout=self.scrape_timeout * 1000)
content = await page.evaluate('''() => {
const selectors = [
'article',
'[itemprop="articleBody"]',
'.article-body',
'.story-body'
];
for (const selector of selectors) {
const element = document.querySelector(selector);
if (element?.textContent?.length > 500) {
return element.innerText;
}
}
return document.body.innerText;
}''')
return re.sub(r'\s+', ' ', content).strip()[:4000] if content else None
finally:
await browser.close()
except Exception as e:
self.logger.error(f"Scraping error for {url}: {str(e)}")
return None
async def process_feeds(self, progress_bar) -> List[Dict]:
"""Process all feeds with improved concurrency and error handling"""
async with aiohttp.ClientSession(headers={'User-Agent': self.user_agent}) as session:
all_articles = []
for idx, (feed_name, feed_url) in enumerate(self.rss_feeds):
try:
progress_bar.progress((idx + 0.5)/len(self.rss_feeds), f"Processing {feed_name}")
feed_data = await self._parse_feed_with_retry(session, feed_url)
if not feed_data:
continue
scrape_tasks = [
self._scrape_article(entry.link)
for entry in feed_data.entries[:3]
]
contents = await asyncio.gather(*scrape_tasks)
valid_articles = [
{
'title': entry.title,
'source': feed_name,
'content': content,
'url': entry.link,
'published': getattr(entry, 'published', 'N/A')
}
for entry, content in zip(feed_data.entries[:3], contents)
if content and len(content) > 500
]
# Store valid articles in database
for article in valid_articles:
if self._store_article(article):
all_articles.append(article)
st.toast(f"Added article from {feed_name}")
progress_bar.progress((idx + 1)/len(self.rss_feeds))
except Exception as e:
self.logger.error(f"Error processing {feed_name}: {str(e)}")
st.error(f"Failed to process {feed_name}")
return await self._generate_reports(all_articles)
async def _generate_reports(self, articles: List[Dict]) -> List[Dict]:
"""Generate reports from collected articles"""
# Group articles by source
source_map = {source: [] for source, _ in self.rss_feeds}
for article in articles:
source_map[article['source']].append(article)
# Find minimum number of articles across sources
min_articles = min(len(articles) for articles in source_map.values() if articles)
reports = []
for i in range(min_articles):
try:
cluster = []
for source in source_map:
if i < len(source_map[source]):
cluster.append(source_map[source][i])
if len(cluster) >= 2: # Only process if we have at least 2 sources
analysis = await self._analyze_cluster(cluster)
if analysis:
report = {
'headline': cluster[0]['title'],
'sources': [a['source'] for a in cluster],
'urls': [a['url'] for a in cluster],
'published_dates': [a['published'] for a in cluster],
'report': analysis
}
if self._store_report(report):
reports.append(report)
except Exception as e:
self.logger.error(f"Report generation error: {str(e)}")
return reports
def display_report(reports: List[Dict]):
# Sidebar
with st.sidebar:
st.image("https://via.placeholder.com/150x150.png?text=NS", width=150)
colored_header(label="Report Information", description="", color_name="blue-70")
st.write(f"📅 **Generated:** {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}")
st.write(f"🤖 **AI Model:** Phi-4")
st.write("📰 **News Sources:**")
for source in ["BBC News", "Reuters", "AP News", "The New York Times", "Al Jazeera"]:
st.write(f" • {source}")
add_vertical_space(2)
with st.expander("ℹ️ About", expanded=False):
st.write("""
This news synthesizer uses advanced AI to analyze and summarize news from multiple sources.
It identifies related stories, cross-references facts, and generates comprehensive reports.
""")
# Main content
col1, col2, col3 = st.columns([1,3,1])
with col2:
st.title("🌐 News Synthesis Hub")
if not reports:
# Display loading animation
lottie_news = load_lottie_url("https://assets4.lottiefiles.com/packages/lf20_9qshezyl.json")
if lottie_news:
st_lottie(lottie_news, height=400)
st.warning("No reports available. Generate a new report to begin.")
return
# Display reports
# ... [rest of display_report function remains the same]
def load_lottie_url(url: str):
try:
r = requests.get(url)
if r.status_code != 200:
return None
return r.json()
except:
return None
async def main():
st.set_page_config(
page_title="News Synthesis Hub",
page_icon="🌐",
layout="wide",
initial_sidebar_state="expanded"
)
if st.button("🔄 Generate New Report"):
progress_bar = st.progress(0)
try:
with st.spinner("Analyzing news feeds..."):
synthesizer = NewsSynthesizer()
reports = await synthesizer.process_feeds(progress_bar)
display_report(reports)
except Exception as e:
st.error(f"Application error: {str(e)[:200]}")
finally:
progress_bar.empty()
else:
display_report([])
if __name__ == "__main__":
asyncio.run(main()). Application error: min() iterable argument is empty
Looks like you've got a well-structured news synthesizer that pulls RSS feeds, scrapes article content, clusters related stories, and generates synthesized reports. Are you looking for feedback on specific aspects, like performance optimizations, better clustering logic, or additional NLP enhancements?
gpt-4o