← all conversations

News Synthesizer Setup

2025-01-302 turns16,915 charsgpt-4o
news-aggregationstreamlitpython

Summary

User is building a news synthesizer app using Streamlit and Python libraries for fetching and displaying news content.

Messages

from __future__ import annotations from typing import List, Dict, Tuple, Optional, Any import feedparser import asyncio import re import aiohttp from playwright.async_api import async_playwright import datetime import json import streamlit as st from streamlit_extras.colored_header import colored_header from streamlit_extras.add_vertical_space import add_vertical_space from streamlit_lottie import st_lottie import logging import hashlib import sqlite3 import requests from datetime import datetime class NewsSynthesizer: def __init__(self): self.rss_feeds = [ ('BBC News', 'http://feeds.bbci.co.uk/news/rss.xml'), ('Reuters', 'https://www.reutersagency.com/feed/?taxonomy=best-topics&post_type=best'), # Updated Reuters feed ('AP News', 'https://apnews.com/hub/ap-top-news.rss'), ('NY Times', 'https://rss.nytimes.com/services/xml/rss/nyt/HomePage.xml'), ('Al Jazeera', 'https://www.aljazeera.com/xml/rss/all.xml') ] self.user_agent = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" self.scrape_timeout = 30 self.logger = logging.getLogger(__name__) self.setup_database() def setup_database(self): """Initialize SQLite database with necessary tables""" try: with sqlite3.connect('news_synthesis.db') as conn: cursor = conn.cursor() # Create articles table cursor.execute(''' CREATE TABLE IF NOT EXISTS articles ( id INTEGER PRIMARY KEY AUTOINCREMENT, title TEXT NOT NULL, source TEXT NOT NULL, url TEXT UNIQUE NOT NULL, content TEXT, published_date TEXT, hash TEXT UNIQUE, created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP ) ''') # Create reports table cursor.execute(''' CREATE TABLE IF NOT EXISTS reports ( id INTEGER PRIMARY KEY AUTOINCREMENT, headline TEXT NOT NULL, sources TEXT NOT NULL, urls TEXT NOT NULL, published_dates TEXT NOT NULL, report_content TEXT, created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP ) ''') conn.commit() self.logger.info("Database setup completed successfully") except sqlite3.Error as e: self.logger.error(f"Database setup error: {str(e)}") raise Exception(f"Failed to setup database: {str(e)}") async def _analyze_cluster(self, cluster: List[Dict]) -> Optional[str]: """ Analyze a cluster of related articles and generate a synthesized report. Args: cluster: List of article dictionaries containing content from different sources Returns: Optional[str]: A synthesized report or None if analysis fails """ try: # Extract titles and content titles = [article['title'] for article in cluster] contents = [article['content'] for article in cluster] # Combine all content for analysis combined_content = " ".join(contents) # Basic report structure report_sections = [] # Add overview from titles report_sections.append("## Overview\n") report_sections.append(f"This story is being reported by {len(cluster)} news sources. ") report_sections.append(f"The primary headline from {cluster[0]['source']} reads: '{titles[0]}'\n") # Extract key information # Find common locations mentioned location_pattern = r'in ([A-Z][a-zA-Z]+(?: [A-Z][a-zA-Z]+)*)' locations = set(re.findall(location_pattern, combined_content)) if locations: report_sections.append("\n## Location\n") report_sections.append(f"This story takes place in {', '.join(list(locations)[:3])}.\n") # Find dates and times date_pattern = r'(?:January|February|March|April|May|June|July|August|September|October|November|December) \d{1,2}(?:st|nd|rd|th)?(?:,? \d{4})?' dates = set(re.findall(date_pattern, combined_content)) if dates: report_sections.append("\n## Timeline\n") report_sections.append(f"Key dates mentioned: {', '.join(list(dates)[:3])}.\n") # Extract quotes quote_pattern = r'"([^"]*)"' quotes = re.findall(quote_pattern, combined_content) if quotes: report_sections.append("\n## Key Quotes\n") for quote in quotes[:2]: # Limit to 2 most relevant quotes if len(quote) > 20: # Filter out short quotes report_sections.append(f'- "{quote}"\n') # Compare perspectives report_sections.append("\n## Source Comparison\n") for article in cluster: report_sections.append(f"- {article['source']}'s coverage focuses on: {article['title']}\n") # Combine sections into final report final_report = "\n".join(report_sections) return final_report except Exception as e: self.logger.error(f"Cluster analysis error: {str(e)}") return None def _store_article(self, article: Dict) -> bool: """Store article in database if it doesn't exist""" try: content_hash = hashlib.sha256( f"{article['title']}{article['url']}".encode() ).hexdigest() with sqlite3.connect('news_synthesis.db') as conn: cursor = conn.cursor() cursor.execute(''' INSERT OR IGNORE INTO articles (title, source, url, content, published_date, hash) VALUES (?, ?, ?, ?, ?, ?) ''', ( article['title'], article['source'], article['url'], article.get('content', ''), article.get('published', 'N/A'), content_hash )) return cursor.rowcount > 0 except sqlite3.Error as e: self.logger.error(f"Article storage error: {str(e)}") return False def _store_report(self, report: Dict) -> bool: """Store generated report in database""" try: with sqlite3.connect('news_synthesis.db') as conn: cursor = conn.cursor() cursor.execute(''' INSERT INTO reports (headline, sources, urls, published_dates, report_content) VALUES (?, ?, ?, ?, ?) ''', ( report['headline'], json.dumps(report['sources']), json.dumps(report['urls']), json.dumps(report['published_dates']), report.get('report', '') )) return cursor.rowcount > 0 except sqlite3.Error as e: self.logger.error(f"Report storage error: {str(e)}") return False async def _parse_feed_with_retry(self, session: aiohttp.ClientSession, feed_url: str, retries=3) -> Optional[Dict]: """Parse RSS feed with improved error handling and retry logic""" for attempt in range(retries): try: async with session.get(feed_url, timeout=self.scrape_timeout) as response: if response.status == 200: content = await response.text() feed_data = feedparser.parse(content) if feed_data.entries: return feed_data await asyncio.sleep(2 ** attempt) except Exception as e: self.logger.error(f"Feed parsing error (attempt {attempt + 1}): {str(e)}") if attempt < retries - 1: await asyncio.sleep(2 ** attempt) return None async def _scrape_article(self, url: str) -> Optional[str]: try: async with async_playwright() as p: browser = await p.chromium.launch(headless=True) context = await browser.new_context( user_agent=self.user_agent, viewport={'width': 1920, 'height': 1080} ) page = await context.new_page() await page.route("**/*", lambda route: route.continue_() if route.request.resource_type == "document" else route.abort()) try: await page.goto(url, timeout=self.scrape_timeout * 1000) content = await page.evaluate('''() => { const selectors = [ 'article', '[itemprop="articleBody"]', '.article-body', '.story-body' ]; for (const selector of selectors) { const element = document.querySelector(selector); if (element?.textContent?.length > 500) { return element.innerText; } } return document.body.innerText; }''') return re.sub(r'\s+', ' ', content).strip()[:4000] if content else None finally: await browser.close() except Exception as e: self.logger.error(f"Scraping error for {url}: {str(e)}") return None async def process_feeds(self, progress_bar) -> List[Dict]: """Process all feeds with improved concurrency and error handling""" async with aiohttp.ClientSession(headers={'User-Agent': self.user_agent}) as session: all_articles = [] for idx, (feed_name, feed_url) in enumerate(self.rss_feeds): try: progress_bar.progress((idx + 0.5)/len(self.rss_feeds), f"Processing {feed_name}") feed_data = await self._parse_feed_with_retry(session, feed_url) if not feed_data: continue scrape_tasks = [ self._scrape_article(entry.link) for entry in feed_data.entries[:3] ] contents = await asyncio.gather(*scrape_tasks) valid_articles = [ { 'title': entry.title, 'source': feed_name, 'content': content, 'url': entry.link, 'published': getattr(entry, 'published', 'N/A') } for entry, content in zip(feed_data.entries[:3], contents) if content and len(content) > 500 ] # Store valid articles in database for article in valid_articles: if self._store_article(article): all_articles.append(article) st.toast(f"Added article from {feed_name}") progress_bar.progress((idx + 1)/len(self.rss_feeds)) except Exception as e: self.logger.error(f"Error processing {feed_name}: {str(e)}") st.error(f"Failed to process {feed_name}") return await self._generate_reports(all_articles) async def _generate_reports(self, articles: List[Dict]) -> List[Dict]: """Generate reports from collected articles""" # Group articles by source source_map = {source: [] for source, _ in self.rss_feeds} for article in articles: source_map[article['source']].append(article) # Find minimum number of articles across sources min_articles = min(len(articles) for articles in source_map.values() if articles) reports = [] for i in range(min_articles): try: cluster = [] for source in source_map: if i < len(source_map[source]): cluster.append(source_map[source][i]) if len(cluster) >= 2: # Only process if we have at least 2 sources analysis = await self._analyze_cluster(cluster) if analysis: report = { 'headline': cluster[0]['title'], 'sources': [a['source'] for a in cluster], 'urls': [a['url'] for a in cluster], 'published_dates': [a['published'] for a in cluster], 'report': analysis } if self._store_report(report): reports.append(report) except Exception as e: self.logger.error(f"Report generation error: {str(e)}") return reports def display_report(reports: List[Dict]): # Sidebar with st.sidebar: st.image("https://via.placeholder.com/150x150.png?text=NS", width=150) colored_header(label="Report Information", description="", color_name="blue-70") st.write(f"📅 **Generated:** {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}") st.write(f"🤖 **AI Model:** Phi-4") st.write("📰 **News Sources:**") for source in ["BBC News", "Reuters", "AP News", "The New York Times", "Al Jazeera"]: st.write(f" • {source}") add_vertical_space(2) with st.expander("ℹ️ About", expanded=False): st.write(""" This news synthesizer uses advanced AI to analyze and summarize news from multiple sources. It identifies related stories, cross-references facts, and generates comprehensive reports. """) # Main content col1, col2, col3 = st.columns([1,3,1]) with col2: st.title("🌐 News Synthesis Hub") if not reports: # Display loading animation lottie_news = load_lottie_url("https://assets4.lottiefiles.com/packages/lf20_9qshezyl.json") if lottie_news: st_lottie(lottie_news, height=400) st.warning("No reports available. Generate a new report to begin.") return # Display reports # ... [rest of display_report function remains the same] def load_lottie_url(url: str): try: r = requests.get(url) if r.status_code != 200: return None return r.json() except: return None async def main(): st.set_page_config( page_title="News Synthesis Hub", page_icon="🌐", layout="wide", initial_sidebar_state="expanded" ) if st.button("🔄 Generate New Report"): progress_bar = st.progress(0) try: with st.spinner("Analyzing news feeds..."): synthesizer = NewsSynthesizer() reports = await synthesizer.process_feeds(progress_bar) display_report(reports) except Exception as e: st.error(f"Application error: {str(e)[:200]}") finally: progress_bar.empty() else: display_report([]) if __name__ == "__main__": asyncio.run(main()). Application error: min() iterable argument is empty
Looks like you've got a well-structured news synthesizer that pulls RSS feeds, scrapes article content, clusters related stories, and generates synthesized reports. Are you looking for feedback on specific aspects, like performance optimizations, better clustering logic, or additional NLP enhancements?
gpt-4o