from __future__ import annotations

import re
from dataclasses import dataclass

import feedparser
import requests


@dataclass
class NewsItem:
    title: str
    summary: str
    link: str
    source: str
    score: int = 0


DEFAULT_FEEDS = [
    # AI & research
    ("OpenAI Blog", "https://openai.com/blog/rss.xml"),
    ("Google AI Blog", "https://blog.google/technology/ai/rss/"),
    ("DeepMind", "https://deepmind.google/blog/rss.xml"),
    ("Hugging Face Blog", "https://huggingface.co/blog/feed.xml"),
    ("TechCrunch AI", "https://techcrunch.com/category/artificial-intelligence/feed/"),
    ("VentureBeat AI", "https://venturebeat.com/category/ai/feed/"),
    ("The Verge AI", "https://www.theverge.com/rss/ai-artificial-intelligence/index.xml"),
    ("MIT Tech Review", "https://www.technologyreview.com/feed/"),
    ("Ars Technica", "https://feeds.arstechnica.com/arstechnica/technology-lab"),
    # Startups & funding
    ("TechCrunch Startups", "https://techcrunch.com/category/startups/feed/"),
    ("TechCrunch", "https://techcrunch.com/feed/"),
    # Cybersecurity
    ("Krebs on Security", "https://krebsonsecurity.com/feed/"),
    ("The Hacker News", "https://feeds.feedburner.com/TheHackersNews"),
    ("BleepingComputer", "https://www.bleepingcomputer.com/feed/"),
    # Developer / open source / cloud
    ("GitHub Blog", "https://github.blog/feed/"),
    ("Dev.to", "https://dev.to/feed"),
    ("InfoQ", "https://feed.infoq.com/"),
    ("AWS News", "https://aws.amazon.com/about-aws/whats-new/recent/feed/"),
    # Hardware / robotics / big tech
    ("The Verge", "https://www.theverge.com/rss/index.xml"),
    ("Wired", "https://www.wired.com/feed/rss"),
    ("NVIDIA Blog", "https://blogs.nvidia.com/feed/"),
]

KEYWORDS = [
    "ai", "artificial intelligence", "llm", "gpt", "claude", "gemini", "openai", "model",
    "agent", "genai", "machine learning", "neural", "research", "paper",
    "startup", "funding", "series", "raised", "valuation", "acquisition", "ipo", "launch",
    "cyber", "security", "breach", "hack", "vulnerability", "cve", "ransomware", "patch",
    "developer", "devops", "api", "sdk", "github", "open source", "framework", "release",
    "cloud", "aws", "azure", "kubernetes", "saas",
    "nvidia", "chip", "gpu", "semiconductor", "hardware",
    "robot", "robotics", "automation",
    "announce", "breakthrough", "product",
]


class NewsService:
    def __init__(self, feeds: list[tuple[str, str]] | None = None, timeout: int = 18):
        self.feeds = feeds or DEFAULT_FEEDS
        self.timeout = timeout

    def fetch(self, max_items: int = 28) -> list[NewsItem]:
        pool: list[NewsItem] = []
        seen: set[str] = set()
        per_feed_cap = 5

        for source, url in self.feeds:
            try:
                resp = requests.get(
                    url,
                    timeout=self.timeout,
                    headers={"User-Agent": "ContentPipeline/2.0 (+premium media cron)"},
                )
                resp.raise_for_status()
                parsed = feedparser.parse(resp.content)
            except Exception:
                continue

            taken = 0
            for entry in parsed.entries:
                title = (getattr(entry, "title", "") or "").strip()
                if not title or title.lower() in seen:
                    continue
                summary = (getattr(entry, "summary", "") or getattr(entry, "description", "") or "")
                summary = _strip_html(summary)[:400]
                link = getattr(entry, "link", "") or ""
                item = NewsItem(title=title, summary=summary, link=link, source=source)
                item.score = _relevance_score(item)
                seen.add(title.lower())
                pool.append(item)
                taken += 1
                if taken >= per_feed_cap:
                    break

        pool.sort(key=lambda x: (-x.score, x.source, x.title.lower()))
        strong = [x for x in pool if x.score > 0]
        chosen = strong if len(strong) >= max_items else strong + [x for x in pool if x.score == 0]
        return chosen[:max_items]

    def as_prompt_block(self, items: list[NewsItem]) -> str:
        if not items:
            return "(No news fetched — use only verifiable recent AI/tech/startup/cyber news.)"
        lines = []
        for i, it in enumerate(items, 1):
            lines.append(
                f"{i}. [{it.source}] score={it.score}\n"
                f"   TITLE: {it.title}\n"
                f"   SUMMARY: {it.summary}\n"
                f"   URL: {it.link}"
            )
        return "\n".join(lines)


def _relevance_score(item: NewsItem) -> int:
    text = f"{item.title} {item.summary}".lower()
    score = 0
    hot = {
        "ai", "openai", "claude", "gpt", "funding", "breach", "launch", "release",
        "startup", "nvidia", "robot", "cve", "model", "agent",
    }
    for kw in KEYWORDS:
        if kw in text:
            score += 4 if kw in hot else 2
    if any(x in item.source.lower() for x in ("crunch", "krebs", "hack", "github", "openai", "nvidia")):
        score += 3
    return score


def _strip_html(html: str) -> str:
    text = re.sub(r"<[^>]+>", " ", html)
    text = re.sub(r"\s+", " ", text)
    return text.strip()
