#!/usr/bin/env python3
"""What changed in the last N days: papers, library releases, Apple developer news.

  fresh-sources science "speculative decoding" "draft model" [--days 30] [--cats cs.LG,cs.CL]
  fresh-sources releases ml-explore/mlx ggml-org/llama.cpp NVIDIA/cutlass [--days 30]
  fresh-sources commits ggml-org/llama.cpp:ggml/src/ggml-metal ml-explore/mlx:mlx/backend/metal
  fresh-sources hf "quantization" "kernel" [--days 30]   # Hugging Face daily papers, with upvotes
  fresh-sources apple [--days 30]

arXiv filters on first-version submission date and publishes daily, so the newest
day can lag. Hugging Face daily papers is a curated subset that catches cs.AI-only
listings a category query can miss; run both for a real scan.

Stdlib only. Prints one line per item, newest first: date, id/tag, title, URL.
Set GITHUB_TOKEN to lift the GitHub rate limit. Exits nonzero when a source fails,
so an empty window and a broken fetch never look the same.
"""
import argparse
import datetime as dt
import email.utils
import json
import os
import re
import sys
import time
import urllib.parse
import urllib.request
import xml.etree.ElementTree as ET

UA = {"User-Agent": "fresh-sources/1.0 (skill: native-inference-perf)"}
ATOM = "{http://www.w3.org/2005/Atom}"
ARXIV_CATS = "cs.LG,cs.DC,cs.AR,cs.PF,cs.CL,cs.CV,cs.AI"


def fetch(url, headers=None):
    req = urllib.request.Request(url, headers={**UA, **(headers or {})})
    with urllib.request.urlopen(req, timeout=60) as resp:
        return resp.read()


def window(days):
    end = dt.datetime.now(dt.timezone.utc)
    return end - dt.timedelta(days=days), end


def science(terms, days, cats, limit):
    start, end = window(days)
    phrase = " OR ".join(
        f'all:"{t}"' if " " in t else f"all:{t}" for t in terms)
    catq = " OR ".join(f"cat:{c}" for c in cats.split(","))
    span = f"submittedDate:[{start:%Y%m%d%H%M} TO {end:%Y%m%d%H%M}]"
    query = f"({phrase}) AND ({catq}) AND {span}"
    url = "https://export.arxiv.org/api/query?" + urllib.parse.urlencode({
        "search_query": query, "sortBy": "submittedDate",
        "sortOrder": "descending", "max_results": limit})
    try:
        root = ET.fromstring(fetch(url))
    except Exception as exc:
        return [], [f"arXiv: {exc}"]
    rows = []
    for e in root.findall(f"{ATOM}entry"):
        arxiv_id = e.findtext(f"{ATOM}id", "").rsplit("/abs/", 1)[-1]
        title = " ".join(e.findtext(f"{ATOM}title", "").split())
        published = e.findtext(f"{ATOM}published", "")[:10]
        prim = e.find("{http://arxiv.org/schemas/atom}primary_category")
        cat = prim.get("term") if prim is not None else ""
        rows.append((published, arxiv_id, f"[{cat}] {title}",
                     f"https://arxiv.org/abs/{arxiv_id}"))
    return rows, []


def hf(terms, days):
    start, end = window(days)
    pattern = re.compile("|".join(rf"\b{re.escape(t)}" for t in terms), re.I) if terms else None
    weeks, day = [], start
    while day <= end + dt.timedelta(days=7):
        iso = day.isocalendar()
        tag = f"{iso[0]}-W{iso[1]:02d}"
        if tag not in weeks:
            weeks.append(tag)
        day += dt.timedelta(days=1)
    seen, rows, failed = set(), [], []
    for week in weeks:
        for page in range(20):
            try:
                data = json.loads(fetch("https://huggingface.co/api/daily_papers?"
                                        f"week={week}&limit=100&p={page}"))
            except Exception as exc:
                failed.append(f"hf {week} p{page}: {exc}")
                break
            for item in data:
                paper = item.get("paper", {})
                pid, stamp = paper.get("id"), paper.get("publishedAt")
                if not pid or not stamp or pid in seen:
                    continue
                when = dt.datetime.fromisoformat(stamp.replace("Z", "+00:00"))
                text = f"{paper.get('title', '')} {paper.get('summary', '')}"
                if start <= when <= end and (pattern is None or pattern.search(text)):
                    seen.add(pid)
                    rows.append((when.date().isoformat(), pid,
                                 f"[{paper.get('upvotes', 0)} up] "
                                 + " ".join(paper.get("title", "").split()),
                                 f"https://arxiv.org/abs/{pid}"))
            if len(data) < 100:
                break
    return rows, failed


def releases(repos, days, per_repo):
    start, _ = window(days)
    headers = {"Accept": "application/vnd.github+json"}
    if os.environ.get("GITHUB_TOKEN"):
        headers["Authorization"] = f"Bearer {os.environ['GITHUB_TOKEN']}"
    rows, failed = [], []
    for repo in repos:
        try:
            data = json.loads(fetch(
                f"https://api.github.com/repos/{repo}/releases?per_page=30", headers))
        except Exception as exc:  # report and keep going with the other repos
            failed.append(f"{repo}: {exc}")
            continue
        kept = 0
        for rel in data:
            if kept >= per_repo:
                break
            stamp = rel.get("published_at") or rel.get("created_at")
            if not stamp:
                continue
            when = dt.datetime.fromisoformat(stamp.replace("Z", "+00:00"))
            if when >= start:
                pre = " (pre)" if rel.get("prerelease") else ""
                rows.append((when.date().isoformat(), f"{repo}@{rel['tag_name']}{pre}",
                             rel.get("name") or "", rel["html_url"]))
                kept += 1
    return rows, failed


def commits(specs, days, per_repo):
    start, _ = window(days)
    headers = {"Accept": "application/vnd.github+json"}
    if os.environ.get("GITHUB_TOKEN"):
        headers["Authorization"] = f"Bearer {os.environ['GITHUB_TOKEN']}"
    rows, failed = [], []
    for spec in specs:
        repo, _, path = spec.partition(":")
        query = {"since": start.strftime("%Y-%m-%dT%H:%M:%SZ"), "per_page": min(per_repo, 100)}
        if path:
            query["path"] = path
        try:
            data = json.loads(fetch(f"https://api.github.com/repos/{repo}/commits?"
                                    + urllib.parse.urlencode(query), headers))
        except Exception as exc:
            failed.append(f"{spec}: {exc}")
            continue
        for c in data:
            info = c.get("commit", {})
            when = (info.get("committer") or info.get("author") or {}).get("date", "")[:10]
            subject = (info.get("message") or "").splitlines()[0] if info.get("message") else ""
            rows.append((when, f"{repo}@{c.get('sha', '')[:10]}", subject, c.get("html_url", "")))
    return rows, failed


def apple(days):
    start, _ = window(days)
    rows, failed = [], []
    for feed in ("https://developer.apple.com/news/releases/rss/releases.rss",
                 "https://developer.apple.com/news/rss/news.rss"):
        try:
            root = ET.fromstring(fetch(feed))
        except Exception as exc:
            failed.append(f"{feed}: {exc}")
            continue
        for item in root.iter("item"):
            raw = item.findtext("pubDate", "")
            try:
                when = email.utils.parsedate_to_datetime(raw)
            except (TypeError, ValueError):
                continue
            if when.tzinfo is None:
                when = when.replace(tzinfo=dt.timezone.utc)
            if when >= start:
                rows.append((when.date().isoformat(), "apple",
                             " ".join(item.findtext("title", "").split()),
                             item.findtext("link", "")))
    return rows, failed


def main():
    ap = argparse.ArgumentParser(description=__doc__,
                                 formatter_class=argparse.RawDescriptionHelpFormatter)
    ap.add_argument("mode", choices=["science", "hf", "releases", "commits", "apple"])
    ap.add_argument("args", nargs="*", help="search terms or owner/repo names")
    ap.add_argument("--days", type=int, default=30)
    ap.add_argument("--cats", default=ARXIV_CATS, help="arXiv categories, comma separated")
    ap.add_argument("--limit", type=int, default=60, help="max arXiv results")
    ap.add_argument("--per-repo", type=int, default=5,
                    help="max releases or commits per repo")
    ap.add_argument("--json", action="store_true")
    a = ap.parse_args()
    failed = []
    if a.mode == "science":
        if not a.args:
            ap.error("science needs at least one search term")
        rows, failed = science(a.args, a.days, a.cats, a.limit)
    elif a.mode == "hf":
        rows, failed = hf(a.args, a.days)
    elif a.mode == "releases":
        if not a.args:
            ap.error("releases needs owner/repo names")
        rows, failed = releases(a.args, a.days, a.per_repo)
    elif a.mode == "commits":
        if not a.args:
            ap.error("commits needs owner/repo[:path] specs")
        rows, failed = commits(a.args, a.days, a.per_repo)
    else:
        rows, failed = apple(a.days)
    rows.sort(key=lambda r: r[0], reverse=True)
    if a.json:
        print(json.dumps([dict(zip(("date", "id", "title", "url"), r)) for r in rows], indent=1))
    else:
        for r in rows:
            print("  ".join(r))
        print(f"# {len(rows)} items, last {a.days} days, "
              f"{time.strftime('%Y-%m-%d')}", file=sys.stderr)
    for f in failed:
        print(f"FAILED {f}", file=sys.stderr)
    return 1 if failed else 0


if __name__ == "__main__":
    sys.exit(main())
