154 lines
5.1 KiB
Python
Executable File
154 lines
5.1 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""RSS fetcher with dedup. Subprocess isolation per feed for true timeout."""
|
|
|
|
import argparse
|
|
import json
|
|
import os
|
|
import re
|
|
import subprocess
|
|
import sys
|
|
import time
|
|
import xml.etree.ElementTree as ET
|
|
from datetime import datetime, timedelta, timezone
|
|
from pathlib import Path
|
|
|
|
SCRIPT_DIR = Path(__file__).resolve().parent
|
|
SKILL_DIR = SCRIPT_DIR.parent
|
|
DEFAULT_OPML = SKILL_DIR / "references" / "feeds.opml"
|
|
DEFAULT_STATE = SKILL_DIR / "references" / "last_seen.json"
|
|
|
|
FETCHER_CODE = r"""
|
|
import sys, json, time
|
|
import requests, feedparser
|
|
from datetime import datetime, timezone
|
|
try:
|
|
resp = requests.get(sys.argv[1], timeout=(5, 15),
|
|
headers={'User-Agent': 'ak-rss-digest/1.0',
|
|
'Accept': 'application/rss+xml, application/atom+xml, text/xml'})
|
|
resp.raise_for_status()
|
|
parsed = feedparser.parse(resp.content)
|
|
entries = []
|
|
for e in parsed.entries:
|
|
link = e.get('link', '')
|
|
if not link: continue
|
|
pub = None
|
|
for df in ['published_parsed', 'updated_parsed']:
|
|
dt = e.get(df)
|
|
if dt: pub = datetime(*dt[:6], tzinfo=timezone.utc); break
|
|
import re
|
|
summary = re.sub(r'<[^>]+>', '', e.get('summary','') or e.get('description','') or '').strip()[:300]
|
|
entries.append({
|
|
'title': e.get('title','(untitled)'), 'url': link,
|
|
'published': pub.isoformat() if pub else None, 'summary': summary,
|
|
})
|
|
print(json.dumps({'status': 'ok', 'feed_title': parsed.feed.get('title',''), 'entries': entries}))
|
|
except Exception as ex:
|
|
print(json.dumps({'status': 'error', 'error': str(ex)}))
|
|
"""
|
|
|
|
|
|
def load_opml(path):
|
|
tree = ET.parse(path)
|
|
feeds = []
|
|
for o in tree.iter("outline"):
|
|
if "xmlUrl" in o.attrib:
|
|
feeds.append({
|
|
"name": o.attrib.get("text") or o.attrib.get("title") or "",
|
|
"url": o.attrib["xmlUrl"],
|
|
"site": o.attrib.get("htmlUrl", ""),
|
|
})
|
|
return feeds
|
|
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser()
|
|
parser.add_argument("--days", type=int, default=7)
|
|
parser.add_argument("--timeout", type=int, default=20)
|
|
parser.add_argument("--feeds", default=str(DEFAULT_OPML))
|
|
parser.add_argument("--state", default=str(DEFAULT_STATE))
|
|
parser.add_argument("--no-dedup", action="store_true")
|
|
args = parser.parse_args()
|
|
|
|
if not os.path.exists(args.feeds):
|
|
print(json.dumps({"error": "OPML not found"}))
|
|
sys.exit(1)
|
|
|
|
feeds = load_opml(args.feeds)
|
|
cutoff = datetime.now(timezone.utc) - timedelta(days=args.days)
|
|
|
|
# Load last_seen
|
|
last_seen = {}
|
|
if not args.no_dedup and os.path.exists(args.state):
|
|
with open(args.state) as f:
|
|
last_seen = json.load(f)
|
|
|
|
all_entries = []
|
|
errors = []
|
|
new_seen = {}
|
|
ok = err = 0
|
|
|
|
for feed in feeds:
|
|
try:
|
|
result = subprocess.run(
|
|
[sys.executable, "-c", FETCHER_CODE, feed["url"]],
|
|
capture_output=True, text=True, timeout=args.timeout
|
|
)
|
|
data = json.loads(result.stdout)
|
|
if data["status"] != "ok":
|
|
err += 1
|
|
errors.append({"feed": feed["name"], "error": data.get("error", "unknown")})
|
|
continue
|
|
|
|
ok += 1
|
|
feed_title = data.get("feed_title", feed["name"])
|
|
for e in data["entries"]:
|
|
key = f'{feed_title}|{e["url"]}'
|
|
new_seen[key] = True
|
|
if not args.no_dedup and key in last_seen:
|
|
continue
|
|
pub = e.get("published")
|
|
if pub:
|
|
pub_dt = datetime.fromisoformat(pub)
|
|
if pub_dt < cutoff:
|
|
continue
|
|
all_entries.append({
|
|
"source": feed_title,
|
|
"title": e["title"],
|
|
"url": e["url"],
|
|
"published": pub,
|
|
"summary": e["summary"],
|
|
})
|
|
except subprocess.TimeoutExpired:
|
|
err += 1
|
|
errors.append({"feed": feed["name"], "error": "timeout"})
|
|
except Exception as ex:
|
|
err += 1
|
|
errors.append({"feed": feed["name"], "error": str(ex)})
|
|
|
|
# Save state
|
|
if not args.no_dedup:
|
|
os.makedirs(os.path.dirname(args.state) or ".", exist_ok=True)
|
|
with open(args.state, "w") as f:
|
|
json.dump(new_seen, f, ensure_ascii=False, indent=2)
|
|
|
|
all_entries.sort(key=lambda e: e.get("published") or "", reverse=True)
|
|
new_by_source = {}
|
|
for e in all_entries:
|
|
s = e["source"]
|
|
new_by_source[s] = new_by_source.get(s, 0) + 1
|
|
|
|
output = {
|
|
"total_feeds": len(feeds),
|
|
"ok": ok, "errors": err,
|
|
"total_entries": len(all_entries),
|
|
"has_new": len(all_entries) > 0,
|
|
"new_by_source": new_by_source,
|
|
"error_details": errors[:5] if errors else None,
|
|
"entries": all_entries[:500],
|
|
}
|
|
print(json.dumps(output, ensure_ascii=False, indent=2))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|