[teamai] Push 1 resource(s) from root
This commit is contained in:
Executable
+153
@@ -0,0 +1,153 @@
|
||||
#!/usr/bin/env python3
|
||||
"""RSS fetcher with dedup. Subprocess isolation per feed for true timeout."""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
import xml.etree.ElementTree as ET
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
|
||||
SCRIPT_DIR = Path(__file__).resolve().parent
|
||||
SKILL_DIR = SCRIPT_DIR.parent
|
||||
DEFAULT_OPML = SKILL_DIR / "references" / "feeds.opml"
|
||||
DEFAULT_STATE = SKILL_DIR / "references" / "last_seen.json"
|
||||
|
||||
FETCHER_CODE = r"""
|
||||
import sys, json, time
|
||||
import requests, feedparser
|
||||
from datetime import datetime, timezone
|
||||
try:
|
||||
resp = requests.get(sys.argv[1], timeout=(5, 15),
|
||||
headers={'User-Agent': 'ak-rss-digest/1.0',
|
||||
'Accept': 'application/rss+xml, application/atom+xml, text/xml'})
|
||||
resp.raise_for_status()
|
||||
parsed = feedparser.parse(resp.content)
|
||||
entries = []
|
||||
for e in parsed.entries:
|
||||
link = e.get('link', '')
|
||||
if not link: continue
|
||||
pub = None
|
||||
for df in ['published_parsed', 'updated_parsed']:
|
||||
dt = e.get(df)
|
||||
if dt: pub = datetime(*dt[:6], tzinfo=timezone.utc); break
|
||||
import re
|
||||
summary = re.sub(r'<[^>]+>', '', e.get('summary','') or e.get('description','') or '').strip()[:300]
|
||||
entries.append({
|
||||
'title': e.get('title','(untitled)'), 'url': link,
|
||||
'published': pub.isoformat() if pub else None, 'summary': summary,
|
||||
})
|
||||
print(json.dumps({'status': 'ok', 'feed_title': parsed.feed.get('title',''), 'entries': entries}))
|
||||
except Exception as ex:
|
||||
print(json.dumps({'status': 'error', 'error': str(ex)}))
|
||||
"""
|
||||
|
||||
|
||||
def load_opml(path):
|
||||
tree = ET.parse(path)
|
||||
feeds = []
|
||||
for o in tree.iter("outline"):
|
||||
if "xmlUrl" in o.attrib:
|
||||
feeds.append({
|
||||
"name": o.attrib.get("text") or o.attrib.get("title") or "",
|
||||
"url": o.attrib["xmlUrl"],
|
||||
"site": o.attrib.get("htmlUrl", ""),
|
||||
})
|
||||
return feeds
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--days", type=int, default=7)
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--feeds", default=str(DEFAULT_OPML))
|
||||
parser.add_argument("--state", default=str(DEFAULT_STATE))
|
||||
parser.add_argument("--no-dedup", action="store_true")
|
||||
args = parser.parse_args()
|
||||
|
||||
if not os.path.exists(args.feeds):
|
||||
print(json.dumps({"error": "OPML not found"}))
|
||||
sys.exit(1)
|
||||
|
||||
feeds = load_opml(args.feeds)
|
||||
cutoff = datetime.now(timezone.utc) - timedelta(days=args.days)
|
||||
|
||||
# Load last_seen
|
||||
last_seen = {}
|
||||
if not args.no_dedup and os.path.exists(args.state):
|
||||
with open(args.state) as f:
|
||||
last_seen = json.load(f)
|
||||
|
||||
all_entries = []
|
||||
errors = []
|
||||
new_seen = {}
|
||||
ok = err = 0
|
||||
|
||||
for feed in feeds:
|
||||
try:
|
||||
result = subprocess.run(
|
||||
[sys.executable, "-c", FETCHER_CODE, feed["url"]],
|
||||
capture_output=True, text=True, timeout=args.timeout
|
||||
)
|
||||
data = json.loads(result.stdout)
|
||||
if data["status"] != "ok":
|
||||
err += 1
|
||||
errors.append({"feed": feed["name"], "error": data.get("error", "unknown")})
|
||||
continue
|
||||
|
||||
ok += 1
|
||||
feed_title = data.get("feed_title", feed["name"])
|
||||
for e in data["entries"]:
|
||||
key = f'{feed_title}|{e["url"]}'
|
||||
new_seen[key] = True
|
||||
if not args.no_dedup and key in last_seen:
|
||||
continue
|
||||
pub = e.get("published")
|
||||
if pub:
|
||||
pub_dt = datetime.fromisoformat(pub)
|
||||
if pub_dt < cutoff:
|
||||
continue
|
||||
all_entries.append({
|
||||
"source": feed_title,
|
||||
"title": e["title"],
|
||||
"url": e["url"],
|
||||
"published": pub,
|
||||
"summary": e["summary"],
|
||||
})
|
||||
except subprocess.TimeoutExpired:
|
||||
err += 1
|
||||
errors.append({"feed": feed["name"], "error": "timeout"})
|
||||
except Exception as ex:
|
||||
err += 1
|
||||
errors.append({"feed": feed["name"], "error": str(ex)})
|
||||
|
||||
# Save state
|
||||
if not args.no_dedup:
|
||||
os.makedirs(os.path.dirname(args.state) or ".", exist_ok=True)
|
||||
with open(args.state, "w") as f:
|
||||
json.dump(new_seen, f, ensure_ascii=False, indent=2)
|
||||
|
||||
all_entries.sort(key=lambda e: e.get("published") or "", reverse=True)
|
||||
new_by_source = {}
|
||||
for e in all_entries:
|
||||
s = e["source"]
|
||||
new_by_source[s] = new_by_source.get(s, 0) + 1
|
||||
|
||||
output = {
|
||||
"total_feeds": len(feeds),
|
||||
"ok": ok, "errors": err,
|
||||
"total_entries": len(all_entries),
|
||||
"has_new": len(all_entries) > 0,
|
||||
"new_by_source": new_by_source,
|
||||
"error_details": errors[:5] if errors else None,
|
||||
"entries": all_entries[:500],
|
||||
}
|
||||
print(json.dumps(output, ensure_ascii=False, indent=2))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user