#!/usr/bin/env python3 """Wrapper for ak-rss-digest: add state-based deduplication on top of the real fetch script.""" import json import subprocess import sys import os SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) REAL_SCRIPT = os.path.join(SCRIPT_DIR, "fetch_today_feed_items.py") STATE_FILE = os.path.join(os.path.dirname(SCRIPT_DIR), "references", "last_seen.json") # Run the real script result = subprocess.run( [sys.executable, REAL_SCRIPT] + sys.argv[1:], capture_output=True, text=True ) if result.returncode != 0: print(result.stderr, file=sys.stderr) sys.exit(result.returncode) data = json.loads(result.stdout) entries = data.get("entries", []) # Load last seen last_seen = {} if os.path.exists(STATE_FILE): with open(STATE_FILE) as f: last_seen = json.load(f) # Filter: keep only entries not seen before (by URL) new_entries = [] new_seen = {} for e in entries: url = e.get("url", "") source = e.get("source", "") if url: key = f"{source}|{url}" new_seen[key] = True if key not in last_seen: new_entries.append(e) # Group new entries by source new_by_source = {} for e in new_entries: src = e.get("source", "unknown") new_by_source[src] = new_by_source.get(src, 0) + 1 # Output filtered results output = { **data, "entries": new_entries, "has_new": len(new_entries) > 0, "new_by_source": new_by_source, "total_fetched": len(entries), } # Save state os.makedirs(os.path.dirname(STATE_FILE), exist_ok=True) with open(STATE_FILE, "w") as f: json.dump(new_seen, f, ensure_ascii=False, indent=2) print(json.dumps(output, ensure_ascii=False, indent=2))