From cb0bbf3e58558584ac2a624d90c9e4be1079f582 Mon Sep 17 00:00:00 2001 From: NightStar Date: Thu, 3 Sep 2026 16:09:54 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E9=80=9A=E7=94=A8=E9=87=87=E9=9B=86+?= =?UTF-8?q?=E6=A8=A1=E6=9D=BF=E6=B3=A8=E5=85=A5=E5=BC=95=E6=93=8E=EF=BC=9A?= =?UTF-8?q?=E9=85=8D=E7=BD=AE=E9=A9=B1=E5=8A=A8=E5=A4=9A=E6=BA=90=E9=87=87?= =?UTF-8?q?=E9=9B=86=EF=BC=8CJinja2=20=E6=A8=A1=E6=9D=BF=E6=B3=A8=E5=85=A5?= =?UTF-8?q?=E6=88=90=E6=96=87=E6=A1=A3=EF=BC=8C=E4=B8=8D=E7=BB=91=E5=AE=9A?= =?UTF-8?q?=E5=8D=95=E6=9C=BA?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .env.example | 13 ++ SKILL.md | 142 ++++++++++++++++++++++ config/morning-digest.yaml | 74 ++++++++++++ scripts/build.py | 237 +++++++++++++++++++++++++++++++++++++ templates/digest.md | 36 ++++++ 5 files changed, 502 insertions(+) create mode 100644 .env.example create mode 100644 SKILL.md create mode 100644 config/morning-digest.yaml create mode 100644 scripts/build.py create mode 100644 templates/digest.md diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..1060b2c --- /dev/null +++ b/.env.example @@ -0,0 +1,13 @@ +# 复制为 .env 并按需填写。机器相关项不写死进仓库,走环境变量。 +# 运行:set -a; . ./.env; set +a; python3 scripts/build.py --config config/morning-digest.yaml + +# 社会热榜服务地址(DailyHotApi 基址,改成你的实际地址) +HOT_API=http://:6688 +# GitHub Trending 采集脚本 +GITHUB_TRENDING_SCRIPT=~/.openclaw/workspace/scripts/fetch_github_trending.py +# RSS 采集脚本 +RSS_SCRIPT=~/.agents/skills/ak-rss-digest/scripts/fetch_today_feed_items.py +# 科技动态注入 JSON 文件(可选,留空则跳过) +TECH_FILE= +# 飞书知识库写入脚本(可选,push 用) +FEISHU_CREATE_SCRIPT=~/.openclaw/workspace/skills/feishu-wiki-tools/scripts/create_digest.py diff --git a/SKILL.md b/SKILL.md new file mode 100644 index 0000000..97db4d1 --- /dev/null +++ b/SKILL.md @@ -0,0 +1,142 @@ +--- +name: "morning-digest" +description: "通用「数据采集 + 模板注入」引擎:用一份 YAML 配置声明多路数据源(http_json/command/static),采集后用 Jinja2 模板注入成 markdown 文档。适合日报/周报/聚合摘要,不绑定单台机器。" +metadata: + author: xf + version: "2.0.0" +--- + +# morning-digest — 通用采集 + 模板注入引擎 + +把「数据采集 → 模板注入 → 输出文档」抽象成一套**配置驱动**的通用 skill,不绑定任何具体机器/域名/IP。早报只是其一例用法,换一份配置即可复用到日报、周报、榜单、聚合摘要。 + +核心思路:**采集用脚本,排版用模板注入**,全程确定性,不依赖 LLM 排版。 + +## 流水线 + +``` +build.py --config config/digest.yaml + ┌─────────────────────────────────────────────┐ + │ 1. 读配置(YAML/JSON) │ + │ 2. ${ENV} 占位符替换(机器相关项从环境变量注入)│ + │ 3. 逐 source 采集(http_json / command / static_json) + │ 4. Jinja2 模板注入 → 输出 markdown │ + └─────────────────────────────────────────────┘ +``` + +## 快速开始 + +```bash +# 1. 准备环境变量(不写死进仓库) +set -a; . ./.env; set +a + +# 2. 一键采集 + 渲染 +python3 scripts/build.py --config config/morning-digest.yaml +# 输出:/tmp/digest_today.md(配置里 render.output 决定) + +# 只看数据不写文件 +python3 scripts/build.py --config config/morning-digest.yaml --dry-run + +# 临时覆盖环境变量 +python3 scripts/build.py --config config/morning-digest.yaml --env HOT_API=http://localhost:6688 +``` + +## 配置文件结构(YAML) + +```yaml +meta: + name: morning-digest + date_format: "%Y-%m-%d" + +sources: + : # 会成为模板变量 + type: http_json | command | static_json + # ...adapter 参数 + map: # 原始 item → 输出字段 + 字段名: 表达式 + +render: + template: templates/digest.md + output: /tmp/xxx.md +``` + +## Source 类型与适配器 + +### `http_json` — 拉一个 JSON 接口取列表 +```yaml +news: + type: http_json + url: "https://host/api/{country}" # {param} 由 param_loop 填充 + param_loop: { key: country, values: [china, us] } # 可选:循环发多请求 + items_path: headlines # 响应里取列表的路径 + limit: 5 + map: + title: "title|headline|name" # | 表示依次回退取第一个非空 + url: "url|link" + source: "sourceLabel|{loop.country}" # {loop.key} = 当前循环值 +``` + +### `command` — 跑 shell 命令,读出 JSON +```yaml +github: + type: command + cmd: "python3 ${GITHUB_TRENDING_SCRIPT} {out}" # {out} = 临时 JSON 文件路径 + items_from_file: "{out}" # 从文件读(默认解析 stdout) + map: { name: name, desc: desc, url: url, lang: lang, stars: stars, stars_today: stars_today } +``` + +### `static_json` — 读本地文件(供 LLM/外部注入) +```yaml +tech: + type: static_json + file: "${TECH_FILE}" # 未设置则跳过(optional: true) + map: { title: title, url: url, summary: summary } +``` + +## map 取值表达式 + +- `a|b|c` — 依次尝试,取第一个非空字段 +- `{loop.key}` — 当前 param_loop 循环值 +- `@index` — 1 起始序号 +- 空串/缺失 → 该字段为空 + +## 模板 + +`templates/digest.md`(Jinja2)。每个 `source` 的 section 名成为一个模板变量(列表)。改排版只改模板,不动脚本。 + +早报模板里 GitHub Trending 已渲染成 markdown 超链接: + +``` +**[org/name](https://github.com/org/name)**(lang):desc +``` + +## 环境变量(`.env.example`) + +机器相关项**一律从环境变量注入**,仓库不写死。 + +| 变量 | 说明 | +|------|------| +| `HOT_API` | DailyHotApi 基址 | +| `GITHUB_TRENDING_SCRIPT` | GitHub Trending 采集脚本 | +| `RSS_SCRIPT` | RSS 采集脚本 | +| `TECH_FILE` | 科技动态注入 JSON(可选) | +| `FEISHU_CREATE_SCRIPT` | 飞书知识库写入脚本(可选) | + +## 失败隔离 + +每个 source 独立 try/except:失败 → 该 section 为空数组,错误写进输出 JSON 顶部的 `_errors`(不影响其他 source,流程不中断)。 + +## 依赖 + +- Python 3.10+(`requests`、`jinja2`、`pyyaml` 可选、`bs4` 视 source) +- source 脚本(`${GITHUB_TRENDING_SCRIPT}`、`${RSS_SCRIPT}` 等)按各自配置 + +## 已知坑 + +- **github.com 直连间歇性超时**:GitHub Trending source 依赖直连官方页面,外网抖动时会空(记入 `_errors` 不中断)。 +- **海外 RSS DNS 吊死**:ak-rss-digest 用子进程隔离 + feedparser,个别源超时不影响整体。 +- **飞书写入幂等**:`create_digest.py` 同标题自动复用节点,不重复建文档。 + +## 扩展 + +想加数据源?在 `sources` 里加一个 section(选一种 adapter,配好 `map`)→ 在模板里用它,即可。无需改引擎。 diff --git a/config/morning-digest.yaml b/config/morning-digest.yaml new file mode 100644 index 0000000..b894172 --- /dev/null +++ b/config/morning-digest.yaml @@ -0,0 +1,74 @@ +# morning-digest 示例配置 +# 用 build.py 跑:python3 scripts/build.py --config config/morning-digest.yaml +# 机器相关项(内部地址/脚本路径)一律走 ${ENV},不写死。 + +meta: + name: morning-digest + date_format: "%Y-%m-%d" + +sources: + # 国际要闻:the-news API,china + us 各前 5 条 + news: + type: http_json + url: "https://www.thehear.org/api/country-view/{country}" + param_loop: + key: country + values: [china, us] + items_path: headlines + limit: 5 + map: + title: "title|headline|name" + url: "url|link" + source: "sourceLabel|sourceName|{loop.country}" + + # 社会热榜:DailyHotApi,weibo + zhihu 各前 10 条 + hot: + type: http_json + url: "${HOT_API}/{platform}" + param_loop: + key: platform + values: [weibo, zhihu] + items_path: data + limit: 10 + map: + title: "title|name" + rank: "@index" + platform: "{loop.platform}" + + # GitHub Trending:复用 fetch_github_trending.py 直连官方页面 + github: + type: command + cmd: "python3 ${GITHUB_TRENDING_SCRIPT} {out}" + items_from_file: "{out}" + map: + name: "name" + desc: "desc" + url: "url" + lang: "lang" + stars: "stars" + stars_today: "stars_today" + + # RSS 精选:复用 ak-rss-digest 的 fetch_today_feed_items.py + rss: + type: command + cmd: "python3 ${RSS_SCRIPT} --format json --days 7 --limit 30" + items_path: items + map: + title: "title" + url: "link" + summary: "summary" + source: "feed_name" + + # 科技动态:默认无;由 LLM 搜好结果注入(见 README) + tech: + type: static_json + file: "${TECH_FILE}" + optional: true + map: + title: "title" + url: "url" + summary: "summary" + +render: + template: "templates/digest.md" + output: "/tmp/digest_today.md" diff --git a/scripts/build.py b/scripts/build.py new file mode 100644 index 0000000..6e9b418 --- /dev/null +++ b/scripts/build.py @@ -0,0 +1,237 @@ +#!/usr/bin/env python3 +# -*- coding: utf-8 -*- +"""通用「数据采集 + 模板注入」引擎。 + +读一份配置(YAML/JSON),按声明采集多路数据源,注入 Jinja2 模板,产出 markdown(或任意文本)。 + +用法: + python3 build.py --config config/digest.yaml [--env K=V ...] [--out PATH] [--dry-run] + +配置结构(示例见 config/morning-digest.yaml): + meta: { name, date_format } + sources: {
: { type, ...adapter 参数, map } } + render: { template, output, [env] } + (可选) sections: 控制模板中板块渲染顺序/说明 + +Source 类型: + http_json GET 一个 JSON 接口,按 items_path 取列表。支持 {param} 循环、| 字段回退、 + @index 序号、{loop.key} 循环值。 + command 跑一条 shell 命令;命令里 {out} 替换为输出 JSON 文件路径(写文件)或直接 stdout JSON。 + static_json 读一个本地 JSON 文件(供 LLM/外部注入)。 +""" +import argparse +import json +import os +import re +import subprocess +import sys +import tempfile +from datetime import datetime + +import requests + +try: + import yaml +except ImportError: # pragma: no cover + yaml = None + +from jinja2 import Environment, FileSystemLoader + +ENV_RE = re.compile(r"\$\{(\w+)\}") + + +def env_sub(value, env): + """把字符串里的 ${VAR} 替换成 env 值,未定义则保留原样。""" + if not isinstance(value, str): + return value + def _repl(m): + return env.get(m.group(1), m.group(0)) + return ENV_RE.sub(_repl, value) + + +def deep_env_sub(obj, env): + if isinstance(obj, dict): + return {k: deep_env_sub(v, env) for k, v in obj.items()} + if isinstance(obj, list): + return [deep_env_sub(v, env) for v in obj] + return env_sub(obj, env) + + +def load_config(path): + with open(path, "rb") as f: + raw = f.read() + if path.endswith((".yaml", ".yml")): + if yaml is None: + raise RuntimeError("需要 pyyaml 才能读 YAML 配置") + return yaml.safe_load(raw) or {} + return json.loads(raw) + + +def get_path(obj, dotted): + """按点分路径取值,支持 dict key 和 list index。""" + if not dotted: + return obj + for part in str(dotted).split("."): + if isinstance(obj, dict): + obj = obj.get(part) + elif isinstance(obj, list) and part.isdigit(): + obj = obj[int(part)] + else: + return None + return obj + + +def resolve_spec(spec, item, ctx): + """解析 map 取值表达式:a|b|c 依次回退;{loop.key} 取循环值;@index 由调用方特殊处理。""" + for token in str(spec).split("|"): + token = token.strip() + if not token: + continue + if token.startswith("{loop.") and token.endswith("}"): + v = ctx.get(token[6:-1].strip()) + if v not in (None, ""): + return v + continue + v = get_path(item, token) + if v not in (None, ""): + return v + return "" + + +def map_item(item, mapping, ctx, index): + out = {} + for field, spec in (mapping or {}).items(): + if str(spec) == "@index": + out[field] = index + else: + out[field] = resolve_spec(spec, item, ctx) + return out + + +# ----------------------------- source adapters ----------------------------- # + +def collect_http_json(src, env): + items = [] + loop = src.get("param_loop") + batches = [{}] if not loop else [{loop["key"]: v} for v in loop.get("values", [])] + for ctx in batches: + url = env_sub(src["url"], env) + for k, v in ctx.items(): + url = url.replace("{%s}" % k, str(v)) + resp = requests.get(url, timeout=src.get("timeout", 20)).json() + raw = get_path(resp, src.get("items_path", "")) or [] + raw = raw[: src.get("limit", 20)] + for i, it in enumerate(raw, 1): + items.append(map_item(it, src.get("map"), ctx, i)) + return items + + +def collect_command(src, env): + cmd = env_sub(src["cmd"], env) + out_file = None + if "{out}" in cmd: + out_file = tempfile.mktemp(suffix=".json") + cmd = cmd.replace("{out}", out_file) + r = subprocess.run(["sh", "-lc", cmd], capture_output=True, text=True, + timeout=src.get("timeout", 180)) + if r.returncode != 0: + raise RuntimeError((r.stderr or r.stdout).strip()[:800]) + file_ref = src.get("items_from_file") + if file_ref: + data_path = env_sub(file_ref, env).replace("{out}", out_file or "") + with open(data_path, encoding="utf-8") as f: + data = json.load(f) + else: + data = json.loads(r.stdout) + raw = get_path(data, src.get("items_path", "")) or data + raw = raw[: src.get("limit", 20)] + return [map_item(it, src.get("map"), {}, i + 1) for i, it in enumerate(raw)] + + +def collect_static(src, env): + path = env_sub(src.get("file", ""), env) + if not path or not os.path.exists(path): + if src.get("optional"): + return [] + raise RuntimeError(f"static 文件不存在: {path}") + with open(path, encoding="utf-8") as f: + data = json.load(f) + raw = get_path(data, src.get("items_path", "")) or data + raw = raw[: src.get("limit", 20)] + return [map_item(it, src.get("map"), {}, i + 1) for i, it in enumerate(raw)] + + +ADAPTERS = {"http_json": collect_http_json, "command": collect_command, + "static_json": collect_static} + + +def render_template(template_file, data, out_path): + # 模板路径解析:绝对路径 > 配置目录 > 配置父目录 > 当前目录 + candidates = [] + if not os.path.isabs(template_file): + base = os.path.dirname(os.path.abspath(config_path if globals().get("config_path") else ".")) + candidates = [os.path.join(base, template_file), + os.path.join(os.path.dirname(base), template_file)] + candidates.append(template_file) + tpl_path = next((p for p in candidates if os.path.exists(p)), candidates[0]) + + env = Environment(loader=FileSystemLoader(os.path.dirname(tpl_path)), + trim_blocks=True, lstrip_blocks=True) + tpl = env.get_template(os.path.basename(tpl_path)) + md = tpl.render(**data) + with open(out_path, "w", encoding="utf-8") as f: + f.write(md) + return out_path + + +def main(): + global config_path + ap = argparse.ArgumentParser() + ap.add_argument("--config", required=True, help="配置文件路径 (YAML/JSON)") + ap.add_argument("--env", action="append", default=[], help="覆盖环境变量 K=V(可多次)") + ap.add_argument("--out", default=None, help="覆盖输出路径") + ap.add_argument("--dry-run", action="store_true", help="只采集不写文件") + args = ap.parse_args() + + config_path = os.path.abspath(args.config) + env = dict(os.environ) + for kv in args.env: + k, _, v = kv.partition("=") + env[k] = v + + cfg = deep_env_sub(load_config(config_path), env) + date = datetime.now() + + data = {"date": cfg.get("meta", {}).get("date_format", "%Y-%m-%d") and + date.strftime(cfg.get("meta", {}).get("date_format", "%Y-%m-%d"))} + + errors = {} + for name, src in (cfg.get("sources") or {}).items(): + adapter = ADAPTERS.get(src.get("type")) + if not adapter: + errors[name] = f"未知 source 类型: {src.get('type')}" + data[name] = [] + continue + try: + data[name] = adapter(src, env) + except Exception as e: + errors[name] = str(e)[:300] + data[name] = [] + + if errors: + data["_errors"] = errors + + if args.dry_run: + print(json.dumps({k: v for k, v in data.items() if k != "_errors"}, + ensure_ascii=False, indent=2)) + return + + out_path = args.out or env_sub(cfg.get("render", {}).get("output", "/tmp/digest.md"), env) + render_template(cfg.get("render", {}).get("template", "templates/digest.md"), data, out_path) + print(f"RENDER_OK: {out_path}") + if errors: + print(f"_errors={json.dumps(errors, ensure_ascii=False)}") + + +if __name__ == "__main__": + main() diff --git a/templates/digest.md b/templates/digest.md new file mode 100644 index 0000000..1848b2f --- /dev/null +++ b/templates/digest.md @@ -0,0 +1,36 @@ +# 📰 早报 · {{ date }} + +## 🌍 国际要闻 +{% for item in news %} +- **{{ item.source }}**:[{{ item.title }}]({{ item.url }}) +{% else %} +- (今日未采集到国际要闻) +{% endfor %} + +## 💻 GitHub Trending +{% for r in github %} +- **[{{ r.name }}]({{ r.url }})**{% if r.lang %}({{ r.lang }}){% endif %}:{{ r.desc }} +{% else %} +- (今日未采集到 GitHub Trending) +{% endfor %} + +## 🔬 科技动态 +{% for t in tech %} +- [{{ t.title }}]({{ t.url }}):{{ t.summary }} +{% else %} +- (今日未采集到科技动态) +{% endfor %} + +## 🔥 社会热榜 +{% for h in hot %} +- {{ h.rank }}. {{ h.title }}({{ h.platform }}) +{% else %} +- (今日未采集到社会热榜) +{% endfor %} + +## 📡 RSS 精选 +{% for r in rss %} +- [{{ r.title }}]({{ r.url }}){% if r.source %}({{ r.source }}){% endif %}:{{ r.summary }} +{% else %} +- (今日未采集到 RSS 精选) +{% endfor %}