[teamai] Push 87 resource(s) from XingfenD

This commit is contained in:
2026-09-10 16:10:45 +08:00
parent 425c9c078a
commit 65c04def51
1314 changed files with 211681 additions and 0 deletions
@@ -0,0 +1,160 @@
#!/usr/bin/env bash
# ─────────────────────────────────────────────────────────────
# html-to-pdf.sh —— 把 Beautiful Article 的单页 HTML 转成 PDF
#
# 用法:
# bash <skill>/scripts/html-to-pdf.sh [input.html] [output.pdf]
# bash <skill>/scripts/html-to-pdf.sh # 默认 article/article.html → article/article.pdf
# bash <skill>/scripts/html-to-pdf.sh --help
#
# 前提:本机已装 Chromium / Google Chrome / Brave / Microsoft Edge 之一
# (脚本会自动探测)。无需 npm 包、无需 Node。
#
# 设计要点(详见 references/pdf-output.md):
# 1. reacticle 默认 TOC 在桌面是左右栅格;PDF 需要 TOC 上、正文下的上下排布。
# 2. 脚本在 HTML 头部注入一段 @media print CSS 覆盖:把 TOC 塌成一列、解除
# sticky、长 TOC 双列省纸、TOC 后强制分页、隐藏 colophon 之外的页眉页脚等。
# 3. 用 headless 浏览器 --print-to-pdf 渲染(执行页面 JS,Raw 交互渲染为初始
# 态);输出标准 A4 / 主题纸色满版背景 + 0.45in 内容留白,无浏览器自带页眉页脚。
# 4. 失败回退:打印"用浏览器手动 Cmd+P → 另存为 PDF"指引,并把注入后的 HTML
# 留在临时目录方便用户自己打印。
# ─────────────────────────────────────────────────────────────
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
CSS_FILE="$SCRIPT_DIR/pdf-print-overrides.css"
INPUT="${1:-article/article.html}"
OUTPUT="${2:-article/article.pdf}"
if [[ "$INPUT" == "--help" || "$INPUT" == "-h" ]]; then
sed -n '2,21p' "$0"
exit 0
fi
if [[ ! -f "$INPUT" ]]; then
echo "✗ 输入文件不存在:$INPUT" >&2
echo " 先在工作区跑 npm run html 产出 article/article.html。" >&2
exit 1
fi
mkdir -p "$(dirname "$OUTPUT")"
# ── 探测可用的 chromium-family 浏览器 ─────────────────────
find_browser() {
local candidates=(
chromium
chromium-browser
google-chrome
google-chrome-stable
chrome
brave-browser
microsoft-edge
"/Applications/Google Chrome.app/Contents/MacOS/Google Chrome"
"/Applications/Google Chrome Canary.app/Contents/MacOS/Google Chrome Canary"
"/Applications/Chromium.app/Contents/MacOS/Chromium"
"/Applications/Microsoft Edge.app/Contents/MacOS/Microsoft Edge"
"/Applications/Brave Browser.app/Contents/MacOS/Brave Browser"
"/Applications/Arc.app/Contents/MacOS/Arc"
"/usr/bin/chromium"
"/usr/bin/google-chrome"
"/snap/bin/chromium"
)
for c in "${candidates[@]}"; do
if command -v "$c" >/dev/null 2>&1; then
echo "$c"; return 0
fi
if [[ -x "$c" ]]; then
echo "$c"; return 0
fi
done
return 1
}
# ── 注入 @media print 覆盖到一个临时 HTML ─────────────────
# 设计:CSS 抽到 scripts/pdf-print-overrides.css(详见该文件顶部注释),这里只
# 负责"把它的内容包在 <style> 里、塞到 </head> 之前"。这样:
# • macOS BSD awk 不接受 -v 传多行字符串("newline in string"),用 awk
# getline 从文件读则两边都吃得下。
# • CSS 文件可独立编辑 / lint / 复用,不被 shell 转义吃掉。
TMP_DIR="$(mktemp -d -t beautiful-article-pdf.XXXXXX)"
TMP_HTML="$TMP_DIR/article-print.html"
if [[ ! -f "$CSS_FILE" ]]; then
echo "✗ 找不到打印覆盖 CSS:$CSS_FILE" >&2
echo " 这个文件应跟脚本同目录(scripts/pdf-print-overrides.css)。" >&2
exit 2
fi
awk -v css_file="$CSS_FILE" '
/<\/head>/ && !done {
print "<style id=\"ra-pdf-overrides\">"
while ((getline line < css_file) > 0) print line
close(css_file)
print "</style>"
done = 1
}
{ print }
' "$INPUT" > "$TMP_HTML"
if ! grep -q 'ra-pdf-overrides' "$TMP_HTML"; then
echo "✗ 注入失败:未在输入 HTML 找到 </head>。" >&2
echo " 你的 article.html 可能不是 Vite + reacticle 单页产物。" >&2
exit 3
fi
# ── 找浏览器 ──────────────────────────────────────────────
BROWSER="$(find_browser || true)"
if [[ -z "$BROWSER" ]]; then
echo "⚠ 未找到任何 chromium-family 浏览器(chromium / google-chrome / brave / edge)。"
echo
echo " 回退方案:注入了打印 CSS 的 HTML 已经放在:"
echo " $TMP_HTML"
echo
echo " 请用浏览器打开它 → Cmd+P / Ctrl+P → 目标改为'另存为 PDF' → 保存。"
echo " 注入的 print 样式会让 TOC 在上、正文在下,与 PDF 阅读习惯对齐。"
exit 3
fi
echo "▸ 用浏览器:$BROWSER"
echo "▸ 输入:$INPUT"
echo "▸ 输出:$OUTPUT"
# ── 渲染 ──────────────────────────────────────────────────
# Chromium 系都支持 --headless --print-to-pdf。
# --no-pdf-header-footer:去掉浏览器自带的 URL / 日期 / 页码(colophon 已有)。
# --virtual-time-budget:给页面 JS 一点时间初始化 Raw 组件(5s 兜底)。
# --hide-scrollbars:避免在 PDF 里看到滚动条残影。
"$BROWSER" \
--headless=new \
--disable-gpu \
--no-sandbox \
--hide-scrollbars \
--no-pdf-header-footer \
--virtual-time-budget=5000 \
--print-to-pdf-no-header \
--print-to-pdf="$OUTPUT" \
"file://$TMP_HTML" 2>/dev/null || {
# 旧版 Chrome 不认 --headless=new,回退老 flag
"$BROWSER" \
--headless \
--disable-gpu \
--no-sandbox \
--hide-scrollbars \
--print-to-pdf-no-header \
--print-to-pdf="$OUTPUT" \
"file://$TMP_HTML" 2>/dev/null
}
# 清理临时(HTML 注入留作回退证据,最后再清)
rm -rf "$TMP_DIR"
if [[ -f "$OUTPUT" ]]; then
SIZE="$(du -h "$OUTPUT" | cut -f1)"
echo "✓ PDF 输出:${OUTPUT} (${SIZE})"
echo
echo " 如果 TOC / 分页不理想,看 references/pdf-output.md 故障排除段。"
else
echo "✗ 浏览器返回成功但输出文件不存在:$OUTPUT" >&2
exit 4
fi
@@ -0,0 +1,192 @@
/*
* PDF print overrides for Beautiful Article.
*
* Injected by scripts/html-to-pdf.sh into article.html's <head> right before
* Chromium headless prints it. reacticle's own print.css already handles
* hiding export bars and break-inside on cards.
*
* This file adds four groups on top of that:
* 0) Theme surface — keep the article theme background / text colors in PDF.
* A) TOC layout — force the sidebar TOC to stack above the article and
* page-break after it, so the article body starts on a fresh page.
* B) Page break behavior — undo reacticle's `.ra-section { break-inside:
* avoid-page }` (which causes huge empty pages for multi-page sections),
* glue headings to the following content, keep Hero / Lead / figures
* atomic, and apply widow/orphan control to paragraphs.
* C) Cover — if the article uses the 3:4 Cover component (default), keep
* its authored geometry intact and only force a page break after it.
* See references/cover.md.
*
* Why a separate file:
* - macOS BSD awk rejects multi-line strings via -v; reading from a file
* with getline sidesteps that.
* - CSS is independently editable / lintable / diff-friendly here.
* - Easy to swap or extend without touching the shell script.
*/
@media print {
@page {
margin: 0;
}
/* ============================================================
* 0 · Theme surface (preserve paper color in PDF)
* ============================================================ */
.ra-root {
background: var(--ra-color-bg, #ffffff) !important;
color: var(--ra-color-text, #111111) !important;
print-color-adjust: exact;
-webkit-print-color-adjust: exact;
position: relative;
z-index: 0;
box-sizing: border-box;
min-height: 100vh;
padding: 0.45in !important;
box-decoration-break: clone;
-webkit-box-decoration-break: clone;
}
.ra-root::before {
content: "";
position: fixed;
inset: 0;
background: var(--ra-color-bg, #ffffff);
z-index: -1;
pointer-events: none;
}
/* ============================================================
* A · TOC layout (TOC above article, not beside it)
* ============================================================ */
/* A1) reacticle's TOC layout is a 2-col grid on desktop and `display: block`
* on mobile (<= 999px). Force the mobile branch for print so the TOC
* sits ABOVE the article column instead of beside it. */
.ra-article-layout--with-toc {
display: block !important;
max-width: none !important;
padding: 0 !important;
}
/* A2) TOC: kill sticky (only paints once on print), add breathing room, and
* push the article to its own pages by breaking after the TOC. */
.ra-toc {
position: static !important;
margin-bottom: 1.5rem !important;
page-break-after: always;
break-after: page;
}
/* A3) Long TOCs save paper as a 2-column layout; short ones collapse
* naturally back to one column. */
.ra-toc__list {
column-count: 2;
column-gap: 1.5rem;
column-fill: balance;
}
/* A4) Never split a TOC item across columns / pages. */
.ra-toc__item {
break-inside: avoid;
page-break-inside: avoid;
}
/* A5) Let the article column flow naturally on the page after the TOC. */
.ra-article-layout--with-toc > .ra-article {
break-before: auto;
}
/* A6) Strip underlines from TOC + body links in print (chrome already shows
* them as link-blue; the colophon footer keeps its underline because it
* sets its own text-decoration inline). */
.ra-toc a,
.ra-article a {
color: inherit;
text-decoration: none;
}
/* ============================================================
* B · Page break behavior (fixes huge empty pages in long sections)
* ============================================================ */
/* B1) reacticle's print.css aggressively sets `.ra-section { break-inside:
* avoid-page }`. For multi-page sections that rule backfires badly:
* the browser pushes the whole oversized section to the next page,
* leaving the previous page nearly empty, then the section overflows
* anyway. UNDO IT — let long sections break naturally across pages. */
.ra-section,
.ra-subsection,
.ra-section__body,
.ra-subsection__body {
break-inside: auto !important;
page-break-inside: auto !important;
}
/* B2) But never strand a heading at the bottom of a page. Glue Section /
* Subsection headings to whatever follows, and keep the heading row
* (index + title) intact. */
.ra-section__head,
.ra-subsection__head,
.ra-hero__title,
.ra-hero__subtitle,
h1,
h2,
h3,
h4 {
break-after: avoid;
page-break-after: avoid;
break-inside: avoid;
page-break-inside: avoid;
}
/* B3) Hero / Lead / Conclusion are short, atomic blocks — never split them
* across pages. (Hero often holds title + subtitle + meta; ugly when
* subtitle ends up alone on next page.) */
.ra-hero,
.ra-lead,
.ra-conclusion {
break-inside: avoid;
page-break-inside: avoid;
}
/* B4) Widows / orphans — never leave 1–2 stranded lines of a paragraph at
* the top / bottom of a page. */
p,
li,
blockquote,
.ra-aside,
.ra-quote {
orphans: 3;
widows: 3;
}
/* B5) Atomic visual blocks: figures, tables, code blocks. Keep them whole
* when reasonable; very long ones still split (the browser falls back). */
figure,
.ra-table,
.ra-codeblock,
.ra-formula,
.ra-image,
.ra-raw {
break-inside: avoid;
page-break-inside: avoid;
}
/* ============================================================
* C · Cover (the 3:4 cover above TOC, see references/cover.md)
* ============================================================ */
/* C1) Keep the cover's screen-authored 3:4 geometry in print. Earlier
* versions stretched .ra-cover to height:100vh, but Chromium print can
* clip absolutely positioned / grid-based cover internals after that
* resize. The stable default is: preserve the cover and start the TOC on
* the next page. Article-specific covers may opt into full-page print
* sizing only after visual PDF verification. */
.ra-cover {
break-inside: avoid;
page-break-inside: avoid;
break-after: page;
page-break-after: always;
}
}
+172
View File
@@ -0,0 +1,172 @@
#!/usr/bin/env bash
# ─────────────────────────────────────────────────────────────
# scaffold.sh —— 一键创建一个 Beautiful Article 工作区。
#
# 用法:
# bash scripts/scaffold.sh <target-dir> [--theme=<id>] [--no-cover]
# bash scripts/scaffold.sh --list-themes
#
# 例子:
# bash <path-to-beautiful-article>/scripts/scaffold.sh ./my-article --theme=tufte
# bash <path-to-beautiful-article>/scripts/scaffold.sh ./brief --theme=press --no-cover
# bash <path-to-beautiful-article>/scripts/scaffold.sh --list-themes
#
# --no-cover:禁用文章封面(默认开 · 屏幕 3:4 / PDF 独占首页)。详见 references/cover.md。
#
# 工作区从 npm 安装**最新发布版的 reacticle**(package.json 里 reacticle: "latest",
# 每次 fresh scaffold 都会取当下最新)。无需本地 reacticle 仓库。
#
# 跑完后看 SKILL.md「Phase 4 First Spread」+ references/component-policy.md /
# raw-policy.md / 选定主题 theme-profiles/<id>.md。
# ─────────────────────────────────────────────────────────────
set -euo pipefail
SKILL_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
TEMPLATE="$SKILL_DIR/assets/scaffold-template"
PROFILES="$SKILL_DIR/theme-profiles/index.json"
DEFAULT_THEME="tufte"
list_themes() {
echo "可用主题(来自 ${PROFILES}):"
echo
# 没有 jq,用 grep + sed 提字段
grep -E '"id"|"label"|"mood"' "$PROFILES" | sed -E \
-e 's/.*"id":[[:space:]]*"([^"]+)".*/ • \1/' \
-e 's/.*"label":[[:space:]]*"([^"]+)".*/ \1/' \
-e 's/.*"mood":[[:space:]]*"([^"]+)".*/ \1/'
echo
echo "用 --theme=<id> 选定一个。默认:${DEFAULT_THEME}。"
}
# 校验主题 id 是否在 theme-profiles/index.json 里
theme_exists() {
grep -Eq "\"id\"[[:space:]]*:[[:space:]]*\"$1\"" "$PROFILES"
}
# ── 解析参数 ──
TARGET=""
THEME="$DEFAULT_THEME"
COVER=1
for arg in "$@"; do
case "$arg" in
--list-themes) list_themes; exit 0 ;;
--theme=*) THEME="${arg#--theme=}" ;;
--no-cover) COVER=0 ;;
--cover) COVER=1 ;;
--*) echo "✗ 未知参数: $arg" >&2; exit 1 ;;
*) [[ -z "$TARGET" ]] && TARGET="$arg" ;;
esac
done
TARGET="${TARGET:-my-article}"
# ── 校验主题 ──
if ! theme_exists "$THEME"; then
echo "✗ 未知主题 '$THEME'。可用主题:" >&2
echo >&2
list_themes >&2
exit 1
fi
# ── 目标目录检查 ──
if [[ -d "$TARGET" && -n "$(ls -A "$TARGET" 2>/dev/null || true)" ]]; then
echo "✗ 目标目录 '$TARGET' 已存在且非空,已中止。" >&2
exit 1
fi
if ! command -v npm >/dev/null; then
echo "✗ 需要 npm,但在 PATH 里没找到。" >&2
exit 1
fi
echo "▸ 在 $TARGET 创建 Beautiful Article 工作区"
echo "▸ 主题:$THEME"
echo "▸ 封面:$([[ "$COVER" == "1" ]] && echo "开(屏幕 3:4 / PDF 独占首页,详见 references/cover.md)" || echo "关")"
echo "▸ reacticle:从 npm 安装最新发布版"
mkdir -p "$TARGET"
# 复制工程模板
cp "$TEMPLATE/package.json" "$TARGET/package.json"
cp "$TEMPLATE/vite.config.ts" "$TARGET/vite.config.ts"
cp "$TEMPLATE/tsconfig.json" "$TARGET/tsconfig.json"
cp "$TEMPLATE/tsconfig.node.json" "$TARGET/tsconfig.node.json"
cp "$TEMPLATE/index.html" "$TARGET/index.html"
# 工作记忆目录 + 文章源目录
mkdir -p "$TARGET/source" "$TARGET/plan" "$TARGET/review" \
"$TARGET/article/sections" "$TARGET/article/raw-blocks" "$TARGET/article/assets"
cp "$TEMPLATE/article/main.tsx" "$TARGET/article/main.tsx"
cp "$TEMPLATE/article/Article.tsx" "$TARGET/article/Article.tsx"
# 一节一文件:assembler + 第一个 section 组件(多 Agent 并行的代码锚点)
cp "$TEMPLATE/article/sections/01-opening.tsx" "$TARGET/article/sections/01-opening.tsx"
# 封面:默认开。--no-cover 时跳过 Cover.tsx 并从 main.tsx 剥掉 __COVER_*__ 段。
if [[ "$COVER" == "1" ]]; then
cp "$TEMPLATE/article/Cover.tsx" "$TARGET/article/Cover.tsx"
fi
# 留住空目录(git 友好)
touch "$TARGET/article/raw-blocks/.gitkeep" "$TARGET/article/assets/.gitkeep"
# ── 注入主题 id(用 perl 避免转义问题)──
# main.tsx: <ThemeProvider theme="__THEME__">
# Article.tsx: colophon "· __THEME__ theme"
export RA_THEME="$THEME"
perl -pi -e 's/__THEME__/$ENV{RA_THEME}/g' "$TARGET/article/main.tsx"
perl -pi -e 's/__THEME__/$ENV{RA_THEME}/g' "$TARGET/article/Article.tsx"
# ── 封面开关:处理 main.tsx 里 __COVER_*__ 标记包裹的区段 ──
# COVER=1 → 去掉两行 __COVER_*_BEGIN__ / __COVER_*_END__ 标记(保留中间的 import 和 <Cover/>)
# COVER=0 → 连标记带中间内容一起剥掉(封面不参与构建)
if [[ "$COVER" == "1" ]]; then
# 删除标记行本身,保留 Cover 引入与渲染
perl -i -ne 'print unless /__COVER_(IMPORT|RENDER)_(BEGIN|END)__/' "$TARGET/article/main.tsx"
else
# 把 BEGIN..END 之间(含两端标记行)整段删掉
perl -i -0pe 's{[^\n]*__COVER_IMPORT_BEGIN__.*?__COVER_IMPORT_END__[^\n]*\n}{}gs' "$TARGET/article/main.tsx"
perl -i -0pe 's{[^\n]*__COVER_RENDER_BEGIN__.*?__COVER_RENDER_END__[^\n]*\n}{}gs' "$TARGET/article/main.tsx"
fi
# 标记起步主题
echo "$THEME" > "$TARGET/.theme"
cd "$TARGET"
echo "▸ 安装依赖(含 reacticle 最新版,可能要等一会)..."
npm install >/dev/null 2>&1
# 确保拿到当下最新(即使将来模板带了 lockfile 也强制刷新到最新)
npm install reacticle@latest >/dev/null 2>&1
INSTALLED_REACTICLE="$(node -e "console.log(JSON.parse(require('fs').readFileSync('node_modules/reacticle/package.json','utf8')).version)" 2>/dev/null || echo '?')"
echo "▸ reacticle 版本:$INSTALLED_REACTICLE"
echo "▸ 跑一次 typecheck 确认接线 OK ..."
if npx tsc --noEmit; then
echo "✓ typecheck 通过"
else
echo "⚠ typecheck 有问题(见上),dev / build 仍可能正常 —— 请人工确认。" >&2
fi
cat <<EOF
✓ 完成。工作区:$TARGET(主题 $THEME,见 .theme;reacticle $INSTALLED_REACTICLE)
下一步:
1. cd $TARGET
2. npm run dev # 预览(Phase 4 先写首屏 + 第一个 Section)
3. 首屏(Hero/Lead)写进 article/Article.tsx(assembler);
第一个 Section 写进 article/sections/01-opening.tsx
—— 铁律:一个 Section 一个文件,坚决不要写进 Article.tsx(多 Agent 并行前提)。
4. $([[ "$COVER" == "1" ]] && echo "封面:替换 article/Cover.tsx 里的 <CoverPlaceholder />,按文章 + 主题做定制(读 references/cover.md)。" || echo "封面:已关闭。如需打开,重新跑 scaffold 时去掉 --no-cover,或手动复制 Cover.tsx 模板。")
5. 把决策落盘到 source/ plan/ review/(Skill 的长期记忆)
构建交付(Phase 8):
• npm run build # 类型检查 + 单页 HTML → dist/index.html(CSS+JS 内联)
• npm run html # 复用 build,再复制为交付物 article/article.html
切主题:改 article/main.tsx 的 <ThemeProvider theme="..."> 一个字(tufte / press)。
升级组件库:npm install reacticle@latest
写作必读(路径在 Skill 仓库内):
• $SKILL_DIR/references/component-policy.md
• $SKILL_DIR/references/raw-policy.md
• $SKILL_DIR/theme-profiles/$THEME.md
EOF
@@ -0,0 +1,100 @@
#!/usr/bin/env python3
"""MarkItDown-backed Source -> Markdown helper for Beautiful Article.
This script intentionally depends on MarkItDown and fails clearly when it is not
available. Use source-to-markdown.py as the lightweight fallback.
Usage:
python3 source-to-markdown-markitdown.py <file-or-url> -o source/source.md
Optional dependency:
python3.10 -m pip install "markitdown[pdf,docx]"
"""
from __future__ import annotations
import argparse
import sys
from pathlib import Path
from urllib.parse import urlparse
INSTALL_HINT = 'python3.10 -m pip install "markitdown[pdf,docx]"'
def load_markitdown():
if sys.version_info < (3, 10):
print(
"✗ MarkItDown requires Python 3.10+; current Python is "
f"{sys.version_info.major}.{sys.version_info.minor}.",
file=sys.stderr,
)
print(f" Install with: {INSTALL_HINT}", file=sys.stderr)
raise SystemExit(2)
try:
from markitdown import MarkItDown # type: ignore
except Exception as exc:
print("✗ MarkItDown is not installed in this Python environment.", file=sys.stderr)
print(f" Install with: {INSTALL_HINT}", file=sys.stderr)
print(" Or use scripts/source-to-markdown.py as the fallback.", file=sys.stderr)
print(f" Import error: {exc}", file=sys.stderr)
raise SystemExit(2)
return MarkItDown
def result_text(result) -> str:
for attr in ("text_content", "markdown"):
value = getattr(result, attr, None)
if isinstance(value, str) and value.strip():
return value
if isinstance(result, str):
return result
raise RuntimeError("MarkItDown returned no text_content/markdown output.")
def is_url(src: str) -> bool:
return urlparse(src).scheme in ("http", "https")
def main() -> None:
parser = argparse.ArgumentParser(description="Convert a source file or URL to Markdown via MarkItDown")
parser.add_argument("input", help="PDF / DOCX / PPTX / HTML / TXT / MD file or URL")
parser.add_argument("-o", "--output", help="write Markdown here (default: stdout)")
parser.add_argument(
"--use-plugins",
action="store_true",
help="enable installed MarkItDown plugins; disabled by default",
)
args = parser.parse_args()
src = args.input
if not is_url(src) and not Path(src).exists():
print(f"✗ 文件不存在:{src}", file=sys.stderr)
raise SystemExit(1)
MarkItDown = load_markitdown()
converter = MarkItDown(enable_plugins=args.use_plugins)
try:
markdown = result_text(converter.convert(src)).strip() + "\n"
except Exception as exc:
print(f"✗ MarkItDown 转换失败:{exc}", file=sys.stderr)
print(" 可改用 scripts/source-to-markdown.py 做轻量 fallback。", file=sys.stderr)
raise SystemExit(2)
if args.output:
out = Path(args.output)
out.parent.mkdir(parents=True, exist_ok=True)
out.write_text(markdown, "utf-8")
print(
f"✓ MarkItDown 写入 {out}({len(markdown)} 字符)。"
"请继续清理噪音并补 extraction-notes.md。",
file=sys.stderr,
)
else:
sys.stdout.write(markdown)
if __name__ == "__main__":
main()
@@ -0,0 +1,174 @@
#!/usr/bin/env python3
"""Source → Markdown extraction helper for the Beautiful Article skill.
Mechanically extracts text from a PDF / DOCX / HTML file (or URL) into rough
Markdown. It does NOT clean editorial noise, mark image placeholders, or record
extraction risk — that judgement stays with the agent (see Phase 1 +
references/source-to-markdown.md). The agent should review and refine the output
into source/source.md and write source/extraction-notes.md.
Usage:
python3 source-to-markdown.py <input.pdf|.docx|.html|.htm|.txt|.md|URL> [-o out.md]
Dependencies are probed at runtime and optional. If a parser is missing the
script prints an install hint and degrades gracefully.
"""
from __future__ import annotations
import argparse
import sys
from pathlib import Path
from urllib.parse import urlparse
def _hint(pkg: str) -> str:
return f" (缺少 {pkg},可安装:pip install {pkg})"
def from_pdf(path: str) -> str:
try:
import pdfplumber # type: ignore
except Exception:
try:
from pdfminer.high_level import extract_text # type: ignore
return extract_text(path)
except Exception:
print(_hint("pdfplumber 或 pdfminer.six"), file=sys.stderr)
raise SystemExit(2)
out = []
with pdfplumber.open(path) as pdf:
for i, page in enumerate(pdf.pages, 1):
out.append(f"\n<!-- page {i} -->\n")
out.append(page.extract_text() or "")
for t in page.extract_tables() or []:
out.append("\n" + _table_to_md(t) + "\n")
return "\n".join(out)
def from_docx(path: str) -> str:
try:
import docx # type: ignore
except Exception:
print(_hint("python-docx"), file=sys.stderr)
raise SystemExit(2)
doc = docx.Document(path)
out = []
for p in doc.paragraphs:
text = p.text.strip()
if not text:
out.append("")
continue
style = (p.style.name or "").lower()
if style.startswith("heading"):
level = "".join(c for c in style if c.isdigit()) or "1"
out.append("#" * min(int(level), 6) + " " + text)
else:
out.append(text)
for table in doc.tables:
rows = [[c.text.strip() for c in r.cells] for r in table.rows]
if rows:
out.append("\n" + _table_to_md(rows) + "\n")
return "\n".join(out)
def from_html(html: str) -> str:
try:
from bs4 import BeautifulSoup # type: ignore
except Exception:
print(_hint("beautifulsoup4"), file=sys.stderr)
# crude fallback: strip tags
import re
return re.sub(r"<[^>]+>", "", html)
soup = BeautifulSoup(html, "html.parser")
for tag in soup(["script", "style", "nav", "footer", "aside", "header"]):
tag.decompose()
main = soup.find("article") or soup.find("main") or soup.body or soup
lines = []
for el in main.find_all(
["h1", "h2", "h3", "h4", "p", "li", "pre", "blockquote"]
):
text = el.get_text(" ", strip=True)
if not text:
continue
name = el.name
if name.startswith("h") and name[1:].isdigit():
lines.append("#" * int(name[1:]) + " " + text)
elif name == "li":
lines.append("- " + text)
elif name == "pre":
lines.append("```\n" + text + "\n```")
elif name == "blockquote":
lines.append("> " + text)
else:
lines.append(text)
return "\n\n".join(lines)
def from_url(url: str) -> str:
try:
import urllib.request
req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
with urllib.request.urlopen(req, timeout=30) as resp: # noqa: S310
html = resp.read().decode("utf-8", "replace")
except Exception as e: # pragma: no cover
print(f"✗ 抓取失败:{e}", file=sys.stderr)
print(" 也可以用 agent 的网页抓取能力获取正文后再清理。", file=sys.stderr)
raise SystemExit(2)
return from_html(html)
def _table_to_md(rows) -> str:
rows = [[("" if c is None else str(c)).replace("\n", " ").strip() for c in r] for r in rows]
if not rows:
return ""
width = max(len(r) for r in rows)
rows = [r + [""] * (width - len(r)) for r in rows]
head = "| " + " | ".join(rows[0]) + " |"
sep = "| " + " | ".join(["---"] * width) + " |"
body = ["| " + " | ".join(r) + " |" for r in rows[1:]]
return "\n".join([head, sep, *body])
def main() -> None:
ap = argparse.ArgumentParser(description="Source → Markdown extraction helper")
ap.add_argument("input", help="PDF / DOCX / HTML / TXT / MD file or URL")
ap.add_argument("-o", "--output", help="write Markdown here (default: stdout)")
args = ap.parse_args()
src = args.input
parsed = urlparse(src)
if parsed.scheme in ("http", "https"):
md = from_url(src)
else:
p = Path(src)
if not p.exists():
print(f"✗ 文件不存在:{src}", file=sys.stderr)
raise SystemExit(1)
ext = p.suffix.lower()
if ext == ".pdf":
md = from_pdf(str(p))
elif ext == ".docx":
md = from_docx(str(p))
elif ext in (".html", ".htm"):
md = from_html(p.read_text("utf-8", "replace"))
elif ext in (".md", ".markdown", ".txt"):
md = p.read_text("utf-8", "replace")
else:
print(f"✗ 不支持的输入类型:{ext}", file=sys.stderr)
raise SystemExit(1)
md = (md or "").strip() + "\n"
if args.output:
out = Path(args.output)
out.parent.mkdir(parents=True, exist_ok=True)
out.write_text(md, "utf-8")
print(f"✓ 写入 {out}({len(md)} 字符)。请人工清理噪音并补 extraction-notes.md。", file=sys.stderr)
else:
sys.stdout.write(md)
if __name__ == "__main__":
main()