Files
feishu-wiki-tools/scripts/md_to_blocks.py
T

141 lines
4.6 KiB
Python

#!/usr/bin/env python3
"""Markdown to Feishu Docx Block JSON converter.
Usage: python3 md_to_blocks.py < input.md > output.json
python3 md_to_blocks.py file.md
"""
import re, json, sys
def parse_inline(text):
"""Parse a line of inline markdown into text_run elements."""
elements = []
pos = 0
# Patterns in order: bold, italic, inline code, link, plain text
while pos < len(text):
# Bold: **text** or __text__
m = re.match(r'\*\*(.+?)\*\*|__(.+?)__', text[pos:])
if m:
elements.append({'text_run': {'content': m.group(1), 'text_element_style': {'bold': True}}})
pos += len(m.group(0))
continue
# Link: [text](url)
m = re.match(r'\[(.+?)\]\((.+?)\)', text[pos:])
if m:
elements.append({'text_run': {'content': m.group(1), 'text_element_style': {'link': {'url': m.group(2)}}}})
pos += len(m.group(0))
continue
# Inline code: `text`
m = re.match(r'`(.+?)`', text[pos:])
if m:
elements.append({'text_run': {'content': m.group(1), 'text_element_style': {'inline_code': True}}})
pos += len(m.group(0))
continue
# Italic: *text*
m = re.match(r'\*(.+?)\*', text[pos:])
if m:
elements.append({'text_run': {'content': m.group(1), 'text_element_style': {'italic': True}}})
pos += len(m.group(0))
continue
# Plain text until next special char
nxt = re.search(r'[\*\[`]', text[pos:])
if nxt:
end = pos + nxt.start()
if end > pos:
elements.append({'text_run': {'content': text[pos:end], 'text_element_style': {}}})
elif end == pos:
# special char at current position but not a match — skip it as literal
elements.append({'text_run': {'content': text[pos:pos+1], 'text_element_style': {}}})
pos += 1
continue
pos = end
else:
elements.append({'text_run': {'content': text[pos:], 'text_element_style': {}}})
break
return elements
def md_to_blocks(md_text):
"""Convert Markdown text to Feishu Block JSON array."""
lines = md_text.strip().split('\n')
blocks = []
i = 0
while i < len(lines):
line = lines[i]
# Empty line
if not line.strip():
# Add empty paragraph only if next line is also text (not heading/list/divider)
if i + 1 < len(lines) and lines[i + 1].strip() and not lines[i + 1].strip().startswith(('#', '-', '*', '`', '---')):
blocks.append({'block_type': 2, 'text': {'elements': [{'text_run': {'content': '', 'text_element_style': {}}}], 'style': {}}})
i += 1
continue
# Divider
if line.strip() == '---' or line.strip() == '***':
blocks.append({'block_type': 22, 'divider': {}})
i += 1
continue
# Headings
h = re.match(r'^(#{1,3})\s+(.+)$', line)
if h:
level = len(h.group(1))
bt = 2 + level # 3=h1, 4=h2, 5=h3
blocks.append({
'block_type': bt,
f'heading{level}': {
'elements': parse_inline(h.group(2)),
'style': {}
}
})
i += 1
continue
# Bullet list
bullet = re.match(r'^[\-\*]\s+(.+)$', line)
if bullet:
blocks.append({
'block_type': 12,
'bullet': {
'elements': parse_inline(bullet.group(1)),
'style': {}
}
})
i += 1
continue
# Numbered list
numbered = re.match(r'^\d+\.\s+(.+)$', line)
if numbered:
blocks.append({
'block_type': 13,
'ordered': {
'elements': parse_inline(numbered.group(1)),
'style': {}
}
})
i += 1
continue
# Regular text
elements = parse_inline(line)
blocks.append({
'block_type': 2,
'text': {
'elements': elements if elements else [{'text_run': {'content': line, 'text_element_style': {}}}],
'style': {}
}
})
i += 1
return blocks
if __name__ == '__main__':
if len(sys.argv) > 1:
with open(sys.argv[1]) as f:
text = f.read()
else:
text = sys.stdin.read()
blocks = md_to_blocks(text)
print(json.dumps(blocks, indent=2, ensure_ascii=False))