141 lines
4.6 KiB
Python
141 lines
4.6 KiB
Python
#!/usr/bin/env python3
|
|
"""Markdown to Feishu Docx Block JSON converter.
|
|
Usage: python3 md_to_blocks.py < input.md > output.json
|
|
python3 md_to_blocks.py file.md
|
|
"""
|
|
|
|
import re, json, sys
|
|
|
|
def parse_inline(text):
|
|
"""Parse a line of inline markdown into text_run elements."""
|
|
elements = []
|
|
pos = 0
|
|
# Patterns in order: bold, italic, inline code, link, plain text
|
|
while pos < len(text):
|
|
# Bold: **text** or __text__
|
|
m = re.match(r'\*\*(.+?)\*\*|__(.+?)__', text[pos:])
|
|
if m:
|
|
elements.append({'text_run': {'content': m.group(1), 'text_element_style': {'bold': True}}})
|
|
pos += len(m.group(0))
|
|
continue
|
|
# Link: [text](url)
|
|
m = re.match(r'\[(.+?)\]\((.+?)\)', text[pos:])
|
|
if m:
|
|
elements.append({'text_run': {'content': m.group(1), 'text_element_style': {'link': {'url': m.group(2)}}}})
|
|
pos += len(m.group(0))
|
|
continue
|
|
# Inline code: `text`
|
|
m = re.match(r'`(.+?)`', text[pos:])
|
|
if m:
|
|
elements.append({'text_run': {'content': m.group(1), 'text_element_style': {'inline_code': True}}})
|
|
pos += len(m.group(0))
|
|
continue
|
|
# Italic: *text*
|
|
m = re.match(r'\*(.+?)\*', text[pos:])
|
|
if m:
|
|
elements.append({'text_run': {'content': m.group(1), 'text_element_style': {'italic': True}}})
|
|
pos += len(m.group(0))
|
|
continue
|
|
# Plain text until next special char
|
|
nxt = re.search(r'[\*\[`]', text[pos:])
|
|
if nxt:
|
|
end = pos + nxt.start()
|
|
if end > pos:
|
|
elements.append({'text_run': {'content': text[pos:end], 'text_element_style': {}}})
|
|
elif end == pos:
|
|
# special char at current position but not a match — skip it as literal
|
|
elements.append({'text_run': {'content': text[pos:pos+1], 'text_element_style': {}}})
|
|
pos += 1
|
|
continue
|
|
pos = end
|
|
else:
|
|
elements.append({'text_run': {'content': text[pos:], 'text_element_style': {}}})
|
|
break
|
|
return elements
|
|
|
|
def md_to_blocks(md_text):
|
|
"""Convert Markdown text to Feishu Block JSON array."""
|
|
lines = md_text.strip().split('\n')
|
|
blocks = []
|
|
i = 0
|
|
while i < len(lines):
|
|
line = lines[i]
|
|
|
|
# Empty line
|
|
if not line.strip():
|
|
# Add empty paragraph only if next line is also text (not heading/list/divider)
|
|
if i + 1 < len(lines) and lines[i + 1].strip() and not lines[i + 1].strip().startswith(('#', '-', '*', '`', '---')):
|
|
blocks.append({'block_type': 2, 'text': {'elements': [{'text_run': {'content': '', 'text_element_style': {}}}], 'style': {}}})
|
|
i += 1
|
|
continue
|
|
|
|
# Divider
|
|
if line.strip() == '---' or line.strip() == '***':
|
|
blocks.append({'block_type': 22, 'divider': {}})
|
|
i += 1
|
|
continue
|
|
|
|
# Headings
|
|
h = re.match(r'^(#{1,3})\s+(.+)$', line)
|
|
if h:
|
|
level = len(h.group(1))
|
|
bt = 2 + level # 3=h1, 4=h2, 5=h3
|
|
blocks.append({
|
|
'block_type': bt,
|
|
f'heading{level}': {
|
|
'elements': parse_inline(h.group(2)),
|
|
'style': {}
|
|
}
|
|
})
|
|
i += 1
|
|
continue
|
|
|
|
# Bullet list
|
|
bullet = re.match(r'^[\-\*]\s+(.+)$', line)
|
|
if bullet:
|
|
blocks.append({
|
|
'block_type': 12,
|
|
'bullet': {
|
|
'elements': parse_inline(bullet.group(1)),
|
|
'style': {}
|
|
}
|
|
})
|
|
i += 1
|
|
continue
|
|
|
|
# Numbered list
|
|
numbered = re.match(r'^\d+\.\s+(.+)$', line)
|
|
if numbered:
|
|
blocks.append({
|
|
'block_type': 13,
|
|
'ordered': {
|
|
'elements': parse_inline(numbered.group(1)),
|
|
'style': {}
|
|
}
|
|
})
|
|
i += 1
|
|
continue
|
|
|
|
# Regular text
|
|
elements = parse_inline(line)
|
|
blocks.append({
|
|
'block_type': 2,
|
|
'text': {
|
|
'elements': elements if elements else [{'text_run': {'content': line, 'text_element_style': {}}}],
|
|
'style': {}
|
|
}
|
|
})
|
|
i += 1
|
|
|
|
return blocks
|
|
|
|
if __name__ == '__main__':
|
|
if len(sys.argv) > 1:
|
|
with open(sys.argv[1]) as f:
|
|
text = f.read()
|
|
else:
|
|
text = sys.stdin.read()
|
|
|
|
blocks = md_to_blocks(text)
|
|
print(json.dumps(blocks, indent=2, ensure_ascii=False))
|