#!/usr/bin/env python3 """Markdown to Feishu Docx Block JSON converter. Usage: python3 md_to_blocks.py < input.md > output.json python3 md_to_blocks.py file.md """ import re, json, sys def parse_inline(text): """Parse a line of inline markdown into text_run elements.""" elements = [] pos = 0 # Patterns in order: bold, italic, inline code, link, plain text while pos < len(text): # Bold: **text** or __text__ m = re.match(r'\*\*(.+?)\*\*|__(.+?)__', text[pos:]) if m: elements.append({'text_run': {'content': m.group(1), 'text_element_style': {'bold': True}}}) pos += len(m.group(0)) continue # Link: [text](url) m = re.match(r'\[(.+?)\]\((.+?)\)', text[pos:]) if m: elements.append({'text_run': {'content': m.group(1), 'text_element_style': {'link': {'url': m.group(2)}}}}) pos += len(m.group(0)) continue # Inline code: `text` m = re.match(r'`(.+?)`', text[pos:]) if m: elements.append({'text_run': {'content': m.group(1), 'text_element_style': {'inline_code': True}}}) pos += len(m.group(0)) continue # Italic: *text* m = re.match(r'\*(.+?)\*', text[pos:]) if m: elements.append({'text_run': {'content': m.group(1), 'text_element_style': {'italic': True}}}) pos += len(m.group(0)) continue # Plain text until next special char nxt = re.search(r'[\*\[`]', text[pos:]) if nxt: end = pos + nxt.start() if end > pos: elements.append({'text_run': {'content': text[pos:end], 'text_element_style': {}}}) elif end == pos: # special char at current position but not a match — skip it as literal elements.append({'text_run': {'content': text[pos:pos+1], 'text_element_style': {}}}) pos += 1 continue pos = end else: elements.append({'text_run': {'content': text[pos:], 'text_element_style': {}}}) break return elements def md_to_blocks(md_text): """Convert Markdown text to Feishu Block JSON array.""" lines = md_text.strip().split('\n') blocks = [] i = 0 while i < len(lines): line = lines[i] # Empty line if not line.strip(): # Add empty paragraph only if next line is also text (not heading/list/divider) if i + 1 < len(lines) and lines[i + 1].strip() and not lines[i + 1].strip().startswith(('#', '-', '*', '`', '---')): blocks.append({'block_type': 2, 'text': {'elements': [{'text_run': {'content': '', 'text_element_style': {}}}], 'style': {}}}) i += 1 continue # Divider if line.strip() == '---' or line.strip() == '***': blocks.append({'block_type': 22, 'divider': {}}) i += 1 continue # Headings h = re.match(r'^(#{1,3})\s+(.+)$', line) if h: level = len(h.group(1)) bt = 2 + level # 3=h1, 4=h2, 5=h3 blocks.append({ 'block_type': bt, f'heading{level}': { 'elements': parse_inline(h.group(2)), 'style': {} } }) i += 1 continue # Bullet list bullet = re.match(r'^[\-\*]\s+(.+)$', line) if bullet: blocks.append({ 'block_type': 12, 'bullet': { 'elements': parse_inline(bullet.group(1)), 'style': {} } }) i += 1 continue # Numbered list numbered = re.match(r'^\d+\.\s+(.+)$', line) if numbered: blocks.append({ 'block_type': 13, 'ordered': { 'elements': parse_inline(numbered.group(1)), 'style': {} } }) i += 1 continue # Regular text elements = parse_inline(line) blocks.append({ 'block_type': 2, 'text': { 'elements': elements if elements else [{'text_run': {'content': line, 'text_element_style': {}}}], 'style': {} } }) i += 1 return blocks if __name__ == '__main__': if len(sys.argv) > 1: with open(sys.argv[1]) as f: text = f.read() else: text = sys.stdin.read() blocks = md_to_blocks(text) print(json.dumps(blocks, indent=2, ensure_ascii=False))