#!/usr/bin/env python3 """Markdown to Feishu Docx Block JSON converter (state-machine based). Usage: python3 md_to_blocks.py < input.md > output.json python3 md_to_blocks.py file.md Design: - Block-level state machine: scans line by line, each line starts a block state (heading / bullet / ordered / quote / code / table / divider / paragraph). - Inline state machine: scans character by character with an explicit state stack so that nested inline markers (e.g. bold-wrapped links) resolve correctly. Inside bold/italic we recursively re-enter the inline machine, then merge the wrapper style onto every produced text_run. - Table blocks (31) are emitted as nested structures with cell blocks (32). """ import re, json, sys TABLE_BLOCK = 31 CELL_BLOCK = 32 TEXT_BLOCK = 2 CODE_BLOCK = 14 PREFIX = 'ecc_' # 全局表格计数器:每个表格用唯一 ID 前缀,避免多表格时 block_id 冲突 _table_counter = 0 LANG_CODES = { '': 1, 'text': 1, 'plain': 1, 'bash': 7, 'sh': 7, 'shell': 7, 'python': 11, 'py': 11, 'go': 5, 'golang': 5, 'js': 12, 'javascript': 12, 'json': 14, 'yaml': 33, 'yml': 33, 'sql': 25, } def lang_code(lang): """Map markdown code fence language to Feishu code block language id.""" return LANG_CODES.get(lang.lower(), 1) # -------------------------------------------------------------------------- # Inline state machine # -------------------------------------------------------------------------- # States of the inline scanner. ST_TEXT = 'TEXT' # plain text / default ST_LINK_TEXT = 'LINK_TEXT' # inside [ ... ] of a link ST_LINK_URL = 'LINK_URL' # inside ( ... ) of a link _INLINE_OPENER = re.compile(r'\*\*|__|\*|`|\[') _INLINE_TOKENS = { '**': 'bold', '__': 'bold', '`': 'code', } def _merge_style(elements, style): """Apply a wrapper style (bold/italic/code) to every produced text_run.""" for el in elements: if 'text_run' in el: el['text_run']['text_element_style'].update(style) return elements def scan_inline(text): """Inline state machine -> list of text_run elements. Handles nesting by recursion: when we hit a bold/italic/code opener we recurse into the inner span, then merge the wrapper style onto the result. Links are handled with an explicit two-phase state (LINK_TEXT then LINK_URL), so `[title](url)` never leaks raw markdown. """ elements = [] pos = 0 n = len(text) def emit(content, style): if content: elements.append({'text_run': {'content': content, 'text_element_style': style}}) while pos < n: ch = text[pos] # --- opener: bold --- if text.startswith('**', pos) or text.startswith('__', pos): width = 2 opener = text[pos:pos + width] end = text.find(opener, pos + width) if end != -1: inner = scan_inline(text[pos + width:end]) _merge_style(inner, {'bold': True}) elements.extend(inner) pos = end + width continue # unmatched -> literal emit(opener, {}) pos += width continue # --- opener: single-star italic --- if ch == '*': end = text.find('*', pos + 1) if end != -1 and not text.startswith('**', pos): inner = scan_inline(text[pos + 1:end]) _merge_style(inner, {'italic': True}) elements.extend(inner) pos = end + 1 continue emit('*', {}) pos += 1 continue # --- opener: inline code --- if ch == '`': end = text.find('`', pos + 1) if end != -1: emit(text[pos + 1:end], {'inline_code': True}) pos = end + 1 continue emit('`', {}) pos += 1 continue # --- link: [text](url) via two-phase state machine --- if ch == '[': # collect link text until matching ']' i = pos + 1 depth = 1 txt = [] while i < n and depth > 0: if text[i] == '[': depth += 1 elif text[i] == ']': depth -= 1 if depth == 0: break txt.append(text[i]) i += 1 if depth == 0 and i + 1 < n and text[i + 1] == '(': # collect url until matching ')' j = i + 2 url_end = text.find(')', j) if url_end != -1: url = text[j:url_end] link_text = ''.join(txt) elements.append({'text_run': { 'content': link_text, 'text_element_style': {'link': {'url': url}}, }}) pos = url_end + 1 continue # not a valid link -> literal '[' emit('[', {}) pos += 1 continue # --- plain text until next inline opener --- m = _INLINE_OPENER.search(text, pos) if m: end = m.start() if end > pos: emit(text[pos:end], {}) pos = end else: emit(text[pos:], {}) break return elements # -------------------------------------------------------------------------- # Table helpers (kept from original, table block-level parsing) # -------------------------------------------------------------------------- def is_sep_line(line): s = line.strip() if not s.startswith('|') or not s.endswith('|'): return False parts = s[1:-1].split('|') return all(re.match(r'^[\s\-:]+$', p) for p in parts) def is_table_row(line): s = line.strip() return s.startswith('|') and s.endswith('|') and '|' in s[1:-1] def split_row(line): return [c.strip() for c in line.strip()[1:-1].split('|')] def parse_table(lines, start_idx): global _table_counter i = start_idx headers = [] has_header = False if i >= len(lines) or not is_table_row(lines[i]): return None, start_idx first_cells = split_row(lines[i]) if not first_cells: return None, start_idx i += 1 if i < len(lines) and is_sep_line(lines[i]): headers = first_cells has_header = True i += 1 data_rows = [] while i < len(lines) and is_table_row(lines[i]): row = split_row(lines[i]) if row: data_rows.append(row) i += 1 if has_header: data_rows.insert(0, headers) else: data_rows.insert(0, first_cells) if not data_rows: return None, start_idx num_rows = len(data_rows) num_cols = max(len(r) for r in data_rows) blocks = [] table_id = f'{PREFIX}t{_table_counter}' _table_counter += 1 cell_ids = [] table_block = { 'block_id': table_id, 'block_type': TABLE_BLOCK, 'table': { 'property': { 'row_size': num_rows, 'column_size': num_cols, 'header_row': has_header } }, 'children': [], '__table': True, '__rows': num_rows, '__cols': num_cols } for row_idx, row in enumerate(data_rows): for col_idx in range(num_cols): cell_text = row[col_idx] if col_idx < len(row) else '' cell_id = f'{table_id}r{row_idx}c{col_idx}' text_id = f'{table_id}r{row_idx}c{col_idx}t' cell_ids.append(cell_id) is_header_cell = has_header and row_idx == 0 elements = scan_inline(cell_text) if not elements: elements = [{'text_run': {'content': cell_text, 'text_element_style': {}}}] if is_header_cell: for el in elements: if 'text_run' in el: el['text_run']['text_element_style']['bold'] = True text_block = { 'block_id': text_id, 'block_type': TEXT_BLOCK, 'text': {'elements': elements, 'style': {}}, 'children': [] } cell_block = { 'block_id': cell_id, 'block_type': CELL_BLOCK, 'table_cell': {}, 'children': [text_id] } table_block['children'].append(cell_id) blocks.append(cell_block) blocks.append(text_block) blocks.insert(0, table_block) return blocks, i # -------------------------------------------------------------------------- # Block-level state machine (line-by-line) # -------------------------------------------------------------------------- def md_to_blocks(md_text): """Convert Markdown text to Feishu Block JSON array.""" lines = md_text.strip().split('\n') blocks = [] i = 0 while i < len(lines): line = lines[i] # Fenced code block: ```lang ... ``` fence = re.match(r'^```(\w*)\s*$', line.strip()) if fence: lang = fence.group(1) code_lines = [] i += 1 while i < len(lines) and not lines[i].strip().startswith('```'): code_lines.append(lines[i]) i += 1 i += 1 # skip closing fence code_text = '\n'.join(code_lines) blocks.append({ 'block_type': CODE_BLOCK, 'code': { 'elements': [{'text_run': {'content': code_text, 'text_element_style': {}}}], 'style': {'language': lang_code(lang)} } }) continue # Empty line if not line.strip(): if i + 1 < len(lines) and lines[i + 1].strip() and not lines[i + 1].strip().startswith(('#', '-', '*', '`', '---', '|')): blocks.append({'block_type': 2, 'text': {'elements': [{'text_run': {'content': '', 'text_element_style': {}}}], 'style': {}}}) i += 1 continue # Divider if line.strip() in ('---', '***'): blocks.append({'block_type': 22, 'divider': {}}) i += 1 continue # Table detection if '|' in line: tbl, ni = parse_table(lines, i) if tbl is not None: blocks.extend(tbl) i = ni continue # Headings h = re.match(r'^(#{1,3})\s+(.+)$', line) if h: level = len(h.group(1)) bt = 2 + level blocks.append({ 'block_type': bt, f'heading{level}': { 'elements': scan_inline(h.group(2)), 'style': {} } }) i += 1 continue # Quote quote = re.match(r'^>\s?(.+)$', line) if quote: blocks.append({ 'block_type': 15, 'quote': { 'elements': scan_inline(quote.group(1)), 'style': {} } }) i += 1 continue # Bullet list bullet = re.match(r'^[\-\*]\s+(.+)$', line) if bullet: blocks.append({ 'block_type': 12, 'bullet': { 'elements': scan_inline(bullet.group(1)), 'style': {} } }) i += 1 continue # Numbered list numbered = re.match(r'^\d+\.\s+(.+)$', line) if numbered: blocks.append({ 'block_type': 13, 'ordered': { 'elements': scan_inline(numbered.group(1)), 'style': {} } }) i += 1 continue # Regular text (paragraph) elements = scan_inline(line) blocks.append({ 'block_type': 2, 'text': { 'elements': elements if elements else [{'text_run': {'content': line, 'text_element_style': {}}}], 'style': {} } }) i += 1 return blocks if __name__ == '__main__': if len(sys.argv) > 1: with open(sys.argv[1]) as f: text = f.read() else: text = sys.stdin.read() blocks = md_to_blocks(text) print(json.dumps(blocks, indent=2, ensure_ascii=False))