311 lines
9.4 KiB
Python
311 lines
9.4 KiB
Python
#!/usr/bin/env python3
|
|
"""Markdown to Feishu Docx Block JSON converter.
|
|
Usage: python3 md_to_blocks.py < input.md > output.json
|
|
python3 md_to_blocks.py file.md
|
|
|
|
Supports: headings, lists, dividers, inline formatting, and markdown tables.
|
|
Table blocks (31) are output as nested structures with cell blocks (32).
|
|
"""
|
|
|
|
import re, json, sys
|
|
|
|
TABLE_BLOCK = 31
|
|
CELL_BLOCK = 32
|
|
TEXT_BLOCK = 2
|
|
CODE_BLOCK = 14
|
|
PREFIX = 'ecc_'
|
|
|
|
LANG_CODES = {
|
|
'': 1, 'text': 1, 'plain': 1,
|
|
'bash': 7, 'sh': 7, 'shell': 7,
|
|
'python': 11, 'py': 11,
|
|
'go': 5, 'golang': 5,
|
|
'js': 12, 'javascript': 12, 'json': 14,
|
|
'yaml': 33, 'yml': 33,
|
|
'sql': 25,
|
|
}
|
|
|
|
def lang_code(lang):
|
|
"""Map markdown code fence language to Feishu code block language id."""
|
|
return LANG_CODES.get(lang.lower(), 1)
|
|
|
|
def parse_inline(text):
|
|
"""Parse a line of inline markdown into text_run elements."""
|
|
elements = []
|
|
pos = 0
|
|
while pos < len(text):
|
|
m = re.match(r'\*\*(.+?)\*\*|__(.+?)__', text[pos:])
|
|
if m:
|
|
elements.append({'text_run': {'content': m.group(1), 'text_element_style': {'bold': True}}})
|
|
pos += len(m.group(0))
|
|
continue
|
|
m = re.match(r'\[(.+?)\]\((.+?)\)', text[pos:])
|
|
if m:
|
|
elements.append({'text_run': {'content': m.group(1), 'text_element_style': {'link': {'url': m.group(2)}}}})
|
|
pos += len(m.group(0))
|
|
continue
|
|
m = re.match(r'`(.+?)`', text[pos:])
|
|
if m:
|
|
elements.append({'text_run': {'content': m.group(1), 'text_element_style': {'inline_code': True}}})
|
|
pos += len(m.group(0))
|
|
continue
|
|
m = re.match(r'\*(.+?)\*', text[pos:])
|
|
if m:
|
|
elements.append({'text_run': {'content': m.group(1), 'text_element_style': {'italic': True}}})
|
|
pos += len(m.group(0))
|
|
continue
|
|
nxt = re.search(r'[\*\[`]', text[pos:])
|
|
if nxt:
|
|
end = pos + nxt.start()
|
|
if end > pos:
|
|
elements.append({'text_run': {'content': text[pos:end], 'text_element_style': {}}})
|
|
elif end == pos:
|
|
elements.append({'text_run': {'content': text[pos:pos+1], 'text_element_style': {}}})
|
|
pos += 1
|
|
continue
|
|
pos = end
|
|
else:
|
|
elements.append({'text_run': {'content': text[pos:], 'text_element_style': {}}})
|
|
break
|
|
return elements
|
|
|
|
def is_sep_line(line):
|
|
"""Check if a line is a markdown table separator, e.g. |---|:---:|---|"""
|
|
s = line.strip()
|
|
if not s.startswith('|') or not s.endswith('|'):
|
|
return False
|
|
parts = s[1:-1].split('|')
|
|
return all(re.match(r'^[\s\-:]+$', p) for p in parts)
|
|
|
|
def is_table_row(line):
|
|
"""Check if a line looks like a markdown table row."""
|
|
s = line.strip()
|
|
return s.startswith('|') and s.endswith('|') and '|' in s[1:-1]
|
|
|
|
def split_row(line):
|
|
return [c.strip() for c in line.strip()[1:-1].split('|')]
|
|
|
|
def parse_table(lines, start_idx):
|
|
"""Parse a markdown table. Returns (blocks, end_idx) or (None, start_idx)."""
|
|
i = start_idx
|
|
headers = []
|
|
has_header = False
|
|
|
|
# Look ahead: need at least 2 lines (header + separator) or 1 plain row
|
|
if i >= len(lines) or not is_table_row(lines[i]):
|
|
return None, start_idx
|
|
|
|
# First line: potential header
|
|
first_cells = split_row(lines[i])
|
|
if not first_cells:
|
|
return None, start_idx
|
|
|
|
i += 1
|
|
# Check for separator
|
|
if i < len(lines) and is_sep_line(lines[i]):
|
|
headers = first_cells
|
|
has_header = True
|
|
i += 1
|
|
|
|
# Collect data rows
|
|
data_rows = []
|
|
while i < len(lines) and is_table_row(lines[i]):
|
|
row = split_row(lines[i])
|
|
if row:
|
|
data_rows.append(row)
|
|
i += 1
|
|
|
|
# No header case: first line was a data row
|
|
if not has_header:
|
|
data_rows.insert(0, first_cells)
|
|
|
|
if not data_rows:
|
|
return None, start_idx
|
|
|
|
num_rows = len(data_rows)
|
|
num_cols = max(len(r) for r in data_rows)
|
|
|
|
# Build nested blocks
|
|
blocks = []
|
|
table_id = f'{PREFIX}t0'
|
|
cell_ids = []
|
|
|
|
# Table container
|
|
table_block = {
|
|
'block_id': table_id,
|
|
'block_type': TABLE_BLOCK,
|
|
'table': {
|
|
'property': {
|
|
'row_size': num_rows,
|
|
'column_size': num_cols,
|
|
'header_row': has_header
|
|
}
|
|
},
|
|
'children': [],
|
|
'__table': True,
|
|
'__rows': num_rows,
|
|
'__cols': num_cols
|
|
}
|
|
|
|
for row_idx, row in enumerate(data_rows):
|
|
for col_idx in range(num_cols):
|
|
cell_text = row[col_idx] if col_idx < len(row) else ''
|
|
cell_id = f'{PREFIX}t0r{row_idx}c{col_idx}'
|
|
text_id = f'{PREFIX}t0r{row_idx}c{col_idx}t'
|
|
cell_ids.append(cell_id)
|
|
|
|
is_header_cell = has_header and row_idx == 0
|
|
elements = parse_inline(cell_text)
|
|
if not elements:
|
|
elements = [{'text_run': {'content': cell_text, 'text_element_style': {}}}]
|
|
if is_header_cell:
|
|
for el in elements:
|
|
if 'text_run' in el:
|
|
el['text_run']['text_element_style']['bold'] = True
|
|
|
|
text_block = {
|
|
'block_id': text_id,
|
|
'block_type': TEXT_BLOCK,
|
|
'text': {'elements': elements, 'style': {}},
|
|
'children': []
|
|
}
|
|
cell_block = {
|
|
'block_id': cell_id,
|
|
'block_type': CELL_BLOCK,
|
|
'table_cell': {},
|
|
'children': [text_id]
|
|
}
|
|
|
|
table_block['children'].append(cell_id)
|
|
blocks.append(cell_block)
|
|
blocks.append(text_block)
|
|
|
|
blocks.insert(0, table_block)
|
|
return blocks, i
|
|
|
|
def md_to_blocks(md_text):
|
|
"""Convert Markdown text to Feishu Block JSON array."""
|
|
lines = md_text.strip().split('\n')
|
|
blocks = []
|
|
i = 0
|
|
while i < len(lines):
|
|
line = lines[i]
|
|
|
|
# Fenced code block: ```lang ... ```
|
|
fence = re.match(r'^```(\w*)\s*$', line.strip())
|
|
if fence:
|
|
lang = fence.group(1)
|
|
code_lines = []
|
|
i += 1
|
|
while i < len(lines) and not lines[i].strip().startswith('```'):
|
|
code_lines.append(lines[i])
|
|
i += 1
|
|
i += 1 # skip closing fence
|
|
code_text = '\n'.join(code_lines)
|
|
blocks.append({
|
|
'block_type': CODE_BLOCK,
|
|
'code': {
|
|
'elements': [{'text_run': {'content': code_text, 'text_element_style': {}}}],
|
|
'style': {'language': lang_code(lang)}
|
|
}
|
|
})
|
|
continue
|
|
|
|
# Empty line
|
|
if not line.strip():
|
|
if i + 1 < len(lines) and lines[i + 1].strip() and not lines[i + 1].strip().startswith(('#', '-', '*', '`', '---', '|')):
|
|
blocks.append({'block_type': 2, 'text': {'elements': [{'text_run': {'content': '', 'text_element_style': {}}}], 'style': {}}})
|
|
i += 1
|
|
continue
|
|
|
|
# Divider
|
|
if line.strip() in ('---', '***'):
|
|
blocks.append({'block_type': 22, 'divider': {}})
|
|
i += 1
|
|
continue
|
|
|
|
# Table detection
|
|
if '|' in line:
|
|
tbl, ni = parse_table(lines, i)
|
|
if tbl is not None:
|
|
blocks.extend(tbl)
|
|
i = ni
|
|
continue
|
|
|
|
# Headings
|
|
h = re.match(r'^(#{1,3})\s+(.+)$', line)
|
|
if h:
|
|
level = len(h.group(1))
|
|
bt = 2 + level
|
|
blocks.append({
|
|
'block_type': bt,
|
|
f'heading{level}': {
|
|
'elements': parse_inline(h.group(2)),
|
|
'style': {}
|
|
}
|
|
})
|
|
i += 1
|
|
continue
|
|
|
|
# Quote
|
|
quote = re.match(r'^>\s?(.+)$', line)
|
|
if quote:
|
|
blocks.append({
|
|
'block_type': 15,
|
|
'quote': {
|
|
'elements': parse_inline(quote.group(1)),
|
|
'style': {}
|
|
}
|
|
})
|
|
i += 1
|
|
continue
|
|
|
|
# Bullet list
|
|
bullet = re.match(r'^[\-\*]\s+(.+)$', line)
|
|
if bullet:
|
|
blocks.append({
|
|
'block_type': 12,
|
|
'bullet': {
|
|
'elements': parse_inline(bullet.group(1)),
|
|
'style': {}
|
|
}
|
|
})
|
|
i += 1
|
|
continue
|
|
|
|
# Numbered list
|
|
numbered = re.match(r'^\d+\.\s+(.+)$', line)
|
|
if numbered:
|
|
blocks.append({
|
|
'block_type': 13,
|
|
'ordered': {
|
|
'elements': parse_inline(numbered.group(1)),
|
|
'style': {}
|
|
}
|
|
})
|
|
i += 1
|
|
continue
|
|
|
|
# Regular text
|
|
elements = parse_inline(line)
|
|
blocks.append({
|
|
'block_type': 2,
|
|
'text': {
|
|
'elements': elements if elements else [{'text_run': {'content': line, 'text_element_style': {}}}],
|
|
'style': {}
|
|
}
|
|
})
|
|
i += 1
|
|
|
|
return blocks
|
|
|
|
if __name__ == '__main__':
|
|
if len(sys.argv) > 1:
|
|
with open(sys.argv[1]) as f:
|
|
text = f.read()
|
|
else:
|
|
text = sys.stdin.read()
|
|
|
|
blocks = md_to_blocks(text)
|
|
print(json.dumps(blocks, indent=2, ensure_ascii=False))
|