Files
md2pdf/md2doc.py
T
sebastian 5a62572f55 feat: --footer/--header with {page}/{pages} placeholders, center-aligned
md2pdf.py:
- Expose --footer and --header (was dead code before)
- Remove --forms flag (B2 fix — weasyprint 68.1 silently drops <input>)
- Substitution support: {page} → counter(page), {pages} → counter(pages)
- Center-aligned via @bottom-center / @top-center

md2doc.py:
- Add --footer and --header support via Word field codes (PAGE, NUMPAGES)
- Center-aligned paragraph in footer/header
- Same {page}/{pages} substitution API as md2pdf

Both tools:
- GFM task list support: - [ ] → ☐, - [x]/- [X] → ☑ (synced _bulletize)
- Help text explains placeholder syntax with examples
2026-07-28 11:02:23 +03:00

712 lines
24 KiB
Python

#!/usr/bin/env python3
"""
Convert markdown files to DOCX (Microsoft Word).
Usage:
md2doc.py input.md # Creates input.docx in same directory
md2doc.py input.md -o output.docx # Specify output file
md2doc.py docs/ # Convert all .md files in directory
md2doc.py docs/ -o docx_output/
md2doc.py input.md --style mono # Use monospace style
Requires: pip install python-docx markdown beautifulsoup4 lxml
"""
import argparse
import sys
from pathlib import Path
import markdown
from bs4 import BeautifulSoup
from docx import Document
from docx.shared import Pt, RGBColor, Cm
from docx.oxml.ns import qn
from docx.oxml import OxmlElement
STYLES = {
"elegant": {
"font": "Helvetica", "size": 11, "heading_color": "1a1a1a",
"link_color": "2c2c2c", "table_header_fill": "e0e0e0",
"code_fill": "f8f8f8", "blockquote_fill": "f8f8f8",
},
"report": {
"font": "Helvetica", "size": 10, "heading_color": "1e3a8a",
"link_color": "1e3a8a", "table_header_fill": "1e3a8a",
"code_fill": "f7f8fa", "blockquote_fill": "f7f8fa",
},
"default": {
"font": "Helvetica", "size": 10, "heading_color": "1a1a1a",
"link_color": "0066cc", "table_header_fill": "f5f5f5",
"code_fill": "f5f5f5", "blockquote_fill": "fafafa",
},
"dark": {
"font": "Helvetica", "size": 11, "heading_color": "818cf8",
"link_color": "818cf8", "table_header_fill": "4f46e5",
"code_fill": "1f2937", "blockquote_fill": "1f2937",
},
"mono": {
"font": "Menlo", "size": 10, "heading_color": "000000",
"link_color": "000000", "table_header_fill": "f0f0f0",
"code_fill": "f5f5f5", "blockquote_fill": "f5f5f5",
},
}
# Common rules applied to all styles via deep-merge.
# Note: python-docx doesn't expose orphans/widows directly via API;
# keep_with_next IS available via paragraph_format.keep_with_next = True.
# page_break_before_h1 is per-style flag (False by default; can be overridden).
COMMON = {
"keep_with_next_headings": True, # applied to H1-H4
"page_break_before_h1": False, # can be overridden per style
"table_cell_valign": "top",
}
def _bulletize(text: str) -> str:
"""Turn markdown list markers into bullets, skipping code/HTML blocks.
Supports GFM task lists:
- [ ] → ☐ (empty checkbox, U+2610)
- [x] → ☑ (checked checkbox, U+2611)
- [X] → ☑
- → • (regular bullet)
Copied from md2pdf.py (kept in sync).
"""
import re
out, in_fence, in_html = [], False, False
for line in text.split("\n"):
stripped = line.lstrip()
if stripped.startswith("```") or stripped.startswith("~~~"):
in_fence = not in_fence
elif not in_fence:
if re.match(r"^<(table|div|section|figure)\b", stripped, re.I):
in_html = True
elif re.match(r"^</(table|div|section|figure)>", stripped, re.I):
in_html = False
if not in_fence and not in_html:
# GFM task list — checked
if line.startswith("- [x] ") or line.startswith("- [X] "):
line = "☑ " + line[6:]
# GFM task list — unchecked
elif line.startswith("- [ ] "):
line = "☐ " + line[6:]
# Regular bullet
elif line.startswith("- "):
line = "• " + line[2:]
out.append(line)
return "\n".join(out)
# ---------------------------------------------------------------------------
# XML manipulation helpers
# Spec ref: docs/specs/2026-07-27-md2doc-design.md §4.1 — verified empirically
# that python-docx 1.2 has no public API for borders/shading/numbering, so
# we manipulate the OOXML directly via OxmlElement.
# ---------------------------------------------------------------------------
def add_paragraph_bottom_border(paragraph, color: str = "000000", size: str = "6"):
"""Attach a bottom border to a paragraph (e.g., Heading 1 underline).
Spec §4.1 H1: SURVIVED — verified via XML inspection.
"""
pPr = paragraph._p.get_or_add_pPr()
pBdr = pPr.find(qn('w:pBdr'))
if pBdr is None:
pBdr = OxmlElement('w:pBdr')
pPr.append(pBdr)
bottom = OxmlElement('w:bottom')
bottom.set(qn('w:val'), 'single')
bottom.set(qn('w:sz'), size)
bottom.set(qn('w:space'), '1')
bottom.set(qn('w:color'), color)
pBdr.append(bottom)
def add_paragraph_shading(paragraph, fill: str):
"""Apply background shading to a paragraph (code, blockquote).
Spec §4.1 H2: SURVIVED.
"""
pPr = paragraph._p.get_or_add_pPr()
shd = OxmlElement('w:shd')
shd.set(qn('w:val'), 'clear')
shd.set(qn('w:color'), 'auto')
shd.set(qn('w:fill'), fill)
pPr.append(shd)
def add_cell_shading(cell, fill: str):
"""Apply background shading to a table cell (header row).
Spec §4.1 H4: SURVIVED. Inserts w:shd inside w:tcPr.
"""
tcPr = cell._tc.get_or_add_tcPr()
shd = OxmlElement('w:shd')
shd.set(qn('w:val'), 'clear')
shd.set(qn('w:color'), 'auto')
shd.set(qn('w:fill'), fill)
tcPr.append(shd)
def register_multilevel_numbering(doc, levels: int = 3, kind: str = "bullet") -> int:
"""Register a multi-level numbering definition in the document.
Returns the numId to reference from paragraphs.
Spec §4.1 H3 Route B: SURVIVED at XML level. Mandatory because style-name
approach ('List Bullet 2') produces flat lists in Word (single-level
abstractNums in default template).
Args:
doc: python-docx Document
levels: number of nesting levels to define (1=flat, 3=three levels deep)
kind: 'bullet' or 'decimal' for numbered lists
"""
numbering = doc.part.numbering_part.element
existing_abstract_ids = [
int(an.get(qn('w:abstractNumId')))
for an in numbering.findall(qn('w:abstractNum'))
]
abstract_num_id = max(existing_abstract_ids, default=-1) + 1
abstract_num = OxmlElement('w:abstractNum')
abstract_num.set(qn('w:abstractNumId'), str(abstract_num_id))
multi_level = OxmlElement('w:multiLevelType')
multi_level.set(qn('w:val'), 'hybridMultilevel')
abstract_num.append(multi_level)
for ilvl in range(levels):
lvl = OxmlElement('w:lvl')
lvl.set(qn('w:ilvl'), str(ilvl))
lvl.set(qn('w:tplc'), '0409000F' if kind == "bullet" else '04070001')
start = OxmlElement('w:start')
start.set(qn('w:val'), '1')
lvl.append(start)
numFmt = OxmlElement('w:numFmt')
numFmt.set(qn('w:val'), 'bullet' if kind == "bullet" else 'decimal')
lvl.append(numFmt)
lvl_text = OxmlElement('w:lvlText')
lvl_text.set(qn('w:val'), '•' if kind == "bullet" else f'%{ilvl + 1}.')
lvl.append(lvl_text)
lvl_jc = OxmlElement('w:lvlJc')
lvl_jc.set(qn('w:val'), 'left')
lvl.append(lvl_jc)
pPr = OxmlElement('w:pPr')
ind = OxmlElement('w:ind')
ind.set(qn('w:left'), str(720 * (ilvl + 1)))
ind.set(qn('w:hanging'), '360')
pPr.append(ind)
lvl.append(pPr)
abstract_num.append(lvl)
numbering.insert(0, abstract_num)
existing_num_ids = [
int(n.get(qn('w:numId')))
for n in numbering.findall(qn('w:num'))
]
num_id = max(existing_num_ids, default=0) + 1
num = OxmlElement('w:num')
num.set(qn('w:numId'), str(num_id))
abstract_ref = OxmlElement('w:abstractNumId')
abstract_ref.set(qn('w:val'), str(abstract_num_id))
num.append(abstract_ref)
numbering.append(num)
return num_id
def attach_list_numbering(paragraph, num_id: int, ilvl: int):
"""Attach numbering to a paragraph at the given indent level.
Spec §4.1 H3 Route B. Must be paired with register_multilevel_numbering.
"""
pPr = paragraph._p.get_or_add_pPr()
existing = pPr.find(qn('w:numPr'))
if existing is not None:
pPr.remove(existing)
numPr = OxmlElement('w:numPr')
ilvl_el = OxmlElement('w:ilvl')
ilvl_el.set(qn('w:val'), str(ilvl))
numId_el = OxmlElement('w:numId')
numId_el.set(qn('w:val'), str(num_id))
numPr.append(ilvl_el)
numPr.append(numId_el)
pPr.append(numPr)
# ---------------------------------------------------------------------------
# Element renderers (walk BeautifulSoup DOM → emit docx elements)
# ---------------------------------------------------------------------------
def _warn_skip_image(node):
"""Warn to stderr when an <img> is encountered (out of scope per spec §10).
Defined early so _render_paragraph (used in Task 4) and convert_md_to_doc
(Task 7) can both reference it without forward-declaration issues.
"""
src = node.get("src", "(no src)")
print(f"Warning: image skipped (out of scope): {src}", file=sys.stderr)
def _add_run_shading(run, fill: str):
"""Apply background shading to a run (for inline <code> highlight).
Inserts w:shd inside w:rPr.
"""
rPr = run._r.get_or_add_rPr()
shd = OxmlElement('w:shd')
shd.set(qn('w:val'), 'clear')
shd.set(qn('w:color'), 'auto')
shd.set(qn('w:fill'), fill)
rPr.append(shd)
def _render_heading(doc, node, level: int, style: dict):
"""Render <h1>-<h4> as Word Heading paragraphs.
H1 also gets a bottom paragraph border (per md2pdf pattern).
Applies COMMON['keep_with_next_headings'] so headings don't orphan at page bottom.
Applies COMMON['page_break_before_h1'] if True (style-overridable).
"""
text = node.get_text(strip=True)
p = doc.add_heading(level=level)
p.add_run(text)
# Apply style-specific color
color_hex = style.get("heading_color", "000000")
for run in p.runs:
run.font.color.rgb = RGBColor.from_string(color_hex)
# Wire COMMON config: keep heading with next paragraph
if COMMON.get("keep_with_next_headings", True):
p.paragraph_format.keep_with_next = True
# H1 page-break-before if style overrides COMMON default
if level == 1 and style.get("page_break_before_h1", COMMON.get("page_break_before_h1", False)):
p.paragraph_format.page_break_before = True
if level == 1:
add_paragraph_bottom_border(p, color=color_hex, size="12")
def _render_paragraph(doc, node, style: dict):
"""Render <p> as a Word paragraph, preserving inline formatting (strong, em, code, a).
Also handles <img> children — prints stderr warning and skips (spec §10).
"""
p = doc.add_paragraph()
code_fill = style.get("code_fill", "f5f5f5")
def _add_runs(element):
for child in element.children:
if isinstance(child, str):
p.add_run(child)
elif child.name == "strong" or child.name == "b":
r = p.add_run(child.get_text())
r.bold = True
elif child.name == "em" or child.name == "i":
r = p.add_run(child.get_text())
r.italic = True
elif child.name == "code":
r = p.add_run(child.get_text())
r.font.name = "Menlo"
_add_run_shading(r, fill=code_fill)
elif child.name == "a":
r = p.add_run(child.get_text())
r.underline = True
r.font.color.rgb = RGBColor.from_string(style.get("link_color", "0000EE"))
elif child.name == "img":
_warn_skip_image(child)
else:
# Unknown inline — recurse
_add_runs(child)
_add_runs(node)
def _render_list(doc, node, style: dict, level: int = 0, num_state: dict = None):
"""Render <ul>/<ol> recursively. Uses register_multilevel_numbering + attach_list_numbering.
`num_state` carries the registered numId across recursion (one numbering definition
per top-level list, all children share it).
Spec §4.2: mandatory manual numbering — List Bullet 2/3 styles don't work.
"""
is_ordered = node.name == "ol"
kind = "decimal" if is_ordered else "bullet"
# Top-level call: register fresh numbering definition
if num_state is None:
num_id = register_multilevel_numbering(doc, levels=3, kind=kind)
num_state = {"num_id": num_id}
for li in node.find_all("li", recursive=False):
# The li's direct text (excluding nested lists)
nested_lists = []
for child in li.children:
if hasattr(child, "name") and child.name in ("ul", "ol"):
nested_lists.append(child)
# Get text content of the li (excluding nested list text)
text_parts = []
for child in li.children:
if hasattr(child, "name") and child.name in ("ul", "ol"):
continue
if isinstance(child, str):
text_parts.append(child.strip())
else:
text_parts.append(child.get_text(strip=True))
text = " ".join(p for p in text_parts if p)
p = doc.add_paragraph(text)
attach_list_numbering(p, num_id=num_state["num_id"], ilvl=level)
# Recurse into nested lists
for nested in nested_lists:
_render_list(doc, nested, style=style, level=level + 1, num_state=num_state)
def _render_table(doc, node, style: dict):
"""Render <table> with header row shading."""
header_fill = style.get("table_header_fill", "1e3a8a")
rows = node.find_all("tr")
if not rows:
return
cols = max(len(row.find_all(["th", "td"])) for row in rows)
table = doc.add_table(rows=len(rows), cols=cols)
table.style = 'Table Grid'
for i, row in enumerate(rows):
cells = row.find_all(["th", "td"])
is_header = row.parent.name == "thead" or any(c.name == "th" for c in cells)
for j, cell in enumerate(cells):
if j >= cols:
break
tc = table.rows[i].cells[j]
tc.text = cell.get_text(strip=True)
if is_header:
add_cell_shading(tc, fill=header_fill)
def _render_pre(doc, node, style: dict):
"""Render <pre> (code block) — shaded background + monospace font.
Preserves newlines via add_break() between lines (Word ignores literal \\n in runs).
Leading whitespace on each line is preserved as literal spaces (monospace font
keeps alignment). Trailing empty line from markdown's final \\n is dropped.
"""
code_fill = style.get("code_fill", "f7f8fa")
text = node.get_text()
lines = text.split("\n")
while lines and lines[-1] == "":
lines.pop()
p = doc.add_paragraph()
for i, line in enumerate(lines):
if i > 0:
p.add_run().add_break()
run = p.add_run(line)
run.font.name = "Menlo"
run.font.size = Pt(8)
add_paragraph_shading(p, fill=code_fill)
def _render_blockquote(doc, node, style: dict):
"""Render <blockquote> — left indent + italic + shading."""
quote_fill = style.get("blockquote_fill", "f7f8fa")
text = node.get_text(strip=True)
p = doc.add_paragraph()
p.paragraph_format.left_indent = Cm(1)
run = p.add_run(text)
run.italic = True
add_paragraph_shading(p, fill=quote_fill)
def _render_hr(doc, node, style: dict):
"""Render <hr> as empty paragraph with bottom border."""
p = doc.add_paragraph()
add_paragraph_bottom_border(p, color="d8d8d8", size="6")
def _set_section_text(section, location: str, text: str, template: str):
"""Apply footer/header text to a docx Section.
location: 'footer' or 'header'.
text: literal text (may be empty — clears the field).
template: original template with {page}/{pages} placeholders (used to detect
whether to inject field codes).
For docx, page numbering uses Word field codes:
{page} → PAGE field
{pages} → NUMPAGES field
Other text is literal.
"""
import re
if location == 'footer':
target = section.footer
else:
target = section.header
# Clear default empty paragraph content (python-docx auto-creates one)
if not target.paragraphs:
target.add_paragraph()
p = target.paragraphs[0]
# Center-align the footer/header paragraph
from docx.enum.text import WD_ALIGN_PARAGRAPH
p.alignment = WD_ALIGN_PARAGRAPH.CENTER
# Clear existing runs
for run in list(p.runs):
run.text = ""
# Split template on placeholders
parts = re.split(r'(\{page\}|\{pages\})', template)
for part in parts:
if part == '{page}':
# Insert PAGE field code
_add_field_code(p, 'PAGE')
elif part == '{pages}':
_add_field_code(p, 'NUMPAGES')
elif part:
p.add_run(part)
def _add_field_code(paragraph, field: str):
"""Insert a Word field code (PAGE, NUMPAGES) into a paragraph via XML."""
run = paragraph.add_run()
# Build the field structure: <w:fldChar begin/> <w:instrText>PAGE</w:instrText> <w:fldChar end/>
fld_begin = OxmlElement('w:fldChar')
fld_begin.set(qn('w:fldCharType'), 'begin')
run._r.append(fld_begin)
instr_run = paragraph.add_run()
instr = OxmlElement('w:instrText')
instr.set(qn('xml:space'), 'preserve')
instr.text = f' {field} '
instr_run._r.append(instr)
sep_run = paragraph.add_run()
fld_sep = OxmlElement('w:fldChar')
fld_sep.set(qn('w:fldCharType'), 'separate')
sep_run._r.append(fld_sep)
# Placeholder for the rendered value (Word will compute on open)
placeholder_run = paragraph.add_run("?")
# Mark as dirty so Word recalculates
fld_dirty = OxmlElement('w:fldChar')
fld_dirty.set(qn('w:fldCharType'), 'end')
placeholder_run._r.append(fld_dirty)
# Set small font for footer/header runs
for r in [run, instr_run, sep_run, placeholder_run]:
r.font.size = Pt(7.5)
r.font.color.rgb = RGBColor.from_string("999999")
def convert_md_to_doc(md_file: Path, output_file: Path, style: str = "elegant",
footer: str = "", header: str = "") -> Path:
"""Convert a single markdown file to DOCX."""
import re
style_cfg = STYLES.get(style, STYLES["elegant"])
with open(md_file, 'r', encoding='utf-8') as f:
md_content = f.read()
# Reuse _bulletize and BLANK_MARKER logic from md2pdf (copied, not imported)
md_content = _bulletize(md_content)
BLANK_MARKER = '<!--blank-line-->'
md_content = re.sub(
r'\n{3,}',
lambda m: '\n\n' + (BLANK_MARKER + '\n\n') * (len(m.group(0)) - 2),
md_content
)
html_content = markdown.markdown(
md_content,
extensions=[
'markdown.extensions.tables',
'markdown.extensions.fenced_code',
'markdown.extensions.toc',
'markdown.extensions.attr_list',
'markdown.extensions.md_in_html',
'markdown.extensions.nl2br',
]
)
html_content = html_content.replace(
BLANK_MARKER, '<div style="height: 0.8em;"></div>'
)
soup = BeautifulSoup(html_content, "lxml")
doc = Document()
# Walk top-level elements in <body>
body = soup.find("body") or soup
for element in body.children:
if not hasattr(element, "name") or element.name is None:
continue
name = element.name
if name in ("h1", "h2", "h3", "h4"):
_render_heading(doc, element, level=int(name[1]), style=style_cfg)
elif name == "p":
_render_paragraph(doc, element, style=style_cfg)
elif name in ("ul", "ol"):
_render_list(doc, element, style=style_cfg)
elif name == "table":
_render_table(doc, element, style=style_cfg)
elif name == "pre":
_render_pre(doc, element, style=style_cfg)
elif name == "blockquote":
_render_blockquote(doc, element, style=style_cfg)
elif name == "hr":
_render_hr(doc, element, style=style_cfg)
elif name == "img":
_warn_skip_image(element)
elif name == "figure":
img = element.find("img")
if img:
_warn_skip_image(img)
else:
_render_paragraph(doc, element, style=style_cfg)
# else: silently skip unknown elements (divs, spans, etc.)
# Apply footer/header to all sections (typically just one)
if footer or header:
for section in doc.sections:
if footer:
_set_section_text(section, 'footer', footer, footer)
if header:
_set_section_text(section, 'header', header, header)
doc.save(output_file)
return output_file
def main():
parser = argparse.ArgumentParser(
description='Convert Markdown files to DOCX',
formatter_class=argparse.RawDescriptionHelpFormatter,
epilog="""
Examples:
%(prog)s document.md Convert single file (elegant style)
%(prog)s document.md -o report.docx Convert with custom output name
%(prog)s docs/ Convert all .md files in directory
%(prog)s doc.md --style mono Use monospace font
%(prog)s doc.md --style dark Use dark theme
%(prog)s contract.md --footer "Confidential" Add footer (static text)
%(prog)s doc.md --header "Project X" Add header (static text)
%(prog)s doc.md --footer "Page {page}" Add footer with page counter
%(prog)s doc.md --footer "Page {page} of {pages}" Page counter with total
Footer/Header placeholders (only when --footer/--header is used):
{page} current page number (e.g. "3")
{pages} total page count (e.g. "12")
Literal text is rendered as-is. Mix freely: "Page {page} of {pages} — Confidential"
Available styles: elegant (default), default, dark, mono, report
"""
)
parser.add_argument('input', help='Input markdown file or directory')
parser.add_argument('-o', '--output', help='Output DOCX file or directory')
parser.add_argument('--style', choices=list(STYLES.keys()), default='elegant',
help='Style template (default: elegant)')
parser.add_argument('-q', '--quiet', action='store_true', help='Suppress output')
parser.add_argument('--footer',
help='Custom footer text (center-aligned). '
'Supports {page} and {pages} placeholders: '
'"Page {page} of {pages}"')
parser.add_argument('--header',
help='Custom header text (center-aligned). '
'Supports {page} and {pages} placeholders: '
'"Section X — page {page}"')
args = parser.parse_args()
input_path = Path(args.input).resolve()
if not input_path.exists():
print(f"Error: '{args.input}' not found", file=sys.stderr)
sys.exit(1)
# Determine files to convert (same logic as md2pdf.py)
if input_path.is_file():
if not input_path.suffix.lower() == '.md':
print(f"Warning: '{input_path.name}' doesn't have .md extension", file=sys.stderr)
files = [input_path]
if args.output:
output_path = Path(args.output).resolve()
if output_path.suffix.lower() == '.docx':
outputs = [output_path]
else:
output_path.mkdir(parents=True, exist_ok=True)
outputs = [output_path / (input_path.stem + '.docx')]
else:
outputs = [input_path.with_suffix('.docx')]
else: # Directory
files = sorted(input_path.glob('**/*.md'))
if not files:
print(f"No .md files found in '{args.input}'", file=sys.stderr)
sys.exit(1)
if args.output:
output_dir = Path(args.output).resolve()
output_dir.mkdir(parents=True, exist_ok=True)
else:
output_dir = input_path
outputs = [output_dir / (f.stem + '.docx') for f in files]
# Convert files
success = 0
errors = 0
for md_file, docx_file in zip(files, outputs):
try:
if not args.quiet:
print(f"Converting: {md_file.name} -> {docx_file.name}...", end=' ', flush=True)
docx_file.parent.mkdir(parents=True, exist_ok=True)
convert_md_to_doc(md_file, docx_file, args.style,
footer=args.footer or "",
header=args.header or "")
if not args.quiet:
size_kb = docx_file.stat().st_size / 1024
print(f"OK ({size_kb:.1f} KB)")
success += 1
except Exception as e:
if not args.quiet:
print(f"FAILED: {e}")
errors += 1
if not args.quiet and len(files) > 1:
print(f"\nDone: {success} converted, {errors} failed")
sys.exit(0 if errors == 0 else 1)
if __name__ == "__main__":
main()