BlogWriter / pipeline /docx_builder.py
vivekchakraverty's picture
Add target word count control and AEO citation guidelines
2559985
Raw
History Blame Contribute Delete
5.75 kB
"""Step 8: assemble the Markdown post + images + captions into a .docx file."""
from __future__ import annotations
import re
from pathlib import Path
from typing import List
from docx import Document
from docx.enum.text import WD_ALIGN_PARAGRAPH
from docx.shared import Inches, Pt, RGBColor
_IMAGE_MARKER = re.compile(r"\[IMAGE:\s*.+?\]", re.DOTALL)
_INLINE = re.compile(r"(\*\*.+?\*\*|\*.+?\*|`.+?`)")
_SEP_CELL = re.compile(r"^:?-{2,}:?$")
_MAX_IMG_WIDTH = Inches(6.0)
def build_docx(markdown: str, images: List[dict], out_path: Path) -> Path:
"""Render `markdown` to a Word document, inserting `images` at [IMAGE:] markers in order."""
doc = Document()
img_iter = iter([im for im in images if im.get("path")])
blocks = _split_blocks(markdown)
i, n = 0, len(blocks)
while i < n:
block = blocks[i]
# Markdown pipe table: a header row followed by a separator row (| --- | --- |).
if _is_table_row(block) and i + 1 < n and _is_separator_row(blocks[i + 1]):
j = i + 2
rows = [block]
while j < n and _is_table_row(blocks[j]):
rows.append(blocks[j])
j += 1
_add_table(doc, rows)
i = j
continue
if _IMAGE_MARKER.fullmatch(block.strip()):
_insert_next_image(doc, img_iter)
i += 1
continue
# a block may still contain an inline marker mixed with text
if _IMAGE_MARKER.search(block):
for piece in _IMAGE_MARKER.split(block):
if piece.strip():
_render_line(doc, piece.strip())
_insert_next_image(doc, img_iter)
i += 1
continue
_render_line(doc, block)
i += 1
# any leftover images that never got placed → append at the end
for im in img_iter:
_add_picture(doc, im)
out_path.parent.mkdir(parents=True, exist_ok=True)
doc.save(out_path)
return out_path
def _is_table_row(line: str) -> bool:
s = line.strip()
return "|" in s and s.count("|") >= 2
def _is_separator_row(line: str) -> bool:
cells = _parse_row(line)
return bool(cells) and all(_SEP_CELL.match(c.replace(" ", "")) for c in cells)
def _parse_row(line: str) -> List[str]:
s = line.strip()
if s.startswith("|"):
s = s[1:]
if s.endswith("|"):
s = s[:-1]
return [c.strip() for c in s.split("|")]
def _add_table(doc: Document, rows: List[str]) -> None:
header = _parse_row(rows[0])
body = [_parse_row(r) for r in rows[1:]]
ncols = max([len(header)] + [len(r) for r in body]) if (header or body) else 0
if ncols == 0:
return
table = doc.add_table(rows=1, cols=ncols)
try:
table.style = "Table Grid"
except Exception:
pass
# header (cells start empty, so add_run yields runs[0])
for c in range(ncols):
run = table.rows[0].cells[c].paragraphs[0].add_run(header[c] if c < len(header) else "")
run.bold = True
# body
for r in body:
cells = table.add_row().cells
for c in range(ncols):
_add_runs(cells[c].paragraphs[0], r[c] if c < len(r) else "")
def _split_blocks(md: str) -> List[str]:
"""Split into logical blocks (headings/paragraphs/markers), preserving order."""
blocks: List[str] = []
for raw in md.split("\n"):
line = raw.rstrip()
if line.strip() == "":
continue
blocks.append(line)
return blocks
def _render_line(doc: Document, line: str) -> None:
stripped = line.strip()
if stripped.startswith("#"):
level = len(stripped) - len(stripped.lstrip("#"))
text = stripped[level:].strip()
doc.add_heading(text, level=min(max(level, 1), 4))
elif re.match(r"^[-*]\s+", stripped):
text = re.sub(r"^[-*]\s+", "", stripped)
p = doc.add_paragraph(style="List Bullet")
_add_runs(p, text)
elif re.match(r"^\d+\.\s+", stripped):
text = re.sub(r"^\d+\.\s+", "", stripped)
p = doc.add_paragraph(style="List Number")
_add_runs(p, text)
elif stripped.startswith(">"):
p = doc.add_paragraph(style="Intense Quote")
_add_runs(p, stripped.lstrip("> ").strip())
else:
p = doc.add_paragraph()
_add_runs(p, stripped)
def _add_runs(paragraph, text: str) -> None:
"""Add text to a paragraph, honoring **bold**, *italic*, and `code`."""
for token in _INLINE.split(text):
if not token:
continue
if token.startswith("**") and token.endswith("**"):
paragraph.add_run(token[2:-2]).bold = True
elif token.startswith("*") and token.endswith("*"):
paragraph.add_run(token[1:-1]).italic = True
elif token.startswith("`") and token.endswith("`"):
run = paragraph.add_run(token[1:-1])
run.font.name = "Consolas"
else:
paragraph.add_run(token)
def _insert_next_image(doc: Document, img_iter) -> None:
im = next(img_iter, None)
if im is not None:
_add_picture(doc, im)
def _add_picture(doc: Document, im: dict) -> None:
path = im.get("path")
if not path or not Path(path).exists():
return
try:
p = doc.add_paragraph()
p.alignment = WD_ALIGN_PARAGRAPH.CENTER
p.add_run().add_picture(str(path), width=_MAX_IMG_WIDTH)
except Exception:
return
caption = (im.get("caption") or "").strip()
if caption:
cap = doc.add_paragraph()
cap.alignment = WD_ALIGN_PARAGRAPH.CENTER
run = cap.add_run(caption)
run.italic = True
run.font.size = Pt(9)
run.font.color.rgb = RGBColor(0x66, 0x66, 0x66)