#!/usr/bin/env python3 """ AI Memo Builder v1.0 — Markdown master -> house-style .docx =========================================================== Deterministic, pure-stdlib (Python 3.8+, no third-party packages) generator that converts a Markdown "master" document into a formatted Word (.docx) memo. Why this exists: LLMs that improvise WordprocessingML produce files that look fine in previews but trigger Word's repair dialog. Three rules decide whether Word accepts a file: 1. Namespace prefixes: every WordprocessingML element needs its `w:` (or `r:`) prefix. A single bare `` makes Word refuse the file even though the XML is well-formed. 2. Schema element order: Word enforces it strictly (pPr, rPr, tcPr, tblPr, sectPr — see ORDER maps below). 3. No duplicates: one element of each kind per container. This builder encodes those rules and validates its own output before writing. Usage: python3 memo_builder.py Fill in the MEMO dict and adapt the house-style constants to your firm. License: MIT """ import re import sys import zipfile import xml.etree.ElementTree as ET # --------------------------------------------------------------------------- # 1. MEMO METADATA — edit these placeholders # --------------------------------------------------------------------------- MEMO = { "title": "Assessment of Regulatory Filing — Example Industries Ltd", "to": "Dr. Jane Doe, General Counsel, Example Industries Ltd", "from": "Dr. John Smith, Outside Counsel", "date": "24 September 2026", "status": "Final", "re": "Regulatory Filing — Preliminary Assessment", "header": "PRIVILEGED & CONFIDENTIAL — PREPARED BY ATTORNEY", } # --------------------------------------------------------------------------- # 2. HOUSE-STYLE CONSTANTS — adapt to your firm's style # --------------------------------------------------------------------------- ORIENTATION = "portrait" # or "landscape" (wide tables) FONT = "Arial" BODY_PT = 10 H1_PT = 18 H2_PT = 18 H3_PT = 14 LINK_BLUE = "0563C1" LABEL_BG = "D5E8F0" # To/From/Date/Status/Re label column BORDER = "CCCCCC" # --------------------------------------------------------------------------- # 3. OOXML SCHEMA ORDER MAPS (rule 2) — do not reorder # --------------------------------------------------------------------------- ORDER = { "pPr": ["pStyle", "keepNext", "pBdr", "shd", "tabs", "spacing", "ind", "jc", "outlineLvl", "rPr"], "rPr": ["rFonts", "b", "i", "color", "u", "sz", "szCs"], "tcPr": ["tcW", "shd", "vAlign"], "tblPr": ["tblW", "tblBorders", "tblLayout", "tblCellMar"], "sectPr": ["headerReference", "footerReference", "pgSz", "pgMar", "cols", "docGrid"], } W_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" R_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" CONTENT_TYPES = ( '\n' '' '' '' '' '' '' '' '' ) ROOT_RELS = ( '\n' '' '' '' ) DOC_RELS = ( '\n' '' '' '' '' '%s' '' ) def esc(text): """XML-escape text content and attribute values.""" return (text.replace("&", "&").replace("<", "<") .replace(">", ">").replace('"', """)) # --------------------------------------------------------------------------- # Link registry: [text](url) -> w:hyperlink r:id, in order of appearance # --------------------------------------------------------------------------- class Links: def __init__(self): self.items = [] # list of urls, index == r:id number def id_for(self, url): self.items.append(url) return len(self.items) - 1 def rels(self): return "".join( '' % (i, esc(url)) for i, url in enumerate(self.items)) def ppr(pstyle=None, keep_next=False, spacing=None, jc=None, outline=None, border_bottom=False, ind=None, sz=None): """Build a w:pPr with correct schema order (rule 2, no duplicates — rule 3).""" parts = [] if pstyle: parts.append('' % pstyle) if keep_next: parts.append("") if border_bottom: parts.append('' % BORDER) if spacing: parts.append('' % spacing) if ind: parts.append('' % ind) if jc: parts.append('' % jc) if outline is not None: parts.append('' % outline) if sz: parts.append('' % (int(sz * 2), int(sz * 2))) return "%s" % "".join(parts) def run(text, bold=False, italic=False, color=None, sz=None, underline=False): """Build a w:r with rPr in schema order.""" rpr = ['' % (FONT, FONT)] if bold: rpr.append("") if italic: rpr.append("") if color: rpr.append('' % color) if underline: rpr.append('') if sz: rpr.append('' % (int(sz * 2), int(sz * 2))) return ('%s%s' % ("".join(rpr), esc(text))) TOKEN_RE = re.compile( r"\*\*(?P.+?)\*\*|\*(?P[^*]+?)\*|" r"\[(?P[^\]]+)\]\((?P[^)]+)\)") def inline_runs(text, links, sz=None): """Markdown inline: **bold**, *italic*, [text](url) -> ordered runs.""" runs, pos = [], 0 for m in TOKEN_RE.finditer(text): if m.start() > pos: runs.append(run(text[pos:m.start()], sz=sz)) if m.group("bold") is not None: runs.append(run(m.group("bold"), bold=True, sz=sz)) elif m.group("ital") is not None: runs.append(run(m.group("ital"), italic=True, sz=sz)) else: rid = links.id_for(m.group("lurl")) half = int((sz or BODY_PT) * 2) runs.append( '' '' '' '' '%s' '' % (rid, FONT, FONT, LINK_BLUE, half, half, esc(m.group("ltxt")))) pos = m.end() if pos < len(text): runs.append(run(text[pos:], sz=sz)) return "".join(runs) def parse_master(md_text): """Parse the Markdown master into blocks: ('h1'|'h2'|'h3', text) | ('p', text) | ('ul'|'ol', [items]) | ('table', [[cells]]).""" blocks = [] lines = md_text.splitlines() i = 0 while i < len(lines): line = lines[i] if line.startswith("|") and i + 1 < len(lines) and \ re.match(r"^\|[\s:|-]+\|$", lines[i + 1].strip()): rows = [line] i += 2 while i < len(lines) and lines[i].startswith("|"): rows.append(lines[i]) i += 1 cells = [[c.strip() for c in r.strip().strip("|").split("|")] for r in rows] blocks.append(("table", cells)) continue m = re.match(r"^(#{1,3})\s+(.*)$", line) if m: blocks.append(("h%d" % len(m.group(1)), m.group(2))) i += 1 elif re.match(r"^\d+\.\s", line): items = [] while i < len(lines) and re.match(r"^\d+\.\s", lines[i]): items.append(re.sub(r"^\d+\.\s+", "", lines[i])) i += 1 blocks.append(("ol", items)) elif line.startswith("- "): items = [] while i < len(lines) and lines[i].startswith("- "): items.append(lines[i][2:]) i += 1 blocks.append(("ul", items)) elif line.strip(): para = [line.strip()] i += 1 while i < len(lines) and lines[i].strip() and \ not re.match(r"^(#{1,3}\s|\d+\.\s|- |\|)", lines[i]): para.append(lines[i].strip()) i += 1 blocks.append(("p", " ".join(para))) else: i += 1 return blocks def heading_xml(level, text): pt = {1: H1_PT, 2: H2_PT, 3: H3_PT}[level] style = {1: "Heading1", 2: "Heading2", 3: "Heading3"}[level] spacing = ('w:before="360" w:after="120"' if level == 1 else 'w:before="280" w:after="100"') return "%s%s" % ( ppr(pstyle=style, keep_next=True, outline=level - 1, spacing=spacing), run(text, bold=True, sz=pt)) TBL_BORDERS = "".join( '' % (side, BORDER) for side in ("top", "left", "bottom", "right", "insideH", "insideV")) TBLPR = ('' "%s" '' '' '' % TBL_BORDERS) def table_xml(cells, links): """GFM table: shaded, repeating header row; 10pt body.""" ncols = len(cells[0]) width = 9072 // max(ncols, 1) rows = [] for r_idx, row in enumerate(cells): header = r_idx == 0 trpr = "%s" % ('' if header else "") tcs = [] for cell in row: shd = ('' % LABEL_BG if header else "") tcpr = ('%s' '' % (width, shd)) content = (run(cell, bold=True, sz=BODY_PT) if header else inline_runs(cell, links, sz=BODY_PT)) tcs.append("%s%s%s" % (tcpr, ppr(spacing='w:after="40"'), content)) rows.append("%s%s" % (trpr, "".join(tcs))) return "%s%s" % (TBLPR, "".join(rows)) def memo_box_xml(): """To/From/Date/Status/Re box: bold labels (no colons), shaded label column, grey borders, bold Re line.""" rows = [] for label, key in (("To", "to"), ("From", "from"), ("Date", "date"), ("Status", "status"), ("Re", "re")): label_cell = ( '' '' '' '%s%s' % (LABEL_BG, ppr(spacing='w:after="40"'), run(label, bold=True, sz=BODY_PT))) value_cell = ( '' '' '%s%s' % (ppr(spacing='w:after="40"'), run(MEMO[key], bold=(key == "re"), sz=BODY_PT))) rows.append('%s%s' % (label_cell, value_cell)) return "%s%s" % (TBLPR, "".join(rows)) def document_xml(blocks, links): if ORIENTATION == "landscape": pg = '' else: pg = '' sect = ('' '' '' '%s' '' '' % pg) body = [] # Substantive title, 18pt bold — never "MEMORANDUM" body.append("%s%s" % ( ppr(keep_next=True, spacing='w:before="120" w:after="160"'), run(MEMO["title"], bold=True, sz=H1_PT))) body.append(memo_box_xml()) body.append("%s" % ppr(spacing='w:after="160"')) for kind, payload in blocks: if kind in ("h1", "h2", "h3"): body.append(heading_xml(int(kind[1]), payload)) elif kind == "p": body.append("%s%s" % ( ppr(spacing='w:after="120"'), inline_runs(payload, links, sz=BODY_PT))) elif kind in ("ul", "ol"): for n, item in enumerate(payload, 1): prefix = "• " if kind == "ul" else "%d. " % n body.append("%s%s%s" % ( ppr(spacing='w:after="60"', ind='w:left="360" w:hanging="360"'), run(prefix, sz=BODY_PT), inline_runs(item, links, sz=BODY_PT))) elif kind == "table": body.append(table_xml(payload, links)) body.append("%s" % ppr(spacing='w:after="120"')) return ('\n' '%s%s' '' % (W_NS, R_NS, "".join(body), sect)) def styles_xml(): normal = ('' '' '' '' '' % (FONT, FONT, int(BODY_PT * 2), int(BODY_PT * 2))) def heading(sid, name, level, sz, before): return ('' '' '' '' '' '' % (sid, name, before, level, FONT, FONT, int(sz * 2), int(sz * 2))) return ('\n' '' '' '' '%s%s%s%s' % (W_NS, FONT, FONT, int(BODY_PT * 2), int(BODY_PT * 2), normal, heading("Heading1", "heading 1", 0, H1_PT, 360), heading("Heading2", "heading 2", 1, H2_PT, 280), heading("Heading3", "heading 3", 2, H3_PT, 280))) def header_xml(): """Confidentiality marking + thin separator line, right-aligned.""" return ('\n' '' '' '' '' '' '%s' % (W_NS, R_NS, BORDER, FONT, FONT, esc(MEMO.get("header", "")))) def footer_xml(): """Centred 'Page x | y' using PAGE / NUMPAGES fields. Each fldSimple carries its own run with literal text — required so Word accepts it.""" def field(instr): return ('' '' '' '1' % (esc(instr), FONT, FONT)) text_run = ('' '' '%%s' % (FONT, FONT)) return ('\n' '' '' '' '%s%s%s%s' % (W_NS, R_NS, text_run % "Page ", field(" PAGE "), text_run % " | ", field(" NUMPAGES "))) # --------------------------------------------------------------------------- # 4. VALIDATOR — must be completely green before delivery # --------------------------------------------------------------------------- BARE_RE = re.compile( r"<(pPr|rPr|tcPr|tblPr|sectPr|spacing|vAlign|jc|ind|tblHeader|keepNext|" r"outlineLvl|pgSz|pgMar|tblW|tcW|shd|tblBorders|pBdr|b|i|u|sz|szCs|" r"color|rFonts|hyperlink|tbl|tc|tr|p|r|t)[ />]") def validate(docx_path): """Well-formedness of ALL parts (including .rels!), schema element order (monotonic), namespace prefixes, no duplicates, ZIP integrity. Well-formed alone is NOT sufficient.""" errors = [] if zipfile.ZipFile(docx_path).testzip() is not None: errors.append("ZIP integrity check failed") ns_w = "{%s}" % W_NS with zipfile.ZipFile(docx_path) as z: for name in z.namelist(): data = z.read(name) try: root = ET.fromstring(data) except ET.ParseError as exc: errors.append("%s: not well-formed: %s" % (name, exc)) continue raw = data.decode("utf-8", "replace") m = BARE_RE.search(raw) if m: errors.append("%s: bare element <%s> — missing w: prefix " "(rule 1)" % (name, m.group(1))) for container, order in ORDER.items(): for el in root.iter(ns_w + container): seen, last = set(), -1 for child in el: tag = child.tag.replace(ns_w, "") if tag in order: idx = order.index(tag) if idx < last: errors.append( "%s: <%s> children out of order in %s " "(rule 2)" % (name, tag, container)) last = idx if tag in seen: errors.append( "%s: duplicate <%s> in %s (rule 3)" % (name, tag, container)) seen.add(tag) return errors # --------------------------------------------------------------------------- # 5. BUILD # --------------------------------------------------------------------------- def build(master_path, out_path): with open(master_path, encoding="utf-8") as fh: md_text = fh.read() blocks = parse_master(md_text) links = Links() doc = document_xml(blocks, links) parts = { "[Content_Types].xml": CONTENT_TYPES, "_rels/.rels": ROOT_RELS, "word/document.xml": doc, "word/_rels/document.xml.rels": DOC_RELS % links.rels(), "word/styles.xml": styles_xml(), "word/header1.xml": header_xml(), "word/footer1.xml": footer_xml(), } with zipfile.ZipFile(out_path, "w", zipfile.ZIP_DEFLATED) as z: for name, content in parts.items(): z.writestr(name, content) errors = validate(out_path) if errors: for e in errors: sys.stderr.write("VALIDATION ERROR: %s\n" % e) sys.exit(1) print("VALIDATION OK — %s" % out_path) if __name__ == "__main__": if len(sys.argv) != 3: sys.exit("usage: python3 memo_builder.py ") build(sys.argv[1], sys.argv[2])