humainflow-web/docs/render-user-guide.py

192 lines
7.8 KiB
Python

#!/usr/bin/env python3
"""Render the user guide to a print-ready HTML page.
From web/, to regenerate docs/user-guide.pdf after editing docs/user-guide.md:
python3 docs/render-user-guide.py docs/user-guide.md /tmp/ug.html \
"HumAIn Flow" "User Guide" "Version of 16 September 2026"
npx playwright pdf file:///tmp/ug.html docs/user-guide.pdf
Deliberately not a general Markdown implementation: it handles the constructs the
guide actually uses, so the output is predictable and the print styling can be
tuned for it (page breaks per chapter, a real table of contents, no orphan
headings). Pandoc is not installed here and pulling one in for one document costs
more than it saves.
"""
import html
import re
import sys
INLINE = [
(re.compile(r"`([^`]+)`"), r"<code>\1</code>"),
(re.compile(r"\*\*([^*]+)\*\*"), r"<strong>\1</strong>"),
(re.compile(r"(?<![\w*])\*([^*\n]+)\*(?![\w*])"), r"<em>\1</em>"),
(re.compile(r"\[([^\]]+)\]\(([^)]+)\)"), r'<a href="\2">\1</a>'),
]
def inline(text):
out = html.escape(text)
for pattern, replacement in INLINE:
out = pattern.sub(replacement, out)
return out
def convert(markdown):
lines = markdown.split("\n")
out, toc = [], []
i = 0
list_stack = []
def close_lists(to_depth=0):
while len(list_stack) > to_depth:
out.append(f"</{list_stack.pop()}>")
while i < len(lines):
line = lines[i]
stripped = line.strip()
if not stripped:
close_lists()
i += 1
continue
if stripped.startswith("```"):
close_lists()
i += 1
block = []
while i < len(lines) and not lines[i].strip().startswith("```"):
block.append(html.escape(lines[i]))
i += 1
i += 1
out.append("<pre><code>" + "\n".join(block) + "</code></pre>")
continue
heading = re.match(r"^(#{1,4})\s+(.*)$", stripped)
if heading:
close_lists()
level = len(heading.group(1))
text = heading.group(2)
anchor = re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-")
# Chapters are h2 and sections h3, so the contents list is those two levels.
if 2 <= level <= 3:
toc.append((level - 1, text, anchor))
out.append(f'<h{level} id="{anchor}">{inline(text)}</h{level}>')
i += 1
continue
if stripped.startswith("|") and i + 1 < len(lines) and re.match(r"^\|[\s:|-]+\|$", lines[i + 1].strip()):
close_lists()
header = [c.strip() for c in stripped.strip("|").split("|")]
i += 2
rows = []
while i < len(lines) and lines[i].strip().startswith("|"):
rows.append([c.strip() for c in lines[i].strip().strip("|").split("|")])
i += 1
head = "".join(f"<th>{inline(c)}</th>" for c in header)
body = "".join("<tr>" + "".join(f"<td>{inline(c)}</td>" for c in r) + "</tr>" for r in rows)
out.append(f"<table><thead><tr>{head}</tr></thead><tbody>{body}</tbody></table>")
continue
if stripped in ("---", "***"):
close_lists()
out.append('<hr class="page-break">')
i += 1
continue
bullet = re.match(r"^(\s*)[-*]\s+(.*)$", line)
number = re.match(r"^(\s*)\d+\.\s+(.*)$", line)
if bullet or number:
match = bullet or number
tag = "ul" if bullet else "ol"
depth = len(match.group(1)) // 2 + 1
while len(list_stack) > depth:
out.append(f"</{list_stack.pop()}>")
while len(list_stack) < depth:
out.append(f"<{tag}>")
list_stack.append(tag)
out.append(f"<li>{inline(match.group(2))}</li>")
i += 1
continue
if stripped.startswith("> "):
close_lists()
out.append(f'<p class="note">{inline(stripped[2:])}</p>')
i += 1
continue
close_lists()
paragraph = [stripped]
i += 1
while i < len(lines) and lines[i].strip() and not re.match(r"^\s*([-*]|\d+\.|#|\||>|```)", lines[i]):
paragraph.append(lines[i].strip())
i += 1
out.append(f"<p>{inline(' '.join(paragraph))}</p>")
close_lists()
return "\n".join(out), toc
CSS = """
@page { size: A4; margin: 18mm 16mm 20mm; }
* { box-sizing: border-box; }
body { font: 10.5pt/1.55 -apple-system, "Segoe UI", Helvetica, Arial, sans-serif;
color: #1e293b; margin: 0; }
h1 { font-size: 21pt; color: #0f172a; margin: 0 0 6pt; line-height: 1.25; }
h2 { font-size: 19pt; color: #0f172a; page-break-before: always; page-break-after: avoid;
border-top: 2px solid #2563eb; padding-top: 6pt; margin: 0 0 10pt; line-height: 1.25; }
h3 { font-size: 12.5pt; color: #1d4ed8; margin: 16pt 0 5pt; page-break-after: avoid; }
h4 { font-size: 10.8pt; color: #0f172a; margin: 12pt 0 3pt; page-break-after: avoid; }
p { margin: 0 0 7pt; orphans: 2; widows: 2; }
ul, ol { margin: 0 0 8pt; padding-left: 18pt; }
li { margin-bottom: 2.5pt; }
code { font-family: "SF Mono", Menlo, Consolas, monospace; font-size: 9pt;
background: #f1f5f9; border: 1px solid #e2e8f0; border-radius: 3px; padding: 0 3px; }
pre { background: #f8fafc; border: 1px solid #e2e8f0; border-left: 3px solid #94a3b8;
border-radius: 4px; padding: 8pt 10pt; overflow-x: auto; page-break-inside: avoid; }
pre code { background: none; border: none; padding: 0; font-size: 8.5pt; line-height: 1.45; }
table { width: 100%; border-collapse: collapse; margin: 0 0 10pt; font-size: 9.5pt;
page-break-inside: avoid; }
th { background: #eff6ff; color: #1e3a8a; text-align: left; font-weight: 600; }
th, td { border: 1px solid #cbd5e1; padding: 4pt 6pt; vertical-align: top; }
.note { border-left: 3px solid #f59e0b; background: #fffbeb; padding: 6pt 9pt;
margin: 0 0 8pt; page-break-inside: avoid; }
hr.page-break { border: 0; page-break-after: always; }
.cover { height: 247mm; display: flex; flex-direction: column; justify-content: center; }
.cover-title { font-size: 34pt; border: 0; padding: 0; }
.cover-sub { font-size: 13pt; color: #475569; margin-top: 2pt; }
.cover-meta { margin-top: 26pt; font-size: 10pt; color: #64748b; }
.toc { page-break-after: always; }
.toc-title { font-size: 19pt; color: #0f172a; border-top: 2px solid #2563eb;
padding-top: 6pt; margin: 0 0 10pt; }
.toc ol { list-style: none; padding-left: 0; counter-reset: chapter; }
.toc li.l1 { counter-increment: chapter; font-weight: 600; margin-top: 6pt; }
.toc li.l1::before { content: counter(chapter) ". "; color: #2563eb; }
.toc li.l2 { padding-left: 16pt; font-weight: 400; font-size: 9.5pt; color: #475569; }
.toc a { color: inherit; text-decoration: none; }
a { color: #1d4ed8; }
"""
def main(md_path, html_path, title, subtitle, meta):
# The cover carries the document title, so the file's own leading H1 would only repeat it.
source = re.sub(r"\A\s*#\s+.*\n", "", open(md_path).read())
body, toc = convert(source)
items = "".join(
f'<li class="l{level}"><a href="#{anchor}">{html.escape(text)}</a></li>'
for level, text, anchor in toc
)
page = f"""<!doctype html><html><head><meta charset="utf-8"><title>{html.escape(title)}</title>
<style>{CSS}</style></head><body>
<div class="cover"><h1 class="cover-title">{html.escape(title)}</h1>
<div class="cover-sub">{html.escape(subtitle)}</div>
<div class="cover-meta">{html.escape(meta)}</div></div>
<div class="toc"><div class="toc-title">Contents</div><ol>{items}</ol></div>
{body}</body></html>"""
open(html_path, "w").write(page)
print(f"html written: {html_path} ({len(page)} bytes, {len(toc)} toc entries)")
if __name__ == "__main__":
main(*sys.argv[1:])