385 lines
17 KiB
Python
385 lines
17 KiB
Python
#!/usr/bin/env python3
|
|||
|
|
# Copyright (c) 2026 Marc Wäckerlin. SPDX-License-Identifier: MIT
|
||
|
|
"""Turn a Markdown file into a PDF in the design of a company.
|
||
|
|
|
||
|
|
The design is not written here. pandoc turns the Markdown into the body of a
|
||
|
|
LaTeX document whose preamble loads the identity of the company and the template
|
||
|
|
of the suite, and xelatex builds it: what comes out is the example document of
|
||
|
|
the suite with a Markdown body, in the font of the company, with its logo in the
|
||
|
|
page header and with the title block of the template.
|
||
|
|
|
||
|
|
md2pdf.py --identity <company-identity> <file.md> [<file.md> …]
|
||
|
|
|
||
|
|
A company links the converter into a directory of the PATH under its own name and
|
||
|
|
hands it its identity there, so that several of them lie side by side and every
|
||
|
|
call says which design it writes:
|
||
|
|
|
||
|
|
#!/bin/sh
|
||
|
|
exec md2pdf.py --identity company-identity "$@"
|
||
|
|
|
||
|
|
The PDF is written next to its source; everything produced on the way lives in a
|
||
|
|
throwaway directory that is removed again. A failed build keeps that directory,
|
||
|
|
names the LaTeX file in it, and prints the lines LaTeX complained about.
|
||
|
|
|
||
|
|
Options are the values that differ per document or per person; nothing else is
|
||
|
|
adjustable, because everything else is the design of the template:
|
||
|
|
|
||
|
|
--identity <name> the package with the colours, fonts and logo of the company
|
||
|
|
--out-dir <dir> write the PDF here instead of beside the source
|
||
|
|
--lang <tag> hyphenation and language of the document. The default
|
||
|
|
`auto` reads it off the words of the text, and leaves it
|
||
|
|
to the document where that carries a `lang:` of its own;
|
||
|
|
a tag given here decides over both
|
||
|
|
--paper <name> a4paper (default), a5paper, letterpaper …
|
||
|
|
--fontsize <pt> 10pt (default, the company size), 11pt, 12pt
|
||
|
|
--toc add a table of contents
|
||
|
|
--highlight colour the code blocks (pandoc's own colours)
|
||
|
|
--keep-tex keep the LaTeX file and everything the build produced,
|
||
|
|
in a build/ beside the PDF — under `--out-dir` where one
|
||
|
|
is given, beside the source where none is
|
||
|
|
"""
|
||
|
|
import argparse
|
||
|
|
import os
|
||
|
|
import re
|
||
|
|
import shutil
|
||
|
|
import subprocess
|
||
|
|
import sys
|
||
|
|
import tempfile
|
||
|
|
|
||
|
|
# Where the suite lies, read through every link on the way. The converter is
|
||
|
|
# meant to be called by its name, so it is linked into a directory of the PATH,
|
||
|
|
# and the name it was started under says where the link stands, never where the
|
||
|
|
# suite is: the filters beside it would then be looked for in the directory of
|
||
|
|
# the link, and the document would be built without them.
|
||
|
|
ROOT = os.path.dirname(os.path.dirname(os.path.realpath(__file__)))
|
||
|
|
FILTERS = [os.path.join(ROOT, "bin", name) for name in
|
||
|
|
("details.lua", "svg.lua", "links.lua",
|
||
|
|
# Before the table filter: that one wraps a table in a group of its
|
||
|
|
# own, and the sentence above would then no longer stand directly in
|
||
|
|
# front of a table.
|
||
|
|
"leadin.lua", "tables.lua", "code.lua", "figures.lua")]
|
||
|
|
|
||
|
|
# Which language a document is written in decides where LaTeX may break a word,
|
||
|
|
# and the wrong patterns break it in the wrong places: German patterns put
|
||
|
|
# "li-nes" into an English table cell and leave words unbroken that then stand
|
||
|
|
# outside the page. Nothing in a Markdown file says the language, so it is read
|
||
|
|
# off the words themselves — the most common words of a language are the ones a
|
||
|
|
# text of any length repeats. The document decides when it carries a `lang:` of
|
||
|
|
# its own, and an explicit --lang decides over both.
|
||
|
|
STOPWORDS = {
|
||
|
|
"de-CH": frozenset((
|
||
|
|
"der die das und ist nicht ein eine den dem des mit für auf von im zu "
|
||
|
|
"sich werden wird sind auch als aus bei nach über oder aber wenn dass "
|
||
|
|
"was noch nur schon kann muss haben hat ich wir sie er es").split()),
|
||
|
|
"en-GB": frozenset((
|
||
|
|
"the and is not are with for from this that of to in it as be by on "
|
||
|
|
"at or but if what still only can must have has we they he she").split()),
|
||
|
|
}
|
||
|
|
WORDS = re.compile(r"[a-zäöüßA-ZÄÖÜ]+")
|
||
|
|
DOCUMENT_LANGUAGE = re.compile(r"^lang:", re.MULTILINE)
|
||
|
|
|
||
|
|
# gfm is what GitHub and the editor preview show. The extensions on top are what
|
||
|
|
# a document of ours uses and gfm does not carry by itself: the YAML block that
|
||
|
|
# gives title, author and date, footnotes, definition lists, and the attributes
|
||
|
|
# behind a picture, `{width=30%}`, which decide how wide it stands. Without that
|
||
|
|
# last one the braces are printed into the text.
|
||
|
|
#
|
||
|
|
# Formulas need no extension here: `--list-extensions=gfm` lists
|
||
|
|
# `tex_math_dollars` and `tex_math_gfm` as ON, so `$x$` and `$$x$$` arrive as
|
||
|
|
# math with this reader. Measured on 2026-09-21 against the same document with
|
||
|
|
# and without the extension written out: the two parse trees are identical.
|
||
|
|
MARKDOWN = ("gfm+yaml_metadata_block+footnotes+definition_lists"
|
||
|
|
"+attributes")
|
||
|
|
|
||
|
|
# Where a fenced code block begins and ends. Inside one, nothing is repaired:
|
||
|
|
# what stands there is an example of itself.
|
||
|
|
FENCE = re.compile(r"^\s*(```|~~~)")
|
||
|
|
|
||
|
|
# A LaTeX error carries its place in this form, because latexmk is called with
|
||
|
|
# -file-line-error: ./file.tex:42: Undefined control sequence.
|
||
|
|
LATEX_ERROR = re.compile(r"^(?:[^\s:]+):\d+: .*$", re.MULTILINE)
|
||
|
|
|
||
|
|
# What the column widths of a table are computed from: how wide the line is and
|
||
|
|
# how much of it one character takes. Both belong to the paper, the margins and
|
||
|
|
# the face of the company, so both are measured in a probe document instead of
|
||
|
|
# being written down — the origin of this family carried 538 and 5.3 as
|
||
|
|
# constants in the filter, and the second company of the family carried 481 and
|
||
|
|
# 4.6 for the same page, a number nobody could account for afterwards.
|
||
|
|
#
|
||
|
|
# The reference line is set once and its width divided by its length. It is the
|
||
|
|
# alphabet twice over with the spaces a text has, so the average is that of a
|
||
|
|
# text and not of one word.
|
||
|
|
REFERENCE = ("the quick brown fox jumps over the lazy dog "
|
||
|
|
"und der flinke braune fuchs springt ueber den faulen hund")
|
||
|
|
PROBE = r"""\documentclass[%(paper)s,%(fontsize)s]{article}
|
||
|
|
\usepackage{%(identity)s}
|
||
|
|
\usepackage{business-suite}
|
||
|
|
\begin{document}
|
||
|
|
\newwrite\businessmetrics
|
||
|
|
\immediate\openout\businessmetrics=metrics.txt
|
||
|
|
\sbox0{%(reference)s}%%
|
||
|
|
\immediate\write\businessmetrics{linewidth \the\linewidth}%%
|
||
|
|
\immediate\write\businessmetrics{reference \the\wd0}%%
|
||
|
|
\immediate\closeout\businessmetrics
|
||
|
|
\end{document}
|
||
|
|
"""
|
||
|
|
LENGTH = re.compile(r"^(\w+) ([0-9.]+)pt$", re.MULTILINE)
|
||
|
|
|
||
|
|
|
||
|
|
def run(command, cwd, environment=None):
|
||
|
|
"""One external command; its output comes back as text."""
|
||
|
|
return subprocess.run(command, cwd=cwd,
|
||
|
|
env=dict(os.environ, **(environment or {})),
|
||
|
|
stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
|
||
|
|
encoding="utf-8", errors="replace")
|
||
|
|
|
||
|
|
|
||
|
|
def language(source, given):
|
||
|
|
"""The language whose hyphenation patterns fit this document.
|
||
|
|
|
||
|
|
None means that nobody has to be told: either the document says it itself,
|
||
|
|
or the text gives no answer and the class keeps its default.
|
||
|
|
"""
|
||
|
|
if given and given != "auto":
|
||
|
|
return given
|
||
|
|
with open(source, encoding="utf-8", errors="replace") as handle:
|
||
|
|
text = handle.read()
|
||
|
|
if DOCUMENT_LANGUAGE.search(text):
|
||
|
|
return None
|
||
|
|
words = [word.lower() for word in WORDS.findall(text)]
|
||
|
|
counted = {tag: sum(word in stopwords for word in words)
|
||
|
|
for tag, stopwords in STOPWORDS.items()}
|
||
|
|
best = max(counted, key=counted.get)
|
||
|
|
return best if counted[best] else None
|
||
|
|
|
||
|
|
|
||
|
|
def searchpath(source_dir, build):
|
||
|
|
"""Where LaTeX looks: the suite, the document, the build directory."""
|
||
|
|
return os.pathsep.join([ROOT + "//", source_dir + "//", build + "//", ""])
|
||
|
|
|
||
|
|
|
||
|
|
def metrics(options, build, source_dir):
|
||
|
|
"""The width of the line and of a character, measured in a probe.
|
||
|
|
|
||
|
|
An empty answer means the probe did not build; the filters then fall back to
|
||
|
|
the measurement of the origin of this family, and the table of a company with
|
||
|
|
another paper comes out too narrow instead of not at all.
|
||
|
|
"""
|
||
|
|
probe = os.path.join(build, "metrics.tex")
|
||
|
|
with open(probe, "w", encoding="utf-8") as handle:
|
||
|
|
handle.write(PROBE % {"paper": options.paper,
|
||
|
|
"fontsize": options.fontsize,
|
||
|
|
"identity": options.identity,
|
||
|
|
"reference": REFERENCE})
|
||
|
|
run(["xelatex", "-no-pdf", "-interaction=nonstopmode", "-halt-on-error",
|
||
|
|
"metrics.tex"], cwd=build,
|
||
|
|
environment={"TEXINPUTS": searchpath(source_dir, build)})
|
||
|
|
written = os.path.join(build, "metrics.txt")
|
||
|
|
if not os.path.exists(written):
|
||
|
|
return {}
|
||
|
|
with open(written, encoding="utf-8", errors="replace") as handle:
|
||
|
|
found = dict(LENGTH.findall(handle.read()))
|
||
|
|
if "linewidth" not in found or "reference" not in found:
|
||
|
|
return {}
|
||
|
|
return {"MD2PDF_LINEWIDTH": found["linewidth"],
|
||
|
|
"MD2PDF_CHARACTER": str(float(found["reference"])
|
||
|
|
/ len(REFERENCE))}
|
||
|
|
|
||
|
|
|
||
|
|
def join_display_math(source, build):
|
||
|
|
"""A formula over several lines, joined into one before pandoc reads it.
|
||
|
|
|
||
|
|
`$$` around a formula is display math, and a writer breaks it over lines to
|
||
|
|
keep it readable. The grammar of the reader looks at those lines first: one
|
||
|
|
that carries nothing but `=` is the UNDERLINE OF A HEADING, so the line
|
||
|
|
above it becomes a heading of the first level and the rest of the formula a
|
||
|
|
paragraph. The document builds green, the page carries the formula as a
|
||
|
|
heading in the colour of a heading, and its first half stands in the running
|
||
|
|
head of the next page — measured on 2026-09-21 in a twelve-page analysis of
|
||
|
|
another company of this family.
|
||
|
|
|
||
|
|
A line break inside display math means nothing to TeX, so the block is
|
||
|
|
joined and reads as the formula that was written. A fenced code block is
|
||
|
|
left alone: the dollars there are an example of themselves.
|
||
|
|
|
||
|
|
What comes back is the file pandoc reads — the source itself where there was
|
||
|
|
nothing to join, a copy in the build directory otherwise — and the number of
|
||
|
|
formulas that were joined.
|
||
|
|
"""
|
||
|
|
with open(source, encoding="utf-8", errors="replace") as handle:
|
||
|
|
lines = handle.read().splitlines()
|
||
|
|
written, joined, index, fenced = [], 0, 0, False
|
||
|
|
while index < len(lines):
|
||
|
|
line = lines[index]
|
||
|
|
if FENCE.match(line):
|
||
|
|
fenced = not fenced
|
||
|
|
stripped = line.strip()
|
||
|
|
open_math = (not fenced and stripped.startswith("$$")
|
||
|
|
and not (len(stripped) > 3 and stripped.endswith("$$")))
|
||
|
|
if not open_math:
|
||
|
|
written.append(line)
|
||
|
|
index += 1
|
||
|
|
continue
|
||
|
|
# The opening line ENDS with the two dollars as well, so the block grows
|
||
|
|
# by one line before the closing dollars are looked for.
|
||
|
|
block, index = [stripped], index + 1
|
||
|
|
while index < len(lines):
|
||
|
|
block.append(lines[index].strip())
|
||
|
|
index += 1
|
||
|
|
if block[-1].endswith("$$"):
|
||
|
|
break
|
||
|
|
if len(block) > 1 and block[-1].endswith("$$"):
|
||
|
|
written.append(" ".join(part for part in block if part))
|
||
|
|
joined += 1
|
||
|
|
else:
|
||
|
|
written += block
|
||
|
|
if not joined:
|
||
|
|
return source, 0
|
||
|
|
copy = os.path.join(build, os.path.basename(source))
|
||
|
|
with open(copy, "w", encoding="utf-8") as handle:
|
||
|
|
handle.write("\n".join(written) + "\n")
|
||
|
|
return copy, joined
|
||
|
|
|
||
|
|
|
||
|
|
def to_latex(source, tex, build, options, measured):
|
||
|
|
"""pandoc: the Markdown body plus the preamble that loads the template."""
|
||
|
|
source_dir = os.path.dirname(os.path.abspath(source))
|
||
|
|
read, joined = join_display_math(source, build)
|
||
|
|
if joined:
|
||
|
|
print(f"{os.path.basename(source)}: {joined} formula(s) over several "
|
||
|
|
"lines joined into one line")
|
||
|
|
tag = language(source, options.lang)
|
||
|
|
filters = []
|
||
|
|
for name in FILTERS:
|
||
|
|
filters += ["--lua-filter", name]
|
||
|
|
command = ["pandoc", os.path.abspath(read),
|
||
|
|
"--standalone", "--from", MARKDOWN, "--to", "latex",
|
||
|
|
"--resource-path", source_dir] + filters + [
|
||
|
|
"-V", "documentclass=article",
|
||
|
|
"-V", "classoption=" + options.paper,
|
||
|
|
"-V", "fontsize=" + options.fontsize,
|
||
|
|
"-V", "header-includes=\\usepackage{" + options.identity + "}",
|
||
|
|
"-V", "header-includes=\\usepackage{business-suite}",
|
||
|
|
"-o", tex]
|
||
|
|
if tag:
|
||
|
|
command += ["-M", "lang=" + tag]
|
||
|
|
print(f"{os.path.basename(source)}: {tag}")
|
||
|
|
if options.toc:
|
||
|
|
command.append("--toc")
|
||
|
|
if not options.highlight:
|
||
|
|
command.append("--no-highlight")
|
||
|
|
return run(command, cwd=build,
|
||
|
|
environment=dict({"MD2PDF_BUILD": build,
|
||
|
|
"MD2PDF_SOURCE_DIR": source_dir}, **measured))
|
||
|
|
|
||
|
|
|
||
|
|
def row_rules(tex):
|
||
|
|
"""A line between two rows of a table, drawn by the package.
|
||
|
|
|
||
|
|
pandoc writes the three rules of a table and nothing between the rows. The
|
||
|
|
rule itself belongs to the package, `\\businessrowrule`, and here it is put
|
||
|
|
behind every row of a table body: a row ends with `\\\\` at the end of a
|
||
|
|
line, the body begins after `\\endlastfoot`, and the last row of it needs
|
||
|
|
none, because the table closes with its own rule underneath.
|
||
|
|
"""
|
||
|
|
with open(tex, encoding="utf-8") as handle:
|
||
|
|
lines = handle.read().splitlines()
|
||
|
|
written, body = [], False
|
||
|
|
for index, line in enumerate(lines):
|
||
|
|
written.append(line)
|
||
|
|
if line.startswith("\\endlastfoot"):
|
||
|
|
body = True
|
||
|
|
elif line.startswith("\\end{longtable}"):
|
||
|
|
body = False
|
||
|
|
elif body and line.rstrip().endswith("\\\\") \
|
||
|
|
and not lines[index + 1].startswith("\\end{longtable}"):
|
||
|
|
written.append("\\businessrowrule")
|
||
|
|
with open(tex, "w", encoding="utf-8") as handle:
|
||
|
|
handle.write("\n".join(written) + "\n")
|
||
|
|
|
||
|
|
|
||
|
|
def to_pdf(tex, build, source_dir):
|
||
|
|
"""xelatex, through latexmk, with as many runs as the document needs."""
|
||
|
|
return run(["latexmk", "-xelatex", "-interaction=nonstopmode",
|
||
|
|
"-file-line-error", "-emulate-aux-dir",
|
||
|
|
"-auxdir=" + build, "-outdir=" + build, tex],
|
||
|
|
cwd=source_dir,
|
||
|
|
environment={"TEXINPUTS": searchpath(source_dir, build)})
|
||
|
|
|
||
|
|
|
||
|
|
def complaints(build, stem, result):
|
||
|
|
"""What LaTeX complained about, for a reader who has to fix the document."""
|
||
|
|
log = os.path.join(build, stem + ".log")
|
||
|
|
text = ""
|
||
|
|
if os.path.exists(log):
|
||
|
|
with open(log, encoding="utf-8", errors="replace") as handle:
|
||
|
|
text = handle.read()
|
||
|
|
found = LATEX_ERROR.findall(text)
|
||
|
|
return found or [line for line in result.stdout.splitlines()
|
||
|
|
if line.strip()][-10:]
|
||
|
|
|
||
|
|
|
||
|
|
def convert(source, options):
|
||
|
|
"""One document; the message of a failure comes back, None means done."""
|
||
|
|
if not os.path.exists(source):
|
||
|
|
return f"{source}: there is no such file"
|
||
|
|
source_dir = os.path.dirname(os.path.abspath(source)) or os.getcwd()
|
||
|
|
stem = os.path.splitext(os.path.basename(source))[0]
|
||
|
|
# What is kept lands where the PDF lands. Whoever names a directory for the
|
||
|
|
# result has said where this document may write; the directory of the source
|
||
|
|
# can belong to somebody else, and a build that keeps its files leaves
|
||
|
|
# twenty of them there.
|
||
|
|
build = os.path.join(os.path.abspath(options.out_dir or source_dir),
|
||
|
|
"build") if options.keep_tex \
|
||
|
|
else tempfile.mkdtemp(prefix="md2pdf-")
|
||
|
|
os.makedirs(build, exist_ok=True)
|
||
|
|
keep = options.keep_tex
|
||
|
|
try:
|
||
|
|
tex = os.path.join(build, stem + ".tex")
|
||
|
|
measured = metrics(options, build, source_dir)
|
||
|
|
written = to_latex(source, tex, build, options, measured)
|
||
|
|
if written.returncode != 0 or not os.path.exists(tex):
|
||
|
|
keep = True
|
||
|
|
return f"{source}: pandoc could not read the document:\n" \
|
||
|
|
f"{written.stdout.strip()}"
|
||
|
|
row_rules(tex)
|
||
|
|
built = to_pdf(tex, build, source_dir)
|
||
|
|
pdf = os.path.join(build, stem + ".pdf")
|
||
|
|
if built.returncode != 0 or not os.path.exists(pdf):
|
||
|
|
keep = True
|
||
|
|
return f"{source}: LaTeX could not build the document:\n " \
|
||
|
|
+ "\n ".join(complaints(build, stem, built)) \
|
||
|
|
+ f"\n the LaTeX file is {tex}"
|
||
|
|
target = os.path.join(options.out_dir or source_dir, stem + ".pdf")
|
||
|
|
os.makedirs(os.path.dirname(os.path.abspath(target)), exist_ok=True)
|
||
|
|
shutil.copyfile(pdf, target)
|
||
|
|
print(target)
|
||
|
|
return None
|
||
|
|
finally:
|
||
|
|
if not keep:
|
||
|
|
shutil.rmtree(build, ignore_errors=True)
|
||
|
|
|
||
|
|
|
||
|
|
def main(arguments=None):
|
||
|
|
parser = argparse.ArgumentParser(
|
||
|
|
description="Markdown to PDF in the design of a company.")
|
||
|
|
parser.add_argument("documents", nargs="+", metavar="file.md")
|
||
|
|
parser.add_argument("--identity", required=True)
|
||
|
|
parser.add_argument("--out-dir")
|
||
|
|
parser.add_argument("--lang", default="auto")
|
||
|
|
parser.add_argument("--paper", default="a4paper")
|
||
|
|
parser.add_argument("--fontsize", default="10pt")
|
||
|
|
parser.add_argument("--toc", action="store_true")
|
||
|
|
parser.add_argument("--highlight", action="store_true")
|
||
|
|
parser.add_argument("--keep-tex", action="store_true")
|
||
|
|
options = parser.parse_args(arguments)
|
||
|
|
failures = [failure for failure in
|
||
|
|
(convert(document, options) for document in options.documents)
|
||
|
|
if failure]
|
||
|
|
for failure in failures:
|
||
|
|
print(failure, file=sys.stderr)
|
||
|
|
return 1 if failures else 0
|
||
|
|
|
||
|
|
|
||
|
|
if __name__ == "__main__":
|
||
|
|
sys.exit(main())
|