#!/usr/bin/env python3 # Copyright (c) 2026 Marc Wäckerlin. SPDX-License-Identifier: MIT """Turn a Markdown file into a PDF in the design of a company. The design is not written here. pandoc turns the Markdown into the body of a LaTeX document whose preamble loads the identity of the company and the template of the suite, and xelatex builds it: what comes out is the example document of the suite with a Markdown body, in the font of the company, with its logo in the page header and with the title block of the template. md2pdf.py --identity [ …] A company links the converter into a directory of the PATH under its own name and hands it its identity there, so that several of them lie side by side and every call says which design it writes: #!/bin/sh exec md2pdf.py --identity company-identity "$@" The PDF is written next to its source; everything produced on the way lives in a throwaway directory that is removed again. A failed build keeps that directory, names the LaTeX file in it, and prints the lines LaTeX complained about. Options are the values that differ per document or per person; nothing else is adjustable, because everything else is the design of the template: --identity the package with the colours, fonts and logo of the company --out-dir write the PDF here instead of beside the source --lang hyphenation and language of the document. The default `auto` reads it off the words of the text, and leaves it to the document where that carries a `lang:` of its own; a tag given here decides over both --paper a4paper (default), a5paper, letterpaper … --fontsize 10pt (default, the company size), 11pt, 12pt --toc add a table of contents --highlight colour the code blocks (pandoc's own colours) --keep-tex keep the LaTeX file and everything the build produced, in a build/ beside the PDF — under `--out-dir` where one is given, beside the source where none is """ import argparse import os import re import shutil import subprocess import sys import tempfile # Where the suite lies, read through every link on the way. The converter is # meant to be called by its name, so it is linked into a directory of the PATH, # and the name it was started under says where the link stands, never where the # suite is: the filters beside it would then be looked for in the directory of # the link, and the document would be built without them. ROOT = os.path.dirname(os.path.dirname(os.path.realpath(__file__))) FILTERS = [os.path.join(ROOT, "bin", name) for name in ("details.lua", "svg.lua", "links.lua", # Before the table filter: that one wraps a table in a group of its # own, and the sentence above would then no longer stand directly in # front of a table. "leadin.lua", "tables.lua", "code.lua", "figures.lua")] # Which language a document is written in decides where LaTeX may break a word, # and the wrong patterns break it in the wrong places: German patterns put # "li-nes" into an English table cell and leave words unbroken that then stand # outside the page. Nothing in a Markdown file says the language, so it is read # off the words themselves — the most common words of a language are the ones a # text of any length repeats. The document decides when it carries a `lang:` of # its own, and an explicit --lang decides over both. STOPWORDS = { "de-CH": frozenset(( "der die das und ist nicht ein eine den dem des mit für auf von im zu " "sich werden wird sind auch als aus bei nach über oder aber wenn dass " "was noch nur schon kann muss haben hat ich wir sie er es").split()), "en-GB": frozenset(( "the and is not are with for from this that of to in it as be by on " "at or but if what still only can must have has we they he she").split()), } WORDS = re.compile(r"[a-zäöüßA-ZÄÖÜ]+") DOCUMENT_LANGUAGE = re.compile(r"^lang:", re.MULTILINE) # gfm is what GitHub and the editor preview show. The extensions on top are what # a document of ours uses and gfm does not carry by itself: the YAML block that # gives title, author and date, footnotes, definition lists, and the attributes # behind a picture, `{width=30%}`, which decide how wide it stands. Without that # last one the braces are printed into the text. # # Formulas need no extension here: `--list-extensions=gfm` lists # `tex_math_dollars` and `tex_math_gfm` as ON, so `$x$` and `$$x$$` arrive as # math with this reader. Measured on 2026-09-21 against the same document with # and without the extension written out: the two parse trees are identical. MARKDOWN = ("gfm+yaml_metadata_block+footnotes+definition_lists" "+attributes") # Where a fenced code block begins and ends. Inside one, nothing is repaired: # what stands there is an example of itself. FENCE = re.compile(r"^\s*(```|~~~)") # A LaTeX error carries its place in this form, because latexmk is called with # -file-line-error: ./file.tex:42: Undefined control sequence. LATEX_ERROR = re.compile(r"^(?:[^\s:]+):\d+: .*$", re.MULTILINE) # What the column widths of a table are computed from: how wide the line is and # how much of it one character takes. Both belong to the paper, the margins and # the face of the company, so both are measured in a probe document instead of # being written down — the origin of this family carried 538 and 5.3 as # constants in the filter, and the second company of the family carried 481 and # 4.6 for the same page, a number nobody could account for afterwards. # # The reference line is set once and its width divided by its length. It is the # alphabet twice over with the spaces a text has, so the average is that of a # text and not of one word. REFERENCE = ("the quick brown fox jumps over the lazy dog " "und der flinke braune fuchs springt ueber den faulen hund") PROBE = r"""\documentclass[%(paper)s,%(fontsize)s]{article} \usepackage{%(identity)s} \usepackage{business-suite} \begin{document} \newwrite\businessmetrics \immediate\openout\businessmetrics=metrics.txt \sbox0{%(reference)s}%% \immediate\write\businessmetrics{linewidth \the\linewidth}%% \immediate\write\businessmetrics{reference \the\wd0}%% \immediate\closeout\businessmetrics \end{document} """ LENGTH = re.compile(r"^(\w+) ([0-9.]+)pt$", re.MULTILINE) def run(command, cwd, environment=None): """One external command; its output comes back as text.""" return subprocess.run(command, cwd=cwd, env=dict(os.environ, **(environment or {})), stdout=subprocess.PIPE, stderr=subprocess.STDOUT, encoding="utf-8", errors="replace") def language(source, given): """The language whose hyphenation patterns fit this document. None means that nobody has to be told: either the document says it itself, or the text gives no answer and the class keeps its default. """ if given and given != "auto": return given with open(source, encoding="utf-8", errors="replace") as handle: text = handle.read() if DOCUMENT_LANGUAGE.search(text): return None words = [word.lower() for word in WORDS.findall(text)] counted = {tag: sum(word in stopwords for word in words) for tag, stopwords in STOPWORDS.items()} best = max(counted, key=counted.get) return best if counted[best] else None def searchpath(source_dir, build): """Where LaTeX looks: the suite, the document, the build directory.""" return os.pathsep.join([ROOT + "//", source_dir + "//", build + "//", ""]) def metrics(options, build, source_dir): """The width of the line and of a character, measured in a probe. An empty answer means the probe did not build; the filters then fall back to the measurement of the origin of this family, and the table of a company with another paper comes out too narrow instead of not at all. """ probe = os.path.join(build, "metrics.tex") with open(probe, "w", encoding="utf-8") as handle: handle.write(PROBE % {"paper": options.paper, "fontsize": options.fontsize, "identity": options.identity, "reference": REFERENCE}) run(["xelatex", "-no-pdf", "-interaction=nonstopmode", "-halt-on-error", "metrics.tex"], cwd=build, environment={"TEXINPUTS": searchpath(source_dir, build)}) written = os.path.join(build, "metrics.txt") if not os.path.exists(written): return {} with open(written, encoding="utf-8", errors="replace") as handle: found = dict(LENGTH.findall(handle.read())) if "linewidth" not in found or "reference" not in found: return {} return {"MD2PDF_LINEWIDTH": found["linewidth"], "MD2PDF_CHARACTER": str(float(found["reference"]) / len(REFERENCE))} def join_display_math(source, build): """A formula over several lines, joined into one before pandoc reads it. `$$` around a formula is display math, and a writer breaks it over lines to keep it readable. The grammar of the reader looks at those lines first: one that carries nothing but `=` is the UNDERLINE OF A HEADING, so the line above it becomes a heading of the first level and the rest of the formula a paragraph. The document builds green, the page carries the formula as a heading in the colour of a heading, and its first half stands in the running head of the next page — measured on 2026-09-21 in a twelve-page analysis of another company of this family. A line break inside display math means nothing to TeX, so the block is joined and reads as the formula that was written. A fenced code block is left alone: the dollars there are an example of themselves. What comes back is the file pandoc reads — the source itself where there was nothing to join, a copy in the build directory otherwise — and the number of formulas that were joined. """ with open(source, encoding="utf-8", errors="replace") as handle: lines = handle.read().splitlines() written, joined, index, fenced = [], 0, 0, False while index < len(lines): line = lines[index] if FENCE.match(line): fenced = not fenced stripped = line.strip() open_math = (not fenced and stripped.startswith("$$") and not (len(stripped) > 3 and stripped.endswith("$$"))) if not open_math: written.append(line) index += 1 continue # The opening line ENDS with the two dollars as well, so the block grows # by one line before the closing dollars are looked for. block, index = [stripped], index + 1 while index < len(lines): block.append(lines[index].strip()) index += 1 if block[-1].endswith("$$"): break if len(block) > 1 and block[-1].endswith("$$"): written.append(" ".join(part for part in block if part)) joined += 1 else: written += block if not joined: return source, 0 copy = os.path.join(build, os.path.basename(source)) with open(copy, "w", encoding="utf-8") as handle: handle.write("\n".join(written) + "\n") return copy, joined def to_latex(source, tex, build, options, measured): """pandoc: the Markdown body plus the preamble that loads the template.""" source_dir = os.path.dirname(os.path.abspath(source)) read, joined = join_display_math(source, build) if joined: print(f"{os.path.basename(source)}: {joined} formula(s) over several " "lines joined into one line") tag = language(source, options.lang) filters = [] for name in FILTERS: filters += ["--lua-filter", name] command = ["pandoc", os.path.abspath(read), "--standalone", "--from", MARKDOWN, "--to", "latex", "--resource-path", source_dir] + filters + [ "-V", "documentclass=article", "-V", "classoption=" + options.paper, "-V", "fontsize=" + options.fontsize, "-V", "header-includes=\\usepackage{" + options.identity + "}", "-V", "header-includes=\\usepackage{business-suite}", "-o", tex] if tag: command += ["-M", "lang=" + tag] print(f"{os.path.basename(source)}: {tag}") if options.toc: command.append("--toc") if not options.highlight: command.append("--no-highlight") return run(command, cwd=build, environment=dict({"MD2PDF_BUILD": build, "MD2PDF_SOURCE_DIR": source_dir}, **measured)) def row_rules(tex): """A line between two rows of a table, drawn by the package. pandoc writes the three rules of a table and nothing between the rows. The rule itself belongs to the package, `\\businessrowrule`, and here it is put behind every row of a table body: a row ends with `\\\\` at the end of a line, the body begins after `\\endlastfoot`, and the last row of it needs none, because the table closes with its own rule underneath. """ with open(tex, encoding="utf-8") as handle: lines = handle.read().splitlines() written, body = [], False for index, line in enumerate(lines): written.append(line) if line.startswith("\\endlastfoot"): body = True elif line.startswith("\\end{longtable}"): body = False elif body and line.rstrip().endswith("\\\\") \ and not lines[index + 1].startswith("\\end{longtable}"): written.append("\\businessrowrule") with open(tex, "w", encoding="utf-8") as handle: handle.write("\n".join(written) + "\n") def to_pdf(tex, build, source_dir): """xelatex, through latexmk, with as many runs as the document needs.""" return run(["latexmk", "-xelatex", "-interaction=nonstopmode", "-file-line-error", "-emulate-aux-dir", "-auxdir=" + build, "-outdir=" + build, tex], cwd=source_dir, environment={"TEXINPUTS": searchpath(source_dir, build)}) def complaints(build, stem, result): """What LaTeX complained about, for a reader who has to fix the document.""" log = os.path.join(build, stem + ".log") text = "" if os.path.exists(log): with open(log, encoding="utf-8", errors="replace") as handle: text = handle.read() found = LATEX_ERROR.findall(text) return found or [line for line in result.stdout.splitlines() if line.strip()][-10:] def convert(source, options): """One document; the message of a failure comes back, None means done.""" if not os.path.exists(source): return f"{source}: there is no such file" source_dir = os.path.dirname(os.path.abspath(source)) or os.getcwd() stem = os.path.splitext(os.path.basename(source))[0] # What is kept lands where the PDF lands. Whoever names a directory for the # result has said where this document may write; the directory of the source # can belong to somebody else, and a build that keeps its files leaves # twenty of them there. build = os.path.join(os.path.abspath(options.out_dir or source_dir), "build") if options.keep_tex \ else tempfile.mkdtemp(prefix="md2pdf-") os.makedirs(build, exist_ok=True) keep = options.keep_tex try: tex = os.path.join(build, stem + ".tex") measured = metrics(options, build, source_dir) written = to_latex(source, tex, build, options, measured) if written.returncode != 0 or not os.path.exists(tex): keep = True return f"{source}: pandoc could not read the document:\n" \ f"{written.stdout.strip()}" row_rules(tex) built = to_pdf(tex, build, source_dir) pdf = os.path.join(build, stem + ".pdf") if built.returncode != 0 or not os.path.exists(pdf): keep = True return f"{source}: LaTeX could not build the document:\n " \ + "\n ".join(complaints(build, stem, built)) \ + f"\n the LaTeX file is {tex}" target = os.path.join(options.out_dir or source_dir, stem + ".pdf") os.makedirs(os.path.dirname(os.path.abspath(target)), exist_ok=True) shutil.copyfile(pdf, target) print(target) return None finally: if not keep: shutil.rmtree(build, ignore_errors=True) def main(arguments=None): parser = argparse.ArgumentParser( description="Markdown to PDF in the design of a company.") parser.add_argument("documents", nargs="+", metavar="file.md") parser.add_argument("--identity", required=True) parser.add_argument("--out-dir") parser.add_argument("--lang", default="auto") parser.add_argument("--paper", default="a4paper") parser.add_argument("--fontsize", default="10pt") parser.add_argument("--toc", action="store_true") parser.add_argument("--highlight", action="store_true") parser.add_argument("--keep-tex", action="store_true") options = parser.parse_args(arguments) failures = [failure for failure in (convert(document, options) for document in options.documents) if failure] for failure in failures: print(failure, file=sys.stderr) return 1 if failures else 0 if __name__ == "__main__": sys.exit(main())