One LaTeX design for a company, in three templates
A company that writes with LaTeX ends up with a document class, a letter class and a presentation theme that share a design and not a line of code. Every fix is then made three times, two of them late and the third never. Here the page is written once. A company declares its colours, faces and logo in one file, and the document template, the letter class and the presentation theme read that one file. Nothing in the suite carries a colour value, a font name or a file name of any company, which is what lets it be published while the companies stay private. Every measure of the page follows from a measurement or from a named definition: the room the head and the foot need is taken from the boxes they really build, one line of the body text stands between the head and the text and between the text and the foot, a heading never stands alone at the foot of a page, a picture takes the size of the family, and a Markdown file reaches the same page as the same document written in LaTeX. The suite ships with a company that does not exist, Nordwind AG, so that it builds and is measured anywhere: three colours, the TeX Gyre families every installation carries, and an icon drawn in TikZ. 574 checks over five measurements, all of them on the rendered page or on the build log rather than on the source. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
commit
adc58e19d7
39 files changed
+6627
No files matched your search
@@ -0,0 +1,55 @@
|
||||
-- Copyright (c) 2026 Marc Wäckerlin. SPDX-License-Identifier: MIT
|
||||
-- Long inline code gets the break opportunities a path or an identifier
|
||||
-- offers, so that it wraps instead of standing outside the page.
|
||||
--
|
||||
-- `payment/stripe_utils/stripe_connector.py:214` is one word to LaTeX: it
|
||||
-- carries no space, and hyphenation patterns do not apply to a typewriter face,
|
||||
-- so the line runs over the right edge and the last characters stand next to
|
||||
-- the paper. Measured over one report: 30 of 44 places where something left the
|
||||
-- type area were spans of this kind, up to 44 points out.
|
||||
--
|
||||
-- A break is offered after the characters that structure such a name, and
|
||||
-- before a capital that follows a small letter, which is where a reader of
|
||||
-- `CreditCardCustomerBasicInfoSerializer` reads a boundary anyway. No hyphen is
|
||||
-- inserted: a hyphen in a path would be read as part of the path.
|
||||
--
|
||||
-- Only a span from 16 characters up is treated. The shortest one that left the
|
||||
-- type area in that document was `plim/settings.py:25` at 19 characters, and a
|
||||
-- short span finds room on the next line by itself, where breaking it would
|
||||
-- only make it harder to read.
|
||||
|
||||
local THRESHOLD = 16
|
||||
|
||||
local function pieces(text)
|
||||
local parts, current = {}, ''
|
||||
for index = 1, #text do
|
||||
local char = text:sub(index, index)
|
||||
local following = text:sub(index + 1, index + 1)
|
||||
current = current .. char
|
||||
local structural = char:match('[/._:%-]') and following ~= ''
|
||||
local camel = char:match('%l') and following:match('%u')
|
||||
if structural or camel then
|
||||
parts[#parts + 1] = current
|
||||
current = ''
|
||||
end
|
||||
end
|
||||
if current ~= '' then
|
||||
parts[#parts + 1] = current
|
||||
end
|
||||
return parts
|
||||
end
|
||||
|
||||
function Code(code)
|
||||
if not FORMAT:match('latex') then return nil end
|
||||
if #code.text < THRESHOLD then return nil end
|
||||
local parts = pieces(code.text)
|
||||
if #parts < 2 then return nil end
|
||||
local result = {}
|
||||
for index, part in ipairs(parts) do
|
||||
result[#result + 1] = pandoc.Code(part)
|
||||
if index < #parts then
|
||||
result[#result + 1] = pandoc.RawInline('latex', '\\allowbreak{}')
|
||||
end
|
||||
end
|
||||
return result
|
||||
end
|
||||
@@ -0,0 +1,32 @@
|
||||
-- Copyright (c) 2026 Marc Wäckerlin. SPDX-License-Identifier: MIT
|
||||
-- A folded block does not belong in a document nobody can click.
|
||||
--
|
||||
-- `<details>` with its `<summary>` is a fold of the browser: the reader opens
|
||||
-- it when he wants it and sees a single line meanwhile. On paper there is
|
||||
-- nothing to open, so the fold arrives unfolded — measured in one report, the
|
||||
-- diagram sources behind it filled whole pages under the pictures they draw.
|
||||
-- The block goes out with everything in it, the summary line included.
|
||||
|
||||
local OPEN = '<details'
|
||||
local CLOSE = '</details>'
|
||||
|
||||
local function marker(block)
|
||||
return block.t == 'RawBlock' and block.format == 'html' and block.text or ''
|
||||
end
|
||||
|
||||
function Blocks(blocks)
|
||||
local kept, depth = {}, 0
|
||||
for _, block in ipairs(blocks) do
|
||||
local raw = marker(block)
|
||||
local opens = raw:find(OPEN, 1, true)
|
||||
local closes = raw:find(CLOSE, 1, true)
|
||||
if opens then
|
||||
depth = depth + (closes and 0 or 1)
|
||||
elseif closes then
|
||||
depth = math.max(0, depth - 1)
|
||||
elseif depth == 0 then
|
||||
kept[#kept + 1] = block
|
||||
end
|
||||
end
|
||||
return kept
|
||||
end
|
||||
@@ -0,0 +1,77 @@
|
||||
-- Copyright (c) 2026 Marc Wäckerlin. SPDX-License-Identifier: MIT
|
||||
-- A picture that stands alone in its paragraph gets its caption under it.
|
||||
--
|
||||
-- `` carries a text between the brackets, and
|
||||
-- every reader who writes it means it as the caption. GFM has no such rule: the
|
||||
-- text becomes the alternative text of the picture, which reaches a screen
|
||||
-- reader and nobody else, and the PDF shows a picture with nothing under it.
|
||||
--
|
||||
-- pandoc has the rule under the name implicit_figures, and its GFM reader does
|
||||
-- not carry it, so the paragraph is turned into a figure here. What the figure
|
||||
-- then looks like — where it stands, how the caption is set — comes from the
|
||||
-- package like everything else.
|
||||
|
||||
-- A picture that stands directly under a heading does not float away from it.
|
||||
--
|
||||
-- A floating picture leaves the page it was written on, and where the heading
|
||||
-- announced nothing but that picture, the heading stays behind alone at the
|
||||
-- foot of the page. Measured over one report: a page ended with its heading,
|
||||
-- 430 points of it still empty, and its picture began on the next. The
|
||||
-- reservation a heading takes cannot answer this — the room was there, and what
|
||||
-- should have filled it had left the page.
|
||||
--
|
||||
-- So this one picture is placed where it stands. It then either fits under its
|
||||
-- heading, or the page breaks BEFORE the heading and the two travel together,
|
||||
-- which is what a reader expects of a heading.
|
||||
local function nailed(figure)
|
||||
return {
|
||||
pandoc.RawBlock('latex', '\\begingroup\\floatplacement{figure}{H}'),
|
||||
figure,
|
||||
pandoc.RawBlock('latex', '\\endgroup'),
|
||||
}
|
||||
end
|
||||
|
||||
function Blocks(blocks)
|
||||
if not FORMAT:match('latex') then return nil end
|
||||
local kept, previous = pandoc.Blocks{}, nil
|
||||
for _, block in ipairs(blocks) do
|
||||
if block.t == 'Figure' and previous and previous.t == 'Header' then
|
||||
for _, piece in ipairs(nailed(block)) do
|
||||
kept:insert(piece)
|
||||
end
|
||||
else
|
||||
kept:insert(block)
|
||||
end
|
||||
previous = block
|
||||
end
|
||||
return kept
|
||||
end
|
||||
|
||||
-- A picture of pixels says so, because only this side knows the file name.
|
||||
--
|
||||
-- How large a picture may be drawn is the package's rule, and it is not the
|
||||
-- same rule for a drawing and for a photograph: a drawing is drawn again at
|
||||
-- every size, a raster of pixels gets coarse. What the package sees is the
|
||||
-- finished box, which carries no file name any more, so the mark is set here
|
||||
-- and read by the picture that follows it.
|
||||
local RASTER = {
|
||||
png = true, jpg = true, jpeg = true, bmp = true,
|
||||
gif = true, tif = true, tiff = true,
|
||||
}
|
||||
|
||||
function Image(image)
|
||||
if not FORMAT:match('latex') then return nil end
|
||||
local extension = image.src:match('%.([%a%d]+)$')
|
||||
if extension and RASTER[extension:lower()] then
|
||||
return { pandoc.RawInline('latex', '\\businessbitmap'), image }
|
||||
end
|
||||
end
|
||||
|
||||
function Para(para)
|
||||
if #para.content ~= 1 then return nil end
|
||||
local image = para.content[1]
|
||||
if image.t ~= 'Image' or #image.caption == 0 then return nil end
|
||||
return pandoc.Figure(
|
||||
pandoc.Blocks{ pandoc.Plain{ image } },
|
||||
{ long = pandoc.Blocks{ pandoc.Plain(image.caption) } })
|
||||
end
|
||||
@@ -0,0 +1,64 @@
|
||||
-- Copyright (c) 2026 Marc Wäckerlin. SPDX-License-Identifier: MIT
|
||||
-- A sentence that announces a table or a list keeps it on its page.
|
||||
--
|
||||
-- "The numbers at a glance:" stands as the last line of a page while the table
|
||||
-- it announces begins on the next one. It is the case of the heading in another
|
||||
-- shape, and the penalty that answers the heading does not answer this one: a
|
||||
-- longtable decides at its own start whether its head and its first row still
|
||||
-- fit and breaks the page itself, past every penalty behind the sentence. The
|
||||
-- room has to stand in FRONT of the sentence, and how much room that is comes
|
||||
-- from the package, which learns it from the table.
|
||||
--
|
||||
-- What marks such a sentence is the colon it ends with together with what
|
||||
-- follows it: a table or a list of any kind. A sentence that ends in a colon
|
||||
-- and is followed by a paragraph announces nothing that can be torn from it.
|
||||
--
|
||||
-- This filter runs BEFORE the one that gives a table its column widths, because
|
||||
-- that one wraps a table in a group of its own and the table would then no
|
||||
-- longer be what follows the sentence.
|
||||
local ANNOUNCED = {
|
||||
Table = true,
|
||||
BulletList = true,
|
||||
OrderedList = true,
|
||||
DefinitionList = true,
|
||||
}
|
||||
|
||||
-- The one thing counted here is how many lines the SENTENCE takes, because a
|
||||
-- sentence is a text and nothing in the package can know it. What the block
|
||||
-- under it needs to begin is measured by the package: a table at the table
|
||||
-- itself, through the aux file, and a list against the floor the package sets.
|
||||
-- So nothing is added for it here.
|
||||
--
|
||||
-- How many characters a line holds is measured and arrives in the environment,
|
||||
-- the same two numbers the table filter uses; the fallback is the measurement
|
||||
-- of the origin of this family.
|
||||
local LINE = (tonumber(os.getenv('MD2PDF_LINEWIDTH')) or 538)
|
||||
/ (tonumber(os.getenv('MD2PDF_CHARACTER')) or 5.3)
|
||||
|
||||
local function lines(block)
|
||||
local length = #pandoc.utils.stringify(block)
|
||||
return math.max(1, math.ceil(length / LINE))
|
||||
end
|
||||
|
||||
local function announces(block, following)
|
||||
if block.t ~= 'Para' or not following or not ANNOUNCED[following.t] then
|
||||
return false
|
||||
end
|
||||
return pandoc.utils.stringify(block):match(':%s*$') ~= nil
|
||||
end
|
||||
|
||||
function Blocks(blocks)
|
||||
if not FORMAT:match('latex') then return nil end
|
||||
local kept = pandoc.Blocks{}
|
||||
for index, block in ipairs(blocks) do
|
||||
if announces(block, blocks[index + 1]) then
|
||||
kept:insert(pandoc.RawBlock(
|
||||
'latex', '\\businessleadin[' .. lines(block) .. ']'))
|
||||
kept:insert(block)
|
||||
kept:insert(pandoc.RawBlock('latex', '\\businesstogether'))
|
||||
else
|
||||
kept:insert(block)
|
||||
end
|
||||
end
|
||||
return kept
|
||||
end
|
||||
@@ -0,0 +1,29 @@
|
||||
-- Copyright (c) 2026 Marc Wäckerlin. SPDX-License-Identifier: MIT
|
||||
-- A link to a Markdown file of the same set points at the PDF of that file.
|
||||
--
|
||||
-- The documents reference each other, `[ARCHITECTURE.md](ARCHITECTURE.md)`, and
|
||||
-- that is right in the repository, where the reader opens the Markdown. In the
|
||||
-- PDF the same target opens nothing: the reader holds a PDF, the file beside it
|
||||
-- is a PDF, and the link reaches for a source he does not have.
|
||||
--
|
||||
-- So a target that names a local Markdown file is rewritten to the PDF that the
|
||||
-- same converter produces from it, with its anchor kept. An address with a
|
||||
-- scheme (https:, mailto:) and an anchor inside the same document stay as they
|
||||
-- are.
|
||||
local function rewritten(target)
|
||||
if target:match('^%a[%w+.-]*:') or target:match('^#') then
|
||||
return nil
|
||||
end
|
||||
local file, anchor = target:match('^([^#]+)(#?.*)$')
|
||||
if not file or not file:lower():match('%.md$') then
|
||||
return nil
|
||||
end
|
||||
return file:sub(1, #file - 3) .. '.pdf' .. anchor
|
||||
end
|
||||
|
||||
function Link(link)
|
||||
local target = rewritten(link.target)
|
||||
if not target then return nil end
|
||||
link.target = target
|
||||
return link
|
||||
end
|
||||
Executable
+385
@@ -0,0 +1,385 @@
|
||||
#!/usr/bin/env python3
|
||||
# Copyright (c) 2026 Marc Wäckerlin. SPDX-License-Identifier: MIT
|
||||
"""Turn a Markdown file into a PDF in the design of a company.
|
||||
|
||||
The design is not written here. pandoc turns the Markdown into the body of a
|
||||
LaTeX document whose preamble loads the identity of the company and the template
|
||||
of the suite, and xelatex builds it: what comes out is the example document of
|
||||
the suite with a Markdown body, in the font of the company, with its logo in the
|
||||
page header and with the title block of the template.
|
||||
|
||||
md2pdf.py --identity <company-identity> <file.md> [<file.md> …]
|
||||
|
||||
A company links the converter into a directory of the PATH under its own name and
|
||||
hands it its identity there, so that several of them lie side by side and every
|
||||
call says which design it writes:
|
||||
|
||||
#!/bin/sh
|
||||
exec md2pdf.py --identity company-identity "$@"
|
||||
|
||||
The PDF is written next to its source; everything produced on the way lives in a
|
||||
throwaway directory that is removed again. A failed build keeps that directory,
|
||||
names the LaTeX file in it, and prints the lines LaTeX complained about.
|
||||
|
||||
Options are the values that differ per document or per person; nothing else is
|
||||
adjustable, because everything else is the design of the template:
|
||||
|
||||
--identity <name> the package with the colours, fonts and logo of the company
|
||||
--out-dir <dir> write the PDF here instead of beside the source
|
||||
--lang <tag> hyphenation and language of the document. The default
|
||||
`auto` reads it off the words of the text, and leaves it
|
||||
to the document where that carries a `lang:` of its own;
|
||||
a tag given here decides over both
|
||||
--paper <name> a4paper (default), a5paper, letterpaper …
|
||||
--fontsize <pt> 10pt (default, the company size), 11pt, 12pt
|
||||
--toc add a table of contents
|
||||
--highlight colour the code blocks (pandoc's own colours)
|
||||
--keep-tex keep the LaTeX file and everything the build produced,
|
||||
in a build/ beside the PDF — under `--out-dir` where one
|
||||
is given, beside the source where none is
|
||||
"""
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
# Where the suite lies, read through every link on the way. The converter is
|
||||
# meant to be called by its name, so it is linked into a directory of the PATH,
|
||||
# and the name it was started under says where the link stands, never where the
|
||||
# suite is: the filters beside it would then be looked for in the directory of
|
||||
# the link, and the document would be built without them.
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.realpath(__file__)))
|
||||
FILTERS = [os.path.join(ROOT, "bin", name) for name in
|
||||
("details.lua", "svg.lua", "links.lua",
|
||||
# Before the table filter: that one wraps a table in a group of its
|
||||
# own, and the sentence above would then no longer stand directly in
|
||||
# front of a table.
|
||||
"leadin.lua", "tables.lua", "code.lua", "figures.lua")]
|
||||
|
||||
# Which language a document is written in decides where LaTeX may break a word,
|
||||
# and the wrong patterns break it in the wrong places: German patterns put
|
||||
# "li-nes" into an English table cell and leave words unbroken that then stand
|
||||
# outside the page. Nothing in a Markdown file says the language, so it is read
|
||||
# off the words themselves — the most common words of a language are the ones a
|
||||
# text of any length repeats. The document decides when it carries a `lang:` of
|
||||
# its own, and an explicit --lang decides over both.
|
||||
STOPWORDS = {
|
||||
"de-CH": frozenset((
|
||||
"der die das und ist nicht ein eine den dem des mit für auf von im zu "
|
||||
"sich werden wird sind auch als aus bei nach über oder aber wenn dass "
|
||||
"was noch nur schon kann muss haben hat ich wir sie er es").split()),
|
||||
"en-GB": frozenset((
|
||||
"the and is not are with for from this that of to in it as be by on "
|
||||
"at or but if what still only can must have has we they he she").split()),
|
||||
}
|
||||
WORDS = re.compile(r"[a-zäöüßA-ZÄÖÜ]+")
|
||||
DOCUMENT_LANGUAGE = re.compile(r"^lang:", re.MULTILINE)
|
||||
|
||||
# gfm is what GitHub and the editor preview show. The extensions on top are what
|
||||
# a document of ours uses and gfm does not carry by itself: the YAML block that
|
||||
# gives title, author and date, footnotes, definition lists, and the attributes
|
||||
# behind a picture, `{width=30%}`, which decide how wide it stands. Without that
|
||||
# last one the braces are printed into the text.
|
||||
#
|
||||
# Formulas need no extension here: `--list-extensions=gfm` lists
|
||||
# `tex_math_dollars` and `tex_math_gfm` as ON, so `$x$` and `$$x$$` arrive as
|
||||
# math with this reader. Measured on 2026-09-21 against the same document with
|
||||
# and without the extension written out: the two parse trees are identical.
|
||||
MARKDOWN = ("gfm+yaml_metadata_block+footnotes+definition_lists"
|
||||
"+attributes")
|
||||
|
||||
# Where a fenced code block begins and ends. Inside one, nothing is repaired:
|
||||
# what stands there is an example of itself.
|
||||
FENCE = re.compile(r"^\s*(```|~~~)")
|
||||
|
||||
# A LaTeX error carries its place in this form, because latexmk is called with
|
||||
# -file-line-error: ./file.tex:42: Undefined control sequence.
|
||||
LATEX_ERROR = re.compile(r"^(?:[^\s:]+):\d+: .*$", re.MULTILINE)
|
||||
|
||||
# What the column widths of a table are computed from: how wide the line is and
|
||||
# how much of it one character takes. Both belong to the paper, the margins and
|
||||
# the face of the company, so both are measured in a probe document instead of
|
||||
# being written down — the origin of this family carried 538 and 5.3 as
|
||||
# constants in the filter, and the second company of the family carried 481 and
|
||||
# 4.6 for the same page, a number nobody could account for afterwards.
|
||||
#
|
||||
# The reference line is set once and its width divided by its length. It is the
|
||||
# alphabet twice over with the spaces a text has, so the average is that of a
|
||||
# text and not of one word.
|
||||
REFERENCE = ("the quick brown fox jumps over the lazy dog "
|
||||
"und der flinke braune fuchs springt ueber den faulen hund")
|
||||
PROBE = r"""\documentclass[%(paper)s,%(fontsize)s]{article}
|
||||
\usepackage{%(identity)s}
|
||||
\usepackage{business-suite}
|
||||
\begin{document}
|
||||
\newwrite\businessmetrics
|
||||
\immediate\openout\businessmetrics=metrics.txt
|
||||
\sbox0{%(reference)s}%%
|
||||
\immediate\write\businessmetrics{linewidth \the\linewidth}%%
|
||||
\immediate\write\businessmetrics{reference \the\wd0}%%
|
||||
\immediate\closeout\businessmetrics
|
||||
\end{document}
|
||||
"""
|
||||
LENGTH = re.compile(r"^(\w+) ([0-9.]+)pt$", re.MULTILINE)
|
||||
|
||||
|
||||
def run(command, cwd, environment=None):
|
||||
"""One external command; its output comes back as text."""
|
||||
return subprocess.run(command, cwd=cwd,
|
||||
env=dict(os.environ, **(environment or {})),
|
||||
stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
|
||||
encoding="utf-8", errors="replace")
|
||||
|
||||
|
||||
def language(source, given):
|
||||
"""The language whose hyphenation patterns fit this document.
|
||||
|
||||
None means that nobody has to be told: either the document says it itself,
|
||||
or the text gives no answer and the class keeps its default.
|
||||
"""
|
||||
if given and given != "auto":
|
||||
return given
|
||||
with open(source, encoding="utf-8", errors="replace") as handle:
|
||||
text = handle.read()
|
||||
if DOCUMENT_LANGUAGE.search(text):
|
||||
return None
|
||||
words = [word.lower() for word in WORDS.findall(text)]
|
||||
counted = {tag: sum(word in stopwords for word in words)
|
||||
for tag, stopwords in STOPWORDS.items()}
|
||||
best = max(counted, key=counted.get)
|
||||
return best if counted[best] else None
|
||||
|
||||
|
||||
def searchpath(source_dir, build):
|
||||
"""Where LaTeX looks: the suite, the document, the build directory."""
|
||||
return os.pathsep.join([ROOT + "//", source_dir + "//", build + "//", ""])
|
||||
|
||||
|
||||
def metrics(options, build, source_dir):
|
||||
"""The width of the line and of a character, measured in a probe.
|
||||
|
||||
An empty answer means the probe did not build; the filters then fall back to
|
||||
the measurement of the origin of this family, and the table of a company with
|
||||
another paper comes out too narrow instead of not at all.
|
||||
"""
|
||||
probe = os.path.join(build, "metrics.tex")
|
||||
with open(probe, "w", encoding="utf-8") as handle:
|
||||
handle.write(PROBE % {"paper": options.paper,
|
||||
"fontsize": options.fontsize,
|
||||
"identity": options.identity,
|
||||
"reference": REFERENCE})
|
||||
run(["xelatex", "-no-pdf", "-interaction=nonstopmode", "-halt-on-error",
|
||||
"metrics.tex"], cwd=build,
|
||||
environment={"TEXINPUTS": searchpath(source_dir, build)})
|
||||
written = os.path.join(build, "metrics.txt")
|
||||
if not os.path.exists(written):
|
||||
return {}
|
||||
with open(written, encoding="utf-8", errors="replace") as handle:
|
||||
found = dict(LENGTH.findall(handle.read()))
|
||||
if "linewidth" not in found or "reference" not in found:
|
||||
return {}
|
||||
return {"MD2PDF_LINEWIDTH": found["linewidth"],
|
||||
"MD2PDF_CHARACTER": str(float(found["reference"])
|
||||
/ len(REFERENCE))}
|
||||
|
||||
|
||||
def join_display_math(source, build):
|
||||
"""A formula over several lines, joined into one before pandoc reads it.
|
||||
|
||||
`$$` around a formula is display math, and a writer breaks it over lines to
|
||||
keep it readable. The grammar of the reader looks at those lines first: one
|
||||
that carries nothing but `=` is the UNDERLINE OF A HEADING, so the line
|
||||
above it becomes a heading of the first level and the rest of the formula a
|
||||
paragraph. The document builds green, the page carries the formula as a
|
||||
heading in the colour of a heading, and its first half stands in the running
|
||||
head of the next page — measured on 2026-09-21 in a twelve-page analysis of
|
||||
another company of this family.
|
||||
|
||||
A line break inside display math means nothing to TeX, so the block is
|
||||
joined and reads as the formula that was written. A fenced code block is
|
||||
left alone: the dollars there are an example of themselves.
|
||||
|
||||
What comes back is the file pandoc reads — the source itself where there was
|
||||
nothing to join, a copy in the build directory otherwise — and the number of
|
||||
formulas that were joined.
|
||||
"""
|
||||
with open(source, encoding="utf-8", errors="replace") as handle:
|
||||
lines = handle.read().splitlines()
|
||||
written, joined, index, fenced = [], 0, 0, False
|
||||
while index < len(lines):
|
||||
line = lines[index]
|
||||
if FENCE.match(line):
|
||||
fenced = not fenced
|
||||
stripped = line.strip()
|
||||
open_math = (not fenced and stripped.startswith("$$")
|
||||
and not (len(stripped) > 3 and stripped.endswith("$$")))
|
||||
if not open_math:
|
||||
written.append(line)
|
||||
index += 1
|
||||
continue
|
||||
# The opening line ENDS with the two dollars as well, so the block grows
|
||||
# by one line before the closing dollars are looked for.
|
||||
block, index = [stripped], index + 1
|
||||
while index < len(lines):
|
||||
block.append(lines[index].strip())
|
||||
index += 1
|
||||
if block[-1].endswith("$$"):
|
||||
break
|
||||
if len(block) > 1 and block[-1].endswith("$$"):
|
||||
written.append(" ".join(part for part in block if part))
|
||||
joined += 1
|
||||
else:
|
||||
written += block
|
||||
if not joined:
|
||||
return source, 0
|
||||
copy = os.path.join(build, os.path.basename(source))
|
||||
with open(copy, "w", encoding="utf-8") as handle:
|
||||
handle.write("\n".join(written) + "\n")
|
||||
return copy, joined
|
||||
|
||||
|
||||
def to_latex(source, tex, build, options, measured):
|
||||
"""pandoc: the Markdown body plus the preamble that loads the template."""
|
||||
source_dir = os.path.dirname(os.path.abspath(source))
|
||||
read, joined = join_display_math(source, build)
|
||||
if joined:
|
||||
print(f"{os.path.basename(source)}: {joined} formula(s) over several "
|
||||
"lines joined into one line")
|
||||
tag = language(source, options.lang)
|
||||
filters = []
|
||||
for name in FILTERS:
|
||||
filters += ["--lua-filter", name]
|
||||
command = ["pandoc", os.path.abspath(read),
|
||||
"--standalone", "--from", MARKDOWN, "--to", "latex",
|
||||
"--resource-path", source_dir] + filters + [
|
||||
"-V", "documentclass=article",
|
||||
"-V", "classoption=" + options.paper,
|
||||
"-V", "fontsize=" + options.fontsize,
|
||||
"-V", "header-includes=\\usepackage{" + options.identity + "}",
|
||||
"-V", "header-includes=\\usepackage{business-suite}",
|
||||
"-o", tex]
|
||||
if tag:
|
||||
command += ["-M", "lang=" + tag]
|
||||
print(f"{os.path.basename(source)}: {tag}")
|
||||
if options.toc:
|
||||
command.append("--toc")
|
||||
if not options.highlight:
|
||||
command.append("--no-highlight")
|
||||
return run(command, cwd=build,
|
||||
environment=dict({"MD2PDF_BUILD": build,
|
||||
"MD2PDF_SOURCE_DIR": source_dir}, **measured))
|
||||
|
||||
|
||||
def row_rules(tex):
|
||||
"""A line between two rows of a table, drawn by the package.
|
||||
|
||||
pandoc writes the three rules of a table and nothing between the rows. The
|
||||
rule itself belongs to the package, `\\businessrowrule`, and here it is put
|
||||
behind every row of a table body: a row ends with `\\\\` at the end of a
|
||||
line, the body begins after `\\endlastfoot`, and the last row of it needs
|
||||
none, because the table closes with its own rule underneath.
|
||||
"""
|
||||
with open(tex, encoding="utf-8") as handle:
|
||||
lines = handle.read().splitlines()
|
||||
written, body = [], False
|
||||
for index, line in enumerate(lines):
|
||||
written.append(line)
|
||||
if line.startswith("\\endlastfoot"):
|
||||
body = True
|
||||
elif line.startswith("\\end{longtable}"):
|
||||
body = False
|
||||
elif body and line.rstrip().endswith("\\\\") \
|
||||
and not lines[index + 1].startswith("\\end{longtable}"):
|
||||
written.append("\\businessrowrule")
|
||||
with open(tex, "w", encoding="utf-8") as handle:
|
||||
handle.write("\n".join(written) + "\n")
|
||||
|
||||
|
||||
def to_pdf(tex, build, source_dir):
|
||||
"""xelatex, through latexmk, with as many runs as the document needs."""
|
||||
return run(["latexmk", "-xelatex", "-interaction=nonstopmode",
|
||||
"-file-line-error", "-emulate-aux-dir",
|
||||
"-auxdir=" + build, "-outdir=" + build, tex],
|
||||
cwd=source_dir,
|
||||
environment={"TEXINPUTS": searchpath(source_dir, build)})
|
||||
|
||||
|
||||
def complaints(build, stem, result):
|
||||
"""What LaTeX complained about, for a reader who has to fix the document."""
|
||||
log = os.path.join(build, stem + ".log")
|
||||
text = ""
|
||||
if os.path.exists(log):
|
||||
with open(log, encoding="utf-8", errors="replace") as handle:
|
||||
text = handle.read()
|
||||
found = LATEX_ERROR.findall(text)
|
||||
return found or [line for line in result.stdout.splitlines()
|
||||
if line.strip()][-10:]
|
||||
|
||||
|
||||
def convert(source, options):
|
||||
"""One document; the message of a failure comes back, None means done."""
|
||||
if not os.path.exists(source):
|
||||
return f"{source}: there is no such file"
|
||||
source_dir = os.path.dirname(os.path.abspath(source)) or os.getcwd()
|
||||
stem = os.path.splitext(os.path.basename(source))[0]
|
||||
# What is kept lands where the PDF lands. Whoever names a directory for the
|
||||
# result has said where this document may write; the directory of the source
|
||||
# can belong to somebody else, and a build that keeps its files leaves
|
||||
# twenty of them there.
|
||||
build = os.path.join(os.path.abspath(options.out_dir or source_dir),
|
||||
"build") if options.keep_tex \
|
||||
else tempfile.mkdtemp(prefix="md2pdf-")
|
||||
os.makedirs(build, exist_ok=True)
|
||||
keep = options.keep_tex
|
||||
try:
|
||||
tex = os.path.join(build, stem + ".tex")
|
||||
measured = metrics(options, build, source_dir)
|
||||
written = to_latex(source, tex, build, options, measured)
|
||||
if written.returncode != 0 or not os.path.exists(tex):
|
||||
keep = True
|
||||
return f"{source}: pandoc could not read the document:\n" \
|
||||
f"{written.stdout.strip()}"
|
||||
row_rules(tex)
|
||||
built = to_pdf(tex, build, source_dir)
|
||||
pdf = os.path.join(build, stem + ".pdf")
|
||||
if built.returncode != 0 or not os.path.exists(pdf):
|
||||
keep = True
|
||||
return f"{source}: LaTeX could not build the document:\n " \
|
||||
+ "\n ".join(complaints(build, stem, built)) \
|
||||
+ f"\n the LaTeX file is {tex}"
|
||||
target = os.path.join(options.out_dir or source_dir, stem + ".pdf")
|
||||
os.makedirs(os.path.dirname(os.path.abspath(target)), exist_ok=True)
|
||||
shutil.copyfile(pdf, target)
|
||||
print(target)
|
||||
return None
|
||||
finally:
|
||||
if not keep:
|
||||
shutil.rmtree(build, ignore_errors=True)
|
||||
|
||||
|
||||
def main(arguments=None):
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Markdown to PDF in the design of a company.")
|
||||
parser.add_argument("documents", nargs="+", metavar="file.md")
|
||||
parser.add_argument("--identity", required=True)
|
||||
parser.add_argument("--out-dir")
|
||||
parser.add_argument("--lang", default="auto")
|
||||
parser.add_argument("--paper", default="a4paper")
|
||||
parser.add_argument("--fontsize", default="10pt")
|
||||
parser.add_argument("--toc", action="store_true")
|
||||
parser.add_argument("--highlight", action="store_true")
|
||||
parser.add_argument("--keep-tex", action="store_true")
|
||||
options = parser.parse_args(arguments)
|
||||
failures = [failure for failure in
|
||||
(convert(document, options) for document in options.documents)
|
||||
if failure]
|
||||
for failure in failures:
|
||||
print(failure, file=sys.stderr)
|
||||
return 1 if failures else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
+33
@@ -0,0 +1,33 @@
|
||||
-- Copyright (c) 2026 Marc Wäckerlin. SPDX-License-Identifier: MIT
|
||||
-- Every drawing a document shows, converted while pandoc reads the document.
|
||||
--
|
||||
-- graphicx cannot read an SVG, and pandoc's own answer to one is the svg
|
||||
-- package, which calls inkscape from inside the LaTeX run and therefore needs
|
||||
-- shell escape — a document could then run any command it likes. So the drawing
|
||||
-- is converted here, before a single line of LaTeX exists, and the picture in
|
||||
-- the document points at the PDF that comes out.
|
||||
--
|
||||
-- The two directories arrive in the environment, from the converter: where the
|
||||
-- document lies, and where the build may write.
|
||||
|
||||
local build = os.getenv('MD2PDF_BUILD')
|
||||
local source = os.getenv('MD2PDF_SOURCE_DIR')
|
||||
|
||||
local function absolute(file)
|
||||
return file:sub(1, 1) == '/' and file or source .. '/' .. file
|
||||
end
|
||||
|
||||
function Image(image)
|
||||
if not image.src:lower():match('%.svg$') then return nil end
|
||||
local name = image.src:gsub('.*/', ''):gsub('%.[Ss][Vv][Gg]$', '') .. '.pdf'
|
||||
local target = build .. '/' .. name
|
||||
local converted, message = pcall(pandoc.pipe, 'inkscape',
|
||||
{'--export-type=pdf', '--export-filename=' .. target,
|
||||
absolute(image.src)}, '')
|
||||
if not converted then
|
||||
error('the drawing ' .. image.src .. ' could not be converted: '
|
||||
.. tostring(message))
|
||||
end
|
||||
image.src = target
|
||||
return image
|
||||
end
|
||||
+152
@@ -0,0 +1,152 @@
|
||||
-- Copyright (c) 2026 Marc Wäckerlin. SPDX-License-Identifier: MIT
|
||||
-- A table that is wider than the line gets relative column widths.
|
||||
--
|
||||
-- A Markdown pipe table says nothing about how wide its columns are, so pandoc
|
||||
-- writes LaTeX columns that never break a line. A table whose content is longer
|
||||
-- than the line then runs out of the type area and the last words stand outside
|
||||
-- the paper. With relative widths LaTeX wraps the cells.
|
||||
--
|
||||
-- The widths are shares of the space the columns have between them; pandoc
|
||||
-- writes every column as `p{(\linewidth - 2(n-1)\tabcolsep) * \real{w}}` and has
|
||||
-- therefore already taken the space between the columns off the line, so the
|
||||
-- shares here add up to one and to nothing else.
|
||||
--
|
||||
-- How wide the line is and how much of it a character takes are MEASURED, and
|
||||
-- they arrive in the environment: the converter builds a probe document with
|
||||
-- the identity of the company and reads `\linewidth` and the width of a reference
|
||||
-- line out of it. Both numbers belong to the paper, the margins and the face of
|
||||
-- the company, and none of the three is the same in two companies. The fallback is
|
||||
-- the measurement of the origin of this family, for a filter that is run
|
||||
-- without the converter.
|
||||
--
|
||||
-- Every table takes the whole line, whether its content needs it or not: a page
|
||||
-- of tables that each end somewhere else has no edge to read along.
|
||||
|
||||
local TYPE_AREA = tonumber(os.getenv('MD2PDF_LINEWIDTH')) or 538
|
||||
local CHARACTER = tonumber(os.getenv('MD2PDF_CHARACTER')) or 5.3
|
||||
-- What LaTeX keeps between any two columns, `2\tabcolsep`, whatever the size of
|
||||
-- the text.
|
||||
local GAP = 12
|
||||
|
||||
-- What a column needs at the very least: its longest word, plus the space that
|
||||
-- keeps it off its neighbour. A column narrower than that pushes its word into
|
||||
-- the next column, measured at eight columns where `Production` printed over
|
||||
-- `Test` and a word stood 23 points outside the page, while the shares still
|
||||
-- added up to one.
|
||||
local SEPARATOR = 3
|
||||
|
||||
-- Where even the least widths do not fit into the line, no share of it can
|
||||
-- help: the table needs more characters than the line holds, and the way a
|
||||
-- typographer gets them is a smaller face. The factors are the sizes of the
|
||||
-- class against the body size — at 10pt, \small is 9pt, \footnotesize 8pt and
|
||||
-- \scriptsize 7pt — so a line holds that much more of them.
|
||||
local SIZES = {
|
||||
{ name = '', factor = 1 },
|
||||
{ name = '\\small', factor = 10 / 9 },
|
||||
{ name = '\\footnotesize', factor = 10 / 8 },
|
||||
{ name = '\\scriptsize', factor = 10 / 7 },
|
||||
}
|
||||
|
||||
local function measure(tbl)
|
||||
local columns = #tbl.colspecs
|
||||
local longest, word = {}, {}
|
||||
for index = 1, columns do
|
||||
longest[index], word[index] = 0, 1
|
||||
end
|
||||
local function scan(rows)
|
||||
for _, row in ipairs(rows) do
|
||||
for index, cell in ipairs(row.cells) do
|
||||
if index <= columns then
|
||||
local text = pandoc.utils.stringify(cell)
|
||||
longest[index] = math.max(longest[index], #text)
|
||||
for piece in text:gmatch('%S+') do
|
||||
word[index] = math.max(word[index], #piece)
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
scan(tbl.head.rows)
|
||||
for _, body in ipairs(tbl.bodies) do
|
||||
scan(body.body)
|
||||
end
|
||||
return longest, word
|
||||
end
|
||||
|
||||
-- The head of a table is marked as the head. pandoc writes the row that names
|
||||
-- the columns in the face of the body, so a head and a first row look the same;
|
||||
-- a document written by hand marks its head cells with `\businesstablehead{…}`,
|
||||
-- and this puts the same command around every head cell of a Markdown table.
|
||||
-- How a head looks is then the package's decision, once for both ways.
|
||||
local function mark_head(tbl)
|
||||
local open = pandoc.RawInline('latex', '\\businesstablehead{')
|
||||
local close = pandoc.RawInline('latex', '}')
|
||||
for _, row in ipairs(tbl.head.rows) do
|
||||
for _, cell in ipairs(row.cells) do
|
||||
local function marked(inlines)
|
||||
return pandoc.Inlines({ open }) .. inlines .. pandoc.Inlines({ close })
|
||||
end
|
||||
cell.contents = cell.contents:walk({
|
||||
Para = function(block) return pandoc.Para(marked(block.content)) end,
|
||||
Plain = function(block) return pandoc.Plain(marked(block.content)) end,
|
||||
})
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
function Table(tbl)
|
||||
if not FORMAT:match('latex') then return nil end
|
||||
local columns = #tbl.colspecs
|
||||
if columns == 0 then return nil end
|
||||
mark_head(tbl)
|
||||
for _, spec in ipairs(tbl.colspecs) do
|
||||
if spec[2] then return nil end
|
||||
end
|
||||
local longest, word = measure(tbl)
|
||||
local content, needed = 0, 0
|
||||
for index = 1, columns do
|
||||
content = content + longest[index]
|
||||
needed = needed + word[index] + SEPARATOR
|
||||
end
|
||||
-- What this table has room for, in characters of the body size: its own
|
||||
-- width, which is the type area minus the space between its columns.
|
||||
local room = (TYPE_AREA - GAP * (columns - 1)) / CHARACTER
|
||||
if content == 0 then
|
||||
return nil
|
||||
end
|
||||
|
||||
-- The face is chosen first: the smallest step at which the columns can hold
|
||||
-- their longest words side by side, and the body size wherever that already
|
||||
-- works.
|
||||
local size = SIZES[#SIZES]
|
||||
for _, candidate in ipairs(SIZES) do
|
||||
if needed <= room * candidate.factor then
|
||||
size = candidate
|
||||
break
|
||||
end
|
||||
end
|
||||
local line = room * size.factor
|
||||
|
||||
-- Then the width: every column keeps its longest word, and what is left over
|
||||
-- goes to the columns in proportion to how much text they carry, so the
|
||||
-- column with the sentences grows and the one with the numbers stays as
|
||||
-- narrow as its heading. Where not even the smallest face gives room for
|
||||
-- every word, the least widths are scaled down together and LaTeX breaks
|
||||
-- inside the words, in the language of the document.
|
||||
local least = needed / line
|
||||
local share = {}
|
||||
for index = 1, columns do
|
||||
local minimum = (word[index] + SEPARATOR) / line
|
||||
share[index] = least >= 1 and minimum / least
|
||||
or minimum + (1 - least) * longest[index] / content
|
||||
end
|
||||
local specs = {}
|
||||
for index, spec in ipairs(tbl.colspecs) do
|
||||
specs[index] = { spec[1], share[index] }
|
||||
end
|
||||
tbl.colspecs = specs
|
||||
if size.name == '' then return tbl end
|
||||
return { pandoc.RawBlock('latex', '\\begingroup' .. size.name),
|
||||
tbl,
|
||||
pandoc.RawBlock('latex', '\\endgroup') }
|
||||
end
|
||||
Reference in new issue
Block a user