diff options
| author | Anders Betts <anders.betts@gmail.com> | 2026-09-20 14:48:36 +0200 |
|---|---|---|
| committer | Anders Betts <anders.betts@gmail.com> | 2026-09-20 14:50:21 +0200 |
| commit | 8607260080dd1eca2d724d6e5b07622fad2d4ac8 (patch) | |
| tree | 8c26f00f82aed77c6d89cfe93751dc0cab9ecf3f /scripts | |
| parent | ea3f9b67e5e4e4b3453df55af9758a94c023d6f7 (diff) | |
| download | bokf-8607260080dd1eca2d724d6e5b07622fad2d4ac8.tar.gz bokf-8607260080dd1eca2d724d6e5b07622fad2d4ac8.zip | |
invoice: extract the Comfortaa wordmark as vector paths
Diffstat (limited to 'scripts')
| -rwxr-xr-x | scripts/extract-wordmark.py | 307 |
1 files changed, 307 insertions, 0 deletions
diff --git a/scripts/extract-wordmark.py b/scripts/extract-wordmark.py new file mode 100755 index 0000000..ab14ad8 --- /dev/null +++ b/scripts/extract-wordmark.py @@ -0,0 +1,307 @@ +#!/usr/bin/env python3 +"""Generate src/wordmark.h from an outgoing invoice PDF. + +The invoice header draws "MAKANDRA AB" and "FAKTURA" in Comfortaa Bold +(SIL OFL 1.1), white on the #314c59 bar. This script extracts the two +words as filled vector paths so the PDF code can draw them without +embedding a font or depending on one at runtime. It is stdlib only and +shells out to poppler for the SVG conversion: + + pdftocairo -svg -f 1 -l 1 7.pdf /tmp/7.svg + scripts/extract-wordmark.py /path/to/7.pdf + +The source invoices are real customer documents and are not distributed +with the repository; only the two wordmark words are extracted. The +generated header stores each word as one path in PDF user space (y up), +normalized to the left baseline, plus its width and height in points. + +Usage: extract-wordmark.py [--check] [--out FILE] PDF +""" + +import argparse +import hashlib +import os +import re +import shutil +import subprocess +import sys +import tempfile + +BAR_COLOR = "rgb(19.215393%, 29.803467%, 34.901428%)" +WHITE_FILL = "rgb(100%, 100%, 100%)" +HEADER_BAND_Y = 90.0 +NUMBER = r"[-+]?(?:\d+\.?\d*|\.\d+)(?:[eE][-+]?\d+)?" +TOKEN = re.compile(r"([MLCZmlcz])|(" + NUMBER + r")") +VALUE = re.compile(NUMBER) +FILL_RE = re.compile(re.escape(WHITE_FILL) + r'"[^>]*>(.*?)</g>', re.S) +GLYPH_RE = re.compile(r'<g id="(glyph-[^"]+)">(.*?)</g>', re.S) +PATH_RE = re.compile(r'<path d="([^"]*)"') +USE_RE = re.compile( + r'<use xlink:href="#(glyph-[^"]+)" x="(' + NUMBER + r')" y="(' + NUMBER + r')"' +) +BAR_RE = re.compile(re.escape(BAR_COLOR) + r'"[^>]*\bd="([^"]*)"') +REPO_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +DEFAULT_OUT = os.path.join(REPO_ROOT, "src", "wordmark.h") + +WORDS = ("MAKANDRA", "FAKTURA") + + +def render_svg(pdf): + """Run pdftocairo on page 1 and return the SVG text.""" + if shutil.which("pdftocairo") is None: + sys.exit("extract-wordmark: pdftocairo not found in PATH") + with tempfile.TemporaryDirectory() as tmp: + out = os.path.join(tmp, "page.svg") + subprocess.run( + ["pdftocairo", "-svg", "-f", "1", "-l", "1", pdf, out], + check=True, + ) + with open(out, encoding="utf-8") as fh: + return fh.read() + + +def parse_glyphs(svg): + """Map glyph id -> path data (empty for blank glyphs such as space).""" + glyphs = {} + for match in GLYPH_RE.finditer(svg): + path = PATH_RE.search(match.group(2)) + glyphs[match.group(1)] = path.group(1) if path else "" + if not glyphs: + sys.exit("extract-wordmark: no glyph definitions in SVG") + return glyphs + + +def header_words(svg, page_height): + """Return the two white word groups inside the dark header bar. + + Each group is a list of (glyph id, x, y) placements in document order, + leftmost word first. + """ + groups = [] + for match in FILL_RE.finditer(svg): + uses = [ + (ref, float(x), float(y)) + for ref, x, y in USE_RE.findall(match.group(1)) + ] + if uses and max(y for _, _, y in uses) < HEADER_BAND_Y: + groups.append(uses) + groups.sort(key=lambda uses: uses[0][1]) + if len(groups) != 2: + sys.exit( + "extract-wordmark: expected 2 white header groups, found %d" % len(groups) + ) + return groups + + +def bar_bbox(svg): + """Bounding box of the #314c59 header bar path.""" + match = BAR_RE.search(svg) + if not match: + sys.exit("extract-wordmark: header bar colour not found in SVG") + nums = [float(v) for v in VALUE.findall(match.group(1))] + xs, ys = nums[0::2], nums[1::2] + return min(xs), min(ys), max(xs), max(ys) + + +def transform_path(data, dx, dy, page_height): + """Clone glyph path data, offset it, and flip y into PDF user space. + + Returns a list of (operator, [(x, y), ...]) segments with absolute + coordinates; only the absolute M/L/C/Z operators are accepted. + """ + tokens = [ + (op, None if op else float(num)) + for op, num in TOKEN.findall(data) + ] + segments = [] + i = 0 + while i < len(tokens): + op = tokens[i][0] + i += 1 + if op == "Z": + segments.append(("Z", [])) + continue + pairs = {"M": 1, "L": 1, "C": 3}.get(op) + if pairs is None: + sys.exit("extract-wordmark: unsupported path operator %r" % op) + points = [] + for _ in range(pairs): + if i + 1 >= len(tokens): + sys.exit("extract-wordmark: truncated path data") + x, y = tokens[i][1], tokens[i + 1][1] + i += 2 + points.append((dx + x, page_height - (dy + y))) + segments.append((op, points)) + return segments + + +def fmt(value): + """Format a coordinate with at most 3 decimals, without trailing zeros.""" + if abs(value) < 0.0005: + value = 0.0 + text = "%.3f" % value + text = text.rstrip("0").rstrip(".") + return text if text not in ("", "-0") else "0" + + +def compose_word(uses, glyphs, page_height): + """Compose one word into a single normalized path string. + + The word's origin becomes the left edge of its ink on the baseline. + Returns (path, width, height, svg bbox, placements, unique glyphs). + """ + baseline = page_height - uses[0][2] + segments = [] + for ref, x, y in uses: + if ref not in glyphs: + sys.exit("extract-wordmark: missing glyph %s" % ref) + segments.extend(transform_path(glyphs[ref], x, y, page_height)) + + points = [point for _, points in segments for point in points] + if not points: + sys.exit("extract-wordmark: word has no outline points") + min_x = min(point[0] for point in points) + max_x = max(point[0] for point in points) + min_y = min(point[1] for point in points) + max_y = max(point[1] for point in points) + svg_bbox = (min_x, max_x, page_height - max_y, page_height - min_y) + + parts = [] + out_min = [None, None] + out_max = [None, None] + for op, points in segments: + if op == "Z": + parts.append("Z") + continue + coords = [] + for x, y in points: + x, y = round(x - min_x, 3), round(y - baseline, 3) + coords.append(x) + coords.append(y) + for i, value in enumerate((x, y)): + if out_min[i] is None: + out_min[i] = out_max[i] = value + else: + out_min[i] = min(out_min[i], value) + out_max[i] = max(out_max[i], value) + parts.append(op + " " + " ".join(fmt(v) for v in coords)) + + return ( + " ".join(parts), + out_max[0] - out_min[0], + out_max[1] - out_min[1], + svg_bbox, + len(uses), + len({ref for ref, _, _ in uses}), + ) + + +def c_literal(name, path, indent=" ", width=76): + """Format a path as one or more adjacent C string literals.""" + lines, current = [], "" + for token in path.split(" "): + if current and len(current) + 1 + len(token) > width: + lines.append(current) + current = token + else: + current = token if not current else current + " " + token + if current: + lines.append(current) + out = ["static const char %s[] =" % name] + for i, line in enumerate(lines): + end = ";" if i == len(lines) - 1 else "" + tail = "" if i == len(lines) - 1 else " " + out.append('%s"%s%s"%s' % (indent, line, tail, end)) + return "\n".join(out) + + +def build_header(words, source_name): + chunks = [] + chunks.append( + "/*\n" + " * Comfortaa Bold wordmark outlines for the invoice header bar.\n" + " *\n" + " * Font: Comfortaa Bold by Johan Aakerlund, SIL Open Font License 1.1\n" + " * (https://scripts.sil.org/OFL). Only the outlines of the two words\n" + " * below are reproduced; no font file is embedded or needed at runtime.\n" + " *\n" + " * Extracted from the #314c59 header bar (page 1, y 53.3-75.7 pt) of\n" + " * the outgoing invoice PDF %s with scripts/extract-wordmark.py.\n" + " * Paths are in PDF user space (y up), origin at the left baseline;\n" + " * the dimensions are the ink width and height in points.\n" + " *\n" + " * Generated by scripts/extract-wordmark.py; do not edit by hand.\n" + " */\n" % source_name + ) + chunks.append("#ifndef BOKF_WORDMARK_H\n#define BOKF_WORDMARK_H\n") + for word, (path, w, h, _bbox, uses, unique) in zip(WORDS, words): + chunks.append( + "#define WORDMARK_%s_W %s\n#define WORDMARK_%s_H %s\n" + % (word, fmt(w), word, fmt(h)) + ) + chunks.append(c_literal("WORDMARK_" + word, path)) + chunks.append("") + chunks.append("#endif\n") + return "\n".join(chunks) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("pdf", help="invoice PDF to extract the wordmark from") + parser.add_argument( + "--check", + action="store_true", + help="print placements, bounding boxes and path digests, write nothing", + ) + parser.add_argument( + "--out", default=DEFAULT_OUT, help="output header (default: src/wordmark.h)" + ) + args = parser.parse_args() + + svg = render_svg(args.pdf) + page_height = float( + re.search(r'<svg[^>]*\bheight="(' + NUMBER + r')"', svg).group(1) + ) + glyphs = parse_glyphs(svg) + words = header_words(svg, page_height) + bar = bar_bbox(svg) + source_name = os.path.basename(args.pdf) + results = [compose_word(uses, glyphs, page_height) for uses in words] + + for word, uses, result in zip(WORDS, words, results): + path, w, h, bbox, count, unique = result + inside = ( + bar[0] <= bbox[0] and bbox[1] <= bar[2] + and bar[1] <= bbox[2] and bbox[3] <= bar[3] + ) + if not inside: + sys.exit("extract-wordmark: %s escapes the header bar" % word) + if args.check: + digest = hashlib.sha256(path.encode()).hexdigest()[:16] + print( + "%s %s: %d placements (%d outlines), bbox x %.3f..%.3f y %.3f..%.3f," + " %.3f x %.3f pt, sha256 %s" + % ( + source_name, + word, + count, + unique, + bbox[0], + bbox[1], + bbox[2], + bbox[3], + w, + h, + digest, + ) + ) + + if args.check: + return + with open(args.out, "w", encoding="utf-8") as fh: + fh.write(build_header(results, source_name)) + print("wrote %s" % os.path.relpath(args.out, REPO_ROOT)) + + +if __name__ == "__main__": + main() |
