#!/usr/bin/env python3 """The manual as a PDF, and as an EPUB: the downloads the page links to. python3 tools/manual/pdf.py docs/manual out.pdf [out.epub] The PDF is docs/manual/index.html through its print stylesheet; the EPUB is README.md through pandoc, which reads Markdown better than a page. Needs WeasyPrint (and so Pillow), and pandoc for the EPUB. The Manual pages workflow runs this and publishes both beside the page. The page's pictures are PNG screenshots and GIF recordings, 16 MB and 46 MB of them, and WeasyPrint and pandoc embed a picture as they find it: losslessly, a GIF whole (63 MB of EPUB, for e-readers that mostly do not animate). So both are made from copies whose pictures are JPEGs (a recording's first frame, which is all a page can show) at the size they were recorded. Full chroma, because a 4:2:0 JPEG smears the coloured noise the AI denoise close-ups exist to show. """ import os import re import shutil import subprocess import sys import tempfile from PIL import Image from weasyprint import HTML QUALITY = 88 def as_jpeg(src, dst): with Image.open(src) as im: im.seek(0) im.convert('RGB').save(dst, 'JPEG', quality=QUALITY, subsampling=0, optimize=True) def main(manual_dir, pdf, epub=None): with tempfile.TemporaryDirectory() as tmp: os.mkdir(os.path.join(tmp, 'media')) renamed = {} for name in sorted(os.listdir(os.path.join(manual_dir, 'media'))): stem, ext = os.path.splitext(name) if ext.lower() not in ('.png', '.gif'): continue jpg = f'{stem}.{ext[1:].lower()}.jpg' # two pictures may share a stem as_jpeg(os.path.join(manual_dir, 'media', name), os.path.join(tmp, 'media', jpg)) renamed[name] = jpg def copy(source, pattern, out): text = open(os.path.join(manual_dir, source), encoding='utf-8').read() text = re.sub(pattern, lambda m: m.group(1) + renamed.get(m.group(2), m.group(2)), text) with open(os.path.join(tmp, out), 'w', encoding='utf-8') as f: f.write(text) return os.path.join(tmp, out) page = copy('index.html', r'(src="media/)([^"]+)', 'index.html') HTML(page).write_pdf(os.path.join(tmp, 'manual.pdf')) shutil.move(os.path.join(tmp, 'manual.pdf'), pdf) if epub: md = copy('README.md', r'(\]\(media/)([^)\s]+)', 'README.md') title = next(l[2:].strip() for l in open(md, encoding='utf-8') if l.startswith('# ')) subprocess.run(['pandoc', 'README.md', '-o', 'manual.epub', '--metadata', f'title={title}', '--toc', '--toc-depth=2', '--epub-chapter-level=2'], cwd=tmp, check=True) shutil.move(os.path.join(tmp, 'manual.epub'), epub) if __name__ == '__main__': if len(sys.argv) not in (3, 4): sys.exit(__doc__) main(*sys.argv[1:])