From 880915e06128db0103944c5cb7a3c00c224026ce Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Fri, 9 Oct 2026 03:03:31 -0400 Subject: [PATCH] Make the manual's EPUB from the same JPEG pictures as its PDF pandoc embedded every recording whole, so the EPUB came out at 63 MB for readers that mostly show a GIF's first frame anyway. tools/manual/pdf.py now makes both downloads from its JPEG copies of the pictures: 7.5 MB. --- .gitea/workflows/manual-pages.yml | 7 ++--- tools/manual/pdf.py | 47 ++++++++++++++++++++----------- 2 files changed, 33 insertions(+), 21 deletions(-) diff --git a/.gitea/workflows/manual-pages.yml b/.gitea/workflows/manual-pages.yml index ecb237c..ea077bc 100644 --- a/.gitea/workflows/manual-pages.yml +++ b/.gitea/workflows/manual-pages.yml @@ -62,7 +62,6 @@ jobs: # A4, a cover and a contents page with page numbers, a chapter a page — # so it says what the page says; tools/manual/pdf.py says why it goes # through JPEG copies of the pictures first. - # The EPUB comes from the Markdown, which pandoc reads better than a page. - name: Make the PDF and the EPUB run: | set -e @@ -73,10 +72,8 @@ jobs: python3 -m venv /tmp/wp /tmp/wp/bin/pip install -q weasyprint mkdir -p site - /tmp/wp/bin/python tools/manual/pdf.py docs/manual site/darkroom-manual.pdf - title=$(sed -n 's/^# //p' docs/manual/README.md | head -1) - (cd docs/manual && pandoc README.md -o ../../site/darkroom-manual.epub \ - --metadata title="$title" --toc --toc-depth=2 --epub-chapter-level=2) + /tmp/wp/bin/python tools/manual/pdf.py docs/manual \ + site/darkroom-manual.pdf site/darkroom-manual.epub ls -l site # One commit, force-pushed: the branch is a build output, and its history diff --git a/tools/manual/pdf.py b/tools/manual/pdf.py index 9a27b62..cc9dae2 100644 --- a/tools/manual/pdf.py +++ b/tools/manual/pdf.py @@ -1,14 +1,17 @@ #!/usr/bin/env python3 -"""The manual as a PDF: docs/manual/index.html through its print stylesheet. +"""The manual as a PDF, and as an EPUB: the downloads the page links to. - python3 tools/manual/pdf.py docs/manual out.pdf + python3 tools/manual/pdf.py docs/manual out.pdf [out.epub] -Needs WeasyPrint (and so Pillow): `pip install weasyprint`. The Manual pages -workflow runs this and publishes the result beside the page. +The PDF is docs/manual/index.html through its print stylesheet; the EPUB is +README.md through pandoc, which reads Markdown better than a page. Needs +WeasyPrint (and so Pillow), and pandoc for the EPUB. The Manual pages +workflow runs this and publishes both beside the page. The page's pictures are PNG screenshots and GIF recordings, 16 MB and 46 MB -of them, and WeasyPrint embeds a picture as it finds it: losslessly, a GIF -whole. So the PDF is made from a copy of the page whose pictures are JPEGs +of them, and WeasyPrint and pandoc embed a picture as they find it: +losslessly, a GIF whole (63 MB of EPUB, for e-readers that mostly do not +animate). So both are made from copies whose pictures are JPEGs (a recording's first frame, which is all a page can show) at the size they were recorded. Full chroma, because a 4:2:0 JPEG smears the coloured noise the AI denoise close-ups exist to show. @@ -16,6 +19,7 @@ the AI denoise close-ups exist to show. import os import re import shutil +import subprocess import sys import tempfile @@ -31,7 +35,7 @@ def as_jpeg(src, dst): im.convert('RGB').save(dst, 'JPEG', quality=QUALITY, subsampling=0, optimize=True) -def main(manual_dir, out): +def main(manual_dir, pdf, epub=None): with tempfile.TemporaryDirectory() as tmp: os.mkdir(os.path.join(tmp, 'media')) renamed = {} @@ -42,16 +46,27 @@ def main(manual_dir, out): jpg = f'{stem}.{ext[1:].lower()}.jpg' # two pictures may share a stem as_jpeg(os.path.join(manual_dir, 'media', name), os.path.join(tmp, 'media', jpg)) renamed[name] = jpg - page = open(os.path.join(manual_dir, 'index.html'), encoding='utf-8').read() - page = re.sub(r'src="media/([^"]+)"', - lambda m: f'src="media/{renamed.get(m.group(1), m.group(1))}"', page) - with open(os.path.join(tmp, 'index.html'), 'w', encoding='utf-8') as f: - f.write(page) - HTML(os.path.join(tmp, 'index.html')).write_pdf(os.path.join(tmp, 'manual.pdf')) - shutil.move(os.path.join(tmp, 'manual.pdf'), out) + + def copy(source, pattern, out): + text = open(os.path.join(manual_dir, source), encoding='utf-8').read() + text = re.sub(pattern, lambda m: m.group(1) + renamed.get(m.group(2), m.group(2)), text) + with open(os.path.join(tmp, out), 'w', encoding='utf-8') as f: + f.write(text) + return os.path.join(tmp, out) + + page = copy('index.html', r'(src="media/)([^"]+)', 'index.html') + HTML(page).write_pdf(os.path.join(tmp, 'manual.pdf')) + shutil.move(os.path.join(tmp, 'manual.pdf'), pdf) + + if epub: + md = copy('README.md', r'(\]\(media/)([^)\s]+)', 'README.md') + title = next(l[2:].strip() for l in open(md, encoding='utf-8') if l.startswith('# ')) + subprocess.run(['pandoc', 'README.md', '-o', 'manual.epub', '--metadata', f'title={title}', + '--toc', '--toc-depth=2', '--epub-chapter-level=2'], cwd=tmp, check=True) + shutil.move(os.path.join(tmp, 'manual.epub'), epub) if __name__ == '__main__': - if len(sys.argv) != 3: + if len(sys.argv) not in (3, 4): sys.exit(__doc__) - main(sys.argv[1], sys.argv[2]) + main(*sys.argv[1:])