Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
The table of contents is too big for display.
Diff view
Diff view
  •  
  •  
  •  
3 changes: 1 addition & 2 deletions .github/CODEOWNERS
Original file line number Diff line number Diff line change
@@ -1,2 +1 @@
* @MatMatt
* @mckeea
- @MatMatt @mckeea
5 changes: 5 additions & 0 deletions .github/non_browsable_doc_map.json
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,11 @@
"source": "CDSE_Migration/CLMS_CDSE_Migration_Dashboard.qmd",
"base": "a3e44009d6b7375f60a58b11278cbe079d93407249c0c5e1c97f00e12e98c09a",
"url": "/a3e44009d6b7375f60a58b11278cbe079d93407249c0c5e1c97f00e12e98c09a.html"
},
{
"source": "tools/parseo-dashboard/index.qmd",
"base": "a4b89445313aeaacde34c37ac1c45d91c7e5a7dbbc9d5ee9b31bbc578e6817bd",
"url": "/a4b89445313aeaacde34c37ac1c45d91c7e5a7dbbc9d5ee9b31bbc578e6817bd.html"
}
]
}
7 changes: 7 additions & 0 deletions .github/runners/Dockerfile.quarto-doc-builder
Original file line number Diff line number Diff line change
Expand Up @@ -6,20 +6,27 @@ ENV DEBIAN_FRONTEND=noninteractive
# Fonts: the OFL families the Typst template falls through to (Lato, JetBrains
# Mono, Carlito for Calibri, Liberation Sans for Arial, DejaVu mono).
# python-is-python3: the workflows invoke bare `python`.
# openssh-client: deploy-docs checks out with ssh-key so its commit-back pushes
# go as the deploy key, which the branch rulesets can bypass.
# unzip: required by `quarto install chrome-headless-shell` (below) to extract.
# rsync: github-pages-deploy-action shells out to it to sync the build.
# imagemagick: build-docs.sh re-encodes media as JPEG for the rendered outputs
# (compress_media.py). Without it a 500MB+ PDF gets rejected by the gh-pages
# push, which caps files at 100MB.
# lib* : runtime deps for chrome-headless-shell, which Quarto uses to
# rasterize mermaid/dot diagrams during the Typst PDF render.
RUN apt-get update && apt-get upgrade -y && apt-get install -y --no-install-recommends \
wget \
git \
openssh-client \
bash \
ca-certificates \
python3 \
python3-pip \
python-is-python3 \
unzip \
rsync \
imagemagick \
fonts-dejavu-core \
fonts-lato \
fonts-jetbrains-mono \
Expand Down
30 changes: 30 additions & 0 deletions .github/scripts/build/build-docs.sh
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,22 @@ step() {
# rm -rf DOCS origin_DOCS && mv source_DOCS DOCS
rm -rf source_DOCS && cp -rp DOCS source_DOCS

# BUILD_ONLY=<substring>: prune the build copy to the matching .qmd(s) so the
# full workflow runs end-to-end on one document. Source stays in source_DOCS.
# url_mapping.json is restored on exit - a one-doc run would prune every other
# doc's entry as "missing".
if [ -n "$BUILD_ONLY" ]; then
# Absolute paths: the script cd's into DOCS/ before the trap fires.
_UM="$PWD/url_mapping.json"
cp "$_UM" "$_UM.bkp" 2>/dev/null || true
trap 'mv -f "$_UM.bkp" "$_UM" 2>/dev/null || true' EXIT
find DOCS -name '*.qmd' ! -path "*$BUILD_ONLY*" -delete
n=$(find DOCS -name '*.qmd' | wc -l)
[ "$n" -gt 0 ] || { echo "ERROR: BUILD_ONLY=$BUILD_ONLY matched no .qmd" >&2; exit 1; }
find DOCS -mindepth 1 -type d -empty -delete
echo "BUILD_ONLY=$BUILD_ONLY -> building $n document(s)"
fi

# Apply cached intros/keywords before the rename - the cache is keyed by original path.
echo "Injecting cached intros & keywords (no API)..."
python3 .github/scripts/build/apply_cached_intros.py DOCS
Expand Down Expand Up @@ -90,6 +106,20 @@ python3 ../.github/scripts/qmd-tools/promote_bare_captions.py .
echo "Baking image descriptions into qmd source..."
python3 ../.github/scripts/build/inject_image_descriptions.py .

# Re-encode media as JPEG for the rendered outputs. Build copy only - DOCS/ in
# git and origin_DOCS/ here keep the lossless originals; only what ships to
# gh-pages is compressed. Must run AFTER the image descriptions above (they are
# keyed by image md5, so re-encoding first would miss every lookup) and before
# the realign below. ~0.03s an image, so there is nothing worth caching.
echo "Compressing media for rendered output..."
python3 ../.github/scripts/build/compress_media.py . -q 92

# Re-pad grid-table rows that the rewrites above (media-dir rename, fig-alt)
# pushed off their column borders - Pandoc mis-parses those and Typst fails
# with "unexpected comma". Must stay the last qmd rewrite before render.
echo "Realigning grid tables..."
python3 ../.github/scripts/qmd-tools/realign_grid_tables.py .

# Render with the no-headers config. The with-headers variant is still on
# disk but nothing activates it anymore.
cp _quarto-no-headers.yml _quarto.yml
Expand Down
278 changes: 278 additions & 0 deletions .github/scripts/build/compress_media.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,278 @@
#!/usr/bin/env python3
"""Re-encode PNG media as JPEG and repoint the .qmd references. Build copy only.

The PDF->qmd intake dumps document images as barely-compressed
``img-<hash>.png``: a 1500x1200 screenshot costs 5-7 MB against 5.4 MB of raw
pixels. Typst embeds them as-is, so Protected_Areas built a 550 MB PDF and
gh-pages rejected the push (GitHub caps files at 100 MB). Lossless
recompression buys ~2x, JPEG buys ~9-16x.

build-docs.sh points this at the render tree only; DOCS/ and origin_DOCS/ keep
the originals. ~0.03s an image, so nothing worth caching.

Both ends of its slot in build-docs.sh matter:
- after inject_image_descriptions.py, which keys alt text on image content
md5. Re-encode first and every lookup misses, silently.
- before realign_grid_tables.py, which wants to be the last qmd rewrite.
Only the extension changes (.png -> .jpg, same length), so grid-table
borders stay put.

It converts every PNG in a document media dir, SKIP_DIRS excluded, then fixes
up the references. Three rules are load-bearing, each of them learned from a
build that died:

- SKIP_DIRS keeps it out of _meta. Theme and template assets
(_meta/theme/typst/logos/*.png) are referenced from .typ and CSS, which
this does not rewrite. Convert one and Typst kills the render with
"file not found".
- It never writes over an existing file. 139 media files ship as both
foo.png and foo.jpg holding DIFFERENT images, with the document pointing
at the .jpg; converting foo.png would clobber it, and the "JPEG was
larger" branch would then delete it outright.
- Reference strings come from the filesystem, not from a regex over the
markdown. One media dir is literally named
"products_Mapping_Guide_Land _Cover_Land_Use_2006-media" - with a space -
which no sane URL pattern matches, so its references were left pointing at
.png files that had already been converted.

Second run is a no-op - what it converted is .jpg and no longer matches.
Images that actually use transparency are left alone.

python3 .github/scripts/build/compress_media.py . --dry-run
python3 .github/scripts/build/compress_media.py . -q 92
"""

import argparse
import subprocess
import sys
import tempfile
from collections import defaultdict
from pathlib import Path

DEFAULT_QUALITY = 92
# All of them. A 1 MB floor sounds sensible but leaves the job half done -
# Protected_Areas' sub-1MB images still add up to ~90 MB. Anything JPEG cannot
# beat is kept as PNG below, so there is no harm in offering the lot.
DEFAULT_MIN_BYTES = 0

# Same set as strip_llms_sidecars.py / strip_unknown_frontmatter.py, which walk
# the same render tree: _site is last render's output (quarto runs --no-clean,
# so it persists), .quarto is the render cache, and _meta and its subdirectories
# hold theme and template assets that .typ and CSS reference by hard-coded path.
SKIP_DIRS = {"_site", ".quarto", "_meta", "templates", "theme", "includes"}


def identify(path, fmt):
"""One ImageMagick -format field, or '' if the file can't be read."""
out = subprocess.run(
["identify", "-format", fmt, str(path)],
capture_output=True, text=True,
)
return out.stdout.strip() if out.returncode == 0 else ""


def is_transparent(path):
"""True only if the image actually uses transparency.

%[opaque] is the one field that reads the same on the ImageMagick 6 most
workstations have and the 7.1 in the build container (Debian trixie). Two
fields that do NOT, each of which shipped a bug:

field IM 6.9.11 IM 7.1.1
%A True Blend <- flattened 135 images to black
%[channels] srgba srgba 5.0 <- a trailing "a" test fails

An alpha channel that is fully opaque is not transparency: ~350 images
carry one and convert to JPEG with no visible change.

If identify fails outright this returns False and the image is converted -
the right way to fail, because convert_to_jpeg flattens onto white.
"""
return identify(path, "%[opaque]").lower() == "false"


def convert_to_jpeg(src, dest, quality):
# "jpg:" prefix, not just the extension: ImageMagick falls back to the INPUT
# format when it doesn't recognise the extension, which would quietly turn
# this into a PNG re-encode.
#
# -background white -alpha remove: JPEG has no alpha, and left to itself
# ImageMagick composites onto black. Anything that slips past
# is_transparent() then lands on a white page as a black box. Explicit here
# so a detection failure degrades to invisible rather than glaring.
subprocess.run(
["convert", str(src), "-background", "white", "-alpha", "remove",
"-alpha", "off", "-quality", str(quality), "-strip", f"jpg:{dest}"],
check=True, capture_output=True,
)


def is_skipped(path, root):
"""True if path sits in a skipped directory, or outside the tree entirely.

Compared relative to the render tree - an absolute path would drag the
checkout's own directory names into the match.
"""
try:
rel = path.resolve().relative_to(root)
except ValueError:
return True
return bool(SKIP_DIRS.intersection(rel.parts))


def find_qmd_files(root_path, root):
return sorted(p for p in root_path.rglob("*.qmd") if not is_skipped(p, root))


def find_media_pngs(root_path, root, min_bytes):
"""Every PNG in a document media dir, skipped dirs excluded.

Driven by the directory, not by what the .qmd files happen to reference:
~250 of these images are referenced by no document, but they are copied
into _site with the rest of their media dir and ship to gh-pages anyway,
so leaving them as PNG wastes ~55 MB for nothing.

Restricted to *-media/ as well as SKIP_DIRS. Document images only ever live
there, and it keeps the walk clear of the assets/ symlink that build-docs.sh
drops into the render tree.
"""
return {p.resolve() for p in root_path.rglob("*")
if p.suffix.lower() == ".png" and p.is_file()
and any(part.endswith("-media") for part in p.parts)
and not is_skipped(p, root)
and p.stat().st_size >= min_bytes}


def build_rename_map(converted):
"""{"<media dir>/<file>.png": "<media dir>/<file>.jpg"}.

The media dir is part of the key because basenames are not unique -
image4.png lives in 8 different media dirs, and keying on the filename
alone rewrote one document's reference when another document's image of
the same name converted.

A key that two different files would share is dropped rather than guessed
at; that needs two media dirs of the same name, which SKIP_DIRS currently
prevents by excluding _site.
"""
by_key = defaultdict(set)
for f in converted:
by_key[f"{f.parent.name}/{f.name}"].add(f)
return {key: f"{key[:key.rfind('.')]}.jpg"
for key, files in by_key.items() if len(files) == 1}


def rewrite_references(qmd_files, renames):
"""Repoint references at the converted files. Returns documents changed.

Plain substring replacement of "<media dir>/<file>", so it does not care
whether the document spells the reference as markdown, an <img src>, with
a ./ or ../ prefix, or with spaces in the path.
"""
changed = 0
for qmd in qmd_files:
text = original = qmd.read_text(encoding="utf-8")
for old, new in renames.items():
if old in text:
text = text.replace(old, new)
if text != original:
qmd.write_text(text, encoding="utf-8")
changed += 1
return changed


def main():
ap = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("target", type=Path,
help="render tree to process, e.g. . from inside DOCS/")
ap.add_argument("-q", "--quality", type=int, default=DEFAULT_QUALITY,
help=f"JPEG quality (default {DEFAULT_QUALITY})")
ap.add_argument("--min-size", type=int, default=DEFAULT_MIN_BYTES,
help="only touch PNGs at least this many bytes (default: all)")
ap.add_argument("--dry-run", action="store_true",
help="report what would change, write nothing")
args = ap.parse_args()

if not args.target.is_dir():
sys.exit(f"error: {args.target} is not a directory")

root = args.target.resolve()
wanted = find_media_pngs(args.target, root, args.min_size)
if not wanted:
print(f"Nothing to do: no PNG in a media dir under {args.target}.")
return 0

before = after = 0
converted = set()
skipped_alpha = skipped_bigger = skipped_collision = 0

for src in sorted(wanted, key=lambda p: p.stat().st_size, reverse=True):
size = src.stat().st_size
before += size

if is_transparent(src):
skipped_alpha += 1
after += size
continue

dest = src.with_suffix(".jpg")

# Never write over a file that already exists - see the module docstring.
if dest.exists():
skipped_collision += 1
after += size
continue

if args.dry_run:
# Encode to a throwaway path just to measure the real saving.
with tempfile.NamedTemporaryFile(suffix=".jpg") as tmp:
convert_to_jpeg(src, tmp.name, args.quality)
new_size = Path(tmp.name).stat().st_size
else:
convert_to_jpeg(src, dest, args.quality)
new_size = dest.stat().st_size

# Tiny palette images can come out bigger as JPEG. Keep the PNG.
if new_size >= size:
skipped_bigger += 1
after += size
if not args.dry_run:
dest.unlink()
continue

after += new_size
converted.add(src)
if not args.dry_run:
src.unlink()

renames = build_rename_map(converted)
qmd_files = find_qmd_files(args.target, root)
if args.dry_run:
touched = sum(1 for q in qmd_files
if any(old in q.read_text(encoding="utf-8") for old in renames))
else:
touched = rewrite_references(qmd_files, renames)

mb = 1024 * 1024
prefix = "[dry-run] would convert" if args.dry_run else "converted"
print(f"{prefix} {len(converted)} image(s) at JPEG q{args.quality}")
if skipped_alpha:
print(f" kept as PNG (uses transparency): {skipped_alpha}")
if skipped_bigger:
print(f" kept as PNG (JPEG was larger): {skipped_bigger}")
if skipped_collision:
print(f" kept as PNG (a .jpg of that name already exists): "
f"{skipped_collision}")
if len(renames) != len(converted):
print(f" warning: {len(converted) - len(renames)} converted image(s) have "
f"an ambiguous media-dir/name key and were NOT repointed")
if after:
print(f" media: {before / mb:.1f} MB -> {after / mb:.1f} MB "
f"({before / after:.1f}x smaller)")
print(f" .qmd files {'to update' if args.dry_run else 'updated'}: {touched}")
return 0


if __name__ == "__main__":
sys.exit(main())
5 changes: 3 additions & 2 deletions .github/scripts/build/inject_image_descriptions.py
Original file line number Diff line number Diff line change
Expand Up @@ -69,8 +69,9 @@ def resolve_image_path(src, qmd_path):


def alt_value(desc):
"""Collapse to one line and drop double quotes (they'd close the attr)."""
return " ".join(desc.split()).replace('"', "'")
"""Collapse to one line; drop double quotes (they'd close the attr) and
pipes (they'd split the cell when the image sits in a table row)."""
return " ".join(desc.split()).replace('"', "'").replace("|", "/")


def is_block_level(text, start, end):
Expand Down
Loading
Loading