Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 4 additions & 2 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -608,12 +608,14 @@ The converter module parses Markdown using [`mistune`](https://github.com/leptur
| `- [x] task` | rendered checkbox |
| `text[^1]` | `#footnote[...]` |
| `![alt](local.png)` | `#image("local.png")` |
| `<img src="local.png" alt="alt">` | Embedded image, using the same path as Markdown images; sizing attributes are ignored |
| `<img src="local.png" width="300">` | Embedded image, using the same path as Markdown images |
| `> blockquote` | `#block(...)` |
| `---` | `#line(...)` |

Special characters (`#`, `$`, `@`, `*`, `_`, etc.) are automatically escaped to prevent Typst interpretation.

A `width` or `height` on an HTML image is read as CSS pixels, a 96th of an inch each, and no picture is drawn wider than the text block however large the number is. Percentages are ignored, since fitting the text block is already the default.

### PDF Templates

Each template lives in `doc_engine/templates/` and exposes the same `setup_doc` entry point, so the compiler can swap between them with `--template`. The default `academic` template provides:
Expand Down Expand Up @@ -744,7 +746,7 @@ docker run --rm -v "$PWD:/workspace" ghcr.io/leonardosalasd/doc-engine-cli build
- [x] Task lists (`- [x]` / `- [ ]`)
- [x] Footnotes (`[^1]`)
- [x] Local images, and remote ones with `--fetch-images`
- [x] Raw HTML `<img>` tags (inline or block); `width`, `height`, and other sizing attributes are not honored
- [x] Raw HTML `<img>` tags (inline or block), sized by `width` and `height` in pixels
- [x] Math blocks (LaTeX `$…$` and `$$…$$`)
- [x] Mermaid and SVG diagram blocks
- [x] GitHub alerts (`> [!NOTE]`, `[!TIP]`, `[!IMPORTANT]`, `[!WARNING]`, `[!CAUTION]`)
Expand Down
21 changes: 15 additions & 6 deletions doc_engine/compiler.py
Original file line number Diff line number Diff line change
Expand Up @@ -139,12 +139,21 @@ def compile_pdf(
typst.compile(str(main_file), output=resolved_output)


# Pictures keep their natural size unless they do not fit the text block, in
# which case they shrink until they do. Forcing every image to full width blows
# up small diagrams, and constraining only the width lets a tall one run past
# the bottom of the page, where Typst clips whatever does not fit.
_FIT_IMAGE = """#let fit-image(path) = context layout(area => {
let img = image(path)
# Pictures keep their natural size, or the size the document asked for, unless
# that does not fit the text block, in which case they shrink until it does.
# Forcing every image to full width blows up small diagrams, and constraining
# only the width lets a tall one run past the bottom of the page, where Typst
# clips whatever does not fit.
_FIT_IMAGE = """#let fit-image(path, width: none, height: none) = context layout(area => {
let img = if width != none and height != none {
image(path, width: width, height: height, fit: "contain")
} else if width != none {
image(path, width: width)
} else if height != none {
image(path, height: height)
} else {
image(path)
}
let natural = measure(img)
let scale = calc.min(1.0, area.width / natural.width, area.height / natural.height)
if scale >= 1.0 { img } else { image(path, width: natural.width * scale) }
Expand Down
47 changes: 39 additions & 8 deletions doc_engine/converter.py
Original file line number Diff line number Diff line change
Expand Up @@ -95,6 +95,33 @@ def parse_inline(inline: Any, match: Any, state: Any) -> int:
md.inline.register("inline_math", _INLINE_MATH, parse_inline, before="link")


def _css_length(value: str | None) -> str | None:
"""A Typst length for an HTML width or height written in pixels.

A CSS pixel is a 96th of an inch, so a stated size converts to a real length
rather than to some guess about how wide a page is. Percentages are left
alone: they measure against a viewport a PDF does not have, and fitting the
picture to the text width is already what the default does.
"""
if not value:
return None
try:
pixels = float(value.strip().removesuffix("px").strip())
except ValueError:
return None
if pixels <= 0:
return None
return f"{pixels * 0.75:g}pt"


def _size_arguments(attrs: dict) -> str:
return "".join(
f", {name}: {length}"
for name in ("width", "height")
if (length := _css_length(attrs.get(name)))
)


def _render_children(renderer: mistune.BaseRenderer, token: dict, state: Any) -> str:
children = token.get("children")
if not children:
Expand Down Expand Up @@ -196,21 +223,21 @@ def _anchor_for(self, url: str) -> str | None:
return self._anchors.get(resolved)

def image(self, token: dict, state: Any) -> str:
url = token.get("attrs", {}).get("url", "")
attrs = token.get("attrs", {})
alt = _render_children(self, token, state)
asset = self._register_image(url)
asset = self._register_image(attrs.get("url", ""))
if asset is None:
return f"[{alt}]" if alt else ""
return self._place(asset)
return self._place(asset, _size_arguments(attrs))

def _place(self, asset: str) -> str:
def _place(self, asset: str, size: str = "") -> str:
"""Emit a picture, cut across pages when it is too tall for one."""
if self._split_tall is None:
return f'#fit-image("{asset}")'
return f'#fit-image("{asset}"{size})'
pieces = self._cut(asset)
if len(pieces) == 1:
return f'#fit-image("{pieces[0]}")'
return "\n#pagebreak(weak: true)\n".join(f'#fit-image("{p}")' for p in pieces)
return f'#fit-image("{pieces[0]}"{size})'
return "\n#pagebreak(weak: true)\n".join(f'#fit-image("{p}"{size})' for p in pieces)

def _cut(self, asset: str) -> list[str]:
# The same picture can appear more than once, and it is registered under
Expand Down Expand Up @@ -261,7 +288,11 @@ def _html_images(self, token: dict, state: Any) -> str:
return "\n".join(
self.image(
{
"attrs": {"url": attrs.get("src") or ""},
"attrs": {
"url": attrs.get("src") or "",
"width": attrs.get("width"),
"height": attrs.get("height"),
},
"children": [{"type": "text", "raw": attrs.get("alt") or ""}],
},
state,
Expand Down
19 changes: 19 additions & 0 deletions tests/test_compiler.py
Original file line number Diff line number Diff line change
@@ -1,3 +1,5 @@
from PIL import Image

from doc_engine.compiler import compile_pdf


Expand All @@ -19,3 +21,20 @@ def test_formatted_title_keeps_custom_template_contract(tmp_path):
title_markup="*Bold* Title",
)
assert output.read_bytes().startswith(b"%PDF-")


def test_a_requested_image_size_compiles(tmp_path):
assets = tmp_path / "assets"
assets.mkdir()
Image.new("RGB", (600, 300), "blue").save(assets / "box.png")
output = tmp_path / "sized.pdf"
compile_pdf(
'#fit-image("assets/box.png", width: 150pt)\n\n'
'#fit-image("assets/box.png", height: 25.5pt)\n\n'
'#fit-image("assets/box.png", width: 1500pt)\n',
"Sizes",
"Author",
str(output),
assets={"assets/box.png": str(assets / "box.png")},
)
assert output.read_bytes().startswith(b"%PDF-")
18 changes: 18 additions & 0 deletions tests/test_converter.py
Original file line number Diff line number Diff line change
Expand Up @@ -230,6 +230,24 @@ def test_comments_and_script_content_are_not_images(self, tmp_path) -> None:
assert not result.assets
assert "#fit-image" not in result.body

def test_pixel_sizes_become_lengths(self, tmp_path) -> None:
(tmp_path / "logo.png").write_bytes(b"image")
body = convert_document('<img src="logo.png" width="820">', base_dir=tmp_path).body
assert ", width: 615pt)" in body
body = convert_document('<img src="logo.png" height="34px">', base_dir=tmp_path).body
assert ", height: 25.5pt)" in body

def test_unusable_sizes_fall_back_to_fitting(self, tmp_path) -> None:
(tmp_path / "logo.png").write_bytes(b"image")
for attribute in ('width="50%"', 'width="auto"', 'width="0"', 'width="-4"', 'width=""'):
body = convert_document(f'<img src="logo.png" {attribute}>', base_dir=tmp_path).body
assert "width:" not in body

def test_markdown_images_ask_for_no_size(self, tmp_path) -> None:
(tmp_path / "logo.png").write_bytes(b"image")
body = convert_document("![alt](logo.png)", base_dir=tmp_path).body
assert body.strip() == '#fit-image("assets/0_logo.png")'

def test_missing_src_and_existing_linebreak_behavior(self) -> None:
assert "[missing]" in convert('<img alt="missing">')
assert convert("a<br>b") == "a\\\nb\n\n"
Expand Down
Loading