Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions docs/src/content/docs/format-guides/html.md
Original file line number Diff line number Diff line change
Expand Up @@ -86,6 +86,7 @@ result.markdown

- **Uses Nokogiri's HTML fragment parser** — handles malformed input without raising.
- **Stateless handlers** — simpler than BBCode's open/close callback API. A handler is an object responding to `#process(element:, parent:)`.
- **Collapses whitespace like a browser** — runs of whitespace become one space, and a space at the start or end of a line (a block boundary) or right after another space is dropped, wherever the inline boundaries fall: `<b>a </b> b` keeps one space. Text inside `<pre>`, `<code>`, `<textarea>` and `<tt>` is kept verbatim. Both tag sets are configurable on the `HandlerRegistry` (`block_level_tags`, `whitespace_preserving_tags`).

```ruby
class AsideHandler < Markbridge::Parsers::HTML::Handlers::BaseHandler
Expand Down
76 changes: 57 additions & 19 deletions lib/markbridge/parsers/html/parser.rb
Original file line number Diff line number Diff line change
Expand Up @@ -70,9 +70,21 @@ def parse(input)
# a <pre> keep their semantics.
@preserve_depth = initial_preserve_depth(doc)

# Whitespace collapsing follows the browser rule (CSS Text
# §4.1.1): a collapsible space is dropped at the start of a
# line, at the end of a line, and right after another
# collapsible space — wherever the inline boundaries fall, so
# `<b>a </b> b` keeps one space and `<p><b>x </b></p>` none.
# A line starts at the parse root and on every block-level tag
# boundary. @space_tail is the element whose last child is a
# collapsible Text ending in a space (the candidate for the
# end-of-line trim), or nil.
@line_start = true
@space_tail = nil

# Process all nodes
children.each { |node| process_node(node, document) }
trim_trailing_whitespace(document)
end_line

document
end
Expand Down Expand Up @@ -117,17 +129,27 @@ def process_text_node(node, parent)

if @preserve_depth.positive?
parent << AST::Text.new(text)
# Preserved whitespace is content: it neither starts a line
# nor collapses against a space that follows it.
@line_start = false
@space_tail = nil
return
end

text = text.gsub(WHITESPACE_RUN, " ") if text.match?(COLLAPSIBLE_WHITESPACE)
# Drop leading whitespace at the start of an element's content,
# matching the browser rule that whitespace at the beginning of a
# block (or before any inline content) is collapsed away.
text = text.lstrip if parent.children.empty?
text = collapse_whitespace(text)
text = text.lstrip if @line_start || @space_tail
return if text.empty?

parent << AST::Text.new(text)
@line_start = false
@space_tail = (parent if text.end_with?(" "))
end

# Collapse runs of whitespace to a single space.
# @param text [String]
# @return [String]
def collapse_whitespace(text)
text.match?(COLLAPSIBLE_WHITESPACE) ? text.gsub(WHITESPACE_RUN, " ") : text
end

# Process an element node
Expand All @@ -137,38 +159,44 @@ def process_element_node(node, parent)
tag_name = node.name
return if IGNORED_TAGS.include?(tag_name)

# Drop whitespace that sits between content and the start of a
# block-level tag, matching browser behavior where such whitespace
# collapses against the block boundary. Applies whether or not a
# handler is registered, so unknown tags like <div> or <section>
# still collapse the whitespace before them.
trim_trailing_whitespace(parent) if @handlers.block_level_tags.include?(tag_name)
# A block-level tag ends the line before it and the line inside
# it, matching browser behavior where whitespace collapses
# against the block boundary. Applies whether or not a handler
# is registered, so unknown tags like <div> or <section> still
# end lines.
block = @handlers.block_level_tags.include?(tag_name)
end_line if block

preserving = @handlers.whitespace_preserving_tags.include?(tag_name)
@preserve_depth += 1 if preserving

dispatch_element(node, tag_name, parent, preserving)
dispatch_element(node, tag_name, parent)

@preserve_depth -= 1 if preserving
end_line if block
end

# Dispatch an element to its handler (or the unknown-tag path) and
# process its children.
# @param node [Nokogiri::XML::Element]
# @param tag_name [String]
# @param parent [AST::Element]
# @param preserving [Boolean] whether this tag preserves whitespace
def dispatch_element(node, tag_name, parent, preserving)
def dispatch_element(node, tag_name, parent)
handler = @handlers[tag_name]
return handle_unknown_tag(node, parent) unless handler

# Handler returns element if children should be processed, nil otherwise
size = parent.children.size
ast_element = handler.process(element: node, parent:)

return unless ast_element

process_children(node, ast_element)
trim_trailing_whitespace(ast_element) unless preserving
if ast_element
process_children(node, ast_element)
elsif parent.children.size > size
# The handler appended a leaf (image, line break, ...), which
# is content on the current line like a word.
@line_start = false
@space_tail = nil
end
end

# Handle unknown tag by tracking it and ignoring the wrapper
Expand All @@ -180,6 +208,16 @@ def handle_unknown_tag(node, parent)
process_children(node, parent)
end

# End the current line: the collapsible space that closed it goes,
# and the next collapsible space starts a line. @space_tail is
# not cleared here: the trim is idempotent, so a stale tail is
# harmless until the next text or leaf on the new line resets it,
# and @line_start already covers the lstrip.
def end_line
trim_trailing_whitespace(@space_tail) if @space_tail
@line_start = true
end

# Number of whitespace-preserving elements enclosing the parse
# root, the root itself included. Computed once per parse; from
# there the counter is maintained during descent. Documents and
Expand Down
13 changes: 10 additions & 3 deletions lib/markbridge/renderers/discourse/tags/url_tag.rb
Original file line number Diff line number Diff line change
Expand Up @@ -28,18 +28,25 @@ def render(element, interface)

if interface.html_mode?
%(<a href="#{HtmlEscaper.escape(href)}">#{text}</a>)
elsif element.bare? || text.empty?
elsif element.bare? || blank?(text)
# Url#bare? judges the AST (so label escaping can't confuse
# it); the rendered-text check additionally catches labels
# that render to nothing (e.g. an empty formatting child).
# that render to nothing (e.g. an empty formatting child)
# or to whitespace only, which wrap_inline would leave
# unlinked.
href
else
"[#{text}](#{markdown_destination(href)})"
interface.wrap_inline(text, "[", "](#{markdown_destination(href)})")
end
end

private

# Unicode-aware, matching the guard in RenderingInterface#wrap_inline.
def blank?(text)
!text.match?(/[^[:space:]]/)
end

# CommonMark link destinations cannot contain whitespace unless
# wrapped in <> — relevant for relative targets like MediaWiki
# page names ("Main Page").
Expand Down
7 changes: 4 additions & 3 deletions mutant.yml
Original file line number Diff line number Diff line change
Expand Up @@ -79,14 +79,15 @@ matcher:
- Markbridge::Parsers::MediaWiki::Parser#normalize_line_endings

# HTML::Parser fast-path guards with output-equivalent slow paths:
# process_text_node's COLLAPSIBLE_WHITESPACE gate (forcing the gsub
# collapse_whitespace's COLLAPSIBLE_WHITESPACE gate (forcing the gsub
# on single-space prose rebuilds an equal string) and
# trim_trailing_whitespace's TRAILING_STRIPPABLE gate (forcing the
# rstrip+pop+re-add on an untrimmed text produces a value-equal
# AST). Both guards exist purely to avoid per-node copies; the
# behavioral branches around them are pinned by the whitespace-
# handling specs. Bucket A.
- Markbridge::Parsers::HTML::Parser#process_text_node
# handling specs. Bucket A. The gate lives in its own method so the
# line-state logic in process_text_node stays under mutation.
- Markbridge::Parsers::HTML::Parser#collapse_whitespace
- Markbridge::Parsers::HTML::Parser#trim_trailing_whitespace

# RenderingInterface#apply_markers' EDGE_WHITESPACE fast path.
Expand Down
86 changes: 86 additions & 0 deletions spec/integration/markbridge/html_inline_whitespace_spec.rb
Original file line number Diff line number Diff line change
@@ -0,0 +1,86 @@
# frozen_string_literal: true

require "spec_helper"

RSpec.describe "HTML inline whitespace", skip: CookedOutput::SKIP_REASON do
def converted_dom(html)
markdown = Markbridge.html_to_markdown(html).markdown
Nokogiri::HTML.fragment(Commonmarker.to_html(markdown, options: CookedOutput::OPTIONS))
end

it "renders a punctuation-ending label identically with whitespace inside or outside strong" do
inside = converted_dom("<p><strong>Tyler: </strong>text</p>")
outside = converted_dom("<p><strong>Tyler:</strong> text</p>")

expect(inside.to_html).to eq(outside.to_html)
expect(inside.at_css("strong").text).to eq("Tyler:")
expect(inside.at_css("p").text).to eq("Tyler: text")
end

it "collapses a space inside and a space outside an inline element into one" do
dom = converted_dom("<p><strong>Tyler: </strong> text</p>")

expect(dom.at_css("p").text).to eq("Tyler: text")
expect(dom.at_css("strong").text).to eq("Tyler:")
end

it "drops the space at the end of a block even when an inline element holds it" do
dom = converted_dom("<p>first</p><p><strong>label </strong></p><p>next</p>")

expect(dom.css("p").map(&:text)).to eq(%w[first label next])
end

it "keeps the separator before an inline label" do
dom = converted_dom("<p>before<strong> label</strong></p>")

expect(dom.at_css("p").text).to eq("before label")
expect(dom.at_css("strong").text).to eq("label")
end

it "preserves spaces across nested and transparent inline containers" do
[
"<strong><span>Tyler: </span></strong>",
"<span style='font-weight:bold'>Tyler: </span>",
"<b>Tyler: </b>",
].each do |label|
dom = converted_dom("<p>#{label}text</p>")
expect(dom.at_css("p").text).to eq("Tyler: text")
expect(dom.at_css("strong").text).to eq("Tyler:")
end
end

it "keeps separation between adjacent formatting elements" do
dom = converted_dom("<p><strong>A </strong><em>B</em></p>")

expect(dom.at_css("p").text).to eq("A B")
expect(dom.at_css("strong").text).to eq("A")
expect(dom.at_css("em").text).to eq("B")
end

it "keeps collapsed spaces inside links and nonbreaking spaces intact" do
dom =
converted_dom("<p>before<a href='/x'> link </a>after<strong> label:&nbsp;</strong>text</p>")

expect(dom.at_css("p").text).to eq("before link after label:\u00a0text")
expect(dom.at_css("a")["href"]).to eq("/x")
end

it "does not invent a separator for intentionally adjacent words" do
dom = converted_dom("<p>pre<strong>fix</strong>suffix</p>")

expect(dom.at_css("p").text).to eq("prefixsuffix")
end

it "does not turn block-edge indentation into code" do
dom = converted_dom("<p> <strong> label </strong> </p><p> next </p>")

expect(dom.css("pre,code")).to be_empty
expect(dom.css("p").map(&:text)).to eq(%w[label next])
end

it "keeps literal code whitespace" do
dom = converted_dom("<pre> literal \nnext </pre>")

expect(dom.at_css("code").text).to eq(" literal \nnext \n")
end
end
Loading
Loading