mirror of
https://github.com/lexmount/moli.git
synced 2026-10-09 08:01:05 +00:00
fix(html2md): preserve code boundaries across empty inline events
This commit is contained in:
+16
-15
@@ -33,6 +33,9 @@ pub(crate) struct Writer<'a> {
|
||||
preserved_spaces: String,
|
||||
breaks: usize,
|
||||
code: Option<String>,
|
||||
// End of the last emitted Markdown code span in the local output. Only
|
||||
// actual output after this position separates it from another code span.
|
||||
markdown_code_end: Option<usize>,
|
||||
line_digits: Option<usize>,
|
||||
heading: bool,
|
||||
single_line_attributes: bool,
|
||||
@@ -176,22 +179,10 @@ impl<'a> Writer<'a> {
|
||||
if text.is_empty() {
|
||||
return;
|
||||
}
|
||||
// Adjacent code elements have separate HTML nodes but no text between
|
||||
// them. Markdown code spans would merge or need an invented space.
|
||||
if self.code.is_some() {
|
||||
self.flush_code_html();
|
||||
}
|
||||
self.flush_code();
|
||||
if preformatted {
|
||||
if !text.is_empty() {
|
||||
if self.code.is_none() {
|
||||
self.prepare_inline('`');
|
||||
self.code = Some(String::new());
|
||||
}
|
||||
self.code
|
||||
.as_mut()
|
||||
.expect("initialized code buffer")
|
||||
.push_str(&text.replace("\r\n", " ").replace(['\n', '\r'], " "));
|
||||
}
|
||||
self.prepare_inline('`');
|
||||
self.code = Some(text.replace("\r\n", " ").replace(['\n', '\r'], " "));
|
||||
return;
|
||||
}
|
||||
let content = text.trim_matches(char::is_whitespace);
|
||||
@@ -265,6 +256,7 @@ impl<'a> Writer<'a> {
|
||||
self.flush_breaks();
|
||||
self.prefix
|
||||
.push_text(std::mem::take(&mut self.output).into());
|
||||
self.markdown_code_end = None;
|
||||
self.prefix.append(output);
|
||||
self.line_digits = None;
|
||||
self.boundary(after);
|
||||
@@ -439,6 +431,13 @@ impl<'a> Writer<'a> {
|
||||
}
|
||||
|
||||
fn flush_code(&mut self) {
|
||||
// Empty text/style events can flush code without adding a separator.
|
||||
// Check the emitted tail after spaces and style markers have been
|
||||
// materialized, so adjacent backtick runs cannot merge code values.
|
||||
if self.markdown_code_end == Some(self.output.len()) {
|
||||
self.flush_code_html();
|
||||
return;
|
||||
}
|
||||
if let Some(code) = self.code.take() {
|
||||
let fence = "`".repeat(longest_run(&code, '`') + 1);
|
||||
let padding = code.starts_with('`')
|
||||
@@ -453,6 +452,7 @@ impl<'a> Writer<'a> {
|
||||
self.output.push(' ');
|
||||
}
|
||||
self.output.push_str(&fence);
|
||||
self.markdown_code_end = Some(self.output.len());
|
||||
self.line_digits = None;
|
||||
}
|
||||
}
|
||||
@@ -476,6 +476,7 @@ impl<'a> Writer<'a> {
|
||||
}
|
||||
}
|
||||
self.output.push_str("</code>");
|
||||
self.markdown_code_end = None;
|
||||
self.line_digits = None;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -160,6 +160,29 @@ fn keeps_adjacent_code_elements_distinct_and_chooses_safe_delimiters() {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_text_nodes_do_not_separate_code_values() {
|
||||
let mut dom = Tree::new();
|
||||
dom.leaf(0, "code", "name");
|
||||
dom.text(0, "");
|
||||
dom.leaf(0, "code", "string");
|
||||
dom.text(0, "");
|
||||
dom.text(0, "");
|
||||
dom.leaf(0, "code", "third");
|
||||
for preformatted_code in [false, true] {
|
||||
let result = Converter::new(Options {
|
||||
preformatted_code,
|
||||
..Options::default()
|
||||
})
|
||||
.convert(&dom, 0);
|
||||
assert_eq!(
|
||||
rendered_html(&result),
|
||||
"<p><code>name</code><code>string</code><code>third</code></p>\n",
|
||||
"preformatted={preformatted_code}: {result}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn preserves_preformatted_text_including_blank_lines_and_nested_elements() {
|
||||
let mut dom = Tree::new();
|
||||
|
||||
@@ -158,8 +158,16 @@ fn adjacent_code_elements_keep_distinct_values() {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_code_between_values_does_not_join_or_invent_code_text() {
|
||||
for empty in ["<code></code>", "<code><!-- source note --></code>"] {
|
||||
fn empty_inline_elements_between_code_do_not_join_or_invent_text() {
|
||||
for empty in [
|
||||
"<code></code>",
|
||||
"<code><!-- source note --></code>",
|
||||
"<em></em>",
|
||||
"<strong></strong>",
|
||||
"<del></del>",
|
||||
"<em><strong></strong></em>",
|
||||
"<span></span>",
|
||||
] {
|
||||
let source = format!("<code>name</code>{empty}<code>string</code>");
|
||||
for preformatted in [false, true] {
|
||||
let result = markdown(&source, preformatted);
|
||||
@@ -172,6 +180,82 @@ fn empty_code_between_values_does_not_join_or_invent_code_text() {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn code_remains_distinct_when_surrounding_styles_coalesce() {
|
||||
for tag in ["em", "strong", "del"] {
|
||||
let source = format!("<{tag}><code>name</code></{tag}><{tag}><code>string</code></{tag}>");
|
||||
for preformatted in [false, true] {
|
||||
let result = markdown(&source, preformatted);
|
||||
assert_eq!(
|
||||
rendered_html(&result),
|
||||
format!("<p><{tag}><code>name</code><code>string</code></{tag}></p>\n"),
|
||||
"{source}, preformatted={preformatted}: {result}"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn actual_output_separates_markdown_code_spans() {
|
||||
for (source, expected) in [
|
||||
(
|
||||
"<code>name</code><em> </em><code>string</code>",
|
||||
"`name` `string`",
|
||||
),
|
||||
(
|
||||
"<code>name</code><strong> </strong><code>string</code>",
|
||||
"`name`\u{a0}`string`",
|
||||
),
|
||||
(
|
||||
"<code>name</code><em>and</em><code>string</code>",
|
||||
"`name`*and*`string`",
|
||||
),
|
||||
(
|
||||
"<em><code>name</code></em><code>string</code>",
|
||||
"*`name`*`string`",
|
||||
),
|
||||
(
|
||||
"<code>name</code><em><code>string</code></em>",
|
||||
"`name`*`string`*",
|
||||
),
|
||||
(
|
||||
"<code>name</code><a href='/a'></a><code>string</code>",
|
||||
"`name`[](/a)`string`",
|
||||
),
|
||||
(
|
||||
"<code>name</code><img src='/i'><code>string</code>",
|
||||
"`name``string`",
|
||||
),
|
||||
(
|
||||
"<code>name</code><br><code>string</code>",
|
||||
"`name` \n`string`",
|
||||
),
|
||||
(
|
||||
"<code>name</code><p></p><code>string</code>",
|
||||
"`name`\n\n`string`",
|
||||
),
|
||||
(
|
||||
"<code>a</code><blockquote><code>b</code></blockquote>x<code>c</code>",
|
||||
"`a`\n\n> `b`\n\nx`c`",
|
||||
),
|
||||
] {
|
||||
for preformatted in [false, true] {
|
||||
assert_eq!(
|
||||
markdown(source, preformatted),
|
||||
expected,
|
||||
"{source}, preformatted={preformatted}"
|
||||
);
|
||||
}
|
||||
}
|
||||
for source in [
|
||||
"<code>name </code><code>string</code>",
|
||||
"<code>name</code><code> string</code>",
|
||||
"<code>name</code><code> </code><code>string</code>",
|
||||
] {
|
||||
assert_eq!(markdown(source, false), "`name` `string`", "{source}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn neighboring_code_values_keep_their_text_and_node_boundaries() {
|
||||
for (source, preformatted, expected) in [
|
||||
@@ -195,6 +279,16 @@ fn neighboring_code_values_keep_their_text_and_node_boundaries() {
|
||||
true,
|
||||
"<p><code>first</code><code>second</code></p>\n",
|
||||
),
|
||||
(
|
||||
"<code>first</code><em></em><code>*[x]|<&>`</code><strong></strong><code>third</code><del></del><code>fourth</code>",
|
||||
false,
|
||||
"<p><code>first</code><code>*[x]|<&>`</code><code>third</code><code>fourth</code></p>\n",
|
||||
),
|
||||
(
|
||||
"<code>word</code><em></em><code> a b </code>",
|
||||
true,
|
||||
"<p><code>word</code><code> a b </code></p>\n",
|
||||
),
|
||||
] {
|
||||
let result = markdown(source, preformatted);
|
||||
assert_eq!(rendered_html(&result), expected, "{source}: {result}");
|
||||
|
||||
@@ -4,8 +4,8 @@
|
||||
"reason": "Moli fences bare pre elements as well as pre/code, preserving preformatted text. Turndown only recognizes a pre/code pair."
|
||||
},
|
||||
"extra_adjacent_code": {
|
||||
"markdown": "<code>a</code>`b`",
|
||||
"reason": "Adjacent code elements retain their semantic boundary without an invented space. Inline HTML preserves the first boundary because adjacent Markdown code spans would merge. Turndown joins their delimiters into one code span whose text contains two extra backticks."
|
||||
"markdown": "`a`<code>b</code>",
|
||||
"reason": "Adjacent code elements retain their semantic boundary without an invented space. Inline HTML preserves the second code element when it immediately follows an emitted Markdown code span. Turndown joins their delimiters into one code span whose text contains two extra backticks."
|
||||
},
|
||||
"extra_code_nested_format": {
|
||||
"markdown": "`ab`",
|
||||
|
||||
Reference in New Issue
Block a user