1//! Structure similar to `*emphasis*` with configurable markers of fixed length.
2//!
3//! There are many structures in various markdown flavors that
4//! can be implemented with this, namely:
5//!
6//! - `*emphasis*` or `_emphasis_` -> `<em>emphasis</em>`
7//! - `**strong**` or `__strong__` -> `<strong>strong</strong>`
8//! - `~~strikethrough~~` -> `<s>strikethrough</s>`
9//! - `==marked==` -> `<mark>marked</mark>`
10//! - `++inserted++` -> `<ins>inserted</ins>`
11//! - `~subscript~` -> `<sub>subscript</sub>`
12//! - `^superscript^` -> `<sup>superscript</sup>`
13//!
14//! You add a custom structure by using [add_with] function, which takes following arguments:
15//! - `MARKER` - marker character
16//! - `LENGTH` - length of the opening/closing marker (can be 1, 2 or 3)
17//! - `CAN_SPLIT_WORD` - whether this structure can be found in the middle of the word
18//! (for example, note the difference between `foo*bar*baz` and `foo_bar_baz`
19//! in CommonMark - first one is an emphasis, second one isn't)
20//! - `md` - parser instance
21//! - `f` - function that should return your custom [Node]
22//!
23//! Here is an example of implementing superscript in your custom code:
24//!
25//! ```rust
26//! use markdown_it::generics::inline::emph_pair;
27//! use markdown_it::{MarkdownIt, Node, NodeValue, Renderer};
28//!
29//! #[derive(Debug)]
30//! struct Superscript;
31//! impl NodeValue for Superscript {
32//! fn render(&self, node: &Node, fmt: &mut dyn Renderer) {
33//! fmt.open("sup", &node.attrs);
34//! fmt.contents(&node.children);
35//! fmt.close("sup");
36//! }
37//! }
38//!
39//! let md = &mut MarkdownIt::new();
40//! emph_pair::add_with::<'^', 1, true>(md, || Node::new(Superscript));
41//!
42//! let html = md.parse("e^iπ^+1=0").render();
43//! assert_eq!(html.trim(), "e<sup>iπ</sup>+1=0");
44//! ```
45//!
46//! Note that these structures have lower priority than the rest of the rules,
47//! e.g. `` *foo`bar*baz` `` is parsed as `*foo<code>bar*baz</code>`.
48//!
49use std::cmp::min;
50
51use crate::common::sourcemap::SourcePos;
52use crate::parser::core::CoreRule;
53use crate::parser::extset::{MarkdownItExt, NodeExt};
54use crate::parser::inline::builtin::InlineParserRule;
55use crate::parser::inline::{InlineRule, InlineState, Text};
56use crate::{MarkdownIt, Node, NodeValue};
57
58#[derive(Debug, Default)]
59struct PairConfig<const MARKER: char> {
60 inserted: bool,
61 fns: [Option<fn() -> Node>; 3],
62}
63impl<const MARKER: char> MarkdownItExt for PairConfig<MARKER> {}
64
65#[derive(Debug, Default)]
66struct OpenersBottom<const MARKER: char>([usize; 6]);
67impl<const MARKER: char> NodeExt for OpenersBottom<MARKER> {}
68
69#[derive(Debug, Clone)]
70#[doc(hidden)]
71pub struct EmphMarker {
72 // Starting marker
73 pub marker: char,
74
75 // Total length of these series of delimiters.
76 pub length: usize,
77
78 // Remaining length that's not already matched to other delimiters.
79 pub remaining: usize,
80
81 // Boolean flags that determine if this delimiter could open or close
82 // an emphasis.
83 pub open: bool,
84 pub close: bool,
85
86 // Inline position (offset into the stripped inline text) where this
87 // delimiter starts. Stored so that `run` can compute the correct
88 // inline-coordinate token length when the delimiter spans multiple
89 // block-quote continuation lines (where srcmap offsets include the
90 // stripped `> ` bytes and therefore exceed the inline position).
91 pub inline_pos: usize,
92}
93
94// this node is supposed to be replaced by actual emph or text node
95impl NodeValue for EmphMarker {}
96
97pub fn add_with<const MARKER: char, const LENGTH: u8, const CAN_SPLIT_WORD: bool>(
98 md: &mut MarkdownIt,
99 f: fn() -> Node,
100) {
101 let pair_config = md.ext.get_or_insert_default::<PairConfig<MARKER>>();
102 pair_config.fns[LENGTH as usize - 1] = Some(f);
103
104 if !pair_config.inserted {
105 pair_config.inserted = true;
106 md.inline
107 .add_rule::<EmphPairScanner<MARKER, CAN_SPLIT_WORD>>();
108 }
109
110 if !md.has_rule::<FragmentsJoin>() {
111 md.add_rule::<FragmentsJoin>()
112 .before_all()
113 .after::<InlineParserRule>();
114 }
115}
116
117#[doc(hidden)]
118pub struct EmphPairScanner<const MARKER: char, const CAN_SPLIT_WORD: bool>;
119impl<const MARKER: char, const CAN_SPLIT_WORD: bool> InlineRule
120 for EmphPairScanner<MARKER, CAN_SPLIT_WORD>
121{
122 const MARKER: char = MARKER;
123
124 // this rule works on a closing marker, so for technical reasons any rules trying to skip it
125 // should see just plain text
126 fn check(_: &mut InlineState) -> Option<usize> {
127 None
128 }
129
130 fn run(state: &mut InlineState) -> Option<(Node, usize)> {
131 let mut chars = state.src[state.pos..state.pos_max].chars();
132 if chars.next().unwrap() != MARKER {
133 return None;
134 }
135
136 let scanned = state.scan_delims(state.pos, CAN_SPLIT_WORD);
137 let inline_closer_start = state.pos;
138 let mut node = Node::new(EmphMarker {
139 marker: MARKER,
140 length: scanned.length,
141 remaining: scanned.length,
142 open: scanned.can_open,
143 close: scanned.can_close,
144 inline_pos: state.pos,
145 });
146 node.srcmap = state.get_map(state.pos, state.pos + scanned.length);
147 let (node, opener_inline_pos) = scan_and_match_delimiters::<MARKER>(state, node);
148 // backtrack to keep correct source maps
149 state.pos += scanned.length;
150 // Compute token_len in inline coordinates. Using srcmap byte offsets
151 // (map.1 - map.0) is incorrect when the token spans multiple blockquote
152 // continuation lines because the srcmap reflects the original file bytes
153 // (which include the stripped `> ` prefixes) rather than the shorter
154 // inline text. When a match was found we use the opener's stored inline
155 // position; when no match was found the token is only the scanned
156 // delimiter itself.
157 let token_len = match opener_inline_pos {
158 Some(oip) => state.pos - oip,
159 None => scanned.length,
160 };
161 // Sanity: keep state.pos valid (should never underflow, but guard anyway)
162 debug_assert!(
163 state.pos >= token_len,
164 "emph backtrack underflow: pos={} token_len={} closer_start={}",
165 state.pos,
166 token_len,
167 inline_closer_start
168 );
169 state.pos -= token_len;
170 Some((node, token_len))
171 }
172}
173
174/// Assuming last token is a closing delimiter we just inserted,
175/// try to find opener(s). If any are found, move stuff to nested emph node.
176///
177/// Returns `(node, opener_inline_pos)`. `opener_inline_pos` is `Some` when a
178/// match was found; the value is the `inline_pos` field of the matched opener.
179fn scan_and_match_delimiters<const MARKER: char>(
180 state: &mut InlineState,
181 mut closer_token: Node,
182) -> (Node, Option<usize>) {
183 if state.node.children.is_empty() {
184 return (closer_token, None);
185 } // must have at least opener and closer
186
187 let mut closer = closer_token.cast_mut::<EmphMarker>().unwrap().clone();
188 if !closer.close {
189 return (closer_token, None);
190 }
191
192 // Previously calculated lower bounds (previous fails)
193 // for each marker, each delimiter length modulo 3,
194 // and for whether this closer can be an opener;
195 // https://github.com/commonmark/cmark/commit/34250e12ccebdc6372b8b49c44fab57c72443460
196 let openers_for_marker = state
197 .node
198 .ext
199 .get_or_insert_default::<OpenersBottom<MARKER>>();
200 let openers_parameter = (closer.open as usize) * 3 + closer.length % 3;
201
202 let min_opener_idx = openers_for_marker.0[openers_parameter];
203
204 let mut idx = state.node.children.len() - 1;
205 let mut new_min_opener_idx = idx;
206 let mut matched_opener_inline_pos: Option<usize> = None;
207 while idx > min_opener_idx {
208 idx -= 1;
209
210 let Some(opener) = state.node.children[idx].cast::<EmphMarker>() else {
211 continue;
212 };
213
214 let mut opener = opener.clone();
215 if opener.open && opener.marker == closer.marker && !is_odd_match(&opener, &closer) {
216 while closer.remaining > 0 && opener.remaining > 0 {
217 let max_marker_len = min(3, min(opener.remaining, closer.remaining));
218 let mut matched_rule = None;
219 let fns = &state.md.ext.get::<PairConfig<MARKER>>().unwrap().fns;
220 for marker_len in (1..=max_marker_len).rev() {
221 if let Some(f) = fns[marker_len - 1] {
222 matched_rule = Some((marker_len, f));
223 break;
224 }
225 }
226
227 // If matched_fn isn't found, it can only mean that function is defined for larger marker
228 // than we have (e.g. function defined for **, we have *).
229 // Treat this as "marker not found".
230 if matched_rule.is_none() {
231 break;
232 }
233
234 let (marker_len, marker_fn) = matched_rule.unwrap();
235
236 closer.remaining -= marker_len;
237 opener.remaining -= marker_len;
238
239 let mut new_token = marker_fn();
240 new_token.children = state.node.children.split_off(idx + 1);
241
242 // cut marker_len chars from start, i.e. "12345" -> "345"
243 let mut end_map_pos = 0;
244 if let Some(map) = closer_token.srcmap {
245 let (start, end) = map.get_byte_offsets();
246 closer_token.srcmap = Some(SourcePos::new(start + marker_len, end));
247 end_map_pos = start + marker_len;
248 }
249
250 // cut marker_len chars from end, i.e. "12345" -> "123"
251 let mut start_map_pos = 0;
252 let opener_token = state.node.children.last_mut().unwrap();
253 if let Some(map) = opener_token.srcmap {
254 let (start, end) = map.get_byte_offsets();
255 opener_token.srcmap = Some(SourcePos::new(start, end - marker_len));
256 start_map_pos = end - marker_len;
257 }
258
259 new_token.srcmap = state.get_map(start_map_pos, end_map_pos);
260
261 // remove empty node as a small optimization so we can do less work later
262 if opener.remaining == 0 {
263 state.node.children.pop();
264 }
265
266 new_min_opener_idx = 0;
267 if matched_opener_inline_pos.is_none() {
268 matched_opener_inline_pos = Some(opener.inline_pos);
269 }
270 state.node.children.push(new_token);
271 }
272 }
273
274 if opener.remaining > 0 {
275 state.node.children[idx].replace(opener);
276 } // otherwise node was already deleted
277 }
278
279 if new_min_opener_idx != 0 {
280 // If match for this delimiter run failed, we want to set lower bound for
281 // future lookups. This is required to make sure algorithm has linear
282 // complexity.
283 //
284 // See details here:
285 // https://github.com/commonmark/cmark/issues/178#issuecomment-270417442
286 //
287 let openers_for_marker = state
288 .node
289 .ext
290 .get_or_insert_default::<OpenersBottom<MARKER>>();
291 openers_for_marker.0[openers_parameter] = new_min_opener_idx;
292 }
293
294 // remove empty node as a small optimization so we can do less work later
295 if closer.remaining > 0 {
296 closer_token.replace(closer);
297 (closer_token, matched_opener_inline_pos)
298 } else {
299 (
300 state.node.children.pop().unwrap(),
301 matched_opener_inline_pos,
302 )
303 }
304}
305
306fn is_odd_match(opener: &EmphMarker, closer: &EmphMarker) -> bool {
307 // from spec:
308 //
309 // If one of the delimiters can both open and close emphasis, then the
310 // sum of the lengths of the delimiter runs containing the opening and
311 // closing delimiters must not be a multiple of 3 unless both lengths
312 // are multiples of 3.
313 //
314 #[allow(clippy::collapsible_if)]
315 if opener.close || closer.open {
316 if (opener.length + closer.length) % 3 == 0 {
317 if opener.length % 3 != 0 || closer.length % 3 != 0 {
318 return true;
319 }
320 }
321 }
322
323 false
324}
325
326#[doc(hidden)]
327pub struct FragmentsJoin;
328impl CoreRule for FragmentsJoin {
329 fn run(node: &mut Node, _: &MarkdownIt) {
330 node.walk_mut(|node, _| fragments_join(node));
331 }
332}
333
334/// Clean up tokens after emphasis and strikethrough postprocessing:
335/// merge adjacent text nodes into one and re-calculate all token levels
336///
337/// This is necessary because initially emphasis delimiter markers (*, _, ~)
338/// are treated as their own separate text tokens. Then emphasis rule either
339/// leaves them as text (needed to merge with adjacent text) or turns them
340/// into opening/closing tags (which messes up levels inside).
341///
342fn fragments_join(node: &mut Node) {
343 // replace all emph markers with text tokens
344 for token in node.children.iter_mut() {
345 if let Some(data) = token.cast::<EmphMarker>() {
346 let content = data.marker.to_string().repeat(data.remaining);
347 token.replace(Text { content });
348 }
349 }
350
351 // collapse adjacent text tokens
352 for idx in 1..node.children.len() {
353 let (tokens1, tokens2) = node.children.split_at_mut(idx);
354
355 let token1 = tokens1.last_mut().unwrap();
356 let Some(t1_data) = token1.cast_mut::<Text>() else {
357 continue;
358 };
359
360 let token2 = tokens2.first_mut().unwrap();
361 let Some(t2_data) = token2.cast_mut::<Text>() else {
362 continue;
363 };
364
365 // concat contents
366 let t2_content = std::mem::take(&mut t2_data.content);
367 t1_data.content += &t2_content;
368
369 // adjust source maps
370 if let Some(map1) = token1.srcmap {
371 if let Some(map2) = token2.srcmap {
372 token1.srcmap = Some(SourcePos::new(
373 map1.get_byte_offsets().0,
374 map2.get_byte_offsets().1,
375 ));
376 }
377 }
378
379 node.children.swap(idx - 1, idx);
380 }
381
382 // remove all empty tokens
383 node.children.retain(|token| {
384 if let Some(data) = token.cast::<Text>() {
385 !data.content.is_empty()
386 } else {
387 true
388 }
389 });
390}