| 1 | //! Structure similar to `*emphasis*` with configurable markers of fixed length. |
| 2 | //! |
| 3 | //! There are many structures in various markdown flavors that |
| 4 | //! can be implemented with this, namely: |
| 5 | //! |
| 6 | //! - `*emphasis*` or `_emphasis_` -> `<em>emphasis</em>` |
| 7 | //! - `**strong**` or `__strong__` -> `<strong>strong</strong>` |
| 8 | //! - `~~strikethrough~~` -> `<s>strikethrough</s>` |
| 9 | //! - `==marked==` -> `<mark>marked</mark>` |
| 10 | //! - `++inserted++` -> `<ins>inserted</ins>` |
| 11 | //! - `~subscript~` -> `<sub>subscript</sub>` |
| 12 | //! - `^superscript^` -> `<sup>superscript</sup>` |
| 13 | //! |
| 14 | //! You add a custom structure by using [add_with] function, which takes following arguments: |
| 15 | //! - `MARKER` - marker character |
| 16 | //! - `LENGTH` - length of the opening/closing marker (can be 1, 2 or 3) |
| 17 | //! - `CAN_SPLIT_WORD` - whether this structure can be found in the middle of the word |
| 18 | //! (for example, note the difference between `foo*bar*baz` and `foo_bar_baz` |
| 19 | //! in CommonMark - first one is an emphasis, second one isn't) |
| 20 | //! - `md` - parser instance |
| 21 | //! - `f` - function that should return your custom [Node] |
| 22 | //! |
| 23 | //! Here is an example of implementing superscript in your custom code: |
| 24 | //! |
| 25 | //! ```rust |
| 26 | //! use markdown_it::generics::inline::emph_pair; |
| 27 | //! use markdown_it::{MarkdownIt, Node, NodeValue, Renderer}; |
| 28 | //! |
| 29 | //! #[derive(Debug)] |
| 30 | //! struct Superscript; |
| 31 | //! impl NodeValue for Superscript { |
| 32 | //! fn render(&self, node: &Node, fmt: &mut dyn Renderer) { |
| 33 | //! fmt.open("sup", &node.attrs); |
| 34 | //! fmt.contents(&node.children); |
| 35 | //! fmt.close("sup"); |
| 36 | //! } |
| 37 | //! } |
| 38 | //! |
| 39 | //! let md = &mut MarkdownIt::new(); |
| 40 | //! emph_pair::add_with::<'^', 1, true>(md, || Node::new(Superscript)); |
| 41 | //! |
| 42 | //! let html = md.parse("e^iπ^+1=0").render(); |
| 43 | //! assert_eq!(html.trim(), "e<sup>iπ</sup>+1=0"); |
| 44 | //! ``` |
| 45 | //! |
| 46 | //! Note that these structures have lower priority than the rest of the rules, |
| 47 | //! e.g. `` *foo`bar*baz` `` is parsed as `*foo<code>bar*baz</code>`. |
| 48 | //! |
| 49 | use std::cmp::min; |
| 50 | |
| 51 | use crate::common::sourcemap::SourcePos; |
| 52 | use crate::parser::core::CoreRule; |
| 53 | use crate::parser::extset::{MarkdownItExt, NodeExt}; |
| 54 | use crate::parser::inline::builtin::InlineParserRule; |
| 55 | use crate::parser::inline::{InlineRule, InlineState, Text}; |
| 56 | use crate::{MarkdownIt, Node, NodeValue}; |
| 57 | |
| 58 | #[derive(Debug, Default)] |
| 59 | struct PairConfig<const MARKER: char> { |
| 60 | inserted: bool, |
| 61 | fns: [Option<fn() -> Node>; 3], |
| 62 | } |
| 63 | impl<const MARKER: char> MarkdownItExt for PairConfig<MARKER> {} |
| 64 | |
| 65 | #[derive(Debug, Default)] |
| 66 | struct OpenersBottom<const MARKER: char>([usize; 6]); |
| 67 | impl<const MARKER: char> NodeExt for OpenersBottom<MARKER> {} |
| 68 | |
| 69 | #[derive(Debug, Clone)] |
| 70 | #[doc(hidden)] |
| 71 | pub struct EmphMarker { |
| 72 | // Starting marker |
| 73 | pub marker: char, |
| 74 | |
| 75 | // Total length of these series of delimiters. |
| 76 | pub length: usize, |
| 77 | |
| 78 | // Remaining length that's not already matched to other delimiters. |
| 79 | pub remaining: usize, |
| 80 | |
| 81 | // Boolean flags that determine if this delimiter could open or close |
| 82 | // an emphasis. |
| 83 | pub open: bool, |
| 84 | pub close: bool, |
| 85 | |
| 86 | // Inline position (offset into the stripped inline text) where this |
| 87 | // delimiter starts. Stored so that `run` can compute the correct |
| 88 | // inline-coordinate token length when the delimiter spans multiple |
| 89 | // block-quote continuation lines (where srcmap offsets include the |
| 90 | // stripped `> ` bytes and therefore exceed the inline position). |
| 91 | pub inline_pos: usize, |
| 92 | } |
| 93 | |
| 94 | // this node is supposed to be replaced by actual emph or text node |
| 95 | impl NodeValue for EmphMarker {} |
| 96 | |
| 97 | pub fn add_with<const MARKER: char, const LENGTH: u8, const CAN_SPLIT_WORD: bool>( |
| 98 | md: &mut MarkdownIt, |
| 99 | f: fn() -> Node, |
| 100 | ) { |
| 101 | let pair_config = md.ext.get_or_insert_default::<PairConfig<MARKER>>(); |
| 102 | pair_config.fns[LENGTH as usize - 1] = Some(f); |
| 103 | |
| 104 | if !pair_config.inserted { |
| 105 | pair_config.inserted = true; |
| 106 | md.inline |
| 107 | .add_rule::<EmphPairScanner<MARKER, CAN_SPLIT_WORD>>(); |
| 108 | } |
| 109 | |
| 110 | if !md.has_rule::<FragmentsJoin>() { |
| 111 | md.add_rule::<FragmentsJoin>() |
| 112 | .before_all() |
| 113 | .after::<InlineParserRule>(); |
| 114 | } |
| 115 | } |
| 116 | |
| 117 | #[doc(hidden)] |
| 118 | pub struct EmphPairScanner<const MARKER: char, const CAN_SPLIT_WORD: bool>; |
| 119 | impl<const MARKER: char, const CAN_SPLIT_WORD: bool> InlineRule |
| 120 | for EmphPairScanner<MARKER, CAN_SPLIT_WORD> |
| 121 | { |
| 122 | const MARKER: char = MARKER; |
| 123 | |
| 124 | // this rule works on a closing marker, so for technical reasons any rules trying to skip it |
| 125 | // should see just plain text |
| 126 | fn check(_: &mut InlineState) -> Option<usize> { |
| 127 | None |
| 128 | } |
| 129 | |
| 130 | fn run(state: &mut InlineState) -> Option<(Node, usize)> { |
| 131 | let mut chars = state.src[state.pos..state.pos_max].chars(); |
| 132 | if chars.next().unwrap() != MARKER { |
| 133 | return None; |
| 134 | } |
| 135 | |
| 136 | let scanned = state.scan_delims(state.pos, CAN_SPLIT_WORD); |
| 137 | let inline_closer_start = state.pos; |
| 138 | let mut node = Node::new(EmphMarker { |
| 139 | marker: MARKER, |
| 140 | length: scanned.length, |
| 141 | remaining: scanned.length, |
| 142 | open: scanned.can_open, |
| 143 | close: scanned.can_close, |
| 144 | inline_pos: state.pos, |
| 145 | }); |
| 146 | node.srcmap = state.get_map(state.pos, state.pos + scanned.length); |
| 147 | let (node, opener_inline_pos) = scan_and_match_delimiters::<MARKER>(state, node); |
| 148 | // backtrack to keep correct source maps |
| 149 | state.pos += scanned.length; |
| 150 | // Compute token_len in inline coordinates. Using srcmap byte offsets |
| 151 | // (map.1 - map.0) is incorrect when the token spans multiple blockquote |
| 152 | // continuation lines because the srcmap reflects the original file bytes |
| 153 | // (which include the stripped `> ` prefixes) rather than the shorter |
| 154 | // inline text. When a match was found we use the opener's stored inline |
| 155 | // position; when no match was found the token is only the scanned |
| 156 | // delimiter itself. |
| 157 | let token_len = match opener_inline_pos { |
| 158 | Some(oip) => state.pos - oip, |
| 159 | None => scanned.length, |
| 160 | }; |
| 161 | // Sanity: keep state.pos valid (should never underflow, but guard anyway) |
| 162 | debug_assert!( |
| 163 | state.pos >= token_len, |
| 164 | "emph backtrack underflow: pos={} token_len={} closer_start={}", |
| 165 | state.pos, |
| 166 | token_len, |
| 167 | inline_closer_start |
| 168 | ); |
| 169 | state.pos -= token_len; |
| 170 | Some((node, token_len)) |
| 171 | } |
| 172 | } |
| 173 | |
| 174 | /// Assuming last token is a closing delimiter we just inserted, |
| 175 | /// try to find opener(s). If any are found, move stuff to nested emph node. |
| 176 | /// |
| 177 | /// Returns `(node, opener_inline_pos)`. `opener_inline_pos` is `Some` when a |
| 178 | /// match was found; the value is the `inline_pos` field of the matched opener. |
| 179 | fn scan_and_match_delimiters<const MARKER: char>( |
| 180 | state: &mut InlineState, |
| 181 | mut closer_token: Node, |
| 182 | ) -> (Node, Option<usize>) { |
| 183 | if state.node.children.is_empty() { |
| 184 | return (closer_token, None); |
| 185 | } // must have at least opener and closer |
| 186 | |
| 187 | let mut closer = closer_token.cast_mut::<EmphMarker>().unwrap().clone(); |
| 188 | if !closer.close { |
| 189 | return (closer_token, None); |
| 190 | } |
| 191 | |
| 192 | // Previously calculated lower bounds (previous fails) |
| 193 | // for each marker, each delimiter length modulo 3, |
| 194 | // and for whether this closer can be an opener; |
| 195 | // https://github.com/commonmark/cmark/commit/34250e12ccebdc6372b8b49c44fab57c72443460 |
| 196 | let openers_for_marker = state |
| 197 | .node |
| 198 | .ext |
| 199 | .get_or_insert_default::<OpenersBottom<MARKER>>(); |
| 200 | let openers_parameter = (closer.open as usize) * 3 + closer.length % 3; |
| 201 | |
| 202 | let min_opener_idx = openers_for_marker.0[openers_parameter]; |
| 203 | |
| 204 | let mut idx = state.node.children.len() - 1; |
| 205 | let mut new_min_opener_idx = idx; |
| 206 | let mut matched_opener_inline_pos: Option<usize> = None; |
| 207 | while idx > min_opener_idx { |
| 208 | idx -= 1; |
| 209 | |
| 210 | let Some(opener) = state.node.children[idx].cast::<EmphMarker>() else { |
| 211 | continue; |
| 212 | }; |
| 213 | |
| 214 | let mut opener = opener.clone(); |
| 215 | if opener.open && opener.marker == closer.marker && !is_odd_match(&opener, &closer) { |
| 216 | while closer.remaining > 0 && opener.remaining > 0 { |
| 217 | let max_marker_len = min(3, min(opener.remaining, closer.remaining)); |
| 218 | let mut matched_rule = None; |
| 219 | let fns = &state.md.ext.get::<PairConfig<MARKER>>().unwrap().fns; |
| 220 | for marker_len in (1..=max_marker_len).rev() { |
| 221 | if let Some(f) = fns[marker_len - 1] { |
| 222 | matched_rule = Some((marker_len, f)); |
| 223 | break; |
| 224 | } |
| 225 | } |
| 226 | |
| 227 | // If matched_fn isn't found, it can only mean that function is defined for larger marker |
| 228 | // than we have (e.g. function defined for **, we have *). |
| 229 | // Treat this as "marker not found". |
| 230 | if matched_rule.is_none() { |
| 231 | break; |
| 232 | } |
| 233 | |
| 234 | let (marker_len, marker_fn) = matched_rule.unwrap(); |
| 235 | |
| 236 | closer.remaining -= marker_len; |
| 237 | opener.remaining -= marker_len; |
| 238 | |
| 239 | let mut new_token = marker_fn(); |
| 240 | new_token.children = state.node.children.split_off(idx + 1); |
| 241 | |
| 242 | // cut marker_len chars from start, i.e. "12345" -> "345" |
| 243 | let mut end_map_pos = 0; |
| 244 | if let Some(map) = closer_token.srcmap { |
| 245 | let (start, end) = map.get_byte_offsets(); |
| 246 | closer_token.srcmap = Some(SourcePos::new(start + marker_len, end)); |
| 247 | end_map_pos = start + marker_len; |
| 248 | } |
| 249 | |
| 250 | // cut marker_len chars from end, i.e. "12345" -> "123" |
| 251 | let mut start_map_pos = 0; |
| 252 | let opener_token = state.node.children.last_mut().unwrap(); |
| 253 | if let Some(map) = opener_token.srcmap { |
| 254 | let (start, end) = map.get_byte_offsets(); |
| 255 | opener_token.srcmap = Some(SourcePos::new(start, end - marker_len)); |
| 256 | start_map_pos = end - marker_len; |
| 257 | } |
| 258 | |
| 259 | new_token.srcmap = state.get_map(start_map_pos, end_map_pos); |
| 260 | |
| 261 | // remove empty node as a small optimization so we can do less work later |
| 262 | if opener.remaining == 0 { |
| 263 | state.node.children.pop(); |
| 264 | } |
| 265 | |
| 266 | new_min_opener_idx = 0; |
| 267 | if matched_opener_inline_pos.is_none() { |
| 268 | matched_opener_inline_pos = Some(opener.inline_pos); |
| 269 | } |
| 270 | state.node.children.push(new_token); |
| 271 | } |
| 272 | } |
| 273 | |
| 274 | if opener.remaining > 0 { |
| 275 | state.node.children[idx].replace(opener); |
| 276 | } // otherwise node was already deleted |
| 277 | } |
| 278 | |
| 279 | if new_min_opener_idx != 0 { |
| 280 | // If match for this delimiter run failed, we want to set lower bound for |
| 281 | // future lookups. This is required to make sure algorithm has linear |
| 282 | // complexity. |
| 283 | // |
| 284 | // See details here: |
| 285 | // https://github.com/commonmark/cmark/issues/178#issuecomment-270417442 |
| 286 | // |
| 287 | let openers_for_marker = state |
| 288 | .node |
| 289 | .ext |
| 290 | .get_or_insert_default::<OpenersBottom<MARKER>>(); |
| 291 | openers_for_marker.0[openers_parameter] = new_min_opener_idx; |
| 292 | } |
| 293 | |
| 294 | // remove empty node as a small optimization so we can do less work later |
| 295 | if closer.remaining > 0 { |
| 296 | closer_token.replace(closer); |
| 297 | (closer_token, matched_opener_inline_pos) |
| 298 | } else { |
| 299 | ( |
| 300 | state.node.children.pop().unwrap(), |
| 301 | matched_opener_inline_pos, |
| 302 | ) |
| 303 | } |
| 304 | } |
| 305 | |
| 306 | fn is_odd_match(opener: &EmphMarker, closer: &EmphMarker) -> bool { |
| 307 | // from spec: |
| 308 | // |
| 309 | // If one of the delimiters can both open and close emphasis, then the |
| 310 | // sum of the lengths of the delimiter runs containing the opening and |
| 311 | // closing delimiters must not be a multiple of 3 unless both lengths |
| 312 | // are multiples of 3. |
| 313 | // |
| 314 | #[allow(clippy::collapsible_if)] |
| 315 | if opener.close || closer.open { |
| 316 | if (opener.length + closer.length) % 3 == 0 { |
| 317 | if opener.length % 3 != 0 || closer.length % 3 != 0 { |
| 318 | return true; |
| 319 | } |
| 320 | } |
| 321 | } |
| 322 | |
| 323 | false |
| 324 | } |
| 325 | |
| 326 | #[doc(hidden)] |
| 327 | pub struct FragmentsJoin; |
| 328 | impl CoreRule for FragmentsJoin { |
| 329 | fn run(node: &mut Node, _: &MarkdownIt) { |
| 330 | node.walk_mut(|node, _| fragments_join(node)); |
| 331 | } |
| 332 | } |
| 333 | |
| 334 | /// Clean up tokens after emphasis and strikethrough postprocessing: |
| 335 | /// merge adjacent text nodes into one and re-calculate all token levels |
| 336 | /// |
| 337 | /// This is necessary because initially emphasis delimiter markers (*, _, ~) |
| 338 | /// are treated as their own separate text tokens. Then emphasis rule either |
| 339 | /// leaves them as text (needed to merge with adjacent text) or turns them |
| 340 | /// into opening/closing tags (which messes up levels inside). |
| 341 | /// |
| 342 | fn fragments_join(node: &mut Node) { |
| 343 | // replace all emph markers with text tokens |
| 344 | for token in node.children.iter_mut() { |
| 345 | if let Some(data) = token.cast::<EmphMarker>() { |
| 346 | let content = data.marker.to_string().repeat(data.remaining); |
| 347 | token.replace(Text { content }); |
| 348 | } |
| 349 | } |
| 350 | |
| 351 | // collapse adjacent text tokens |
| 352 | for idx in 1..node.children.len() { |
| 353 | let (tokens1, tokens2) = node.children.split_at_mut(idx); |
| 354 | |
| 355 | let token1 = tokens1.last_mut().unwrap(); |
| 356 | let Some(t1_data) = token1.cast_mut::<Text>() else { |
| 357 | continue; |
| 358 | }; |
| 359 | |
| 360 | let token2 = tokens2.first_mut().unwrap(); |
| 361 | let Some(t2_data) = token2.cast_mut::<Text>() else { |
| 362 | continue; |
| 363 | }; |
| 364 | |
| 365 | // concat contents |
| 366 | let t2_content = std::mem::take(&mut t2_data.content); |
| 367 | t1_data.content += &t2_content; |
| 368 | |
| 369 | // adjust source maps |
| 370 | if let Some(map1) = token1.srcmap { |
| 371 | if let Some(map2) = token2.srcmap { |
| 372 | token1.srcmap = Some(SourcePos::new( |
| 373 | map1.get_byte_offsets().0, |
| 374 | map2.get_byte_offsets().1, |
| 375 | )); |
| 376 | } |
| 377 | } |
| 378 | |
| 379 | node.children.swap(idx - 1, idx); |
| 380 | } |
| 381 | |
| 382 | // remove all empty tokens |
| 383 | node.children.retain(|token| { |
| 384 | if let Some(data) = token.cast::<Text>() { |
| 385 | !data.content.is_empty() |
| 386 | } else { |
| 387 | true |
| 388 | } |
| 389 | }); |
| 390 | } |