Skip to main content

forge_doc/
markdown.rs

1//! Shared Markdown code-region and MDX ESM handling.
2//!
3//! README and NatSpec escaping retain their own policies for malformed fences and prose.
4
5use markdown::{ParseOptions, mdast::Node, to_mdast};
6use std::ops::Range;
7
8/// Byte ranges that Markdown parses as code under `options`. An HTML entity would render literally
9/// inside these ranges, so neutralization skips them. If malformed MDX cannot be parsed, returning
10/// no ranges favors neutralizing possible ESM over preserving an invalid code example
11/// byte-for-byte.
12pub(crate) fn code_regions(text: &str, options: &ParseOptions) -> Vec<Range<usize>> {
13    let Ok(tree) = to_mdast(text, options) else { return Vec::new() };
14    let mut regions = Vec::new();
15    collect_code_regions(&tree, &mut regions);
16    regions
17}
18
19/// Collect fenced and inline code positions from the MDX-aware syntax tree.
20fn collect_code_regions(node: &Node, regions: &mut Vec<Range<usize>>) {
21    if matches!(node, Node::Code(_) | Node::InlineCode(_))
22        && let Some(position) = node.position()
23    {
24        regions.push(position.start.offset..position.end.offset);
25    }
26    if let Some(children) = node.children() {
27        for child in children {
28            collect_code_regions(child, regions);
29        }
30    }
31}
32
33/// Logical lines and their byte offsets in the original text. CRLF is one separator; lone CR and
34/// LF are separators too. The separator bytes are excluded from the returned slices and preserved
35/// in the source string.
36pub(crate) fn logical_lines(text: &str) -> impl Iterator<Item = (usize, &str)> {
37    let bytes = text.as_bytes();
38    let mut offset = 0;
39    std::iter::from_fn(move || {
40        if offset >= bytes.len() {
41            return None;
42        }
43        let start = offset;
44        let end = bytes[start..]
45            .iter()
46            .position(|&byte| byte == b'\n' || byte == b'\r')
47            .map_or(bytes.len(), |position| start + position);
48        offset = if end == bytes.len() {
49            end
50        } else if bytes[end] == b'\r' && bytes.get(end + 1) == Some(&b'\n') {
51            end + 2
52        } else {
53            end + 1
54        };
55        Some((start, &text[start..end]))
56    })
57}
58
59/// Check a position against sorted, merged ranges while advancing monotonically.
60pub(crate) fn region_contains(
61    regions: &[Range<usize>],
62    cursor: &mut usize,
63    position: usize,
64) -> bool {
65    while regions.get(*cursor).is_some_and(|region| region.end <= position) {
66        *cursor += 1;
67    }
68    regions.get(*cursor).is_some_and(|region| region.start <= position)
69}
70
71/// Neutralize any line MDX would parse as an ESM statement (`import ` or `export ` at column one):
72/// the keyword's prefix becomes HTML entities, so the line renders the same but no longer
73/// begins with an ESM token. NatSpec text can be inherited from a dependency via `@inheritdoc`, so
74/// this must run wherever displayed prose is assembled. A keyword that falls inside a Markdown
75/// code span or fenced code block is left untouched (see `code_regions`): the entity would render
76/// literally and corrupt the example, and MDX would not execute it there.
77pub(crate) fn neutralize_esm(text: &str) -> String {
78    let regions = code_regions(text, &ParseOptions::mdx());
79    let mut region_cursor = 0;
80    let mut copied = 0;
81    let mut out = String::with_capacity(text.len());
82
83    for (line_start, line) in logical_lines(text) {
84        let replacement = if line.starts_with("import ") {
85            Some(("&#105;&#109;", 2))
86        } else if line.starts_with("export ") {
87            Some(("&#101;", 1))
88        } else {
89            None
90        };
91        let Some((entity, prefix_len)) = replacement else { continue };
92        if region_contains(&regions, &mut region_cursor, line_start) {
93            continue;
94        }
95        out.push_str(&text[copied..line_start]);
96        out.push_str(entity);
97        copied = line_start + prefix_len;
98    }
99
100    out.push_str(&text[copied..]);
101    out
102}