Skip to main content

rustdoc/passes/lint/
bare_urls.rs

1//! Detects links that are not linkified, e.g., in Markdown such as `Go to https://example.com/.`
2//! Suggests wrapping the link with angle brackets: `Go to <https://example.com/>.` to linkify it.
3
4use core::ops::Range;
5use std::sync::LazyLock;
6
7use regex::Regex;
8use rustc_errors::{Applicability, DiagDecorator};
9use rustc_hir::HirId;
10use rustc_resolve::rustdoc::pulldown_cmark::{
11    DefaultBrokenLinkCallback, Event, Tag, TextMergeWithOffset,
12};
13use rustc_resolve::rustdoc::source_span_for_markdown_range;
14use tracing::trace;
15
16use crate::clean::*;
17use crate::core::DocContext;
18use crate::html::markdown::main_body_opts;
19
20pub(super) fn visit_item(cx: &DocContext<'_>, item: &Item, hir_id: HirId, dox: &str) {
21    let report_diag = |cx: &DocContext<'_>,
22                       msg: &'static str,
23                       range: Range<usize>,
24                       without_brackets: Option<&str>| {
25        let maybe_sp = source_span_for_markdown_range(cx.tcx, dox, &range, &item.attrs.doc_strings)
26            .map(|(sp, _)| sp);
27        let sp = maybe_sp.unwrap_or_else(|| item.attr_span(cx.tcx));
28        cx.tcx.emit_node_span_lint(
29            crate::lint::BARE_URLS,
30            hir_id,
31            sp,
32            DiagDecorator(|lint| {
33                lint.primary_message(msg)
34                    .note("bare URLs are not automatically turned into clickable links");
35                // The fallback of using the attribute span is suitable for
36                // highlighting where the error is, but not for placing the < and >
37                if let Some(sp) = maybe_sp {
38                    if let Some(without_brackets) = without_brackets {
39                        lint.multipart_suggestion(
40                            "use an automatic link instead",
41                            vec![(sp, format!("<{without_brackets}>"))],
42                            Applicability::MachineApplicable,
43                        );
44                    } else {
45                        lint.multipart_suggestion(
46                            "use an automatic link instead",
47                            vec![
48                                (sp.shrink_to_lo(), "<".to_string()),
49                                (sp.shrink_to_hi(), ">".to_string()),
50                            ],
51                            Applicability::MachineApplicable,
52                        );
53                    }
54                }
55            }),
56        );
57    };
58
59    // pulldown-cmark can split a URL into multiple `Text` events while processing
60    // characters such as `_` according to CommonMark's emphasis rules.
61    // `TextMergeWithOffset` merges these events so we can check the complete URL.
62    let mut p = TextMergeWithOffset::<DefaultBrokenLinkCallback>::new_ext(dox, main_body_opts());
63
64    while let Some((event, range)) = p.next() {
65        match event {
66            Event::Text(_s) => find_raw_urls(cx, dox, range, &report_diag),
67            // We don't want to check the text inside code blocks or links.
68            Event::Start(tag @ (Tag::CodeBlock(_) | Tag::Link { .. })) => {
69                let end = tag.to_end();
70                for (event, _) in p.by_ref() {
71                    if matches!(event, Event::End(tag) if tag == end) {
72                        break;
73                    }
74                }
75            }
76            _ => {}
77        }
78    }
79}
80
81static URL_SCHEME_HOST_REGEX: LazyLock<Regex> = LazyLock::new(|| {
82    Regex::new(concat!(
83        r"https?://",                          // url scheme
84        r"([-a-zA-Z0-9@:%._\+~#=]{2,256}\.)+", // one or more subdomains
85        r"[a-zA-Z]{2,63}",                     // root domain
86    ))
87    .expect("failed to build regex")
88});
89
90fn find_raw_urls(
91    cx: &DocContext<'_>,
92    dox: &str,
93    range: Range<usize>,
94    f: &impl Fn(&DocContext<'_>, &'static str, Range<usize>, Option<&str>),
95) {
96    trace!("looking for raw urls in {text}", text = &dox[range.clone()]);
97    // For now, we only check "full" URLs (meaning, starting with "http://" or "https://").
98    for match_ in URL_SCHEME_HOST_REGEX.find_iter(&dox[range.clone()]) {
99        let mut url_range = match_.range();
100        // We have a range within `dox[range]`.
101        // We need a range within `dox` to report the diagnostic.
102        url_range.start += range.start;
103        url_range.end += range.start;
104        // We found the scheme and host. Find the path, query, or fragment.
105        // We want to check for matching, balanced parens,
106        // but regex isn't powerful enough for that.
107        let mut paren_stack = Vec::with_capacity(3);
108        'parts: while let Some(&sep) = dox.as_bytes().get(url_range.end) {
109            // The hostname must be immediately followed by a path, query,
110            // or fragment-declaring separator.
111            if !matches!(sep, b'/' | b'?' | b'#') {
112                break;
113            }
114            url_range.end += 1;
115            while let Some(&c) = dox.as_bytes().get(url_range.end) {
116                if c == b'(' {
117                    paren_stack.push(url_range.end);
118                } else if c == b')' {
119                    // We assume the first unmatched parenthesis marks the end of the url,
120                    // as urls rarely contain unbalanced parenthesis in practice.
121                    if paren_stack.pop().is_none() {
122                        break 'parts;
123                    }
124                } else if !matches!(
125                    c,
126                    b'-'
127                    | b'a'..=b'z'
128                    | b'A'..=b'Z'
129                    | b'0'..=b'9'
130                    | b'@'
131                    | b':'
132                    | b'%'
133                    | b'_'
134                    | b'\\'
135                    | b'+'
136                    | b'.'
137                    | b'~'
138                    | b'&'
139                    | b'='
140                ) {
141                    break;
142                }
143                url_range.end += 1;
144            }
145        }
146        // We assume the first unmatched parenthesis marks the end of the url,
147        // as urls rarely contain unbalanced parenthesis in practice.
148        if let Some(&end) = paren_stack.first() {
149            url_range.end = end;
150        }
151        let mut without_brackets = None;
152        // If the link is contained inside `[]`, then we need to replace the brackets and
153        // not just add `<>`.
154        if dox[..url_range.start].ends_with('[')
155            && url_range.end <= dox.len()
156            && dox[url_range.end..].starts_with(']')
157        {
158            url_range.start -= 1;
159            url_range.end += 1;
160            without_brackets = Some(match_.as_str());
161        } else {
162            // Periods are valid in URLs, but very uncommon as the last character of one, while
163            // being very common as sentence punctuation right after one. Leave any trailing
164            // period out of the link, so that `Visit https://example.com/docs.` is linkified as
165            // `Visit <https://example.com/docs>.`.
166            let trailing_periods =
167                dox[url_range.clone()].len() - dox[url_range.clone()].trim_end_matches('.').len();
168            url_range.end -= trailing_periods;
169        }
170        f(cx, "this URL is not a hyperlink", url_range, without_brackets);
171    }
172}