Skip to main content

rustc_parse/lexer/
mod.rs

1use diagnostics::make_errors_for_mismatched_closing_delims;
2use rustc_ast::ast::{self, AttrStyle};
3use rustc_ast::token::{self, CommentKind, Delimiter, IdentKind, Token, TokenKind};
4use rustc_ast::tokenstream::TokenStream;
5use rustc_ast::util::unicode::{TEXT_FLOW_CONTROL_CHARS, contains_text_flow_control_chars};
6use rustc_errors::codes::*;
7use rustc_errors::{Applicability, Diag, DiagCtxtHandle, Diagnostic, StashKey};
8use rustc_lexer::{
9    Base, Cursor, DocStyle, FrontmatterAllowed, LiteralKind, RawStrError, is_horizontal_whitespace,
10};
11use rustc_lint_defs::builtin::{
12    RUST_2021_PREFIXES_INCOMPATIBLE_SYNTAX, RUST_2024_GUARDED_STRING_INCOMPATIBLE_SYNTAX,
13    TEXT_DIRECTION_CODEPOINT_IN_COMMENT, TEXT_DIRECTION_CODEPOINT_IN_LITERAL,
14};
15use rustc_literal_escaper::{EscapeError, Mode, check_for_errors};
16use rustc_session::parse::ParseSess;
17use rustc_span::edition::Edition;
18use rustc_span::{BytePos, Pos, Span, Symbol, sym};
19use tracing::debug;
20
21use crate::lexer::diagnostics::TokenTreeDiagInfo;
22use crate::lexer::unicode_chars::UNICODE_ARRAY;
23
24mod diagnostics;
25mod tokentrees;
26mod unescape_error_reporting;
27mod unicode_chars;
28
29use unescape_error_reporting::{emit_unescape_error, escaped_char};
30
31// This type is used a lot. Make sure it doesn't unintentionally get bigger.
32//
33// This assertion is in this crate, rather than in `rustc_lexer`, because that
34// crate cannot depend on `rustc_data_structures`.
35#[cfg(target_pointer_width = "64")]
36const _: [(); 12] = [(); ::std::mem::size_of::<rustc_lexer::Token>()];rustc_data_structures::static_assert_size!(rustc_lexer::Token, 12);
37
38const INVISIBLE_CHARACTERS: [char; 8] = [
39    '\u{200b}', '\u{200c}', '\u{2060}', '\u{2061}', '\u{2062}', '\u{00ad}', '\u{034f}', '\u{061c}',
40];
41
42#[derive(#[automatically_derived]
impl ::core::clone::Clone for UnmatchedDelim {
    #[inline]
    fn clone(&self) -> Self {
        Self {
            found_delim: ::core::clone::Clone::clone(&self.found_delim),
            found_span: ::core::clone::Clone::clone(&self.found_span),
            unclosed_span: ::core::clone::Clone::clone(&self.unclosed_span),
            candidate_span: ::core::clone::Clone::clone(&self.candidate_span),
        }
    }
}Clone, #[automatically_derived]
impl ::core::fmt::Debug for UnmatchedDelim {
    #[inline]
    fn fmt(&self, f: &mut ::core::fmt::Formatter) -> ::core::fmt::Result {
        ::core::fmt::Formatter::debug_struct_field4_finish(f,
            "UnmatchedDelim", "found_delim", &self.found_delim, "found_span",
            &self.found_span, "unclosed_span", &self.unclosed_span,
            "candidate_span", &&self.candidate_span)
    }
}Debug)]
43pub(crate) struct UnmatchedDelim {
44    pub found_delim: Option<Delimiter>,
45    pub found_span: Span,
46    pub unclosed_span: Option<Span>,
47    pub candidate_span: Option<Span>,
48}
49
50/// Which tokens should be stripped before lexing the tokens.
51pub enum StripTokens {
52    /// Strip both shebang and frontmatter.
53    ShebangAndFrontmatter,
54    /// Strip the shebang but not frontmatter.
55    ///
56    /// That means that char sequences looking like frontmatter are simply
57    /// interpreted as regular Rust lexemes.
58    Shebang,
59    /// Strip nothing.
60    ///
61    /// In other words, char sequences looking like a shebang or frontmatter
62    /// are simply interpreted as regular Rust lexemes.
63    Nothing,
64}
65
66pub(crate) fn lex_token_trees<'psess, 'src>(
67    psess: &'psess ParseSess,
68    mut src: &'src str,
69    mut start_pos: BytePos,
70    override_span: Option<Span>,
71    strip_tokens: StripTokens,
72) -> Result<TokenStream, Vec<Diag<'psess>>> {
73    match strip_tokens {
74        StripTokens::Shebang | StripTokens::ShebangAndFrontmatter => {
75            if let Some(shebang_len) = rustc_lexer::strip_shebang(src) {
76                src = &src[shebang_len..];
77                start_pos = start_pos + BytePos::from_usize(shebang_len);
78            }
79        }
80        StripTokens::Nothing => {}
81    }
82
83    let frontmatter_allowed = match strip_tokens {
84        StripTokens::ShebangAndFrontmatter => FrontmatterAllowed::Yes,
85        StripTokens::Shebang | StripTokens::Nothing => FrontmatterAllowed::No,
86    };
87
88    let cursor = Cursor::new(src, frontmatter_allowed);
89    let mut lexer = Lexer {
90        psess,
91        start_pos,
92        pos: start_pos,
93        src,
94        cursor,
95        override_span,
96        nbsp_is_whitespace: false,
97        last_lifetime: None,
98        token: Token::dummy(),
99        diag_info: TokenTreeDiagInfo::default(),
100    };
101    let res = lexer.lex_token_trees(/* is_delimited */ false);
102
103    let mut unmatched_closing_delims: Vec<_> =
104        make_errors_for_mismatched_closing_delims(&lexer.diag_info.unmatched_delims, psess);
105
106    match res {
107        Ok((_open_spacing, stream)) => {
108            if unmatched_closing_delims.is_empty() {
109                Ok(stream)
110            } else {
111                // Return error if there are unmatched delimiters or unclosed delimiters.
112                Err(unmatched_closing_delims)
113            }
114        }
115        Err(errs) => {
116            // We emit delimiter mismatch errors first, then emit the unclosing delimiter mismatch
117            // because the delimiter mismatch is more likely to be the root cause of error
118            unmatched_closing_delims.push(errs);
119            Err(unmatched_closing_delims)
120        }
121    }
122}
123
124struct Lexer<'psess, 'src> {
125    psess: &'psess ParseSess,
126    /// Initial position, read-only.
127    start_pos: BytePos,
128    /// The absolute offset within the source_map of the current character.
129    pos: BytePos,
130    /// Source text to tokenize.
131    src: &'src str,
132    /// Cursor for getting lexer tokens.
133    cursor: Cursor<'src>,
134    override_span: Option<Span>,
135    /// When a "unknown start of token: \u{a0}" has already been emitted earlier
136    /// in this file, it's safe to treat further occurrences of the non-breaking
137    /// space character as whitespace.
138    nbsp_is_whitespace: bool,
139
140    /// Track the `Span` for the leading `'` of the last lifetime. Used for
141    /// diagnostics to detect possible typo where `"` was meant.
142    last_lifetime: Option<Span>,
143
144    /// The current token.
145    token: Token,
146
147    diag_info: TokenTreeDiagInfo,
148}
149
150impl<'psess, 'src> Lexer<'psess, 'src> {
151    fn dcx(&self) -> DiagCtxtHandle<'psess> {
152        self.psess.dcx()
153    }
154
155    fn mk_sp(&self, lo: BytePos, hi: BytePos) -> Span {
156        self.override_span.unwrap_or_else(|| Span::with_root_ctxt(lo, hi))
157    }
158
159    /// Returns the next token, paired with a bool indicating if the token was
160    /// preceded by whitespace.
161    fn next_token_from_cursor(&mut self) -> (Token, bool) {
162        let mut preceded_by_whitespace = false;
163        let mut swallow_next_invalid = 0;
164        // Skip trivial (whitespace & comments) tokens
165        loop {
166            let str_before = self.cursor.as_str();
167            let token = self.cursor.advance_token();
168            let start = self.pos;
169            self.pos = self.pos + BytePos(token.len);
170
171            {
    use ::tracing::__macro_support::Callsite as _;
    static __CALLSITE: ::tracing::callsite::DefaultCallsite =
        {
            static META: ::tracing::Metadata<'static> =
                {
                    ::tracing_core::metadata::Metadata::new("event /rustc-dev/8d1a76430406c877b35d0b627e7f796dcf0dfeca/compiler/rustc_parse/src/lexer/mod.rs:171",
                        "rustc_parse::lexer", ::tracing::Level::DEBUG,
                        ::tracing_core::__macro_support::Option::Some("/rustc-dev/8d1a76430406c877b35d0b627e7f796dcf0dfeca/compiler/rustc_parse/src/lexer/mod.rs"),
                        ::tracing_core::__macro_support::Option::Some(171u32),
                        ::tracing_core::__macro_support::Option::Some("rustc_parse::lexer"),
                        ::tracing_core::field::FieldSet::new(&["message"],
                            ::tracing_core::callsite::Identifier(&__CALLSITE)),
                        ::tracing::metadata::Kind::EVENT)
                };
            ::tracing::callsite::DefaultCallsite::new(&META)
        };
    let enabled =
        ::tracing::Level::DEBUG <= ::tracing::level_filters::STATIC_MAX_LEVEL
                &&
                ::tracing::Level::DEBUG <=
                    ::tracing::level_filters::LevelFilter::current() &&
            {
                let interest = __CALLSITE.interest();
                !interest.is_never() &&
                    ::tracing::__macro_support::__is_enabled(__CALLSITE.metadata(),
                        interest)
            };
    if enabled {
        (|value_set: ::tracing::field::ValueSet|
                    {
                        let meta = __CALLSITE.metadata();
                        ::tracing::Event::dispatch(meta, &value_set);
                        ;
                    })({
                #[allow(unused_imports)]
                use ::tracing::field::{debug, display, Value};
                __CALLSITE.metadata().fields().value_set_all(&[(::tracing::__macro_support::Option::Some(&format_args!("next_token: {0:?}({1:?})",
                                                    token.kind, self.str_from(start)) as
                                            &dyn ::tracing::field::Value))])
            });
    } else { ; }
};debug!("next_token: {:?}({:?})", token.kind, self.str_from(start));
172
173            if let rustc_lexer::TokenKind::Semi
174            | rustc_lexer::TokenKind::LineComment { .. }
175            | rustc_lexer::TokenKind::BlockComment { .. }
176            | rustc_lexer::TokenKind::CloseParen
177            | rustc_lexer::TokenKind::CloseBrace
178            | rustc_lexer::TokenKind::CloseBracket = token.kind
179            {
180                // Heuristic: we assume that it is unlikely we're dealing with an unterminated
181                // string surrounded by single quotes.
182                self.last_lifetime = None;
183            }
184
185            // Now "cook" the token, converting the simple `rustc_lexer::TokenKind` enum into a
186            // rich `rustc_ast::TokenKind`. This turns strings into interned symbols and runs
187            // additional validation.
188            let kind = match token.kind {
189                rustc_lexer::TokenKind::LineComment { doc_style } => {
190                    // Skip non-doc comments
191                    let Some(doc_style) = doc_style else {
192                        self.lint_unicode_text_flow(start);
193                        preceded_by_whitespace = true;
194                        continue;
195                    };
196
197                    // Opening delimiter of the length 3 is not included into the symbol.
198                    let content_start = start + BytePos(3);
199                    let content = self.str_from(content_start);
200                    self.lint_doc_comment_unicode_text_flow(start, content);
201                    self.cook_doc_comment(content_start, content, CommentKind::Line, doc_style)
202                }
203                rustc_lexer::TokenKind::BlockComment { doc_style, terminated } => {
204                    if !terminated {
205                        self.report_unterminated_block_comment(start, doc_style);
206                    }
207
208                    // Skip non-doc comments
209                    let Some(doc_style) = doc_style else {
210                        self.lint_unicode_text_flow(start);
211                        preceded_by_whitespace = true;
212                        continue;
213                    };
214
215                    // Opening delimiter of the length 3 and closing delimiter of the length 2
216                    // are not included into the symbol.
217                    let content_start = start + BytePos(3);
218                    let content_end = self.pos - BytePos(if terminated { 2 } else { 0 });
219                    let content = self.str_from_to(content_start, content_end);
220                    self.lint_doc_comment_unicode_text_flow(start, content);
221                    self.cook_doc_comment(content_start, content, CommentKind::Block, doc_style)
222                }
223                rustc_lexer::TokenKind::Frontmatter { has_invalid_preceding_whitespace, invalid_infostring } => {
224                    self.validate_frontmatter(start, has_invalid_preceding_whitespace, invalid_infostring);
225                    preceded_by_whitespace = true;
226                    continue;
227                }
228                rustc_lexer::TokenKind::Whitespace => {
229                    preceded_by_whitespace = true;
230                    continue;
231                }
232                rustc_lexer::TokenKind::Ident => self.ident(start),
233                rustc_lexer::TokenKind::RawIdent => {
234                    let sym = nfc_normalize(self.str_from(start + BytePos(2)));
235                    let span = self.mk_sp(start, self.pos);
236                    self.psess.symbol_gallery.insert(sym, span);
237                    if !sym.can_be_raw() {
238                        self.dcx().emit_err(crate::diagnostics::CannotBeRawIdent { span, ident: sym });
239                    }
240                    self.psess.raw_identifier_spans.push(span);
241                    token::Ident(sym, IdentKind::Raw)
242                }
243                rustc_lexer::TokenKind::ForcedKeywordIdent => {
244                    let span = self.mk_sp(start, self.pos);
245
246                    if span.edition().at_least_rust_2021() {
247                        let sym = nfc_normalize(self.str_from(start + BytePos(2)));
248                        self.psess.symbol_gallery.insert(sym, span);
249                        self.psess.gated_spans.gate(sym::forced_keywords, span);
250                        token::Ident(sym, IdentKind::ForcedKeyword)
251                    } else {
252                        // Reset the state so that only the `k` was consumed.
253                        self.pos = start + BytePos(1);
254                        self.cursor = Cursor::new(&str_before[1..], FrontmatterAllowed::No);
255
256                        self.psess.buffer_lint(
257                            RUST_2021_PREFIXES_INCOMPATIBLE_SYNTAX,
258                            span,
259                            ast::CRATE_NODE_ID,
260                            crate::diagnostics::ReservedPrefixLint {
261                                subject: "this".into(),
262                                kind: "forced keyword",
263                                edition: Edition::Edition2021,
264                                sugg: self.mk_sp(start, self.pos).shrink_to_hi(),
265                            }
266                        );
267                        self.ident(start)
268                    }
269                }
270                rustc_lexer::TokenKind::UnknownPrefix => {
271                    self.report_unknown_prefix(start);
272                    self.ident(start)
273                }
274                rustc_lexer::TokenKind::UnknownPrefixLifetime => {
275                    self.report_unknown_prefix(start);
276                    // Include the leading `'` in the real identifier, for macro
277                    // expansion purposes. See #12512 for the gory details of why
278                    // this is necessary.
279                    let lifetime_name = self.str_from(start);
280                    self.last_lifetime = Some(self.mk_sp(start, start + BytePos(1)));
281                    let ident = Symbol::intern(lifetime_name);
282                    token::Lifetime(ident, IdentKind::Normal)
283                }
284                rustc_lexer::TokenKind::InvalidIdent
285                    // Do not recover an identifier with emoji if the codepoint is a confusable
286                    // with a recoverable substitution token, like `➖`.
287                    if !UNICODE_ARRAY.iter().any(|&(c, _, _)| {
288                        let sym = self.str_from(start);
289                        sym.chars().count() == 1 && c == sym.chars().next().unwrap()
290                    }) =>
291                {
292                    let sym = nfc_normalize(self.str_from(start));
293                    let span = self.mk_sp(start, self.pos);
294                    self.psess
295                        .bad_unicode_identifiers
296                        .borrow_mut()
297                        .entry(sym)
298                        .or_default()
299                        .push(span);
300                    token::Ident(sym, IdentKind::Normal)
301                }
302                // split up (raw) c string literals to an ident and a string literal when edition <
303                // 2021.
304                rustc_lexer::TokenKind::Literal {
305                    kind: kind @ (LiteralKind::CStr { .. } | LiteralKind::RawCStr { .. }),
306                    suffix_start: _,
307                } if let span = self.mk_sp(start, self.pos) && !span.edition().at_least_rust_2021() => {
308                    let (prefix_len, kind) = match kind {
309                        LiteralKind::CStr { .. } => (1, "C string literal"),
310                        LiteralKind::RawCStr { .. } => (2, "raw C string literal"),
311                        _ => ::core::panicking::panic("internal error: entered unreachable code")unreachable!(),
312                    };
313
314                    // reset the state so that only the prefix ("c" or "cr") was consumed.
315                    self.pos = start + BytePos(prefix_len);
316                    self.cursor = Cursor::new(&str_before[prefix_len as usize..], FrontmatterAllowed::No);
317
318                    self.psess.buffer_lint(
319                        RUST_2021_PREFIXES_INCOMPATIBLE_SYNTAX,
320                        span,
321                        ast::CRATE_NODE_ID,
322                        crate::diagnostics::ReservedPrefixLint {
323                            subject: "this".into(),
324                            kind,
325                            edition: Edition::Edition2021,
326                            sugg: self.mk_sp(start, self.pos).shrink_to_hi(),
327                        },
328                    );
329
330                    self.ident(start)
331                }
332                rustc_lexer::TokenKind::GuardedStrPrefix => {
333                    self.maybe_report_guarded_str(start, str_before)
334                }
335                rustc_lexer::TokenKind::Literal { kind, suffix_start } => {
336                    let suffix_start = start + BytePos(suffix_start);
337                    let (kind, symbol) = self.cook_lexer_literal(start, suffix_start, kind);
338                    let suffix = if suffix_start < self.pos {
339                        let string = self.str_from(suffix_start);
340                        if string == "_" {
341                            self.dcx().emit_err(crate::diagnostics::UnderscoreLiteralSuffix {
342                                span: self.mk_sp(suffix_start, self.pos),
343                            });
344                            None
345                        } else {
346                            Some(Symbol::intern(string))
347                        }
348                    } else {
349                        None
350                    };
351                    self.lint_literal_unicode_text_flow(symbol, kind, self.mk_sp(start, self.pos), "literal");
352                    token::Literal(token::Lit { kind, symbol, suffix })
353                }
354                rustc_lexer::TokenKind::Lifetime { starts_with_number } => {
355                    // Include the leading `'` in the real identifier, for macro
356                    // expansion purposes. See #12512 for the gory details of why
357                    // this is necessary.
358                    let lifetime_name = nfc_normalize(self.str_from(start));
359                    self.last_lifetime = Some(self.mk_sp(start, start + BytePos(1)));
360                    if starts_with_number {
361                        let span = self.mk_sp(start, self.pos);
362                        self.dcx()
363                            .struct_err("lifetimes cannot start with a number")
364                            .with_span(span)
365                            .stash(span, StashKey::LifetimeIsChar);
366                    }
367                    token::Lifetime(lifetime_name, IdentKind::Normal)
368                }
369                rustc_lexer::TokenKind::RawLifetime => {
370                    self.last_lifetime = Some(self.mk_sp(start, start + BytePos(1)));
371
372                    let ident_start = start + BytePos(3);
373                    if self.mk_sp(start, ident_start).at_least_rust_2021() {
374                        // If the raw lifetime is followed by \' then treat it a normal
375                        // lifetime followed by a \', which is to interpret it as a character
376                        // literal. In this case, it's always an invalid character literal
377                        // since the literal must necessarily have >3 characters (r#...) inside
378                        // of it, which is invalid.
379                        if self.cursor.as_str().starts_with('\'') {
380                            let lit_span = self.mk_sp(start, self.pos + BytePos(1));
381                            let contents = self.str_from_to(start + BytePos(1), self.pos);
382                            emit_unescape_error(
383                                self.dcx(),
384                                contents,
385                                lit_span,
386                                lit_span,
387                                Mode::Char,
388                                0..contents.len(),
389                                EscapeError::MoreThanOneChar,
390                            )
391                            .expect("expected error");
392                        }
393
394                        let span = self.mk_sp(start, self.pos);
395
396                        let lifetime_name_without_tick =
397                            Symbol::intern(&self.str_from(ident_start));
398                        if !lifetime_name_without_tick.can_be_raw() {
399                            self.dcx().emit_err(
400                                crate::diagnostics::CannotBeRawLifetime {
401                                    span,
402                                    ident: lifetime_name_without_tick
403                                }
404                            );
405                        }
406
407                        // Put the `'` back onto the lifetime name.
408                        let mut lifetime_name =
409                            String::with_capacity(lifetime_name_without_tick.as_str().len() + 1);
410                        lifetime_name.push('\'');
411                        lifetime_name += lifetime_name_without_tick.as_str();
412                        let sym = nfc_normalize(&lifetime_name);
413
414                        // Make sure we mark this as a raw identifier.
415                        self.psess.raw_identifier_spans.push(span);
416
417                        token::Lifetime(sym, IdentKind::Raw)
418                    } else {
419                        // Reset the state so we just lex the `'r`.
420                        self.pos = start + BytePos(2);
421                        self.cursor = Cursor::new(&str_before[2 as usize..], FrontmatterAllowed::No);
422
423                        let prefix_span = self.mk_sp(start, self.pos);
424                        self.psess.buffer_lint(
425                            RUST_2021_PREFIXES_INCOMPATIBLE_SYNTAX,
426                            prefix_span,
427                            ast::CRATE_NODE_ID,
428                            crate::diagnostics::ReservedPrefixLint {
429                                subject: "`r`".into(),
430                                kind: "prefix",
431                                edition: Edition::Edition2021,
432                                sugg: prefix_span.shrink_to_hi(),
433                            }
434                        );
435
436                        let lifetime_name = nfc_normalize(self.str_from(start));
437                        token::Lifetime(lifetime_name, IdentKind::Normal)
438                    }
439                }
440                rustc_lexer::TokenKind::Semi => token::Semi,
441                rustc_lexer::TokenKind::Comma => token::Comma,
442                rustc_lexer::TokenKind::Dot => token::Dot,
443                rustc_lexer::TokenKind::OpenParen => token::OpenParen,
444                rustc_lexer::TokenKind::CloseParen => token::CloseParen,
445                rustc_lexer::TokenKind::OpenBrace => token::OpenBrace,
446                rustc_lexer::TokenKind::CloseBrace => token::CloseBrace,
447                rustc_lexer::TokenKind::OpenBracket => token::OpenBracket,
448                rustc_lexer::TokenKind::CloseBracket => token::CloseBracket,
449                rustc_lexer::TokenKind::At => token::At,
450                rustc_lexer::TokenKind::Pound => token::Pound,
451                rustc_lexer::TokenKind::Tilde => token::Tilde,
452                rustc_lexer::TokenKind::Question => token::Question,
453                rustc_lexer::TokenKind::Colon => token::Colon,
454                rustc_lexer::TokenKind::Dollar => token::Dollar,
455                rustc_lexer::TokenKind::Eq => token::Eq,
456                rustc_lexer::TokenKind::Bang => token::Bang,
457                rustc_lexer::TokenKind::Lt => token::Lt,
458                rustc_lexer::TokenKind::Gt => token::Gt,
459                rustc_lexer::TokenKind::Minus => token::Minus,
460                rustc_lexer::TokenKind::And => token::And,
461                rustc_lexer::TokenKind::Or => token::Or,
462                rustc_lexer::TokenKind::Plus => token::Plus,
463                rustc_lexer::TokenKind::Star => token::Star,
464                rustc_lexer::TokenKind::Slash => token::Slash,
465                rustc_lexer::TokenKind::Caret => token::Caret,
466                rustc_lexer::TokenKind::Percent => token::Percent,
467
468                rustc_lexer::TokenKind::Unknown | rustc_lexer::TokenKind::InvalidIdent => {
469                    // Don't emit diagnostics for sequences of the same invalid token
470                    if swallow_next_invalid > 0 {
471                        swallow_next_invalid -= 1;
472                        continue;
473                    }
474                    let mut it = self.str_from_to_end(start).chars();
475                    let c = it.next().unwrap();
476                    if c == '\u{00a0}' {
477                        // If an error has already been reported on non-breaking
478                        // space characters earlier in the file, treat all
479                        // subsequent occurrences as whitespace.
480                        if self.nbsp_is_whitespace {
481                            preceded_by_whitespace = true;
482                            continue;
483                        }
484                        self.nbsp_is_whitespace = true;
485                    }
486                    let repeats = it.take_while(|c1| *c1 == c).count();
487                    // FIXME: the lexer could be used to turn the ASCII version of unicode
488                    // homoglyphs, instead of keeping a table in `check_for_substitution`into the
489                    // token. Ideally, this should be inside `rustc_lexer`. However, we should
490                    // first remove compound tokens like `<<` from `rustc_lexer`, and then add
491                    // fancier error recovery to it, as there will be less overall work to do this
492                    // way.
493                    let (token, sugg) =
494                        unicode_chars::check_for_substitution(self, start, c, repeats + 1);
495                    self.dcx().emit_err(crate::diagnostics::UnknownTokenStart {
496                        span: self.mk_sp(start, self.pos + Pos::from_usize(repeats * c.len_utf8())),
497                        escaped: escaped_char(c),
498                        sugg,
499                        null: c == '\x00',
500                        invisible: INVISIBLE_CHARACTERS.contains(&c),
501                        repeat: if repeats > 0 {
502                            swallow_next_invalid = repeats;
503                            Some(crate::diagnostics::UnknownTokenRepeat { repeats })
504                        } else {
505                            None
506                        },
507                    });
508
509                    if let Some(token) = token {
510                        token
511                    } else {
512                        preceded_by_whitespace = true;
513                        continue;
514                    }
515                }
516                rustc_lexer::TokenKind::Eof => token::Eof,
517            };
518            let span = self.mk_sp(start, self.pos);
519            return (Token::new(kind, span), preceded_by_whitespace);
520        }
521    }
522
523    fn ident(&self, start: BytePos) -> TokenKind {
524        let sym = nfc_normalize(self.str_from(start));
525        let span = self.mk_sp(start, self.pos);
526        self.psess.symbol_gallery.insert(sym, span);
527        token::Ident(sym, IdentKind::Normal)
528    }
529
530    /// Detect usages of Unicode codepoints changing the direction of the text on screen and loudly
531    /// complain about it.
532    fn lint_unicode_text_flow(&self, start: BytePos) {
533        // Opening delimiter of the length 2 is not included into the comment text.
534        let content_start = start + BytePos(2);
535        let content = self.str_from(content_start);
536        if contains_text_flow_control_chars(content) {
537            let span = self.mk_sp(start, self.pos);
538            let content = content.to_string();
539            self.psess.dyn_buffer_lint(
540                TEXT_DIRECTION_CODEPOINT_IN_COMMENT,
541                span,
542                ast::CRATE_NODE_ID,
543                move |dcx, level| {
544                    let spans: Vec<_> = content
545                        .char_indices()
546                        .filter_map(|(i, c)| {
547                            TEXT_FLOW_CONTROL_CHARS.contains(&c).then(|| {
548                                let lo = span.lo() + BytePos(2 + i as u32);
549                                (c, span.with_lo(lo).with_hi(lo + BytePos(c.len_utf8() as u32)))
550                            })
551                        })
552                        .collect();
553                    let characters = spans
554                        .iter()
555                        .map(|&(c, span)| crate::diagnostics::UnicodeCharNoteSub {
556                            span,
557                            c_debug: ::alloc::__export::must_use({
        ::alloc::fmt::format(format_args!("{0:?}", c))
    })format!("{c:?}"),
558                        })
559                        .collect();
560                    let suggestions = (!spans.is_empty()).then_some(
561                        crate::diagnostics::UnicodeTextFlowSuggestion {
562                            spans: spans.iter().map(|(_c, span)| *span).collect(),
563                        },
564                    );
565
566                    crate::diagnostics::UnicodeTextFlow {
567                        comment_span: span,
568                        characters,
569                        suggestions,
570                        num_codepoints: spans.len(),
571                    }
572                    .into_diag(dcx, level)
573                },
574            );
575        }
576    }
577
578    fn lint_doc_comment_unicode_text_flow(&mut self, start: BytePos, content: &str) {
579        if contains_text_flow_control_chars(content) {
580            self.report_text_direction_codepoint(
581                content,
582                self.mk_sp(start, self.pos),
583                0,
584                false,
585                true,
586                "doc comment",
587            );
588        }
589    }
590
591    fn lint_literal_unicode_text_flow(
592        &mut self,
593        text: Symbol,
594        lit_kind: token::LitKind,
595        span: Span,
596        label: &'static str,
597    ) {
598        if !contains_text_flow_control_chars(text.as_str()) {
599            return;
600        }
601        let (padding, point_at_inner_spans) = match lit_kind {
602            // account for `"` or `'`
603            token::LitKind::Str | token::LitKind::Char => (1, true),
604            // account for `c"`
605            token::LitKind::CStr => (2, true),
606            // account for `r###"`
607            token::LitKind::StrRaw(n) => (n as u32 + 2, true),
608            // account for `cr###"`
609            token::LitKind::CStrRaw(n) => (n as u32 + 3, true),
610            // suppress bad literals.
611            token::LitKind::Err(_) => return,
612            // Be conservative just in case new literals do support these.
613            _ => (0, false),
614        };
615        self.report_text_direction_codepoint(
616            text.as_str(),
617            span,
618            padding,
619            point_at_inner_spans,
620            false,
621            label,
622        );
623    }
624
625    fn report_text_direction_codepoint(
626        &self,
627        text: &str,
628        span: Span,
629        padding: u32,
630        point_at_inner_spans: bool,
631        is_doc_comment: bool,
632        label: &str,
633    ) {
634        // Obtain the `Span`s for each of the forbidden chars.
635        let spans: Vec<_> = text
636            .char_indices()
637            .filter_map(|(i, c)| {
638                TEXT_FLOW_CONTROL_CHARS.contains(&c).then(|| {
639                    let lo = span.lo() + BytePos(i as u32 + padding);
640                    (c, span.with_lo(lo).with_hi(lo + BytePos(c.len_utf8() as u32)))
641                })
642            })
643            .collect();
644
645        let label = label.to_string();
646        let count = spans.len();
647        let labels =
648            point_at_inner_spans.then_some(crate::diagnostics::HiddenUnicodeCodepointsDiagLabels {
649                spans: spans.clone(),
650            });
651        let sub = if point_at_inner_spans && !spans.is_empty() {
652            crate::diagnostics::HiddenUnicodeCodepointsDiagSub::Escape { spans }
653        } else {
654            crate::diagnostics::HiddenUnicodeCodepointsDiagSub::NoEscape { spans, is_doc_comment }
655        };
656
657        self.psess.buffer_lint(
658            TEXT_DIRECTION_CODEPOINT_IN_LITERAL,
659            span,
660            ast::CRATE_NODE_ID,
661            crate::diagnostics::HiddenUnicodeCodepointsDiag {
662                label,
663                count,
664                span_label: span,
665                labels,
666                sub,
667            },
668        );
669    }
670
671    fn validate_frontmatter(
672        &self,
673        start: BytePos,
674        has_invalid_preceding_whitespace: bool,
675        invalid_infostring: bool,
676    ) {
677        let s = self.str_from(start);
678        let real_start = s.find("---").unwrap();
679        let frontmatter_opening_pos = BytePos(real_start as u32) + start;
680        let real_s = &s[real_start..];
681        let within = real_s.trim_start_matches('-');
682        let len_opening = real_s.len() - within.len();
683
684        let frontmatter_opening_end_pos = frontmatter_opening_pos + BytePos(len_opening as u32);
685        if has_invalid_preceding_whitespace {
686            let line_start =
687                BytePos(s[..real_start].rfind("\n").map_or(0, |i| i as u32 + 1)) + start;
688            let span = self.mk_sp(line_start, frontmatter_opening_end_pos);
689            let label_span = self.mk_sp(line_start, frontmatter_opening_pos);
690            self.dcx().emit_err(crate::diagnostics::FrontmatterInvalidOpeningPrecedingWhitespace {
691                span,
692                note_span: label_span,
693            });
694        }
695
696        let line_end = real_s.find('\n').unwrap_or(real_s.len());
697        if invalid_infostring {
698            let span = self.mk_sp(
699                frontmatter_opening_end_pos,
700                frontmatter_opening_pos + BytePos(line_end as u32),
701            );
702            self.dcx().emit_err(crate::diagnostics::FrontmatterInvalidInfostring { span });
703        }
704
705        let last_line_start = real_s.rfind('\n').map_or(line_end, |i| i + 1);
706
707        let content = &real_s[line_end..last_line_start];
708        if let Some(cr_offset) = content.find('\r') {
709            let cr_pos = start + BytePos((real_start + line_end + cr_offset) as u32);
710            let span = self.mk_sp(cr_pos, cr_pos + BytePos(1 as u32));
711            self.dcx().emit_err(crate::diagnostics::BareCrFrontmatter { span });
712        }
713
714        let last_line = &real_s[last_line_start..];
715        let last_line_trimmed = last_line.trim_start_matches(is_horizontal_whitespace);
716        let last_line_start_pos = frontmatter_opening_pos + BytePos(last_line_start as u32);
717
718        let frontmatter_span = self.mk_sp(frontmatter_opening_pos, self.pos);
719        self.psess.gated_spans.gate(sym::frontmatter, frontmatter_span);
720
721        if !last_line_trimmed.starts_with("---") {
722            let label_span = self.mk_sp(frontmatter_opening_pos, frontmatter_opening_end_pos);
723            self.dcx().emit_err(crate::diagnostics::FrontmatterUnclosed {
724                span: frontmatter_span,
725                note_span: label_span,
726            });
727            return;
728        }
729
730        if last_line_trimmed.len() != last_line.len() {
731            let line_end = last_line_start_pos + BytePos(last_line.len() as u32);
732            let span = self.mk_sp(last_line_start_pos, line_end);
733            let whitespace_end =
734                last_line_start_pos + BytePos((last_line.len() - last_line_trimmed.len()) as u32);
735            let label_span = self.mk_sp(last_line_start_pos, whitespace_end);
736            self.dcx().emit_err(crate::diagnostics::FrontmatterInvalidClosingPrecedingWhitespace {
737                span,
738                note_span: label_span,
739            });
740        }
741
742        let rest = last_line_trimmed.trim_start_matches('-');
743        let len_close = last_line_trimmed.len() - rest.len();
744        if len_close != len_opening {
745            let span = self.mk_sp(frontmatter_opening_pos, self.pos);
746            let opening = self.mk_sp(frontmatter_opening_pos, frontmatter_opening_end_pos);
747            let last_line_close_pos = last_line_start_pos + BytePos(len_close as u32);
748            let close = self.mk_sp(last_line_start_pos, last_line_close_pos);
749            self.dcx().emit_err(crate::diagnostics::FrontmatterLengthMismatch {
750                span,
751                opening,
752                close,
753                len_opening,
754                len_close,
755            });
756        }
757
758        // Only up to 255 `-`s are allowed in code fences
759        if u8::try_from(len_opening).is_err() {
760            self.dcx().emit_err(crate::diagnostics::FrontmatterTooManyDashes { len_opening });
761        }
762
763        if !rest.trim_matches(is_horizontal_whitespace).is_empty() {
764            let span = self.mk_sp(last_line_start_pos, self.pos);
765            self.dcx().emit_err(crate::diagnostics::FrontmatterExtraCharactersAfterClose { span });
766        }
767    }
768
769    fn cook_doc_comment(
770        &self,
771        content_start: BytePos,
772        content: &str,
773        comment_kind: CommentKind,
774        doc_style: DocStyle,
775    ) -> TokenKind {
776        if content.contains('\r') {
777            for (idx, _) in content.char_indices().filter(|&(_, c)| c == '\r') {
778                let span = self.mk_sp(
779                    content_start + BytePos(idx as u32),
780                    content_start + BytePos(idx as u32 + 1),
781                );
782                let block = #[allow(non_exhaustive_omitted_patterns)] match comment_kind {
    CommentKind::Block => true,
    _ => false,
}matches!(comment_kind, CommentKind::Block);
783                self.dcx().emit_err(crate::diagnostics::CrDocComment { span, block });
784            }
785        }
786
787        let attr_style = match doc_style {
788            DocStyle::Outer => AttrStyle::Outer,
789            DocStyle::Inner => AttrStyle::Inner,
790        };
791
792        token::DocComment(comment_kind, attr_style, Symbol::intern(content))
793    }
794
795    fn cook_lexer_literal(
796        &self,
797        start: BytePos,
798        end: BytePos,
799        kind: rustc_lexer::LiteralKind,
800    ) -> (token::LitKind, Symbol) {
801        match kind {
802            rustc_lexer::LiteralKind::Char { terminated } => {
803                if !terminated {
804                    let mut err = self
805                        .dcx()
806                        .struct_span_fatal(self.mk_sp(start, end), "unterminated character literal")
807                        .with_code(E0762);
808                    if let Some(lt_sp) = self.last_lifetime {
809                        err.multipart_suggestion(
810                            "if you meant to write a string literal, use double quotes",
811                            ::alloc::boxed::box_assume_init_into_vec_unsafe(::alloc::intrinsics::write_box_via_move(::alloc::boxed::Box::new_uninit(),
        [(lt_sp, "\"".to_string()),
                (self.mk_sp(start, start + BytePos(1)), "\"".to_string())]))vec![
812                                (lt_sp, "\"".to_string()),
813                                (self.mk_sp(start, start + BytePos(1)), "\"".to_string()),
814                            ],
815                            Applicability::MaybeIncorrect,
816                        );
817                    }
818                    err.emit()
819                }
820                self.cook_quoted(token::Char, Mode::Char, start, end, 1, 1) // ' '
821            }
822            rustc_lexer::LiteralKind::Byte { terminated } => {
823                if !terminated {
824                    self.dcx()
825                        .struct_span_fatal(
826                            self.mk_sp(start + BytePos(1), end),
827                            "unterminated byte constant",
828                        )
829                        .with_code(E0763)
830                        .emit()
831                }
832                self.cook_quoted(token::Byte, Mode::Byte, start, end, 2, 1) // b' '
833            }
834            rustc_lexer::LiteralKind::Str { terminated } => {
835                if !terminated {
836                    self.dcx()
837                        .struct_span_fatal(
838                            self.mk_sp(start, end),
839                            "unterminated double quote string",
840                        )
841                        .with_code(E0765)
842                        .emit()
843                }
844                self.cook_quoted(token::Str, Mode::Str, start, end, 1, 1) // " "
845            }
846            rustc_lexer::LiteralKind::ByteStr { terminated } => {
847                if !terminated {
848                    self.dcx()
849                        .struct_span_fatal(
850                            self.mk_sp(start + BytePos(1), end),
851                            "unterminated double quote byte string",
852                        )
853                        .with_code(E0766)
854                        .emit()
855                }
856                self.cook_quoted(token::ByteStr, Mode::ByteStr, start, end, 2, 1)
857                // b" "
858            }
859            rustc_lexer::LiteralKind::CStr { terminated } => {
860                if !terminated {
861                    self.dcx()
862                        .struct_span_fatal(
863                            self.mk_sp(start + BytePos(1), end),
864                            "unterminated C string",
865                        )
866                        .with_code(E0767)
867                        .emit()
868                }
869                self.cook_quoted(token::CStr, Mode::CStr, start, end, 2, 1) // c" "
870            }
871            rustc_lexer::LiteralKind::RawStr { n_hashes } => {
872                if let Some(n_hashes) = n_hashes {
873                    let n = u32::from(n_hashes);
874                    let kind = token::StrRaw(n_hashes);
875                    self.cook_quoted(kind, Mode::RawStr, start, end, 2 + n, 1 + n)
876                // r##" "##
877                } else {
878                    self.report_raw_str_error(start, 1);
879                }
880            }
881            rustc_lexer::LiteralKind::RawByteStr { n_hashes } => {
882                if let Some(n_hashes) = n_hashes {
883                    let n = u32::from(n_hashes);
884                    let kind = token::ByteStrRaw(n_hashes);
885                    self.cook_quoted(kind, Mode::RawByteStr, start, end, 3 + n, 1 + n)
886                // br##" "##
887                } else {
888                    self.report_raw_str_error(start, 2);
889                }
890            }
891            rustc_lexer::LiteralKind::RawCStr { n_hashes } => {
892                if let Some(n_hashes) = n_hashes {
893                    let n = u32::from(n_hashes);
894                    let kind = token::CStrRaw(n_hashes);
895                    self.cook_quoted(kind, Mode::RawCStr, start, end, 3 + n, 1 + n)
896                // cr##" "##
897                } else {
898                    self.report_raw_str_error(start, 2);
899                }
900            }
901            rustc_lexer::LiteralKind::Int { base, empty_int } => {
902                let mut kind = token::Integer;
903                if empty_int {
904                    let span = self.mk_sp(start, end);
905                    let guar = self.dcx().emit_err(crate::diagnostics::NoDigitsLiteral { span });
906                    kind = token::Err(guar);
907                } else if #[allow(non_exhaustive_omitted_patterns)] match base {
    Base::Binary | Base::Octal => true,
    _ => false,
}matches!(base, Base::Binary | Base::Octal) {
908                    let base = base as u32;
909                    let s = self.str_from_to(start + BytePos(2), end);
910                    for (idx, c) in s.char_indices() {
911                        let span = self.mk_sp(
912                            start + BytePos::from_usize(2 + idx),
913                            start + BytePos::from_usize(2 + idx + c.len_utf8()),
914                        );
915                        if c != '_' && c.to_digit(base).is_none() {
916                            let guar = self
917                                .dcx()
918                                .emit_err(crate::diagnostics::InvalidDigitLiteral { span, base });
919                            kind = token::Err(guar);
920                        }
921                    }
922                }
923                (kind, self.symbol_from_to(start, end))
924            }
925            rustc_lexer::LiteralKind::Float { base, empty_exponent } => {
926                let mut kind = token::Float;
927                if empty_exponent {
928                    let span = self.mk_sp(start, self.pos);
929                    let guar = self.dcx().emit_err(crate::diagnostics::EmptyExponentFloat { span });
930                    kind = token::Err(guar);
931                }
932                let base = match base {
933                    Base::Hexadecimal => Some("hexadecimal"),
934                    Base::Octal => Some("octal"),
935                    Base::Binary => Some("binary"),
936                    _ => None,
937                };
938                if let Some(base) = base {
939                    let span = self.mk_sp(start, end);
940                    let guar = self
941                        .dcx()
942                        .emit_err(crate::diagnostics::FloatLiteralUnsupportedBase { span, base });
943                    kind = token::Err(guar)
944                }
945                (kind, self.symbol_from_to(start, end))
946            }
947        }
948    }
949
950    #[inline]
951    fn src_index(&self, pos: BytePos) -> usize {
952        (pos - self.start_pos).to_usize()
953    }
954
955    /// Slice of the source text from `start` up to but excluding `self.pos`,
956    /// meaning the slice does not include the character `self.ch`.
957    fn str_from(&self, start: BytePos) -> &'src str {
958        self.str_from_to(start, self.pos)
959    }
960
961    /// As symbol_from, with an explicit endpoint.
962    fn symbol_from_to(&self, start: BytePos, end: BytePos) -> Symbol {
963        {
    use ::tracing::__macro_support::Callsite as _;
    static __CALLSITE: ::tracing::callsite::DefaultCallsite =
        {
            static META: ::tracing::Metadata<'static> =
                {
                    ::tracing_core::metadata::Metadata::new("event /rustc-dev/8d1a76430406c877b35d0b627e7f796dcf0dfeca/compiler/rustc_parse/src/lexer/mod.rs:963",
                        "rustc_parse::lexer", ::tracing::Level::DEBUG,
                        ::tracing_core::__macro_support::Option::Some("/rustc-dev/8d1a76430406c877b35d0b627e7f796dcf0dfeca/compiler/rustc_parse/src/lexer/mod.rs"),
                        ::tracing_core::__macro_support::Option::Some(963u32),
                        ::tracing_core::__macro_support::Option::Some("rustc_parse::lexer"),
                        ::tracing_core::field::FieldSet::new(&["message"],
                            ::tracing_core::callsite::Identifier(&__CALLSITE)),
                        ::tracing::metadata::Kind::EVENT)
                };
            ::tracing::callsite::DefaultCallsite::new(&META)
        };
    let enabled =
        ::tracing::Level::DEBUG <= ::tracing::level_filters::STATIC_MAX_LEVEL
                &&
                ::tracing::Level::DEBUG <=
                    ::tracing::level_filters::LevelFilter::current() &&
            {
                let interest = __CALLSITE.interest();
                !interest.is_never() &&
                    ::tracing::__macro_support::__is_enabled(__CALLSITE.metadata(),
                        interest)
            };
    if enabled {
        (|value_set: ::tracing::field::ValueSet|
                    {
                        let meta = __CALLSITE.metadata();
                        ::tracing::Event::dispatch(meta, &value_set);
                        ;
                    })({
                #[allow(unused_imports)]
                use ::tracing::field::{debug, display, Value};
                __CALLSITE.metadata().fields().value_set_all(&[(::tracing::__macro_support::Option::Some(&format_args!("taking an ident from {0:?} to {1:?}",
                                                    start, end) as &dyn ::tracing::field::Value))])
            });
    } else { ; }
};debug!("taking an ident from {:?} to {:?}", start, end);
964        Symbol::intern(self.str_from_to(start, end))
965    }
966
967    /// Slice of the source text spanning from `start` up to but excluding `end`.
968    fn str_from_to(&self, start: BytePos, end: BytePos) -> &'src str {
969        &self.src[self.src_index(start)..self.src_index(end)]
970    }
971
972    /// Slice of the source text spanning from `start` until the end
973    fn str_from_to_end(&self, start: BytePos) -> &'src str {
974        &self.src[self.src_index(start)..]
975    }
976
977    fn report_raw_str_error(&self, start: BytePos, prefix_len: u32) -> ! {
978        match rustc_lexer::validate_raw_str(self.str_from(start), prefix_len) {
979            Err(RawStrError::InvalidStarter { bad_char }) => {
980                self.report_non_started_raw_string(start, bad_char)
981            }
982            Err(RawStrError::NoTerminator { expected, found, possible_terminator_offset }) => self
983                .report_unterminated_raw_string(start, expected, possible_terminator_offset, found),
984            Err(RawStrError::TooManyDelimiters { found }) => {
985                self.report_too_many_hashes(start, found)
986            }
987            Ok(()) => {
    ::core::panicking::panic_fmt(format_args!("no error found for supposedly invalid raw string literal"));
}panic!("no error found for supposedly invalid raw string literal"),
988        }
989    }
990
991    fn report_non_started_raw_string(&self, start: BytePos, bad_char: char) -> ! {
992        self.dcx()
993            .struct_span_fatal(
994                self.mk_sp(start, self.pos),
995                ::alloc::__export::must_use({
        ::alloc::fmt::format(format_args!("found invalid character; only `#` is allowed in raw string delimitation: {0}",
                escaped_char(bad_char)))
    })format!(
996                    "found invalid character; only `#` is allowed in raw string delimitation: {}",
997                    escaped_char(bad_char)
998                ),
999            )
1000            .emit_fatal()
1001    }
1002
1003    fn report_unterminated_raw_string(
1004        &self,
1005        start: BytePos,
1006        n_hashes: u32,
1007        possible_offset: Option<u32>,
1008        found_terminators: u32,
1009    ) -> ! {
1010        let mut err =
1011            self.dcx().struct_span_fatal(self.mk_sp(start, start), "unterminated raw string");
1012        err.code(E0748);
1013        err.span_label(self.mk_sp(start, start), "unterminated raw string");
1014
1015        if n_hashes > 0 {
1016            err.note(::alloc::__export::must_use({
        ::alloc::fmt::format(format_args!("this raw string should be terminated with `\"{0}`",
                "#".repeat(n_hashes as usize)))
    })format!(
1017                "this raw string should be terminated with `\"{}`",
1018                "#".repeat(n_hashes as usize)
1019            ));
1020        }
1021
1022        if let Some(possible_offset) = possible_offset {
1023            let lo = start + BytePos(possible_offset);
1024            let hi = lo + BytePos(found_terminators);
1025            let span = self.mk_sp(lo, hi);
1026            err.span_suggestion_verbose(
1027                span,
1028                "consider terminating the string here",
1029                "#".repeat(n_hashes as usize),
1030                Applicability::MaybeIncorrect,
1031            );
1032        }
1033
1034        err.emit_fatal()
1035    }
1036
1037    fn report_unterminated_block_comment(&self, start: BytePos, doc_style: Option<DocStyle>) -> ! {
1038        let msg = match doc_style {
1039            Some(_) => "unterminated block doc-comment",
1040            None => "unterminated block comment",
1041        };
1042        let last_bpos = self.pos;
1043        let mut err = self.dcx().struct_span_fatal(self.mk_sp(start, last_bpos), msg);
1044        err.code(E0758);
1045        let mut nested_block_comment_open_idxs = ::alloc::vec::Vec::new()vec![];
1046        let mut last_nested_block_comment_idxs = None;
1047        let mut content_chars = self.str_from(start).char_indices().peekable();
1048
1049        while let Some((idx, current_char)) = content_chars.next() {
1050            match content_chars.peek() {
1051                Some((_, '*')) if current_char == '/' => {
1052                    nested_block_comment_open_idxs.push(idx);
1053                }
1054                Some((_, '/')) if current_char == '*' => {
1055                    last_nested_block_comment_idxs =
1056                        nested_block_comment_open_idxs.pop().map(|open_idx| (open_idx, idx));
1057                }
1058                _ => {}
1059            };
1060        }
1061
1062        if let Some((nested_open_idx, nested_close_idx)) = last_nested_block_comment_idxs {
1063            err.span_label(self.mk_sp(start, start + BytePos(2)), msg)
1064                .span_label(
1065                    self.mk_sp(
1066                        start + BytePos(nested_open_idx as u32),
1067                        start + BytePos(nested_open_idx as u32 + 2),
1068                    ),
1069                    "...as last nested comment starts here, maybe you want to close this instead?",
1070                )
1071                .span_label(
1072                    self.mk_sp(
1073                        start + BytePos(nested_close_idx as u32),
1074                        start + BytePos(nested_close_idx as u32 + 2),
1075                    ),
1076                    "...and last nested comment terminates here.",
1077                );
1078        }
1079
1080        err.emit_fatal();
1081    }
1082
1083    // RFC 3101 introduced the idea of (reserved) prefixes. As of Rust 2021,
1084    // using a (unknown) prefix is an error. In earlier editions, however, they
1085    // only result in a (allowed by default) lint, and are treated as regular
1086    // identifier tokens.
1087    fn report_unknown_prefix(&self, start: BytePos) {
1088        let prefix_span = self.mk_sp(start, self.pos);
1089        let prefix = self.str_from_to(start, self.pos);
1090        let expn_data = prefix_span.ctxt().outer_expn_data();
1091
1092        if expn_data.edition.at_least_rust_2021() {
1093            // In Rust 2021, this is a hard error.
1094            let sugg = if prefix == "rb" {
1095                Some(crate::diagnostics::UnknownPrefixSugg::UseBr(prefix_span))
1096            } else if prefix == "rc" {
1097                Some(crate::diagnostics::UnknownPrefixSugg::UseCr(prefix_span))
1098            } else if expn_data.is_root() {
1099                if self.cursor.first() == '\''
1100                    && let Some(start) = self.last_lifetime
1101                    && self.cursor.third() != '\''
1102                    && let end = self.mk_sp(self.pos, self.pos + BytePos(1))
1103                    && !self.psess.source_map().is_multiline(start.until(end))
1104                {
1105                    // FIXME: An "unclosed `char`" error will be emitted already in some cases,
1106                    // but it's hard to silence this error while not also silencing important cases
1107                    // too. We should use the error stashing machinery instead.
1108                    Some(crate::diagnostics::UnknownPrefixSugg::MeantStr { start, end })
1109                } else {
1110                    Some(crate::diagnostics::UnknownPrefixSugg::Whitespace(
1111                        prefix_span.shrink_to_hi(),
1112                    ))
1113                }
1114            } else {
1115                None
1116            };
1117            self.dcx().emit_err(crate::diagnostics::UnknownPrefix {
1118                span: prefix_span,
1119                prefix,
1120                sugg,
1121            });
1122        } else {
1123            // Before Rust 2021, only emit a lint for migration.
1124            self.psess.buffer_lint(
1125                RUST_2021_PREFIXES_INCOMPATIBLE_SYNTAX,
1126                prefix_span,
1127                ast::CRATE_NODE_ID,
1128                crate::diagnostics::ReservedPrefixLint {
1129                    subject: ::alloc::__export::must_use({
        ::alloc::fmt::format(format_args!("`{0}`", prefix))
    })format!("`{prefix}`"),
1130                    kind: "prefix",
1131                    edition: Edition::Edition2021,
1132                    sugg: prefix_span.shrink_to_hi(),
1133                },
1134            );
1135        }
1136    }
1137
1138    /// Detect guarded string literal syntax
1139    ///
1140    /// RFC 3593 reserved this syntax for future use. As of Rust 2024,
1141    /// using this syntax produces an error. In earlier editions, however, it
1142    /// only results in an (allowed by default) lint, and is treated as
1143    /// separate tokens.
1144    fn maybe_report_guarded_str(&mut self, start: BytePos, str_before: &'src str) -> TokenKind {
1145        let span = self.mk_sp(start, self.pos);
1146        let edition2024 = span.edition().at_least_rust_2024();
1147
1148        let space_pos = start + BytePos(1);
1149        let space_span = self.mk_sp(space_pos, space_pos);
1150
1151        let mut cursor = Cursor::new(str_before, FrontmatterAllowed::No);
1152
1153        let (is_string, span, unterminated) = match cursor.guarded_double_quoted_string() {
1154            Some(rustc_lexer::GuardedStr { n_hashes, terminated, token_len }) => {
1155                let end = start + BytePos(token_len);
1156                let span = self.mk_sp(start, end);
1157                let str_start = start + BytePos(n_hashes);
1158
1159                if edition2024 {
1160                    self.cursor = cursor;
1161                    self.pos = end;
1162                }
1163
1164                let unterminated = if terminated { None } else { Some(str_start) };
1165
1166                (true, span, unterminated)
1167            }
1168            None => {
1169                // We should only get here in the `##+` case.
1170                if true {
    {
        match (&self.str_from_to(start, start + BytePos(2)), &"##") {
            (left_val, right_val) => {
                if !(*left_val == *right_val) {
                    let kind = ::core::panicking::AssertKind::Eq;
                    ::core::panicking::assert_failed(kind, &*left_val,
                        &*right_val, ::core::option::Option::None);
                }
            }
        }
    };
};debug_assert_eq!(self.str_from_to(start, start + BytePos(2)), "##");
1171
1172                (false, span, None)
1173            }
1174        };
1175        if edition2024 {
1176            if let Some(str_start) = unterminated {
1177                // Only a fatal error if string is unterminated.
1178                self.dcx()
1179                    .struct_span_fatal(
1180                        self.mk_sp(str_start, self.pos),
1181                        "unterminated double quote string",
1182                    )
1183                    .with_code(E0765)
1184                    .emit()
1185            }
1186
1187            let sugg = if span.from_expansion() {
1188                None
1189            } else {
1190                Some(crate::diagnostics::GuardedStringSugg(space_span))
1191            };
1192
1193            // In Edition 2024 and later, emit a hard error.
1194            let err = if is_string {
1195                self.dcx().emit_err(crate::diagnostics::ReservedString { span, sugg })
1196            } else {
1197                self.dcx().emit_err(crate::diagnostics::ReservedMultihash { span, sugg })
1198            };
1199
1200            token::Literal(token::Lit {
1201                kind: token::Err(err),
1202                symbol: self.symbol_from_to(start, self.pos),
1203                suffix: None,
1204            })
1205        } else {
1206            // Before Rust 2024, only emit a lint for migration.
1207            self.psess.buffer_lint(
1208                RUST_2024_GUARDED_STRING_INCOMPATIBLE_SYNTAX,
1209                span,
1210                ast::CRATE_NODE_ID,
1211                crate::diagnostics::ReservedPrefixLint {
1212                    subject: "this".into(),
1213                    kind: if is_string { "guarded string literal" } else { "reserved token" },
1214                    edition: Edition::Edition2024,
1215                    sugg: space_span,
1216                },
1217            );
1218
1219            // For backwards compatibility, roll back to after just the first `#`
1220            // and return the `Pound` token.
1221            self.pos = start + BytePos(1);
1222            self.cursor = Cursor::new(&str_before[1..], FrontmatterAllowed::No);
1223            token::Pound
1224        }
1225    }
1226
1227    fn report_too_many_hashes(&self, start: BytePos, num: u32) -> ! {
1228        self.dcx().emit_fatal(crate::diagnostics::TooManyHashes {
1229            span: self.mk_sp(start, self.pos),
1230            num,
1231        });
1232    }
1233
1234    fn cook_quoted(
1235        &self,
1236        mut kind: token::LitKind,
1237        mode: Mode,
1238        start: BytePos,
1239        end: BytePos,
1240        prefix_len: u32,
1241        postfix_len: u32,
1242    ) -> (token::LitKind, Symbol) {
1243        let content_start = start + BytePos(prefix_len);
1244        let content_end = end - BytePos(postfix_len);
1245        let lit_content = self.str_from_to(content_start, content_end);
1246        check_for_errors(lit_content, mode, |range, err| {
1247            let span_with_quotes = self.mk_sp(start, end);
1248            let (start, end) = (range.start as u32, range.end as u32);
1249            let lo = content_start + BytePos(start);
1250            let hi = lo + BytePos(end - start);
1251            let span = self.mk_sp(lo, hi);
1252            let is_fatal = err.is_fatal();
1253            if let Some(guar) = emit_unescape_error(
1254                self.dcx(),
1255                lit_content,
1256                span_with_quotes,
1257                span,
1258                mode,
1259                range,
1260                err,
1261            ) {
1262                if !is_fatal { ::core::panicking::panic("assertion failed: is_fatal") };assert!(is_fatal);
1263                kind = token::Err(guar);
1264            }
1265        });
1266
1267        // We normally exclude the quotes for the symbol, but for errors we
1268        // include it because it results in clearer error messages.
1269        let sym = if !#[allow(non_exhaustive_omitted_patterns)] match kind {
    token::Err(_) => true,
    _ => false,
}matches!(kind, token::Err(_)) {
1270            Symbol::intern(lit_content)
1271        } else {
1272            self.symbol_from_to(start, end)
1273        };
1274        (kind, sym)
1275    }
1276}
1277
1278pub fn nfc_normalize(string: &str) -> Symbol {
1279    use unicode_normalization::{IsNormalized, UnicodeNormalization, is_nfc_quick};
1280    match is_nfc_quick(string.chars()) {
1281        IsNormalized::Yes => Symbol::intern(string),
1282        _ => {
1283            let normalized_str: String = string.chars().nfc().collect();
1284            Symbol::intern(&normalized_str)
1285        }
1286    }
1287}