From 8535baa02181efad7020dd5b9d09e8f5f805eeeb Mon Sep 17 00:00:00 2001 From: A4-Tacks Date: Thu, 13 Aug 2026 23:01:34 +0800 Subject: [PATCH 01/18] fix: don't error on tail comma for some macro Example --- ```rust const _: &str = env!("PATH",); ``` **Before this PR** ``` /* expand error: expected string literal */ ``` **After this PR** ```rust const _: &str = "/usr/bin:/bin"; ``` --- .../macro_expansion_tests/builtin_fn_macro.rs | 14 +++++++++-- .../crates/hir-expand/src/builtin/fn_macro.rs | 23 +++++++++++++++++-- 2 files changed, 33 insertions(+), 4 deletions(-) diff --git a/src/tools/rust-analyzer/crates/hir-def/src/macro_expansion_tests/builtin_fn_macro.rs b/src/tools/rust-analyzer/crates/hir-def/src/macro_expansion_tests/builtin_fn_macro.rs index 46cdb39c5b46b..d6ccf9ca51ab0 100644 --- a/src/tools/rust-analyzer/crates/hir-def/src/macro_expansion_tests/builtin_fn_macro.rs +++ b/src/tools/rust-analyzer/crates/hir-def/src/macro_expansion_tests/builtin_fn_macro.rs @@ -132,13 +132,23 @@ fn test_env_expand() { #[rustc_builtin_macro] macro_rules! env {() => {}} -fn main() { env!("TEST_ENV_VAR"); } +fn main() { + env!("TEST_ENV_VAR"); + env!("TEST_ENV_VAR",); + env!("TEST_ENV_VAR", "error"); + env!("TEST_ENV_VAR", "error",); +} "#, expect![[r##" #[rustc_builtin_macro] macro_rules! env {() => {}} -fn main() { "UNRESOLVED_ENV_VAR"; } +fn main() { + "UNRESOLVED_ENV_VAR"; + "UNRESOLVED_ENV_VAR"; + "UNRESOLVED_ENV_VAR"; + "UNRESOLVED_ENV_VAR"; +} "##]], ); } diff --git a/src/tools/rust-analyzer/crates/hir-expand/src/builtin/fn_macro.rs b/src/tools/rust-analyzer/crates/hir-expand/src/builtin/fn_macro.rs index b173f44f34ab1..7a162fcf4bcb7 100644 --- a/src/tools/rust-analyzer/crates/hir-expand/src/builtin/fn_macro.rs +++ b/src/tools/rust-analyzer/crates/hir-expand/src/builtin/fn_macro.rs @@ -767,7 +767,26 @@ fn relative_file( } fn parse_string(tt: &tt::TopSubtree) -> Result<(Symbol, Span), ExpandError> { - let mut tt = TtElement::Subtree(tt.top_subtree(), tt.iter()); + let expect_literal = |span| ExpandError::other(span, "expected string literal"); + let mut tt = { + let mut tt_iter = tt.iter(); + let extracted = + tt_iter.next().ok_or_else(|| expect_literal(tt.top_subtree().delimiter.close))?; + + match tt_iter.next() { + None => {} + Some(TtElement::Leaf(tt::Leaf::Punct(it))) if it.char == ',' => { + // Tail comma + // FIXME: Ignored like env!("NAME", "compile_error message") + } + Some(tt) => { + return Err(ExpandError::other(tt.first_span(), "unexpected input")); + } + } + + extracted + }; + (|| { // FIXME: We wrap expression fragments in parentheses which can break this expectation // here @@ -795,7 +814,7 @@ fn parse_string(tt: &tt::TopSubtree) -> Result<(Symbol, Span), ExpandError> { TtElement::Subtree(tt, _) => Err(tt.delimiter.open.cover(tt.delimiter.close)), } })() - .map_err(|span| ExpandError::other(span, "expected string literal")) + .map_err(expect_literal) } fn include_expand( From 84e7c4c29bbebe06b4563e62a9eee3624dafcc57 Mon Sep 17 00:00:00 2001 From: Chayim Refael Friedman Date: Sun, 9 Aug 2026 04:35:47 +0300 Subject: [PATCH 02/18] Optimize the heck out of the storage of token trees This is basically the most optimized (memory-wise) storage possible, found after multiple measurements. The price we pay for this ultra-extra optimization is a bunch of unsafe, encapsulated in `tt/src/storage.rs`. Since macros and therefore token trees are so common in Rust code, I think this is worth it. Some stats: - On rust-analyzer itself, memory usage is reduced by 30mb. rust-analyzer doesn't use macros a lot and the previous optimization already took the most, but when considering that *all* token trees in r-a now consumes only about 40mb, this is still surprising. - On buck2, ~143mb is saved. - On omicron, ~436mb is saved, and this is after the previous optimization already ripped 880mb! It is only using ~180mb for token trees now, in total! The basic idea is to use a variable-length encoding into a bytes array. Multiple measurements were done in order to determine the most common forms of token trees along with their frequencies, and to find the best encoding. In addition, we also now sort the compressed spans by their frequencies (in a descending order), so that even if a `TopSubtree` has more than 2^4 unique compressed spans, we will still use the more efficient encoding for the biggest number of spans possible. This is made possible by the fact that unlike the previous encoding, now we don't force one span encoding for all tokens (or in fact even for the two spans in one subtree). --- src/tools/rust-analyzer/Cargo.lock | 1 - src/tools/rust-analyzer/crates/cfg/Cargo.toml | 5 +- .../crates/hir-expand/src/fixup.rs | 8 +- .../rust-analyzer/crates/intern/src/symbol.rs | 29 +- src/tools/rust-analyzer/crates/mbe/src/lib.rs | 2 +- src/tools/rust-analyzer/crates/tt/Cargo.toml | 1 - .../rust-analyzer/crates/tt/src/buffer.rs | 99 +- src/tools/rust-analyzer/crates/tt/src/iter.rs | 88 +- src/tools/rust-analyzer/crates/tt/src/lib.rs | 243 +-- .../rust-analyzer/crates/tt/src/storage.rs | 1875 ++++++++++------- 10 files changed, 1351 insertions(+), 1000 deletions(-) diff --git a/src/tools/rust-analyzer/Cargo.lock b/src/tools/rust-analyzer/Cargo.lock index 19bdd0c7635a5..f1ea05a5ec366 100644 --- a/src/tools/rust-analyzer/Cargo.lock +++ b/src/tools/rust-analyzer/Cargo.lock @@ -3096,7 +3096,6 @@ name = "tt" version = "0.0.0" dependencies = [ "arrayvec", - "indexmap", "intern", "ra-ap-rustc_lexer", "rustc-hash 2.1.2", diff --git a/src/tools/rust-analyzer/crates/cfg/Cargo.toml b/src/tools/rust-analyzer/crates/cfg/Cargo.toml index 15de1f329385d..7759fc4ae1948 100644 --- a/src/tools/rust-analyzer/crates/cfg/Cargo.toml +++ b/src/tools/rust-analyzer/crates/cfg/Cargo.toml @@ -28,10 +28,9 @@ arbitrary = { version = "1.4.1", features = ["derive"] } # local deps syntax-bridge.workspace = true -syntax.workspace = true -# tt is needed for testing -cfg = { path = ".", default-features = false, features = ["tt"] } +# tt and syntax are needed for testing +cfg = { path = ".", default-features = false, features = ["tt", "syntax"] } [features] default = [] diff --git a/src/tools/rust-analyzer/crates/hir-expand/src/fixup.rs b/src/tools/rust-analyzer/crates/hir-expand/src/fixup.rs index 939104b709163..8781358822064 100644 --- a/src/tools/rust-analyzer/crates/hir-expand/src/fixup.rs +++ b/src/tools/rust-analyzer/crates/hir-expand/src/fixup.rs @@ -419,9 +419,11 @@ mod tests { } fn check_subtree_eq(a: &tt::TopSubtree, b: &tt::TopSubtree) -> bool { - let a = a.view().as_token_trees().iter_flat_tokens(); - let b = b.view().as_token_trees().iter_flat_tokens(); - a.len() == b.len() && std::iter::zip(a, b).all(|(a, b)| check_tt_eq(&a, &b)) + let a = a.view().as_token_trees(); + let b = b.view().as_token_trees(); + a.len() == b.len() + && std::iter::zip(a.iter_flat_tokens(), b.iter_flat_tokens()) + .all(|(a, b)| check_tt_eq(&a, &b)) } fn check_tt_eq(a: &tt::TokenTree, b: &tt::TokenTree) -> bool { diff --git a/src/tools/rust-analyzer/crates/intern/src/symbol.rs b/src/tools/rust-analyzer/crates/intern/src/symbol.rs index 72d32d1017747..cf41db85a163d 100644 --- a/src/tools/rust-analyzer/crates/intern/src/symbol.rs +++ b/src/tools/rust-analyzer/crates/intern/src/symbol.rs @@ -174,6 +174,19 @@ impl Symbol { self.repr.as_str() } + #[inline] + pub fn into_raw(self) -> NonNull<*const str> { + ManuallyDrop::new(self).repr.packed + } + + /// # Safety + /// + /// The pointer must have come from [`Symbol::into_raw()`]. + #[inline] + pub unsafe fn from_raw(ptr: NonNull<*const str>) -> Symbol { + Symbol { repr: TaggedArcPtr { packed: ptr } } + } + #[inline] fn select_shard( storage: &'static Map, @@ -217,11 +230,12 @@ impl Symbol { shard.shrink_to(len, |(x, _)| Self::hash(storage, x.as_str())); } } -} -impl Drop for Symbol { + /// # Safety + /// + /// You must know that you have a `Symbol` instance that won't be dropped, so decreasing the refcount is valid. #[inline] - fn drop(&mut self) { + pub unsafe fn decrease_refcount(&mut self) { // SAFETY: We're dropping, we have ownership. let Some(arc) = (unsafe { self.repr.try_as_arc_owned() }) else { return; @@ -237,6 +251,15 @@ impl Drop for Symbol { } } +impl Drop for Symbol { + #[inline] + fn drop(&mut self) { + unsafe { + self.decrease_refcount(); + } + } +} + impl Clone for Symbol { fn clone(&self) -> Self { Self { repr: increase_arc_refcount(self.repr) } diff --git a/src/tools/rust-analyzer/crates/mbe/src/lib.rs b/src/tools/rust-analyzer/crates/mbe/src/lib.rs index 76fdac097ff71..6de9b4275ce2d 100644 --- a/src/tools/rust-analyzer/crates/mbe/src/lib.rs +++ b/src/tools/rust-analyzer/crates/mbe/src/lib.rs @@ -442,7 +442,7 @@ pub fn expect_fragment<'t>( } let res = cursor.crossed(); - tt_iter.flat_advance(res.len()); + tt_iter.flat_advance_to(&cursor); ExpandResult { value: res, err } } diff --git a/src/tools/rust-analyzer/crates/tt/Cargo.toml b/src/tools/rust-analyzer/crates/tt/Cargo.toml index 9a798b592d03c..bd8f740b2f50d 100644 --- a/src/tools/rust-analyzer/crates/tt/Cargo.toml +++ b/src/tools/rust-analyzer/crates/tt/Cargo.toml @@ -16,7 +16,6 @@ doctest = false arrayvec.workspace = true text-size.workspace = true rustc-hash.workspace = true -indexmap.workspace = true span = { path = "../span", version = "0.0", default-features = false } stdx.workspace = true diff --git a/src/tools/rust-analyzer/crates/tt/src/buffer.rs b/src/tools/rust-analyzer/crates/tt/src/buffer.rs index 78cf4b956d0ce..03dcb1b405378 100644 --- a/src/tools/rust-analyzer/crates/tt/src/buffer.rs +++ b/src/tools/rust-analyzer/crates/tt/src/buffer.rs @@ -1,41 +1,51 @@ //! Stateful iteration over token trees. //! //! We use this as the source of tokens for parser. -use crate::{Leaf, Subtree, TokenTree, TokenTreesView, dispatch_ref}; +use crate::{Leaf, Subtree, TokenTree, TokenTreesView, storage::TokenTreesSlice}; pub struct Cursor<'a> { - buffer: TokenTreesView<'a>, - index: usize, - subtrees_stack: Vec, + origin: TokenTreesSlice<'a>, + buffer_before_current: TokenTreesSlice<'a>, + buffer_after_current: TokenTreesSlice<'a>, + /// The number of times we called [`Self::advance()`]. Also the index of [`Self::next`]. + advances_count: usize, + len: usize, + next: Option, + subtrees_stack: Vec<(usize, Subtree)>, } impl<'a> Cursor<'a> { - pub fn new(buffer: TokenTreesView<'a>) -> Self { - Self { buffer, index: 0, subtrees_stack: Vec::new() } + pub fn new(origin: TokenTreesView<'a>) -> Self { + let mut buffer_after_current = origin.slice; + let buffer_before_current = buffer_after_current; + let len = origin.len; + let next = if len >= 1 { buffer_after_current.advance() } else { None }; + Self { + origin: origin.slice, + buffer_after_current, + buffer_before_current, + advances_count: 0, + len, + next, + subtrees_stack: Vec::new(), + } } /// Check whether it is eof pub fn eof(&self) -> bool { - self.index == self.buffer.len() && self.subtrees_stack.is_empty() + self.next.is_none() && self.subtrees_stack.is_empty() } pub fn is_root(&self) -> bool { self.subtrees_stack.is_empty() } - fn at(&self, idx: usize) -> Option { - dispatch_ref! { - match self.buffer.repr => tt => Some(tt.get(idx)?.to_api(self.buffer.span_parts)) - } + fn last_subtree(&self) -> Option<(usize, Subtree)> { + self.subtrees_stack.last().copied() } - fn last_subtree(&self) -> Option<(usize, Subtree)> { - self.subtrees_stack.last().map(|&subtree_idx| { - let Some(TokenTree::Subtree(subtree)) = self.at(subtree_idx) else { - panic!("subtree pointing to non-subtree"); - }; - (subtree_idx, subtree) - }) + pub(crate) fn remaining(&self) -> TokenTreesView<'a> { + TokenTreesView { slice: self.buffer_before_current, len: self.len - self.advances_count } } pub fn end(&mut self) -> Subtree { @@ -44,7 +54,7 @@ impl<'a> Cursor<'a> { // +1 because `Subtree.len` excludes the subtree itself. assert_eq!( last_subtree_idx + last_subtree.usize_len() + 1, - self.index, + self.advances_count, "called `Cursor::end()` without finishing a subtree" ); self.subtrees_stack.pop(); @@ -55,11 +65,24 @@ impl<'a> Cursor<'a> { pub fn token_tree(&self) -> Option { if let Some((last_subtree_idx, last_subtree)) = self.last_subtree() { // +1 because `Subtree.len` excludes the subtree itself. - if last_subtree_idx + last_subtree.usize_len() + 1 == self.index { + if last_subtree_idx + last_subtree.usize_len() + 1 == self.advances_count { return None; } } - self.at(self.index) + self.next.clone() + } + + fn advance(&mut self) { + if self.advances_count >= self.len { + return; + } + if let Some(TokenTree::Subtree(subtree)) = self.next { + self.subtrees_stack.push((self.advances_count, subtree)); + } + self.advances_count += 1; + self.buffer_before_current = self.buffer_after_current; + self.next = + if self.advances_count < self.len { self.buffer_after_current.advance() } else { None }; } /// Bump the cursor, and enters a subtree if it is on one. @@ -68,40 +91,35 @@ impl<'a> Cursor<'a> { // +1 because `Subtree.len` excludes the subtree itself. assert_ne!( last_subtree_idx + last_subtree.usize_len() + 1, - self.index, + self.advances_count, "called `Cursor::bump()` when at the end of a subtree" ); } - if let Some(TokenTree::Subtree(_)) = self.at(self.index) { - self.subtrees_stack.push(self.index); - } - self.index += 1; + self.advance(); } pub fn bump_or_end(&mut self) { - if let Some((last_subtree_idx, last_subtree)) = self.last_subtree() { - // +1 because `Subtree.len` excludes the subtree itself. - if last_subtree_idx + last_subtree.usize_len() + 1 == self.index { - self.subtrees_stack.pop(); - return; - } - } // +1 because `Subtree.len` excludes the subtree itself. - if let Some(TokenTree::Subtree(_)) = self.at(self.index) { - self.subtrees_stack.push(self.index); + if let Some((last_subtree_idx, last_subtree)) = self.last_subtree() + && last_subtree_idx + last_subtree.usize_len() + 1 == self.advances_count + { + self.subtrees_stack.pop(); + return; } - self.index += 1; + self.advance(); } pub fn peek_two_leaves(&self) -> Option<[Leaf; 2]> { if let Some((last_subtree_idx, last_subtree)) = self.last_subtree() { // +1 because `Subtree.len` excludes the subtree itself. let last_end = last_subtree_idx + last_subtree.usize_len() + 1; - if last_end == self.index || last_end == self.index + 1 { + if last_end == self.advances_count || last_end == self.advances_count + 1 { return None; } } - self.at(self.index).zip(self.at(self.index + 1)).and_then(|it| match it { + let mut buffer = self.buffer_after_current; + let next_next = if self.advances_count + 1 < self.len { buffer.advance() } else { None }; + self.next.clone().zip(next_next).and_then(|it| match it { (TokenTree::Leaf(a), TokenTree::Leaf(b)) => Some([a, b]), _ => None, }) @@ -109,9 +127,6 @@ impl<'a> Cursor<'a> { pub fn crossed(&self) -> TokenTreesView<'a> { assert!(self.is_root()); - TokenTreesView { - repr: self.buffer.repr.get(..self.index).unwrap(), - span_parts: self.buffer.span_parts, - } + TokenTreesView { slice: self.origin, len: self.advances_count } } } diff --git a/src/tools/rust-analyzer/crates/tt/src/iter.rs b/src/tools/rust-analyzer/crates/tt/src/iter.rs index 7caacd40dd7e3..9419a9e8534eb 100644 --- a/src/tools/rust-analyzer/crates/tt/src/iter.rs +++ b/src/tools/rust-analyzer/crates/tt/src/iter.rs @@ -8,8 +8,8 @@ use intern::sym; use span::Span; use crate::{ - Ident, Leaf, MAX_GLUED_PUNCT_LEN, Punct, Spacing, Subtree, TokenTree, TokenTreesReprRef, - TokenTreesView, dispatch_ref, + Ident, Leaf, MAX_GLUED_PUNCT_LEN, Punct, Spacing, Subtree, TokenTree, TokenTreesView, + buffer::Cursor, }; #[derive(Clone)] @@ -126,13 +126,13 @@ impl<'a> TtIter<'a> { return Ok(res); } - let (second, third) = match (self.peek_n(0), self.peek_n(1)) { - (Some(TokenTree::Leaf(Leaf::Punct(p2))), Some(TokenTree::Leaf(Leaf::Punct(p3)))) + let (second, third) = match self.peek_two() { + [Some(TokenTree::Leaf(Leaf::Punct(p2))), Some(TokenTree::Leaf(Leaf::Punct(p3)))] if p2.spacing == Spacing::Joint => { (p2, Some(p3)) } - (Some(TokenTree::Leaf(Leaf::Punct(p2))), _) => (p2, None), + [Some(TokenTree::Leaf(Leaf::Punct(p2))), _] => (p2, None), _ => { res.push(first); return Ok(res); @@ -165,20 +165,21 @@ impl<'a> TtIter<'a> { } /// This method won't check for subtrees, so the nth token tree may not be the nth sibling of the current tree. - fn peek_n(&self, n: usize) -> Option { - dispatch_ref! { - match self.inner.repr => tt => Some(tt.get(n)?.to_api(self.inner.span_parts)) - } + fn peek_two(&self) -> [Option; 2] { + let mut iter = self.inner.iter_flat_tokens(); + [iter.next(), iter.next()] } pub fn peek(&self) -> Option> { - match self.peek_n(0)? { + if self.inner.is_empty() { + return None; + } + let mut slice = self.inner.slice; + match slice.advance()? { TokenTree::Leaf(leaf) => Some(TtElement::Leaf(leaf)), TokenTree::Subtree(subtree) => { - let nested_repr = self.inner.repr.get(1..subtree.usize_len() + 1).unwrap(); - let nested_iter = TtIter { - inner: TokenTreesView { repr: nested_repr, span_parts: self.inner.span_parts }, - }; + let nested_iter = + TtIter { inner: TokenTreesView { len: subtree.usize_len(), slice } }; Some(TtElement::Subtree(subtree, nested_iter)) } } @@ -186,7 +187,7 @@ impl<'a> TtIter<'a> { /// Equivalent to `peek().is_none()`, but a bit faster. pub fn is_empty(&self) -> bool { - self.inner.len() == 0 + self.inner.is_empty() } pub fn next_span(&self) -> Option { @@ -197,9 +198,9 @@ impl<'a> TtIter<'a> { self.inner } - /// **Warning**: This advances `skip` **flat** token trees, subtrees account for children+1! - pub fn flat_advance(&mut self, skip: usize) { - self.inner.repr = self.inner.repr.get(skip..).unwrap(); + /// **Warning**: This advances **flat** token trees, subtrees account for children+1! + pub fn flat_advance_to(&mut self, up_to: &Cursor<'a>) { + self.inner = up_to.remaining(); } pub fn savepoint(&self) -> TtIterSavepoint<'a> { @@ -207,34 +208,7 @@ impl<'a> TtIter<'a> { } pub fn from_savepoint(&self, savepoint: TtIterSavepoint<'a>) -> TokenTreesView<'a> { - let len = match (self.inner.repr, savepoint.0.repr) { - ( - TokenTreesReprRef::SpanStorage32(this), - TokenTreesReprRef::SpanStorage32(savepoint), - ) => { - (this.as_ptr() as usize - savepoint.as_ptr() as usize) - / size_of::>() - } - ( - TokenTreesReprRef::SpanStorage64(this), - TokenTreesReprRef::SpanStorage64(savepoint), - ) => { - (this.as_ptr() as usize - savepoint.as_ptr() as usize) - / size_of::>() - } - ( - TokenTreesReprRef::SpanStorage96(this), - TokenTreesReprRef::SpanStorage96(savepoint), - ) => { - (this.as_ptr() as usize - savepoint.as_ptr() as usize) - / size_of::>() - } - _ => panic!("savepoint did not originate from this TtIter"), - }; - TokenTreesView { - repr: savepoint.0.repr.get(..len).unwrap(), - span_parts: savepoint.0.span_parts, - } + TokenTreesView { slice: savepoint.0.slice, len: savepoint.0.len - self.inner.len } } pub fn next_as_view(&mut self) -> Option> { @@ -274,12 +248,20 @@ impl TtElement<'_> { impl<'a> Iterator for TtIter<'a> { type Item = TtElement<'a>; fn next(&mut self) -> Option { - let result = self.peek()?; - let skip = match &result { - TtElement::Leaf(_) => 1, - TtElement::Subtree(subtree, _) => subtree.usize_len() + 1, - }; - self.inner.repr = self.inner.repr.get(skip..).unwrap(); - Some(result) + if self.inner.is_empty() { + return None; + } + self.inner.len -= 1; + let (tt, subtree_slice) = self.inner.slice.advance_skip_subtree()?; + match tt { + TokenTree::Leaf(leaf) => Some(TtElement::Leaf(leaf)), + TokenTree::Subtree(subtree) => { + self.inner.len -= subtree.usize_len(); + let nested_iter = TtIter { + inner: TokenTreesView { len: subtree.usize_len(), slice: subtree_slice }, + }; + Some(TtElement::Subtree(subtree, nested_iter)) + } + } } } diff --git a/src/tools/rust-analyzer/crates/tt/src/lib.rs b/src/tools/rust-analyzer/crates/tt/src/lib.rs index 7b46c33596441..2bc2b64cd4fd0 100644 --- a/src/tools/rust-analyzer/crates/tt/src/lib.rs +++ b/src/tools/rust-analyzer/crates/tt/src/lib.rs @@ -17,7 +17,7 @@ pub mod buffer; pub mod iter; mod storage; -use std::{fmt, slice::SliceIndex}; +use std::fmt; use arrayvec::ArrayString; use buffer::Cursor; @@ -27,7 +27,7 @@ use stdx::{impl_from, itertools::Itertools as _}; pub use span::Span; pub use text_size::{TextRange, TextSize}; -use crate::storage::{CompressedSpanPart, SpanStorage}; +use crate::storage::TokenTreesSlice; pub use self::iter::{TtElement, TtIter}; pub use self::storage::{TopSubtree, TopSubtreeBuilder}; @@ -42,9 +42,11 @@ pub struct Lit { } #[derive(Debug, Copy, Clone, PartialEq, Eq, Hash)] +#[repr(u8)] +// The discriminants are important for `storage.rs` decoding. pub enum IdentIsRaw { - No, - Yes, + No = 0, + Yes = 1, } impl IdentIsRaw { pub fn yes(self) -> bool { @@ -113,6 +115,14 @@ impl Leaf { Leaf::Ident(it) => &it.span, } } + + fn symbol(&self) -> Option<&Symbol> { + match self { + Leaf::Literal(Literal { text_and_suffix: symbol, .. }) + | Leaf::Ident(Ident { sym: symbol, .. }) => Some(symbol), + Leaf::Punct(_) => None, + } + } } impl_from!(Literal, Punct, Ident for Leaf); @@ -129,67 +139,16 @@ impl Subtree { } } -#[rust_analyzer::macro_style(braces)] -macro_rules! dispatch_ref { - ( - match $scrutinee:expr => $tt:ident => $body:expr - ) => { - match $scrutinee { - $crate::TokenTreesReprRef::SpanStorage32($tt) => $body, - $crate::TokenTreesReprRef::SpanStorage64($tt) => $body, - $crate::TokenTreesReprRef::SpanStorage96($tt) => $body, - } - }; -} -use dispatch_ref; - -#[derive(Clone, Copy)] -enum TokenTreesReprRef<'a> { - SpanStorage32(&'a [crate::storage::TokenTree]), - SpanStorage64(&'a [crate::storage::TokenTree]), - SpanStorage96(&'a [crate::storage::TokenTree]), -} - -impl<'a> TokenTreesReprRef<'a> { - #[inline] - fn get(&self, index: I) -> Option - where - I: SliceIndex< - [crate::storage::TokenTree], - Output = [crate::storage::TokenTree], - >, - I: SliceIndex< - [crate::storage::TokenTree], - Output = [crate::storage::TokenTree], - >, - I: SliceIndex< - [crate::storage::TokenTree], - Output = [crate::storage::TokenTree], - >, - { - Some(match self { - TokenTreesReprRef::SpanStorage32(tt) => { - TokenTreesReprRef::SpanStorage32(tt.get(index)?) - } - TokenTreesReprRef::SpanStorage64(tt) => { - TokenTreesReprRef::SpanStorage64(tt.get(index)?) - } - TokenTreesReprRef::SpanStorage96(tt) => { - TokenTreesReprRef::SpanStorage96(tt.get(index)?) - } - }) - } -} - #[derive(Clone, Copy)] pub struct TokenTreesView<'a> { - repr: TokenTreesReprRef<'a>, - span_parts: &'a [CompressedSpanPart], + slice: TokenTreesSlice<'a>, + len: usize, } impl<'a> TokenTreesView<'a> { + #[inline] pub fn empty() -> Self { - Self { repr: TokenTreesReprRef::SpanStorage32(&[]), span_parts: &[] } + Self { slice: TokenTreesSlice::empty(), len: 0 } } pub fn iter(&self) -> TtIter<'a> { @@ -201,9 +160,7 @@ impl<'a> TokenTreesView<'a> { } pub fn len(&self) -> usize { - dispatch_ref! { - match self.repr => tt => tt.len() - } + self.len } pub fn is_empty(&self) -> bool { @@ -211,12 +168,9 @@ impl<'a> TokenTreesView<'a> { } pub fn try_into_subtree(self) -> Option> { - let is_subtree = dispatch_ref! { - match self.repr => tt => matches!( - tt.first(), - Some(crate::storage::TokenTree::Subtree { len, .. }) if (*len as usize) == (tt.len() - 1) - ) - }; + let is_subtree = self.iter_flat_tokens().next().is_some_and( + |it| matches!(it, TokenTree::Subtree(subtree) if subtree.usize_len() == self.len - 1), + ); if is_subtree { Some(SubtreeView(self)) } else { None } } @@ -251,23 +205,29 @@ impl<'a> TokenTreesView<'a> { } pub fn first_span(&self) -> Option { - Some(dispatch_ref! { - match self.repr => tt => tt.first()?.first_span().span(self.span_parts) - }) + self.iter_flat_tokens().next().map(|it| it.first_span()) } + /// Note: this is quite expensive, this needs to decode the whole view, + /// although it "tricks" by skipping subtrees (since we know their byte length). pub fn last_span(&self) -> Option { - Some(dispatch_ref! { - match self.repr => tt => tt.last()?.last_span().span(self.span_parts) - }) + let mut iter = self.iter(); + loop { + match iter.last()? { + TtElement::Leaf(leaf) => return Some(*leaf.span()), + TtElement::Subtree(subtree, tt_iter) => { + if subtree.len == 0 { + return Some(subtree.delimiter.close); + } else { + iter = tt_iter; + } + } + } + } } - pub fn iter_flat_tokens(self) -> impl ExactSizeIterator + use<'a> { - (0..self.len()).map(move |idx| { - dispatch_ref! { - match self.repr => tt => tt[idx].to_api(self.span_parts) - } - }) + pub fn iter_flat_tokens(&self) -> impl Iterator + use<'a> { + self.slice.iter().take(self.len) } } @@ -343,23 +303,10 @@ impl<'a> SubtreeView<'a> { } pub fn top_subtree(&self) -> Subtree { - dispatch_ref! { - match self.0.repr => tt => { - let crate::storage::TokenTree::Subtree { len, delim_kind, open_span, close_span } = - &tt[0] - else { - unreachable!("the first token tree is always the top subtree"); - }; - Subtree { - delimiter: Delimiter { - open: open_span.span(self.0.span_parts), - close: close_span.span(self.0.span_parts), - kind: *delim_kind, - }, - len: *len, - } - } - } + let Some(TokenTree::Subtree(subtree)) = self.0.iter_flat_tokens().next() else { + unreachable!("the first token tree is always the top subtree"); + }; + subtree } pub fn strip_invisible(&self) -> TokenTreesView<'a> { @@ -371,18 +318,10 @@ impl<'a> SubtreeView<'a> { } pub fn token_trees(&self) -> TokenTreesView<'a> { - let repr = match self.0.repr { - TokenTreesReprRef::SpanStorage32(token_trees) => { - TokenTreesReprRef::SpanStorage32(&token_trees[1..]) - } - TokenTreesReprRef::SpanStorage64(token_trees) => { - TokenTreesReprRef::SpanStorage64(&token_trees[1..]) - } - TokenTreesReprRef::SpanStorage96(token_trees) => { - TokenTreesReprRef::SpanStorage96(&token_trees[1..]) - } - }; - TokenTreesView { repr, ..self.0 } + let mut result = self.0; + result.slice.advance(); + result.len -= 1; + result } } @@ -435,11 +374,13 @@ impl Delimiter { } #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +#[repr(u8)] +// The discriminants are important for decoding for `storage.rs`. pub enum DelimiterKind { - Parenthesis, - Brace, - Bracket, - Invisible, + Parenthesis = 0, + Brace = 1, + Bracket = 2, + Invisible = 3, } #[derive(Debug, Clone, PartialEq, Eq, Hash)] @@ -555,7 +496,9 @@ pub struct Punct { /// compound token. Used for conversions to `proc_macro::Spacing`. Also used to /// guide pretty-printing, which is where the `JointHidden` value (which isn't /// part of `proc_macro::Spacing`) comes in useful. +// The discriminants are important for decoding for `storage.rs`. #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +#[repr(u8)] pub enum Spacing { /// The token cannot join with the following token to form a compound /// token. @@ -572,7 +515,7 @@ pub enum Spacing { /// /// Converts to `proc_macro::Spacing::Alone`, and /// `proc_macro::Spacing::Alone` converts back to this. - Alone, + Alone = 0, /// The token can join with the following token to form a compound token. /// @@ -586,7 +529,7 @@ pub enum Spacing { /// /// Converts to `proc_macro::Spacing::Joint`, and /// `proc_macro::Spacing::Joint` converts back to this. - Joint, + Joint = 1, /// The token can join with the following token to form a compound token, /// but this will not be visible at the proc macro level. (This is what the @@ -608,7 +551,7 @@ pub enum Spacing { /// pretty-printing of `TokenStream`'s produced by other means (i.e. parsed /// source code, internally constructed token streams, and token streams /// produced by declarative macros). - JointHidden, + JointHidden = 2, } /// Identifier or keyword. @@ -764,36 +707,16 @@ impl Subtree { } pub fn pretty(tkns: TokenTreesView<'_>) -> String { - return dispatch_ref! { - match tkns.repr => tt => pretty_impl(tkns, tt) - }; - - use crate::storage::TokenTree; + return pretty_impl(tkns.iter()); - fn tokentree_to_text( - tkns_view: TokenTreesView<'_>, - tkn: &TokenTree, - tkns: &mut &[TokenTree], - ) -> String { + fn tokentree_to_text(tkn: TtElement<'_>) -> String { match tkn { - TokenTree::Ident { sym, is_raw, .. } => format!("{}{}", is_raw.as_str(), sym), - &TokenTree::Literal { ref text_and_suffix, kind, suffix_len, span } => { - format!( - "{}", - Literal { - text_and_suffix: text_and_suffix.clone(), - span: span.span(tkns_view.span_parts), - kind, - suffix_len - } - ) + TtElement::Leaf(leaf) => { + format!("{}", leaf) } - TokenTree::Punct { char, .. } => format!("{}", char), - TokenTree::Subtree { len, delim_kind, .. } => { - let (subtree_content, rest) = tkns.split_at(*len as usize); - let content = pretty_impl(tkns_view, subtree_content); - *tkns = rest; - let (open, close) = match *delim_kind { + TtElement::Subtree(Subtree { delimiter, .. }, subtree_content) => { + let content = pretty_impl(subtree_content); + let (open, close) = match delimiter.kind { DelimiterKind::Brace => ("{", "}"), DelimiterKind::Bracket => ("[", "]"), DelimiterKind::Parenthesis => ("(", ")"), @@ -804,23 +727,16 @@ pub fn pretty(tkns: TokenTreesView<'_>) -> String { } } - fn pretty_impl( - tkns_view: TokenTreesView<'_>, - mut tkns: &[TokenTree], - ) -> String { + fn pretty_impl(tkns: TtIter<'_>) -> String { let mut last = String::new(); let mut last_to_joint = true; - while let Some((tkn, rest)) = tkns.split_first() { - tkns = rest; - last = [last, tokentree_to_text(tkns_view, tkn, &mut tkns)].join(if last_to_joint { - "" - } else { - " " - }); + for tkn in tkns { + last = + [last, tokentree_to_text(tkn.clone())].join(if last_to_joint { "" } else { " " }); last_to_joint = false; - if let TokenTree::Punct { spacing, .. } = tkn - && *spacing == Spacing::Joint + if let TtElement::Leaf(Leaf::Punct(Punct { spacing, .. })) = tkn + && spacing == Spacing::Joint { last_to_joint = true; } @@ -847,7 +763,7 @@ impl TransformTtAction<'_> { /// tts view. pub fn transform_tt<'b>( tt: &mut TopSubtree, - mut callback: impl FnMut(TokenTree) -> TransformTtAction<'b>, + mut callback: impl FnMut(&TokenTree) -> TransformTtAction<'b>, ) { let mut tt_vec = tt.as_token_trees().iter_flat_tokens().collect::>(); @@ -867,27 +783,20 @@ pub fn transform_tt<'b>( } } - let current = match &tt_vec[i] { - TokenTree::Leaf(leaf) => TokenTree::Leaf(match leaf { - Leaf::Literal(leaf) => Leaf::Literal(leaf.clone()), - Leaf::Punct(leaf) => Leaf::Punct(*leaf), - Leaf::Ident(leaf) => Leaf::Ident(leaf.clone()), - }), - TokenTree::Subtree(subtree) => TokenTree::Subtree(*subtree), - }; + let current = &tt_vec[i]; let action = callback(current); match action { TransformTtAction::Keep => { // This cannot be shared with the replaced case, because then we may push the same subtree // twice, and will update it twice which will lead to errors. - if let TokenTree::Subtree(_) = &tt_vec[i] { + if let TokenTree::Subtree(_) = current { subtrees_stack.push(i); } i += 1; } TransformTtAction::ReplaceWith(replacement) => { - let old_len = 1 + match &tt_vec[i] { + let old_len = 1 + match current { TokenTree::Leaf(_) => 0, TokenTree::Subtree(subtree) => subtree.usize_len(), }; diff --git a/src/tools/rust-analyzer/crates/tt/src/storage.rs b/src/tools/rust-analyzer/crates/tt/src/storage.rs index 50a1106175ab3..150777cc39e55 100644 --- a/src/tools/rust-analyzer/crates/tt/src/storage.rs +++ b/src/tools/rust-analyzer/crates/tt/src/storage.rs @@ -2,40 +2,32 @@ //! will waste a lot of memory. So instead we implement a clever compression mechanism: //! //! A `TopSubtree` has a list of [`CompressedSpanPart`], which are the parts of a span -//! that tend to be shared between tokens - namely, without the range. The main list -//! of token trees is kept in one of three versions, where we use the smallest version -//! we can for this tree: +//! that tend to be shared between tokens - namely, without the range. //! -//! 1. In the most common version a span is just a `u32`. The bits are divided as follows: -//! there are 4 bits that index into the [`CompressedSpanPart`] list. 20 bits -//! store the range start, and 8 bits store the range length. In experiments, -//! this accounts for 75%-85% of the spans. -//! 2. In the second version a span is 64 bits. 32 bits for the range start, 16 bits -//! for the range length, and 16 bits for the span parts index. This is used in -//! less than 2% of all `TopSubtree`s, but they account for 15%-25% of the spans: -//! those are mostly token tree munchers, that generate a lot of `SyntaxContext`s -//! (because they recurse a lot), which is why they can't fit in the first version, -//! and tend to generate a lot of code. -//! 3. The third version is practically unused; 65,535 bytes for a token and 65,535 -//! unique span parts is more than enough for everybody. However, someone may still -//! create a macro that requires more, therefore we have this version as a backup: -//! it uses 96 bits, 32 for each of the range start, length and span parts index. - -use std::fmt; +//! The main list of token trees is stored in a variable-length encoding as bytes. +//! The encoding is documented in the [`decode()`] function (which decodes one [`TokenTree`]). + +use std::{assert_matches, collections::hash_map, fmt::Debug, hint::cold_path, mem::transmute}; + +#[cfg(all(debug_assertions, not(miri)))] +use std::cell::Cell; + +#[cfg(not(all(debug_assertions, not(miri))))] +use std::mem::MaybeUninit; use intern::Symbol; -use rustc_hash::FxBuildHasher; +use rustc_hash::FxHashMap; use span::{Span, SpanAnchor, SyntaxContext, TextRange, TextSize}; use crate::{ - DelimSpan, DelimiterKind, IdentIsRaw, LitKind, Spacing, SubtreeView, TokenTreesReprRef, - TokenTreesView, TtIter, dispatch_ref, + DelimSpan, Delimiter, DelimiterKind, Ident, IdentIsRaw, Leaf, LitKind, Literal, Punct, Spacing, + Subtree, SubtreeView, TokenTree, TokenTreesView, TtIter, }; #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] -pub(crate) struct CompressedSpanPart { - pub(crate) anchor: SpanAnchor, - pub(crate) ctx: SyntaxContext, +struct CompressedSpanPart { + anchor: SpanAnchor, + ctx: SyntaxContext, } impl CompressedSpanPart { @@ -50,376 +42,1086 @@ impl CompressedSpanPart { } } -pub(crate) trait SpanStorage: Copy { - fn can_hold(text_range: TextRange, span_parts_index: usize) -> bool; - - fn new(text_range: TextRange, span_parts_index: usize) -> Self; - - fn text_range(&self) -> TextRange; +trait Encodable: Sized { + #[cfg(all(debug_assertions, not(miri)))] + fn write(self, buffer: &[Cell]); + #[cfg(all(debug_assertions, not(miri)))] + fn read(buffer: &[u8]) -> Self; +} - fn span_parts_index(&self) -> usize; +impl Encodable for u8 { + #[cfg(all(debug_assertions, not(miri)))] + fn write(self, buffer: &[Cell]) { + buffer[0].set(self); + } + #[cfg(all(debug_assertions, not(miri)))] + fn read(buffer: &[u8]) -> Self { + buffer[0] + } +} - #[inline] - fn span(&self, span_parts: &[CompressedSpanPart]) -> Span { - span_parts[self.span_parts_index()].recombine(self.text_range()) +impl Encodable for u16 { + #[cfg(all(debug_assertions, not(miri)))] + fn write(self, buffer: &[Cell]) { + let value = self.to_ne_bytes(); + let buffer: &[Cell; size_of::()] = buffer.try_into().unwrap(); + for (b, v) in std::iter::zip(buffer, value) { + b.set(v); + } + } + #[cfg(all(debug_assertions, not(miri)))] + fn read(buffer: &[u8]) -> Self { + Self::from_ne_bytes(buffer.try_into().unwrap()) } } -#[inline] -const fn n_bits_mask(n: u32) -> u32 { - (1 << n) - 1 +impl Encodable for u32 { + #[cfg(all(debug_assertions, not(miri)))] + fn write(self, buffer: &[Cell]) { + let value = self.to_ne_bytes(); + let buffer: &[Cell; size_of::()] = buffer.try_into().unwrap(); + for (b, v) in std::iter::zip(buffer, value) { + b.set(v); + } + } + #[cfg(all(debug_assertions, not(miri)))] + fn read(buffer: &[u8]) -> Self { + Self::from_ne_bytes(buffer.try_into().unwrap()) + } } -#[derive(Clone, Copy, PartialEq, Eq, Hash)] -pub(crate) struct SpanStorage32(u32); +impl Encodable for char { + #[cfg(all(debug_assertions, not(miri)))] + fn write(self, buffer: &[Cell]) { + u32::from(self).write(buffer) + } + #[cfg(all(debug_assertions, not(miri)))] + fn read(buffer: &[u8]) -> Self { + char::from_u32(u32::read(buffer)).unwrap() + } +} -impl SpanStorage32 { - const SPAN_PARTS_BIT: u32 = 4; - const LEN_BITS: u32 = 8; - const OFFSET_BITS: u32 = 20; +struct UninitBuffer { + #[cfg(all(debug_assertions, not(miri)))] + buffer: Box<[u8]>, + #[cfg(not(all(debug_assertions, not(miri))))] + buffer: Box<[MaybeUninit]>, } -const _: () = assert!( - (SpanStorage32::SPAN_PARTS_BIT + SpanStorage32::LEN_BITS + SpanStorage32::OFFSET_BITS) - == u32::BITS -); +impl UninitBuffer { + #[inline] + fn new(capacity: usize) -> Self { + Self { + #[cfg(all(debug_assertions, not(miri)))] + buffer: vec![0; capacity].into_boxed_slice(), + #[cfg(not(all(debug_assertions, not(miri))))] + buffer: Box::new_uninit_slice(capacity), + } + } -impl SpanStorage for SpanStorage32 { #[inline] - fn can_hold(text_range: TextRange, span_parts_index: usize) -> bool { - let offset = u32::from(text_range.start()); - let len = u32::from(text_range.len()); - let span_parts_index = span_parts_index as u32; + fn writer(&mut self) -> BufferWriter<'_> { + BufferWriter { + #[cfg(all(debug_assertions, not(miri)))] + buffer: Cell::from_mut(&mut *self.buffer).as_slice_of_cells(), + #[cfg(not(all(debug_assertions, not(miri))))] + ptr: self.buffer.as_mut_ptr_range().end.cast::(), + #[cfg(not(all(debug_assertions, not(miri))))] + _marker: stdx::variance::PhantomCovariantLifetime::new(), + } + } - offset <= n_bits_mask(Self::OFFSET_BITS) - && len <= n_bits_mask(Self::LEN_BITS) - && span_parts_index <= n_bits_mask(Self::SPAN_PARTS_BIT) + #[cfg(all(debug_assertions, not(miri)))] + unsafe fn finish(self, writer_finish: usize) -> Box<[u8]> { + self.buffer[writer_finish..].into() } + #[cfg(not(all(debug_assertions, not(miri))))] #[inline] - fn new(text_range: TextRange, span_parts_index: usize) -> Self { - let offset = u32::from(text_range.start()); - let len = u32::from(text_range.len()); - let span_parts_index = span_parts_index as u32; - - debug_assert!(offset <= n_bits_mask(Self::OFFSET_BITS)); - debug_assert!(len <= n_bits_mask(Self::LEN_BITS)); - debug_assert!(span_parts_index <= n_bits_mask(Self::SPAN_PARTS_BIT)); - - Self( - (offset << (Self::LEN_BITS + Self::SPAN_PARTS_BIT)) - | (len << Self::SPAN_PARTS_BIT) - | span_parts_index, - ) + unsafe fn finish(mut self, writer_finish: *mut u8) -> Box<[u8]> { + let end = self.buffer.as_mut_ptr_range().end.cast::(); + unsafe { + let bytes_len = end.offset_from_unsigned(writer_finish); + let mut buffer = Box::<[u8]>::new_uninit_slice(bytes_len); + buffer.as_mut_ptr().cast::().copy_from_nonoverlapping(writer_finish, bytes_len); + buffer.assume_init() + } } +} +#[derive(Clone, Copy)] +struct BufferWriter<'a> { + #[cfg(all(debug_assertions, not(miri)))] + buffer: &'a [Cell], + #[cfg(not(all(debug_assertions, not(miri))))] + ptr: *mut u8, + #[cfg(not(all(debug_assertions, not(miri))))] + _marker: stdx::variance::PhantomCovariantLifetime<'a>, +} + +impl<'a> BufferWriter<'a> { #[inline] - fn text_range(&self) -> TextRange { - let offset = TextSize::new(self.0 >> (Self::SPAN_PARTS_BIT + Self::LEN_BITS)); - let len = TextSize::new((self.0 >> Self::SPAN_PARTS_BIT) & n_bits_mask(Self::LEN_BITS)); - TextRange::at(offset, len) + unsafe fn new(buffer: &'a mut [u8]) -> Self { + BufferWriter { + #[cfg(all(debug_assertions, not(miri)))] + buffer: Cell::from_mut(buffer).as_slice_of_cells(), + #[cfg(not(all(debug_assertions, not(miri))))] + ptr: buffer.as_mut_ptr_range().end, + #[cfg(not(all(debug_assertions, not(miri))))] + _marker: stdx::variance::PhantomCovariantLifetime::new(), + } + } + + #[cfg_attr(not(all(debug_assertions, not(miri))), inline(always))] + unsafe fn write(&mut self, value: T) { + #[cfg(all(debug_assertions, not(miri)))] + { + let write_at = self.buffer.split_off(self.buffer.len() - size_of::()..).unwrap(); + value.write(write_at); + } + #[cfg(not(all(debug_assertions, not(miri))))] + unsafe { + self.ptr = self.ptr.sub(size_of::()); + self.ptr.cast::().write_unaligned(value); + } } + fn len_since(self, other: BufferWriter<'_>) -> usize { + #[cfg(all(debug_assertions, not(miri)))] + { + other.buffer.len() - self.buffer.len() + } + #[cfg(not(all(debug_assertions, not(miri))))] + { + other.ptr.addr() - self.ptr.addr() + } + } + + #[cfg(all(debug_assertions, not(miri)))] + fn finish(self) -> usize { + self.buffer.len() + } + + #[cfg(not(all(debug_assertions, not(miri))))] #[inline] - fn span_parts_index(&self) -> usize { - (self.0 & n_bits_mask(Self::SPAN_PARTS_BIT)) as usize + fn finish(self) -> *mut u8 { + self.ptr } } -impl fmt::Debug for SpanStorage32 { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - f.debug_struct("SpanStorage32") - .field("text_range", &self.text_range()) - .field("span_parts_index", &self.span_parts_index()) - .finish() - } +#[inline] +const fn n_bits_mask(n: u32) -> u32 { + (1 << n) - 1 } -#[derive(Clone, Copy, PartialEq, Eq, Hash)] -pub(crate) struct SpanStorage64 { - offset: u32, - len_and_parts: u32, +#[inline] +const fn n_bits_mask_u64(n: u32) -> u64 { + (1 << n) - 1 } -impl SpanStorage64 { - const SPAN_PARTS_BIT: u32 = 16; - const LEN_BITS: u32 = 16; +// Encoding is done in reverse, from the end to the beginning. This is in order +// to be able to tell how many bytes the children of a `Subtree` occupy, and store +// *that* efficiently, with the smallest possible type. + +#[must_use] +unsafe fn encode_span<'a>( + ptr: BufferWriter<'a>, + span: &Span, + span_parts_map: &FxHashMap, + extra_two_bits: u32, + force_heavy_encoding: bool, +) -> BufferWriter<'a> { + let span_parts_index = span_parts_map[&CompressedSpanPart::from_span(span)] as u32; + let offset = u32::from(span.range.start()); + let len = u32::from(span.range.len()); + unsafe { + encode_span_no_map(ptr, span_parts_index, offset, len, extra_two_bits, force_heavy_encoding) + } } -const _: () = assert!((SpanStorage64::SPAN_PARTS_BIT + SpanStorage64::LEN_BITS) == u32::BITS); +#[must_use] +unsafe fn encode_span_no_map( + mut ptr: BufferWriter<'_>, + mut span_parts_index: u32, + mut offset: u32, + mut len: u32, + extra_two_bits: u32, + force_heavy_encoding: bool, +) -> BufferWriter<'_> { + debug_assert!(extra_two_bits & !0b11 == 0); + let mut first_u32 = extra_two_bits; + first_u32 |= (span_parts_index & n_bits_mask(4)) << 2; + span_parts_index >>= 4; + first_u32 |= (len & n_bits_mask(8)) << (2 + 4); + len >>= 8; + first_u32 |= (offset & n_bits_mask(17)) << (2 + 4 + 8 + 1); + offset >>= 17; + + let extends_to_next = span_parts_index != 0 || len != 0 || offset != 0 || force_heavy_encoding; + first_u32 |= u32::from(extends_to_next) << (2 + 4 + 8); + + if extends_to_next { + ptr = unsafe { + encode_extended_span(ptr, span_parts_index, len, offset, force_heavy_encoding) + }; + } + unsafe { ptr.write::(first_u32) }; -impl SpanStorage for SpanStorage64 { - #[inline] - fn can_hold(text_range: TextRange, span_parts_index: usize) -> bool { - let len = u32::from(text_range.len()); - let span_parts_index = span_parts_index as u32; + ptr +} - len <= n_bits_mask(Self::LEN_BITS) && span_parts_index <= n_bits_mask(Self::SPAN_PARTS_BIT) +#[cold] +#[must_use] +unsafe fn encode_extended_span( + mut ptr: BufferWriter<'_>, + span_parts_index: u32, + len: u32, + offset: u32, + force_heavy_encoding: bool, +) -> BufferWriter<'_> { + if span_parts_index <= n_bits_mask(11) + && len <= n_bits_mask(10) + && offset <= n_bits_mask(10) + && !force_heavy_encoding + { + let mut second_u32 = span_parts_index; + second_u32 |= len << 11; + second_u32 |= offset << (11 + 10); + second_u32 <<= 1; + unsafe { ptr.write::(second_u32) }; + } else { + assert!(span_parts_index <= n_bits_mask(24), "too big `span_parts_index`"); + + let mut u64 = u64::from(span_parts_index); + u64 |= u64::from(len) << 24; + u64 |= u64::from(offset) << (24 + 24); + let third_u32 = u64 as u32; + let mut second_u32 = (u64 >> u32::BITS) as u32; + second_u32 <<= 1; + second_u32 |= 0b1; + unsafe { + ptr.write::(third_u32); + ptr.write::(second_u32); + } } - #[inline] - fn new(text_range: TextRange, span_parts_index: usize) -> Self { - let offset = u32::from(text_range.start()); - let len = u32::from(text_range.len()); - let span_parts_index = span_parts_index as u32; + ptr +} + +#[must_use] +unsafe fn encode_symbol<'a>( + mut ptr: BufferWriter<'a>, + symbol: &Symbol, + tag: u32, + symbols_map: &FxHashMap, +) -> BufferWriter<'a> { + let symbol_idx = symbols_map[symbol] as u32; + unsafe { + if symbol_idx <= n_bits_mask(6) { + ptr.write::(((symbol_idx << 2) | tag) as u8); + } else if symbol_idx <= n_bits_mask(13) { + ptr.write::(((symbol_idx >> 6) << 1) as u8); + ptr.write::(((symbol_idx << 2) | 0b10 | tag) as u8); + } else { + ptr.write::((symbol_idx >> 13) as u16); + ptr.write::((((symbol_idx >> 6) << 1) | 0b1) as u8); + ptr.write::(((symbol_idx << 2) | 0b10 | tag) as u8); + } + } + ptr +} - debug_assert!(len <= n_bits_mask(Self::LEN_BITS)); - debug_assert!(span_parts_index <= n_bits_mask(Self::SPAN_PARTS_BIT)); +#[must_use] +unsafe fn encode<'a>( + mut ptr: BufferWriter<'a>, + tt: TokenTree, + span_parts_map: &FxHashMap, + byte_size_after: &mut [u32], + symbols_map: &FxHashMap, +) -> BufferWriter<'a> { + let before_ptr = ptr; + unsafe { + match tt { + TokenTree::Leaf(Leaf::Punct(Punct { char, spacing, span })) => { + if char.is_ascii() { + let spacing = spacing as u8; + let span_extra = 0b1 | (u32::from(spacing) & 0b10); + let char = ((char as u8) << 1) | (spacing & 0b1); + ptr.write::(char); + ptr = encode_span(ptr, &span, span_parts_map, span_extra, false); + } else { + let mut control_byte = 0b110; + control_byte |= (spacing as u8) << 3; + ptr.write::(char); + ptr.write::(control_byte); + ptr = encode_span(ptr, &span, span_parts_map, 0b00, false); + } + } + TokenTree::Leaf(Leaf::Ident(Ident { sym, span, is_raw })) => { + ptr = encode_symbol(ptr, &sym, is_raw as u32, symbols_map); + ptr = encode_span(ptr, &span, span_parts_map, 0b10, false); + } + TokenTree::Leaf(Leaf::Literal(Literal { text_and_suffix, span, kind, suffix_len })) => { + if matches!(kind, LitKind::Str | LitKind::StrRaw(0 | 1) | LitKind::Integer) + && u32::from(suffix_len) <= n_bits_mask(4) + { + // Literal, format 1. + let mut control_byte = match kind { + LitKind::Str => 0b0_011, + LitKind::StrRaw(0) => 0b0_100, + LitKind::StrRaw(1) => 0b1_011, + LitKind::Integer => 0b1_100, + _ => unreachable!(), + }; + control_byte |= suffix_len << 4; + ptr = encode_symbol(ptr, &text_and_suffix, 0, symbols_map); + ptr.write::(control_byte); + } else { + // Literal, format 2. + let mut control_byte = 0b111; + let (kind, raw_count) = match kind { + LitKind::Byte => (0, None), + LitKind::Char => (1, None), + LitKind::Integer => (2, None), + LitKind::Float => (3, None), + LitKind::Str => (4, None), + LitKind::StrRaw(count) => (5, Some(count)), + LitKind::ByteStr => (6, None), + LitKind::ByteStrRaw(count) => (7, Some(count)), + LitKind::CStr => (8, None), + LitKind::CStrRaw(count) => (9, Some(count)), + LitKind::Err(()) => (10, None), + }; + control_byte |= kind << 3; + ptr = encode_symbol(ptr, &text_and_suffix, 0, symbols_map); + ptr.write::(suffix_len); + if let Some(raw_count) = raw_count { + ptr.write::(raw_count); + } + ptr.write::(control_byte); + } + ptr = encode_span(ptr, &span, span_parts_map, 0b00, false); + } + TokenTree::Subtree(Subtree { delimiter, len }) => { + let open_span_parts_index = + span_parts_map[&CompressedSpanPart::from_span(&delimiter.open)] as u32; + let close_span_parts_index = + span_parts_map[&CompressedSpanPart::from_span(&delimiter.close)] as u32; + let close_span_offset_from_open = delimiter + .close + .range + .start() + .checked_sub(delimiter.open.range.start()) + .map_or(u32::MAX, u32::from); + let children_byte_len = byte_size_after[1] - byte_size_after[1 + len as usize]; + if open_span_parts_index == close_span_parts_index + && len <= n_bits_mask(u8::BITS) + && children_byte_len <= n_bits_mask(u8::BITS) + && delimiter.open.range.len() == TextSize::new(1) + && delimiter.close.range.len() == TextSize::new(1) + && close_span_offset_from_open <= n_bits_mask(11) + { + // Subtree, format 1. + let span = Span { + range: TextRange::at( + delimiter.open.range.start(), + TextSize::new(close_span_offset_from_open), + ), + anchor: delimiter.open.anchor, + ctx: delimiter.open.ctx, + }; + let mut control_byte = 0b000; + control_byte |= (delimiter.kind as u8) << 3; + control_byte |= ((close_span_offset_from_open >> 8) << (3 + 2)) as u8; + ptr.write::(children_byte_len as u8); + ptr.write::(len as u8); + ptr.write::(control_byte); + ptr = encode_span(ptr, &span, span_parts_map, 0b00, false); + } else if open_span_parts_index == close_span_parts_index + && len <= n_bits_mask(u8::BITS) + && children_byte_len <= n_bits_mask(u8::BITS) + && delimiter.open.range.len() == delimiter.close.range.len() + && close_span_offset_from_open <= n_bits_mask(3) + { + // Subtree, format 2. + let mut control_byte = 0b001; + control_byte |= (delimiter.kind as u8) << 3; + control_byte |= (close_span_offset_from_open << (3 + 2)) as u8; + ptr.write::(children_byte_len as u8); + ptr.write::(len as u8); + ptr.write::(control_byte); + ptr = encode_span(ptr, &delimiter.open, span_parts_map, 0b00, false); + } else if len <= n_bits_mask(u8::BITS) + && children_byte_len <= n_bits_mask(12) + && delimiter.open.range.len() == delimiter.close.range.len() + && close_span_offset_from_open <= n_bits_mask(8) + && close_span_parts_index <= n_bits_mask(7) + { + // Subtree, format 3. + let mut control_byte = 0b010; + control_byte |= (delimiter.kind as u8) << 3; + control_byte |= (close_span_parts_index << (3 + 2)) as u8; + let mut children_byte_len = children_byte_len << 4; + children_byte_len |= close_span_parts_index >> 3; + ptr.write::(children_byte_len as u16); + ptr.write::(len as u8); + ptr.write::(close_span_offset_from_open as u8); + ptr.write::(control_byte); + ptr = encode_span(ptr, &delimiter.open, span_parts_map, 0b00, false); + } else { + // Subtree, format 4. + let mut control_byte = 0b101; + control_byte |= (delimiter.kind as u8) << 3; + ptr.write::(children_byte_len); + ptr.write::(len); + ptr = encode_span(ptr, &delimiter.close, span_parts_map, 0b00, false); + ptr.write::(control_byte); + ptr = encode_span(ptr, &delimiter.open, span_parts_map, 0b00, false); + } + } + } - Self { offset, len_and_parts: (len << Self::SPAN_PARTS_BIT) | span_parts_index } + let element_byte_size: u32 = ptr.len_since(before_ptr).try_into().unwrap(); + byte_size_after[0] = byte_size_after[1] + element_byte_size; } - #[inline] - fn text_range(&self) -> TextRange { - let offset = TextSize::new(self.offset); - let len = TextSize::new(self.len_and_parts >> Self::SPAN_PARTS_BIT); - TextRange::at(offset, len) + ptr +} + +/// We always encode the top subtree with the heaviest encoding because we sometimes want to change it. +unsafe fn encode_top_subtree<'a>( + mut ptr: BufferWriter<'a>, + top_subtree: Subtree, + open_span_parts_index: u32, + byte_size_after: &[u32], +) -> BufferWriter<'a> { + unsafe { + let Subtree { delimiter, len } = top_subtree; + let children_byte_len = byte_size_after[1] - byte_size_after[1 + len as usize]; + + let mut control_byte = 0b101; + control_byte |= (delimiter.kind as u8) << 3; + ptr.write::(children_byte_len); + ptr.write::(len); + ptr = encode_span_no_map( + ptr, + open_span_parts_index + 1, + delimiter.close.range.start().into(), + delimiter.close.range.len().into(), + 0b00, + true, + ); + ptr.write::(control_byte); + ptr = encode_span_no_map( + ptr, + open_span_parts_index, + delimiter.open.range.start().into(), + delimiter.open.range.len().into(), + 0b00, + true, + ); } + ptr +} - #[inline] - fn span_parts_index(&self) -> usize { - (self.len_and_parts & n_bits_mask(Self::SPAN_PARTS_BIT)) as usize +fn change_root_delimiter(buffer: &mut [u8], new_delim: DelimiterKind) { + // The span is 3*u32 and then the control byte, in which the delimiter comes. + let control_byte_index = 3 * size_of::(); + let mut control_byte = buffer[control_byte_index]; + control_byte &= 0b111; // Remove previous delimiter. + control_byte |= (new_delim as u8) << 3; + buffer[control_byte_index] = control_byte; +} + +unsafe fn change_root_spans( + buffer: &mut [u8], + open_span_parts_index: u32, + close_span_parts_index: u32, + open_range: TextRange, + close_range: TextRange, +) { + // Remember we write in reverse, so we add `3 * size_of::()`. + unsafe { + _ = encode_span_no_map( + BufferWriter::new(&mut buffer[..3 * size_of::()]), + open_span_parts_index, + u32::from(open_range.start()), + u32::from(open_range.len()), + 0b00, + true, + ); + _ = encode_span_no_map( + BufferWriter::new( + &mut buffer[..3 * size_of::() + size_of::() + 3 * size_of::()], + ), + close_span_parts_index, + u32::from(close_range.start()), + u32::from(close_range.len()), + 0b00, + true, + ); } } -impl fmt::Debug for SpanStorage64 { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - f.debug_struct("SpanStorage64") - .field("text_range", &self.text_range()) - .field("span_parts_index", &self.span_parts_index()) - .finish() +/// This is subtree in format 4: two spans, each at most 3*u32, a u8 control byte, a u32 length and a u32 bytes length. +const BIGGEST_POSSIBLE_TT_ENCODING: usize = + 2 * 3 * size_of::() + size_of::() + size_of::() + size_of::(); + +fn encode_all( + tts: std::vec::IntoIter, + mut compressed_span_frequencies: FxHashMap, + mut symbol_frequencies: FxHashMap, +) -> TopSubtree { + let tts_len = tts.len(); + let mut token_trees = tts.enumerate(); + let Some((_, TokenTree::Subtree(top_subtree))) = token_trees.next() else { + panic!("must always have a top subtree"); + }; + + let (span_parts, span_parts_map) = { + let mut compressed_spans = compressed_span_frequencies + .keys() + .copied() + .chain([ + CompressedSpanPart::from_span(&top_subtree.delimiter.open), + CompressedSpanPart::from_span(&top_subtree.delimiter.close), + ]) + .collect::>(); + { + // For this purpose, do not consider the top delimiters. They should stay last and not affect the other spans, + // since we might want to change them. + let len = compressed_spans.len(); + let compressed_spans = &mut compressed_spans[..len - 2]; + // No need sort if there is already enough space for everyone to be encoded efficiently. + if compressed_span_frequencies.len() > n_bits_mask(4) as usize { + // We want more used spans to have lower indices, so they can be encoded more efficiently. + compressed_spans.sort_unstable_by_key(|span| { + std::cmp::Reverse(compressed_span_frequencies[span]) + }); + } + for (index, span) in compressed_spans.iter().enumerate() { + *compressed_span_frequencies.get_mut(span).unwrap() = index; + } + } + (compressed_spans, compressed_span_frequencies) + }; + + let (symbols, symbols_map) = { + let mut symbols = symbol_frequencies.keys().cloned().collect::>(); + // No need sort if there is already enough space for everyone to be encoded efficiently. + if symbol_frequencies.len() > n_bits_mask(6) as usize { + // We want more used spans to have lower indices, so they can be encoded more efficiently. + symbols.sort_unstable_by_key(|symbol| std::cmp::Reverse(symbol_frequencies[symbol])); + } + for (index, span) in symbols.iter().enumerate() { + *symbol_frequencies.get_mut(span).unwrap() = index; + } + (symbols, symbol_frequencies) + }; + + // +1 because each `encode()` calls reads the previous value. + let mut byte_size_after = vec![0u32; tts_len + 1]; + let bytes_capacity = tts_len * BIGGEST_POSSIBLE_TT_ENCODING; + unsafe { + let mut temp_buffer = UninitBuffer::new(bytes_capacity); + let mut ptr = temp_buffer.writer(); + for (index, tt) in token_trees.rev() { + ptr = encode(ptr, tt, &span_parts_map, &mut byte_size_after[index..], &symbols_map); + } + ptr = encode_top_subtree(ptr, top_subtree, (span_parts.len() - 2) as u32, &byte_size_after); + + let writer_finish = ptr.finish(); + let buffer = temp_buffer.finish(writer_finish); + + TopSubtree { buffer, span_parts, len: tts_len, symbols } } } -impl From for SpanStorage64 { - #[inline] - fn from(value: SpanStorage32) -> Self { - SpanStorage64::new(value.text_range(), value.span_parts_index()) +fn compute_span_frequencies_and_symbols( + tts: &[TokenTree], +) -> (FxHashMap, FxHashMap) { + let mut span_frequencies = FxHashMap::default(); + let mut symbols = FxHashMap::default(); + for tt in tts { + match tt { + TokenTree::Leaf(leaf) => { + if let Some(symbol) = leaf.symbol() { + *symbols.entry(symbol.clone()).or_insert(0) += 1; + } + + *span_frequencies.entry(CompressedSpanPart::from_span(leaf.span())).or_insert(0) += + 1; + } + TokenTree::Subtree(subtree) => { + *span_frequencies + .entry(CompressedSpanPart::from_span(&subtree.delimiter.open)) + .or_insert(0) += 1; + *span_frequencies + .entry(CompressedSpanPart::from_span(&subtree.delimiter.close)) + .or_insert(0) += 1; + } + } } + (span_frequencies, symbols) } -#[derive(Clone, Copy, PartialEq, Eq, Hash)] -pub(crate) struct SpanStorage96 { - offset: u32, - len: u32, - parts: u32, +#[derive(Clone, Copy)] +struct BufferReader<'a> { + #[cfg(all(debug_assertions, not(miri)))] + buffer: &'a [u8], + #[cfg(not(all(debug_assertions, not(miri))))] + ptr: *const u8, + #[cfg(not(all(debug_assertions, not(miri))))] + _marker: stdx::variance::PhantomCovariantLifetime<'a>, } -impl SpanStorage for SpanStorage96 { +impl<'a> BufferReader<'a> { #[inline] - fn can_hold(_text_range: TextRange, _span_parts_index: usize) -> bool { - true + fn start_end(slice: &'a [u8]) -> (Self, Self) { + #[cfg(all(debug_assertions, not(miri)))] + { + let start = Self { buffer: slice }; + let end = Self { buffer: &slice[slice.len()..] }; + (start, end) + } + #[cfg(not(all(debug_assertions, not(miri))))] + { + let ptrs = slice.as_ptr_range(); + let start = Self { + ptr: ptrs.start, + #[cfg(not(all(debug_assertions, not(miri))))] + _marker: stdx::variance::PhantomCovariantLifetime::new(), + }; + let end = Self { + ptr: ptrs.end, + #[cfg(not(all(debug_assertions, not(miri))))] + _marker: stdx::variance::PhantomCovariantLifetime::new(), + }; + (start, end) + } } - #[inline] - fn new(text_range: TextRange, span_parts_index: usize) -> Self { - let offset = u32::from(text_range.start()); - let len = u32::from(text_range.len()); - let span_parts_index = span_parts_index as u32; - - Self { offset, len, parts: span_parts_index } + #[cfg_attr(not(all(debug_assertions, not(miri))), inline(always))] + unsafe fn read(&mut self) -> T { + #[cfg(all(debug_assertions, not(miri)))] + { + let read_at = self.buffer.split_off(..size_of::()).unwrap(); + T::read(read_at) + } + #[cfg(not(all(debug_assertions, not(miri))))] + unsafe { + let result = self.ptr.cast::().read_unaligned(); + self.ptr = self.ptr.add(size_of::()); + result + } } - #[inline] - fn text_range(&self) -> TextRange { - let offset = TextSize::new(self.offset); - let len = TextSize::new(self.len); - TextRange::at(offset, len) + #[cfg_attr(not(all(debug_assertions, not(miri))), inline(always))] + unsafe fn skip(&mut self, amount: usize) { + #[cfg(all(debug_assertions, not(miri)))] + { + self.buffer = &self.buffer[amount..]; + } + #[cfg(not(all(debug_assertions, not(miri))))] + unsafe { + self.ptr = self.ptr.add(amount); + } } +} +impl PartialEq for BufferReader<'_> { #[inline] - fn span_parts_index(&self) -> usize { - self.parts as usize + fn eq(&self, other: &Self) -> bool { + #[cfg(all(debug_assertions, not(miri)))] + let self_ptr = self.buffer.as_ptr(); + #[cfg(not(all(debug_assertions, not(miri))))] + let self_ptr = self.ptr; + #[cfg(all(debug_assertions, not(miri)))] + let other_ptr = other.buffer.as_ptr(); + #[cfg(not(all(debug_assertions, not(miri))))] + let other_ptr = other.ptr; + + self_ptr == other_ptr } } -impl fmt::Debug for SpanStorage96 { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - f.debug_struct("SpanStorage96") - .field("text_range", &self.text_range()) - .field("span_parts_index", &self.span_parts_index()) - .finish() - } +struct InlineSpanParts { + span_parts_index: usize, + text_range: TextRange, } -impl From for SpanStorage96 { - #[inline] - fn from(value: SpanStorage32) -> Self { - SpanStorage96::new(value.text_range(), value.span_parts_index()) +#[inline] +unsafe fn decode_span( + ptr: BufferReader<'_>, + mut first_u32: u32, +) -> (BufferReader<'_>, InlineSpanParts) { + first_u32 >>= 2; + let span_parts_index = first_u32 & n_bits_mask(4); + let len = (first_u32 >> 4) & n_bits_mask(8); + let extends_to_next = (first_u32 & (1 << (4 + 8))) != 0; + let offset = first_u32 >> (4 + 8 + 1); + if !extends_to_next { + let result = InlineSpanParts { + span_parts_index: span_parts_index as usize, + text_range: TextRange::at(TextSize::new(offset), TextSize::new(len)), + }; + (ptr, result) + } else { + unsafe { decode_extended_span(ptr, span_parts_index, len, offset) } } } -impl From for SpanStorage96 { - #[inline] - fn from(value: SpanStorage64) -> Self { - SpanStorage96::new(value.text_range(), value.span_parts_index()) +#[cold] +unsafe fn decode_extended_span( + mut ptr: BufferReader<'_>, + mut span_parts_index: u32, + mut len: u32, + mut offset: u32, +) -> (BufferReader<'_>, InlineSpanParts) { + unsafe { + let mut second_u32 = ptr.read::(); + let extends_to_next = (second_u32 & 0b1) != 0; + second_u32 >>= 1; + if extends_to_next { + let third_u32 = ptr.read::(); + let u64 = u64::from(third_u32) | (u64::from(second_u32) << u32::BITS); + let rest_span_parts_index = (u64 & n_bits_mask_u64(24)) as u32; + span_parts_index |= rest_span_parts_index << 4; + let rest_len = ((u64 >> 24) & n_bits_mask_u64(24)) as u32; + len |= rest_len << 8; + let rest_offset = (u64 >> (24 + 24)) as u32; + offset |= rest_offset << 17; + } else { + let rest_span_parts_index = second_u32 & n_bits_mask(11); + span_parts_index |= rest_span_parts_index << 4; + let rest_len = (second_u32 >> 11) & n_bits_mask(10); + len |= rest_len << 8; + let rest_offset = second_u32 >> (11 + 10); + offset |= rest_offset << 17; + }; + let result = InlineSpanParts { + span_parts_index: span_parts_index as usize, + text_range: TextRange::at(TextSize::new(offset), TextSize::new(len)), + }; + (ptr, result) } } -// We don't use structs or enum nesting here to save padding. -#[derive(Debug, Clone, PartialEq, Eq, Hash)] -pub(crate) enum TokenTree { - Literal { text_and_suffix: Symbol, span: S, kind: LitKind, suffix_len: u8 }, - Punct { char: char, spacing: Spacing, span: S }, - Ident { sym: Symbol, span: S, is_raw: IdentIsRaw }, - Subtree { len: u32, delim_kind: DelimiterKind, open_span: S, close_span: S }, +// FIXME: It'll probably be better to ensure this ourselves via a `#[repr(C, align(4))]` wrapper, even though practically +// this holds for all 32- and 64-bit targets (Rust does not guarantee this). +const _: () = assert!(align_of::<*const *const str>() >= 4); // Needed for the tagging of idents. + +unsafe fn decode_symbol<'a>( + mut ptr: BufferReader<'a>, + first_byte: u8, + symbols: &[Symbol], +) -> (BufferReader<'a>, Symbol) { + let mut symbol_idx = u32::from(first_byte) >> 1; + let extends_to_next = symbol_idx & 0b1 == 0b1; + symbol_idx >>= 1; + if extends_to_next { + let mut second_byte = unsafe { ptr.read::() }; + let rest_bytes = + if second_byte & 0b1 == 0b1 { u32::from(unsafe { ptr.read::() }) } else { 0 }; + second_byte >>= 1; + symbol_idx |= u32::from(second_byte) << 6; + symbol_idx |= rest_bytes << 13; + } + (ptr, symbols[symbol_idx as usize].clone()) } -impl TokenTree { - #[inline] - pub(crate) fn first_span(&self) -> &S { - match self { - TokenTree::Literal { span, .. } => span, - TokenTree::Punct { span, .. } => span, - TokenTree::Ident { span, .. } => span, - TokenTree::Subtree { open_span, .. } => open_span, +/// We need `MaybeUninit` to preserve provenance. +/// +/// The returned `u32` is the length of the children *in bytes*, if we read a subtree. Otherwise it's zero. +unsafe fn decode<'a>( + mut ptr: BufferReader<'a>, + compressed: &[CompressedSpanPart], + symbols: &[Symbol], +) -> (BufferReader<'a>, TokenTree, u32) { + unsafe { + let span_and_extra = ptr.read::(); + let span; + (ptr, span) = decode_span(ptr, span_and_extra); + let span = compressed[span.span_parts_index].recombine(span.text_range); + + if span_and_extra & 0b1 == 0b1 { + // An ASCII punct. + let char_and_half_spacing = ptr.read::(); + + let spacing = (span_and_extra & 0b10) | (u32::from(char_and_half_spacing) & 0b1); + let spacing = transmute::(spacing as u8); + + let char = char::from(char_and_half_spacing >> 1); + + return (ptr, TokenTree::Leaf(Leaf::Punct(Punct { char, spacing, span })), 0); + } else if span_and_extra & 0b10 == 0b10 { + // An ident. + let symbol_first_byte = ptr.read::(); + let is_raw = symbol_first_byte & 0b1; + let symbol; + (ptr, symbol) = decode_symbol(ptr, symbol_first_byte, symbols); + let is_raw = transmute::(is_raw); + + return (ptr, TokenTree::Leaf(Leaf::Ident(Ident { sym: symbol, span, is_raw })), 0); } - } - #[inline] - pub(crate) fn last_span(&self) -> &S { - match self { - TokenTree::Literal { span, .. } => span, - TokenTree::Punct { span, .. } => span, - TokenTree::Ident { span, .. } => span, - TokenTree::Subtree { close_span, .. } => close_span, - } - } + let mut children_byte_len = 0; + + let control_byte = u32::from(ptr.read::()); + let control_byte_extra_data = control_byte >> 3; + let result = match control_byte & 0b111 { + 0b000 => { + // Subtree, format 1: + // - Same span_parts_index for open and close span. + // - Subtree length is a u8. + // - Subtree length in bytes is a u8. + // - The length of both the open and close span is 1 - so we use the already-parsed length for the open span + // for other things (we can only assume it has 8 bits available, the minimum format for a length). + // - The offset between the open span's start and the close span's start is stored in 11 bits. + let kind = transmute::((control_byte_extra_data & 0b11) as u8); + let mut open_span = span; + let mut close_span_offset_from_open = u32::from(span.range.len()); + open_span.range = TextRange::at(open_span.range.start(), TextSize::new(1)); + close_span_offset_from_open |= (control_byte_extra_data >> 2) << 8; + let close_span = Span { + range: open_span.range + TextSize::new(close_span_offset_from_open), + anchor: open_span.anchor, + ctx: open_span.ctx, + }; + let len = u32::from(ptr.read::()); + children_byte_len = u32::from(ptr.read::()); - #[inline] - pub(crate) fn to_api(&self, span_parts: &[CompressedSpanPart]) -> crate::TokenTree { - match self { - TokenTree::Literal { text_and_suffix, span, kind, suffix_len } => { - crate::TokenTree::Leaf(crate::Leaf::Literal(crate::Literal { - text_and_suffix: text_and_suffix.clone(), - span: span.span(span_parts), - kind: *kind, - suffix_len: *suffix_len, - })) + TokenTree::Subtree(Subtree { + delimiter: Delimiter { open: open_span, close: close_span, kind }, + len, + }) } - TokenTree::Punct { char, spacing, span } => { - crate::TokenTree::Leaf(crate::Leaf::Punct(crate::Punct { - char: *char, - spacing: *spacing, - span: span.span(span_parts), - })) + 0b001 => { + // Subtree, format 2: + // - Same span_parts_index for open and close span. + // - Subtree length is a u8. + // - Subtree length in bytes is a u8. + // - The open and close span have the same length. This covers many cases because most cases either give + // both length 1 (the brackets themselves) or the same span (usually, the whole range they encompass). + // - The offset between the open span's start and the close span's start is stored in 3 bits. + let kind = transmute::((control_byte_extra_data & 0b11) as u8); + let open_span = span; + let close_span_offset_from_open = control_byte_extra_data >> 2; + let close_span = Span { + range: open_span.range + TextSize::new(close_span_offset_from_open), + anchor: open_span.anchor, + ctx: open_span.ctx, + }; + let len = u32::from(ptr.read::()); + children_byte_len = u32::from(ptr.read::()); + + TokenTree::Subtree(Subtree { + delimiter: Delimiter { open: open_span, close: close_span, kind }, + len, + }) } - TokenTree::Ident { sym, span, is_raw } => { - crate::TokenTree::Leaf(crate::Leaf::Ident(crate::Ident { - sym: sym.clone(), - span: span.span(span_parts), - is_raw: *is_raw, - })) + 0b010 => { + // Subtree, format 3: + // - Subtree length is a u8. + // - Subtree length in bytes is 12 bits. + // - The open and close span have the same length. + // - The offset between the open span's start and the close span's start is stored in 8 bits. + // - The close span's span_parts_index is stored in 7 bits. + // This is less efficient than formats 1 and 2 (requires two more bytes), but more efficient than the general format. + let kind = transmute::((control_byte_extra_data & 0b11) as u8); + let open_span = span; + let close_span_offset_from_open = u32::from(ptr.read::()); + let len = u32::from(ptr.read::()); + children_byte_len = u32::from(ptr.read::()); + let mut close_span_parts_index = control_byte_extra_data >> 2; + close_span_parts_index |= (children_byte_len & 0b1111) << 3; + children_byte_len >>= 4; + let close_span = compressed[close_span_parts_index as usize] + .recombine(open_span.range + TextSize::new(close_span_offset_from_open)); + + TokenTree::Subtree(Subtree { + delimiter: Delimiter { open: open_span, close: close_span, kind }, + len, + }) } - TokenTree::Subtree { len, delim_kind, open_span, close_span } => { - crate::TokenTree::Subtree(crate::Subtree { - delimiter: crate::Delimiter { - open: open_span.span(span_parts), - close: close_span.span(span_parts), - kind: *delim_kind, - }, - len: *len, + 0b101 => { + cold_path(); + + // Subtree, format 4 - most general format: subtree length is u32, subtree length in bytes is u32, full decoded + // span for close span, delimiter kind is 5 bits (not needed now, will be needed when we have different kinds of + // invisible delimiters). + let kind = transmute::(control_byte_extra_data as u8); + let open_span = span; + let close_span_start = ptr.read::(); + let close_span; + (ptr, close_span) = decode_span(ptr, close_span_start); + let close_span = + compressed[close_span.span_parts_index].recombine(close_span.text_range); + let len = ptr.read::(); + children_byte_len = ptr.read::(); + + TokenTree::Subtree(Subtree { + delimiter: Delimiter { open: open_span, close: close_span, kind }, + len, }) } - } - } + 0b110 => { + cold_path(); - #[inline] - fn convert>(self) -> TokenTree { - match self { - TokenTree::Literal { text_and_suffix, span, kind, suffix_len } => { - TokenTree::Literal { text_and_suffix, span: span.into(), kind, suffix_len } + // Non-ASCII punct. Extremely rare but technically possible. + let spacing = transmute::(control_byte_extra_data as u8); + let char = ptr.read::(); + + TokenTree::Leaf(Leaf::Punct(Punct { char, spacing, span })) } - TokenTree::Punct { char, spacing, span } => { - TokenTree::Punct { char, spacing, span: span.into() } + 0b011 | 0b100 => { + // Literal, format 1: the 6 bits remaining from `control_byte` decide the kind and the suffix len from a constant set + // (of the most common). + let control_byte_extra_data = control_byte >> 2; + let text_and_suffix_first_byte = ptr.read::(); + let text_and_suffix; + (ptr, text_and_suffix) = decode_symbol(ptr, text_and_suffix_first_byte, symbols); + let kind = match control_byte_extra_data & 0b11 { + 0b00 => LitKind::Str, + 0b01 => LitKind::StrRaw(0), + 0b10 => LitKind::StrRaw(1), + 0b11 => LitKind::Integer, + _ => unreachable!(), + }; + let suffix_len = (control_byte_extra_data >> 2) as u8; + + TokenTree::Leaf(Leaf::Literal(Literal { text_and_suffix, span, kind, suffix_len })) } - TokenTree::Ident { sym, span, is_raw } => { - TokenTree::Ident { sym, span: span.into(), is_raw } + 0b111 => { + cold_path(); + + // Literal, format 2: most general format. + let kind = match control_byte_extra_data { + 0 => LitKind::Byte, + 1 => LitKind::Char, + 2 => LitKind::Integer, + 3 => LitKind::Float, + 4 => LitKind::Str, + 5 => LitKind::StrRaw(ptr.read::()), + 6 => LitKind::ByteStr, + 7 => LitKind::ByteStrRaw(ptr.read::()), + 8 => LitKind::CStr, + 9 => LitKind::CStrRaw(ptr.read::()), + 10 => LitKind::Err(()), + _ => unreachable!(), + }; + let suffix_len = ptr.read::(); + let text_and_suffix_first_byte = ptr.read::(); + let text_and_suffix; + (ptr, text_and_suffix) = decode_symbol(ptr, text_and_suffix_first_byte, symbols); + + TokenTree::Leaf(Leaf::Literal(Literal { text_and_suffix, span, kind, suffix_len })) } - TokenTree::Subtree { len, delim_kind, open_span, close_span } => TokenTree::Subtree { - len, - delim_kind, - open_span: open_span.into(), - close_span: close_span.into(), - }, - } + _ => unreachable!(), + }; + (ptr, result, children_byte_len) } } -// This is used a lot, make sure it doesn't grow unintentionally. -const _: () = { - assert!(size_of::>() == 16); - assert!(size_of::>() == 24); - assert!(size_of::>() == 32); -}; +#[derive(Clone, Copy)] +pub(crate) struct TokenTreesSlice<'a> { + current: BufferReader<'a>, + end: BufferReader<'a>, + span_parts: &'a [CompressedSpanPart], + symbols: &'a [Symbol], +} -#[rust_analyzer::macro_style(braces)] -macro_rules! dispatch { - ( - match $scrutinee:expr => $tt:ident => $body:expr - ) => { - match $scrutinee { - TopSubtreeRepr::SpanStorage32($tt) => $body, - TopSubtreeRepr::SpanStorage64($tt) => $body, - TopSubtreeRepr::SpanStorage96($tt) => $body, +unsafe impl Send for TokenTreesSlice<'_> {} +unsafe impl Sync for TokenTreesSlice<'_> {} + +impl<'a> TokenTreesSlice<'a> { + #[inline] + fn new(top_subtree: &'a TopSubtree) -> Self { + let (current, end) = BufferReader::start_end(&top_subtree.buffer); + Self { current, end, span_parts: &top_subtree.span_parts, symbols: &top_subtree.symbols } + } + + #[inline] + pub(crate) fn empty() -> Self { + let (current, end) = BufferReader::start_end(&[]); + Self { current, end, span_parts: &[], symbols: &[] } + } + + pub(crate) fn advance(&mut self) -> Option { + if self.current == self.end { + return None; } - }; -} -#[derive(Debug, Clone, PartialEq, Eq, Hash)] -pub(crate) enum TopSubtreeRepr { - SpanStorage32(Box<[TokenTree]>), - SpanStorage64(Box<[TokenTree]>), - SpanStorage96(Box<[TokenTree]>), + let (new_current, token_tree, _children_byte_len) = + unsafe { decode(self.current, self.span_parts, self.symbols) }; + self.current = new_current; + Some(token_tree) + } + + /// This is like `advance()`, but when encountering a subtree, it changes `self` to skip it and returns a + /// slice into it (what `advance()` would have done to `self`). + pub(crate) fn advance_skip_subtree(&mut self) -> Option<(TokenTree, TokenTreesSlice<'a>)> { + if self.current == self.end { + return None; + } + + let (new_current, token_tree, children_byte_len) = + unsafe { decode(self.current, self.span_parts, self.symbols) }; + self.current = new_current; + let subtree_slice = *self; + unsafe { self.current.skip(children_byte_len as usize) }; + Some((token_tree, subtree_slice)) + } + + pub(crate) fn iter(mut self) -> impl Iterator { + std::iter::from_fn(move || self.advance()) + } } #[derive(Clone, PartialEq, Eq, Hash)] pub struct TopSubtree { - repr: TopSubtreeRepr, + buffer: Box<[u8]>, + /// The last two are the top subtree's open and close span, in this order. span_parts: Box<[CompressedSpanPart]>, + symbols: Box<[Symbol]>, + len: usize, } impl TopSubtree { pub fn empty(span: DelimSpan) -> Self { - Self { - repr: TopSubtreeRepr::SpanStorage96(Box::new([TokenTree::Subtree { + encode_all( + vec![TokenTree::Subtree(Subtree { + delimiter: Delimiter::invisible_delim_spanned(span), len: 0, - delim_kind: DelimiterKind::Invisible, - open_span: SpanStorage96::new(span.open.range, 0), - close_span: SpanStorage96::new(span.close.range, 1), - }])), - span_parts: Box::new([ - CompressedSpanPart::from_span(&span.open), - CompressedSpanPart::from_span(&span.close), - ]), - } - } - - pub fn invisible_from_leaves( - delim_span: Span, - leaves: [crate::Leaf; N], - ) -> Self { - let mut builder = TopSubtreeBuilder::new(crate::Delimiter::invisible_spanned(delim_span)); - builder.extend(leaves); - builder.build() + })] + .into_iter(), + FxHashMap::default(), + FxHashMap::default(), + ) } - pub fn from_token_trees(delimiter: crate::Delimiter, token_trees: TokenTreesView<'_>) -> Self { + pub fn invisible_from_leaves(delim_span: Span, leaves: [Leaf; N]) -> Self { + Self::from_serialized( + std::iter::chain( + [TokenTree::Subtree(Subtree { + delimiter: Delimiter::invisible_spanned(delim_span), + len: leaves.len() as u32, + })], + leaves.into_iter().map(TokenTree::Leaf), + ) + .collect(), + ) + } + + pub fn from_token_trees(delimiter: Delimiter, token_trees: TokenTreesView<'_>) -> Self { let mut builder = TopSubtreeBuilder::new(delimiter); builder.extend_with_tt(token_trees); builder.build() } - pub fn from_serialized(tt: Vec) -> Self { - let mut tt = tt.into_iter(); - let Some(crate::TokenTree::Subtree(top_subtree)) = tt.next() else { - panic!("first must always come the top subtree") - }; - let mut builder = TopSubtreeBuilder::new(top_subtree.delimiter); - for tt in tt { - builder.push_token_tree(tt); - } - builder.build() + pub fn from_serialized(tts: Vec) -> Self { + let (span_frequencies, symbols) = compute_span_frequencies_and_symbols(&tts[1..]); // Do not include the top subtree. + encode_all(tts.into_iter(), span_frequencies, symbols) } pub fn from_subtree(subtree: SubtreeView<'_>) -> Self { @@ -429,121 +1131,44 @@ impl TopSubtree { } pub fn view(&self) -> SubtreeView<'_> { - let repr = match &self.repr { - TopSubtreeRepr::SpanStorage32(token_trees) => { - TokenTreesReprRef::SpanStorage32(token_trees) - } - TopSubtreeRepr::SpanStorage64(token_trees) => { - TokenTreesReprRef::SpanStorage64(token_trees) - } - TopSubtreeRepr::SpanStorage96(token_trees) => { - TokenTreesReprRef::SpanStorage96(token_trees) - } - }; - SubtreeView(TokenTreesView { repr, span_parts: &self.span_parts }) + let slice = TokenTreesSlice::new(self); + SubtreeView(TokenTreesView { slice, len: self.len }) } pub fn iter(&self) -> TtIter<'_> { self.view().iter() } - pub fn top_subtree(&self) -> crate::Subtree { + pub fn top_subtree(&self) -> Subtree { self.view().top_subtree() } pub fn set_top_subtree_delimiter_kind(&mut self, kind: DelimiterKind) { - dispatch! { - match &mut self.repr => tt => { - let TokenTree::Subtree { delim_kind, .. } = &mut tt[0] else { - unreachable!("the first token tree is always the top subtree"); - }; - *delim_kind = kind; - } - } - } - - fn ensure_can_hold(&mut self, range: TextRange) { - fn can_hold(_: &[TokenTree], range: TextRange) -> bool { - S::can_hold(range, 0) - } - let can_hold = dispatch! { - match &self.repr => tt => can_hold(tt, range) - }; - if can_hold { - return; - } - - // Otherwise, we do something very junky: recreate the entire tree. Hopefully this should be rare. - let mut builder = TopSubtreeBuilder::new(self.top_subtree().delimiter); - builder.extend_with_tt(self.token_trees()); - builder.ensure_can_hold(range, 0); - *self = builder.build(); + change_root_delimiter(&mut self.buffer, kind); } pub fn set_top_subtree_delimiter_span(&mut self, span: DelimSpan) { - self.ensure_can_hold(span.open.range); - self.ensure_can_hold(span.close.range); - fn do_it(tt: &mut [TokenTree], span: DelimSpan) { - let TokenTree::Subtree { open_span, close_span, .. } = &mut tt[0] else { - unreachable!() - }; - *open_span = S::new(span.open.range, 0); - *close_span = S::new(span.close.range, 1); - } - dispatch! { - match &mut self.repr => tt => do_it(tt, span) - } - self.span_parts[0] = CompressedSpanPart::from_span(&span.open); - self.span_parts[1] = CompressedSpanPart::from_span(&span.close); - } - - /// Note: this cannot change spans. - pub fn set_token(&mut self, idx: usize, leaf: crate::Leaf) { - fn do_it( - tt: &mut [TokenTree], - idx: usize, - span_parts: &[CompressedSpanPart], - leaf: crate::Leaf, - ) { - assert!( - !matches!(tt[idx], TokenTree::Subtree { .. }), - "`TopSubtree::set_token()` must be called on a leaf" - ); - let existing_span_compressed = *tt[idx].first_span(); - let existing_span = existing_span_compressed.span(span_parts); - assert_eq!( - *leaf.span(), - existing_span, - "`TopSubtree::set_token()` cannot change spans" + let open_span_idx = self.span_parts.len() - 2; + let close_span_idx = open_span_idx + 1; + unsafe { + change_root_spans( + &mut self.buffer, + open_span_idx as u32, + close_span_idx as u32, + span.open.range, + span.close.range, ); - match leaf { - crate::Leaf::Literal(leaf) => { - tt[idx] = TokenTree::Literal { - text_and_suffix: leaf.text_and_suffix, - span: existing_span_compressed, - kind: leaf.kind, - suffix_len: leaf.suffix_len, - } - } - crate::Leaf::Punct(leaf) => { - tt[idx] = TokenTree::Punct { - char: leaf.char, - spacing: leaf.spacing, - span: existing_span_compressed, - } - } - crate::Leaf::Ident(leaf) => { - tt[idx] = TokenTree::Ident { - sym: leaf.sym, - span: existing_span_compressed, - is_raw: leaf.is_raw, - } - } - } - } - dispatch! { - match &mut self.repr => tt => do_it(tt, idx, &self.span_parts, leaf) } + self.span_parts[open_span_idx] = CompressedSpanPart::from_span(&span.open); + self.span_parts[close_span_idx] = CompressedSpanPart::from_span(&span.close); + } + + /// **Warning**: This is very expensive, this rebuilds the whole tree. Avoid using this if you can. + pub fn set_token(&mut self, idx: usize, leaf: Leaf) { + let mut tts = TokenTreesSlice::new(self).iter().collect::>(); + assert_matches!(tts[idx], TokenTree::Leaf(_), "cannot change a subtree to a leaf"); + tts[idx] = leaf.into(); + *self = TopSubtree::from_serialized(tts); } pub fn token_trees(&self) -> TokenTreesView<'_> { @@ -561,368 +1186,172 @@ impl TopSubtree { } } -#[rust_analyzer::macro_style(braces)] -macro_rules! dispatch_builder { - ( - match $scrutinee:expr => $tt:ident => $body:expr - ) => { - match $scrutinee { - TopSubtreeBuilderRepr::SpanStorage32($tt) => $body, - TopSubtreeBuilderRepr::SpanStorage64($tt) => $body, - TopSubtreeBuilderRepr::SpanStorage96($tt) => $body, - } - }; -} - -#[derive(Debug, Clone, PartialEq, Eq, Hash)] -enum TopSubtreeBuilderRepr { - SpanStorage32(Vec>), - SpanStorage64(Vec>), - SpanStorage96(Vec>), -} - -type FxIndexSet = indexmap::IndexSet; - -/// In any tree, the first two subtree parts are reserved for the top subtree. -/// -/// We do it because `TopSubtree` exposes an API to modify the top subtree, therefore it's more convenient -/// this way, and it's unlikely to affect memory usage. -const RESERVED_SPAN_PARTS_LEN: usize = 2; - #[derive(Debug, Clone)] pub struct TopSubtreeBuilder { unclosed_subtree_indices: Vec, - token_trees: TopSubtreeBuilderRepr, - span_parts: FxIndexSet, + token_trees: Vec, + span_parts_frequencies: FxHashMap, last_closed_subtree: Option, - /// We need to keep those because they are not inside `span_parts`, see [`RESERVED_SPAN_PARTS_LEN`]. - top_subtree_spans: DelimSpan, + symbol_frequencies: FxHashMap, } impl TopSubtreeBuilder { - pub fn new(top_delimiter: crate::Delimiter) -> Self { - let mut result = Self { - unclosed_subtree_indices: Vec::new(), - token_trees: TopSubtreeBuilderRepr::SpanStorage32(Vec::new()), - span_parts: FxIndexSet::default(), - last_closed_subtree: None, - top_subtree_spans: top_delimiter.delim_span(), + fn insert_span(&mut self, span: &Span) { + *self.span_parts_frequencies.entry(CompressedSpanPart::from_span(span)).or_insert(0) += 1; + } + + fn remove_span(span_parts_frequencies: &mut FxHashMap, span: &Span) { + let hash_map::Entry::Occupied(mut entry) = + span_parts_frequencies.entry(CompressedSpanPart::from_span(span)) + else { + panic!("span not present"); }; - result.ensure_can_hold(top_delimiter.open.range, 0); - result.ensure_can_hold(top_delimiter.close.range, 1); - fn push_first(tt: &mut Vec>, top_delimiter: crate::Delimiter) { - tt.push(TokenTree::Subtree { - len: 0, - delim_kind: top_delimiter.kind, - open_span: S::new(top_delimiter.open.range, 0), - close_span: S::new(top_delimiter.close.range, 1), - }); + *entry.get_mut() -= 1; + if *entry.get() == 0 { + entry.remove(); } - dispatch_builder! { - match &mut result.token_trees => tt => push_first(tt, top_delimiter) - } - result } - fn span_part_index(&mut self, part: CompressedSpanPart) -> usize { - self.span_parts.insert_full(part).0 + RESERVED_SPAN_PARTS_LEN + fn insert_symbol(&mut self, symbol: Symbol) { + *self.symbol_frequencies.entry(symbol).or_insert(0) += 1; } - fn switch_repr>(repr: &mut Vec>) -> Vec> { - let repr = std::mem::take(repr); - repr.into_iter().map(|tt| tt.convert()).collect() + fn remove_symbol(symbol_frequencies: &mut FxHashMap, symbol: Symbol) { + let hash_map::Entry::Occupied(mut entry) = symbol_frequencies.entry(symbol) else { + panic!("span not present"); + }; + *entry.get_mut() -= 1; + if *entry.get() == 0 { + entry.remove(); + } } - /// Ensures we have a representation that can hold these values. - fn ensure_can_hold(&mut self, text_range: TextRange, span_parts_index: usize) { - match &mut self.token_trees { - TopSubtreeBuilderRepr::SpanStorage32(token_trees) => { - if SpanStorage32::can_hold(text_range, span_parts_index) { - // Can hold. - } else if SpanStorage64::can_hold(text_range, span_parts_index) { - self.token_trees = - TopSubtreeBuilderRepr::SpanStorage64(Self::switch_repr(token_trees)); - } else { - self.token_trees = - TopSubtreeBuilderRepr::SpanStorage96(Self::switch_repr(token_trees)); - } - } - TopSubtreeBuilderRepr::SpanStorage64(token_trees) => { - if SpanStorage64::can_hold(text_range, span_parts_index) { - // Can hold. - } else { - self.token_trees = - TopSubtreeBuilderRepr::SpanStorage96(Self::switch_repr(token_trees)); - } - } - TopSubtreeBuilderRepr::SpanStorage96(_) => { - // Can hold anything. - } - } + pub fn new(top_delimiter: Delimiter) -> Self { + let mut result = Self { + unclosed_subtree_indices: Vec::new(), + token_trees: Vec::new(), + span_parts_frequencies: FxHashMap::default(), + last_closed_subtree: None, + symbol_frequencies: FxHashMap::default(), + }; + // Do not insert the top delimiters, they have their own place because we sometimes need to change them. + result.token_trees.push(TokenTree::Subtree(Subtree { delimiter: top_delimiter, len: 0 })); + result } /// Not to be exposed, this assumes the subtree's children will be filled in immediately. - fn push_subtree(&mut self, subtree: crate::Subtree) { - let open_span_parts_index = - self.span_part_index(CompressedSpanPart::from_span(&subtree.delimiter.open)); - self.ensure_can_hold(subtree.delimiter.open.range, open_span_parts_index); - let close_span_parts_index = - self.span_part_index(CompressedSpanPart::from_span(&subtree.delimiter.close)); - self.ensure_can_hold(subtree.delimiter.close.range, close_span_parts_index); - fn do_it( - tt: &mut Vec>, - open_span_parts_index: usize, - close_span_parts_index: usize, - subtree: crate::Subtree, - ) { - let open_span = S::new(subtree.delimiter.open.range, open_span_parts_index); - let close_span = S::new(subtree.delimiter.close.range, close_span_parts_index); - tt.push(TokenTree::Subtree { - len: subtree.len, - delim_kind: subtree.delimiter.kind, - open_span, - close_span, - }); - } - dispatch_builder! { - match &mut self.token_trees => tt => do_it(tt, open_span_parts_index, close_span_parts_index, subtree) - } + fn push_subtree(&mut self, subtree: Subtree) { + self.insert_span(&subtree.delimiter.open); + self.insert_span(&subtree.delimiter.close); + self.token_trees.push(subtree.into()); } pub fn open(&mut self, delimiter_kind: DelimiterKind, open_span: Span) { - let span_parts_index = self.span_part_index(CompressedSpanPart::from_span(&open_span)); - self.ensure_can_hold(open_span.range, span_parts_index); - fn do_it( - token_trees: &mut Vec>, - delimiter_kind: DelimiterKind, - range: TextRange, - span_parts_index: usize, - ) -> usize { - let open_span = S::new(range, span_parts_index); - token_trees.push(TokenTree::Subtree { - len: 0, - delim_kind: delimiter_kind, - open_span, - close_span: open_span, // Will be overwritten on close. - }); - token_trees.len() - 1 - } - let subtree_idx = dispatch_builder! { - match &mut self.token_trees => tt => do_it(tt, delimiter_kind, open_span.range, span_parts_index) - }; + self.insert_span(&open_span); + let subtree_idx = self.token_trees.len(); + self.token_trees.push(TokenTree::Subtree(Subtree { + delimiter: Delimiter { open: open_span, close: open_span, kind: delimiter_kind }, + len: 0, // Will be overwritten on close. + })); self.unclosed_subtree_indices.push(subtree_idx); } pub fn close(&mut self, close_span: Span) { - let span_parts_index = self.span_part_index(CompressedSpanPart::from_span(&close_span)); - let range = close_span.range; - self.ensure_can_hold(range, span_parts_index); + self.insert_span(&close_span); let last_unclosed_index = self .unclosed_subtree_indices .pop() .expect("attempt to close a `tt::Subtree` when none is open"); - fn do_it( - token_trees: &mut [TokenTree], - last_unclosed_index: usize, - range: TextRange, - span_parts_index: usize, - ) { - let token_trees_len = token_trees.len(); - let TokenTree::Subtree { len, delim_kind: _, open_span: _, close_span } = - &mut token_trees[last_unclosed_index] - else { - unreachable!("unclosed token tree is always a subtree"); - }; - *len = (token_trees_len - last_unclosed_index - 1) as u32; - *close_span = S::new(range, span_parts_index); - } - dispatch_builder! { - match &mut self.token_trees => tt => do_it(tt, last_unclosed_index, range, span_parts_index) - } + let token_trees_len = self.token_trees.len(); + let TokenTree::Subtree(Subtree { delimiter: Delimiter { open: _, close, kind: _ }, len }) = + &mut self.token_trees[last_unclosed_index] + else { + unreachable!("unclosed token tree is always a subtree"); + }; + *len = (token_trees_len - last_unclosed_index - 1) as u32; + *close = close_span; self.last_closed_subtree = Some(last_unclosed_index); } /// You cannot call this consecutively, it will only work once after close. pub fn remove_last_subtree_if_invisible(&mut self) { let Some(last_subtree_idx) = self.last_closed_subtree else { return }; - fn do_it(tt: &mut Vec>, last_subtree_idx: usize) { - if let TokenTree::Subtree { delim_kind: DelimiterKind::Invisible, .. } = - tt[last_subtree_idx] - { - tt.remove(last_subtree_idx); - } - } - dispatch_builder! { - match &mut self.token_trees => tt => do_it(tt, last_subtree_idx) + if let TokenTree::Subtree(Subtree { + delimiter: Delimiter { kind: DelimiterKind::Invisible, .. }, + .. + }) = self.token_trees[last_subtree_idx] + { + self.token_trees.remove(last_subtree_idx); } self.last_closed_subtree = None; } - fn push_literal(&mut self, leaf: crate::Literal) { - let span_parts_index = self.span_part_index(CompressedSpanPart::from_span(&leaf.span)); - let range = leaf.span.range; - self.ensure_can_hold(range, span_parts_index); - fn do_it( - tt: &mut Vec>, - range: TextRange, - span_parts_index: usize, - leaf: crate::Literal, - ) { - tt.push(TokenTree::Literal { - text_and_suffix: leaf.text_and_suffix, - span: S::new(range, span_parts_index), - kind: leaf.kind, - suffix_len: leaf.suffix_len, - }) - } - dispatch_builder! { - match &mut self.token_trees => tt => do_it(tt, range, span_parts_index, leaf) - } - } - - fn push_punct(&mut self, leaf: crate::Punct) { - let span_parts_index = self.span_part_index(CompressedSpanPart::from_span(&leaf.span)); - let range = leaf.span.range; - self.ensure_can_hold(range, span_parts_index); - fn do_it( - tt: &mut Vec>, - range: TextRange, - span_parts_index: usize, - leaf: crate::Punct, - ) { - tt.push(TokenTree::Punct { - char: leaf.char, - spacing: leaf.spacing, - span: S::new(range, span_parts_index), - }) - } - dispatch_builder! { - match &mut self.token_trees => tt => do_it(tt, range, span_parts_index, leaf) - } - } - - fn push_ident(&mut self, leaf: crate::Ident) { - let span_parts_index = self.span_part_index(CompressedSpanPart::from_span(&leaf.span)); - let range = leaf.span.range; - self.ensure_can_hold(range, span_parts_index); - fn do_it( - tt: &mut Vec>, - range: TextRange, - span_parts_index: usize, - leaf: crate::Ident, - ) { - tt.push(TokenTree::Ident { - sym: leaf.sym, - span: S::new(range, span_parts_index), - is_raw: leaf.is_raw, - }) - } - dispatch_builder! { - match &mut self.token_trees => tt => do_it(tt, range, span_parts_index, leaf) - } - } - - pub fn push(&mut self, leaf: crate::Leaf) { - match leaf { - crate::Leaf::Literal(leaf) => self.push_literal(leaf), - crate::Leaf::Punct(leaf) => self.push_punct(leaf), - crate::Leaf::Ident(leaf) => self.push_ident(leaf), - } - } - - fn push_token_tree(&mut self, tt: crate::TokenTree) { + pub fn push(&mut self, leaf: Leaf) { + self.insert_span(leaf.span()); + if let Some(symbol) = leaf.symbol() { + self.insert_symbol(symbol.clone()); + } + self.token_trees.push(leaf.into()); + } + + fn push_token_tree(&mut self, tt: TokenTree) { match tt { - crate::TokenTree::Leaf(leaf) => self.push(leaf), - crate::TokenTree::Subtree(subtree) => self.push_subtree(subtree), + TokenTree::Leaf(leaf) => self.push(leaf), + TokenTree::Subtree(subtree) => self.push_subtree(subtree), } } - pub fn extend(&mut self, leaves: impl IntoIterator) { + pub fn extend(&mut self, leaves: impl IntoIterator) { leaves.into_iter().for_each(|leaf| self.push(leaf)); } pub fn extend_with_tt(&mut self, tt: TokenTreesView<'_>) { - fn do_it( - this: &mut TopSubtreeBuilder, - tt: &[TokenTree], - span_parts: &[CompressedSpanPart], - ) { - for tt in tt { - this.push_token_tree(tt.to_api(span_parts)); - } - } - dispatch_ref! { - match tt.repr => tt_repr => do_it(self, tt_repr, tt.span_parts) - } + tt.iter_flat_tokens().for_each(|tt| self.push_token_tree(tt)); } /// Like [`Self::extend_with_tt()`], but makes sure the new tokens will never be /// joint with whatever comes after them. pub fn extend_with_tt_alone(&mut self, tt: TokenTreesView<'_>) { self.extend_with_tt(tt); - fn do_it(tt: &mut [TokenTree]) { - if let Some(TokenTree::Punct { spacing, .. }) = tt.last_mut() { - *spacing = Spacing::Alone; - } - } - if !tt.is_empty() { - dispatch_builder! { - match &mut self.token_trees => tt => do_it(tt) - } + if !tt.is_empty() + && let Some(TokenTree::Leaf(Leaf::Punct(Punct { spacing, .. }))) = + self.token_trees.last_mut() + { + *spacing = Spacing::Alone; } } pub fn expected_delimiters(&self) -> impl Iterator { self.unclosed_subtree_indices.iter().rev().map(|&subtree_idx| { - dispatch_builder! { - match &self.token_trees => tt => { - let TokenTree::Subtree { delim_kind, .. } = tt[subtree_idx] else { - unreachable!("unclosed token tree is always a subtree") - }; - delim_kind - } - } + let TokenTree::Subtree(Subtree { delimiter, .. }) = self.token_trees[subtree_idx] + else { + unreachable!("unclosed token tree is always a subtree") + }; + delimiter.kind }) } /// Builds, and remove the top subtree if it has only one subtree child. pub fn build_skip_top_subtree(mut self) -> TopSubtree { - fn remove_first_if_needed( - tt: &mut Vec>, - top_delim_span: &mut DelimSpan, - span_parts: &FxIndexSet, - ) { - let tt_len = tt.len(); - let Some(TokenTree::Subtree { len, open_span, close_span, .. }) = tt.get_mut(1) else { - return; - }; - if (*len as usize) != (tt_len - 2) { - // Subtree does not cover the whole tree (minus 2; itself, and the top span). - return; - } - - // Now we need to adjust the spans, because we assume that the first two spans are always reserved. - let top_open_span = span_parts - .get_index(open_span.span_parts_index() - RESERVED_SPAN_PARTS_LEN) - .unwrap() - .recombine(open_span.text_range()); - let top_close_span = span_parts - .get_index(close_span.span_parts_index() - RESERVED_SPAN_PARTS_LEN) - .unwrap() - .recombine(close_span.text_range()); - *top_delim_span = DelimSpan { open: top_open_span, close: top_close_span }; - // Can't remove the top spans from the map, as maybe they're used by other things as well. - // Now we need to reencode the spans, because their parts index changed: - *open_span = S::new(open_span.text_range(), 0); - *close_span = S::new(close_span.text_range(), 1); - - tt.remove(0); - } - dispatch_builder! { - match &mut self.token_trees => tt => remove_first_if_needed(tt, &mut self.top_subtree_spans, &self.span_parts) + assert!( + self.unclosed_subtree_indices.is_empty(), + "attempt to build an unbalanced `TopSubtreeBuilder`" + ); + let tt_len = self.token_trees.len(); + if let Some(&TokenTree::Subtree(Subtree { len, delimiter })) = self.token_trees.get(1) + && (len as usize) == (tt_len - 2) + { + // The top subtree's delimiters should not be included. + Self::remove_span(&mut self.span_parts_frequencies, &delimiter.open); + Self::remove_span(&mut self.span_parts_frequencies, &delimiter.close); + + let mut token_trees = self.token_trees.into_iter(); + token_trees.next(); // Remove the first subtree. + encode_all(token_trees, self.span_parts_frequencies, self.symbol_frequencies) + } else { + self.build() } - self.build() } pub fn build(mut self) -> TopSubtree { @@ -930,57 +1359,52 @@ impl TopSubtreeBuilder { self.unclosed_subtree_indices.is_empty(), "attempt to build an unbalanced `TopSubtreeBuilder`" ); - fn finish_top_len(tt: &mut [TokenTree]) { - let total_len = tt.len() as u32; - let TokenTree::Subtree { len, .. } = &mut tt[0] else { - unreachable!("first token tree is always a subtree"); - }; - *len = total_len - 1; - } - dispatch_builder! { - match &mut self.token_trees => tt => finish_top_len(tt) - } - - let span_parts = [ - CompressedSpanPart::from_span(&self.top_subtree_spans.open), - CompressedSpanPart::from_span(&self.top_subtree_spans.close), - ] - .into_iter() - .chain(self.span_parts.iter().copied()) - .collect(); - - let repr = match self.token_trees { - TopSubtreeBuilderRepr::SpanStorage32(tt) => { - TopSubtreeRepr::SpanStorage32(tt.into_boxed_slice()) - } - TopSubtreeBuilderRepr::SpanStorage64(tt) => { - TopSubtreeRepr::SpanStorage64(tt.into_boxed_slice()) - } - TopSubtreeBuilderRepr::SpanStorage96(tt) => { - TopSubtreeRepr::SpanStorage96(tt.into_boxed_slice()) - } + let tts_len = self.token_trees.len(); + let TokenTree::Subtree(top_subtree) = &mut self.token_trees[0] else { + panic!("first token tree must be a subtree"); }; - - TopSubtree { repr, span_parts } + top_subtree.len = (tts_len - 1).try_into().unwrap(); + encode_all( + self.token_trees.into_iter(), + self.span_parts_frequencies, + self.symbol_frequencies, + ) } - pub fn restore_point(&self) -> SubtreeBuilderRestorePoint { - let token_trees_len = dispatch_builder! { - match &self.token_trees => tt => tt.len() - }; + pub fn restore_point(&mut self) -> SubtreeBuilderRestorePoint { + // We reset the `last_closed_subtree`, since restoring from a restore point doesn't play well with removing the last subtree. + self.last_closed_subtree = None; SubtreeBuilderRestorePoint { unclosed_subtree_indices_len: self.unclosed_subtree_indices.len(), - token_trees_len, - last_closed_subtree: self.last_closed_subtree, + token_trees_len: self.token_trees.len(), } } pub fn restore(&mut self, restore_point: SubtreeBuilderRestorePoint) { - self.unclosed_subtree_indices.truncate(restore_point.unclosed_subtree_indices_len); - dispatch_builder! { - match &mut self.token_trees => tt => tt.truncate(restore_point.token_trees_len) + if restore_point.token_trees_len >= self.token_trees.len() { + // This means we restored twice, potentially with an earlier restore point first. + return; + } + + for tt in &self.token_trees[restore_point.token_trees_len..] { + match tt { + TokenTree::Leaf(leaf) => { + Self::remove_span(&mut self.span_parts_frequencies, leaf.span()); + + if let Some(symbol) = leaf.symbol() { + Self::remove_symbol(&mut self.symbol_frequencies, symbol.clone()); + } + } + TokenTree::Subtree(subtree) => { + Self::remove_span(&mut self.span_parts_frequencies, &subtree.delimiter.open); + Self::remove_span(&mut self.span_parts_frequencies, &subtree.delimiter.close); + } + } } - self.last_closed_subtree = restore_point.last_closed_subtree; + + self.unclosed_subtree_indices.truncate(restore_point.unclosed_subtree_indices_len); + self.token_trees.truncate(restore_point.token_trees_len); + self.last_closed_subtree = None; } } @@ -988,5 +1412,4 @@ impl TopSubtreeBuilder { pub struct SubtreeBuilderRestorePoint { unclosed_subtree_indices_len: usize, token_trees_len: usize, - last_closed_subtree: Option, } From 798009c502496ce5689184ee1846ed7408a7e6fc Mon Sep 17 00:00:00 2001 From: A4-Tacks Date: Mon, 17 Aug 2026 09:19:32 +0800 Subject: [PATCH 03/18] Add a flag do not parse rest arguments --- .../src/macro_expansion_tests/builtin_fn_macro.rs | 12 ++++++++++-- .../crates/hir-expand/src/builtin/fn_macro.rs | 15 ++++++++++----- 2 files changed, 20 insertions(+), 7 deletions(-) diff --git a/src/tools/rust-analyzer/crates/hir-def/src/macro_expansion_tests/builtin_fn_macro.rs b/src/tools/rust-analyzer/crates/hir-def/src/macro_expansion_tests/builtin_fn_macro.rs index d6ccf9ca51ab0..7120980dd30cf 100644 --- a/src/tools/rust-analyzer/crates/hir-def/src/macro_expansion_tests/builtin_fn_macro.rs +++ b/src/tools/rust-analyzer/crates/hir-def/src/macro_expansion_tests/builtin_fn_macro.rs @@ -160,13 +160,21 @@ fn test_option_env_expand() { #[rustc_builtin_macro] macro_rules! option_env {() => {}} -fn main() { option_env!("TEST_ENV_VAR"); } +fn main() { + option_env!("TEST_ENV_VAR"); + option_env!("TEST_ENV_VAR",); + option_env!("TEST_ENV_VAR", "invalid"); +} "#, expect![[r#" #[rustc_builtin_macro] macro_rules! option_env {() => {}} -fn main() { $crate::option::Option::None:: < &str>; } +fn main() { + $crate::option::Option::None:: < &str>; + $crate::option::Option::None:: < &str>; + /* error: unexpected input */; +} "#]], ); } diff --git a/src/tools/rust-analyzer/crates/hir-expand/src/builtin/fn_macro.rs b/src/tools/rust-analyzer/crates/hir-expand/src/builtin/fn_macro.rs index 7a162fcf4bcb7..a91a1b08b6963 100644 --- a/src/tools/rust-analyzer/crates/hir-expand/src/builtin/fn_macro.rs +++ b/src/tools/rust-analyzer/crates/hir-expand/src/builtin/fn_macro.rs @@ -766,7 +766,7 @@ fn relative_file( } } -fn parse_string(tt: &tt::TopSubtree) -> Result<(Symbol, Span), ExpandError> { +fn parse_string(tt: &tt::TopSubtree, rest: bool) -> Result<(Symbol, Span), ExpandError> { let expect_literal = |span| ExpandError::other(span, "expected string literal"); let mut tt = { let mut tt_iter = tt.iter(); @@ -778,6 +778,11 @@ fn parse_string(tt: &tt::TopSubtree) -> Result<(Symbol, Span), ExpandError> { Some(TtElement::Leaf(tt::Leaf::Punct(it))) if it.char == ',' => { // Tail comma // FIXME: Ignored like env!("NAME", "compile_error message") + if let Some(tt) = tt_iter.next() + && !rest + { + return Err(ExpandError::other(tt.first_span(), "unexpected input")); + } } Some(tt) => { return Err(ExpandError::other(tt.first_span(), "unexpected input")); @@ -846,7 +851,7 @@ pub fn include_input_to_file_id( arg_id: MacroCallId, arg: &tt::TopSubtree, ) -> Result { - let (s, span) = parse_string(arg)?; + let (s, span) = parse_string(arg, false)?; relative_file(db, arg_id, s.as_str(), false, span) } @@ -870,7 +875,7 @@ fn include_str_expand( tt: &tt::TopSubtree, call_site: Span, ) -> ExpandResult { - let (path, input_span) = match parse_string(tt) { + let (path, input_span) = match parse_string(tt, false) { Ok(it) => it, Err(e) => { return ExpandResult::new( @@ -908,7 +913,7 @@ fn env_expand( tt: &tt::TopSubtree, span: Span, ) -> ExpandResult { - let (key, span) = match parse_string(tt) { + let (key, span) = match parse_string(tt, true) { Ok(it) => it, Err(e) => { return ExpandResult::new( @@ -946,7 +951,7 @@ fn option_env_expand( tt: &tt::TopSubtree, call_site: Span, ) -> ExpandResult { - let (key, span) = match parse_string(tt) { + let (key, span) = match parse_string(tt, false) { Ok(it) => it, Err(e) => { return ExpandResult::new( From c6e5196a0bf667b32916820a61cc4eccbd9b5570 Mon Sep 17 00:00:00 2001 From: A4-Tacks Date: Mon, 17 Aug 2026 09:45:48 +0800 Subject: [PATCH 04/18] Rename 'rest' to 'allow_rest_args' --- .../rust-analyzer/crates/hir-expand/src/builtin/fn_macro.rs | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/tools/rust-analyzer/crates/hir-expand/src/builtin/fn_macro.rs b/src/tools/rust-analyzer/crates/hir-expand/src/builtin/fn_macro.rs index a91a1b08b6963..44579304d3646 100644 --- a/src/tools/rust-analyzer/crates/hir-expand/src/builtin/fn_macro.rs +++ b/src/tools/rust-analyzer/crates/hir-expand/src/builtin/fn_macro.rs @@ -766,7 +766,7 @@ fn relative_file( } } -fn parse_string(tt: &tt::TopSubtree, rest: bool) -> Result<(Symbol, Span), ExpandError> { +fn parse_string(tt: &tt::TopSubtree, allow_rest_args: bool) -> Result<(Symbol, Span), ExpandError> { let expect_literal = |span| ExpandError::other(span, "expected string literal"); let mut tt = { let mut tt_iter = tt.iter(); @@ -779,7 +779,7 @@ fn parse_string(tt: &tt::TopSubtree, rest: bool) -> Result<(Symbol, Span), Expan // Tail comma // FIXME: Ignored like env!("NAME", "compile_error message") if let Some(tt) = tt_iter.next() - && !rest + && !allow_rest_args { return Err(ExpandError::other(tt.first_span(), "unexpected input")); } From e5cb09e62bfd654113338bd8188a5551db39b03e Mon Sep 17 00:00:00 2001 From: kivancgnlp Date: Mon, 17 Aug 2026 03:38:59 +0000 Subject: [PATCH 05/18] ide-diagnostics: emit E0600 for unary operator on unsupported type Addresses the FIXME in `hir-ty/src/infer/op.rs` inside `infer_user_unop`, which previously silently discarded operator method resolution failures for `!x` and `-x` expressions. When the operand's type does not implement `std::ops::Not` (for `!`) or `std::ops::Neg` (for `-`), rust-analyzer now reports the same E0600 error that rustc produces: cannot apply unary operator `!` to type `Question` Wired through the standard inference diagnostic pipeline: new `InferenceDiagnostic::UnaryOperatorCannotBeApplied` variant in hir-ty, matching `UnaryOperatorCannotBeApplied` struct plus conversion in hir, and a handler in ide-diagnostics using `DiagnosticCode::RustcHardError("E0600")`. Filtering for unresolved / error-typed operands is done in `resolve_diagnostics()` (crates/hir-ty/src/infer/unify.rs) alongside the existing `references_non_lt_error()` filter chain for other diagnostics that carry a type. This keeps `infer_user_unop` free of callsite guards and lets the natural inference pipeline suppress spurious reports on incomplete code and on macro expansions that infer to `{unknown}`. The `unary_ops` region of `test-utils/src/minicore.rs` also gains builtin `Not` and `Neg` impls, mirroring how `add_impl!` provides them in the `add` region. Without these, the diagnostic test fixture would incorrectly flag `!true`, `!0i32` and similar builtin uses as errors, because `lookup_op_method` would find no impl in the minicore fixture even though real `core` has one. With the impls present, primitives resolve normally and only genuinely unsupported operators trigger the diagnostic. This also lets us correctly report `-1u32` as E0600, since real `core` does not implement `Neg` for unsigned integers. Because the new `not_impl!` / `neg_impl!` blocks live in a nested `region:builtin_impls` inside `region:unary_ops`, the new tests opt into both flags via `//- minicore: unary_ops, builtin_impls`. The existing `legacy_const_generics` test in `mismatched_arg_count` uses `-1i32` / `-1i8` inline and now needs the same directive so that `core::ops::Neg` is in scope for its operands. Minicore `region:eq` and `region:float_consts` now depend on `unary_ops, builtin_impls` so their smoke tests resolve `Not`/`Neg` without per-callsite guards. The `Clone for [T; 1]` impl inside `region:builtin_impls` uses `self[0]`, so it is scoped to a nested `region:index` and only compiles when `index` is also enabled. The `UnaryOp::Deref` case is left unchanged; it is already handled by the `CannotBeDereferenced` diagnostic (E0614) and `infer_user_unop` is never called for `Deref`. Part of rust-lang/rust-analyzer#22140. --- .../rust-analyzer/crates/hir-ty/src/infer.rs | 9 +- .../crates/hir-ty/src/infer/op.rs | 8 +- .../crates/hir-ty/src/infer/unify.rs | 1 + .../crates/hir/src/diagnostics.rs | 12 ++ .../src/handlers/mismatched_arg_count.rs | 1 + .../unary_operator_cannot_be_applied.rs | 158 ++++++++++++++++++ .../crates/ide-diagnostics/src/lib.rs | 2 + .../crates/test-utils/src/minicore.rs | 30 +++- 8 files changed, 216 insertions(+), 5 deletions(-) create mode 100644 src/tools/rust-analyzer/crates/ide-diagnostics/src/handlers/unary_operator_cannot_be_applied.rs diff --git a/src/tools/rust-analyzer/crates/hir-ty/src/infer.rs b/src/tools/rust-analyzer/crates/hir-ty/src/infer.rs index a719b364a872e..1b0eaffbdeaea 100644 --- a/src/tools/rust-analyzer/crates/hir-ty/src/infer.rs +++ b/src/tools/rust-analyzer/crates/hir-ty/src/infer.rs @@ -47,7 +47,7 @@ use hir_def::{ TupleFieldId, TupleId, VariantId, attrs::AttrFlags, expr_store::{Body, ExpressionStore, HygieneId, body::Param, path::Path}, - hir::{BindingId, ExprId, ExprOrPatId, ExprOrPatIdPacked, LabelId, PatId}, + hir::{BindingId, ExprId, ExprOrPatId, ExprOrPatIdPacked, LabelId, PatId, UnaryOp}, lang_item::LangItems, layout::Integer, resolver::{HasResolver, ResolveValueResult, Resolver, TypeNs, ValueNs}, @@ -434,6 +434,13 @@ pub enum InferenceDiagnostic { expr: ExprId, found: StoredTy, }, + UnaryOperatorCannotBeApplied { + #[type_visitable(ignore)] + expr: ExprId, + #[type_visitable(ignore)] + op: UnaryOp, + found: StoredTy, + }, MutRefInImmRefPat { #[type_visitable(ignore)] pat: PatId, diff --git a/src/tools/rust-analyzer/crates/hir-ty/src/infer/op.rs b/src/tools/rust-analyzer/crates/hir-ty/src/infer/op.rs index 5fd4e830fb172..16e62e9dace48 100644 --- a/src/tools/rust-analyzer/crates/hir-ty/src/infer/op.rs +++ b/src/tools/rust-analyzer/crates/hir-ty/src/infer/op.rs @@ -9,7 +9,7 @@ use syntax::ast::{ArithOp, BinaryOp, UnaryOp}; use tracing::debug; use crate::{ - Adjust, Adjustment, AutoBorrow, + Adjust, Adjustment, AutoBorrow, InferenceDiagnostic, infer::{AllowTwoPhase, AutoBorrowMutability, Expectation, InferenceContext, expr::ExprIsRead}, method_resolution::{MethodCallee, TreatNotYetDefinedOpaques}, next_solver::{ @@ -271,7 +271,11 @@ impl<'db> InferenceContext<'db> { method.sig.output() } Err(_errors) => { - // FIXME: Report diagnostic. + self.push_diagnostic(InferenceDiagnostic::UnaryOperatorCannotBeApplied { + expr: ex, + op, + found: operand_ty.store(), + }); self.types.types.error } } diff --git a/src/tools/rust-analyzer/crates/hir-ty/src/infer/unify.rs b/src/tools/rust-analyzer/crates/hir-ty/src/infer/unify.rs index 7b589efba2f63..8070ed8788977 100644 --- a/src/tools/rust-analyzer/crates/hir-ty/src/infer/unify.rs +++ b/src/tools/rust-analyzer/crates/hir-ty/src/infer/unify.rs @@ -586,6 +586,7 @@ pub(super) mod resolve_completely { | InferenceDiagnostic::CannotIndexInto { found: ty, .. } | InferenceDiagnostic::ExpectedFunction { found: ty, .. } | InferenceDiagnostic::ExpectedArrayOrSlicePat { found: ty, .. } + | InferenceDiagnostic::UnaryOperatorCannotBeApplied { found: ty, .. } | InferenceDiagnostic::UnresolvedField { receiver: ty, .. } | InferenceDiagnostic::UnresolvedMethodCall { receiver: ty, .. } = diagnostic && ty.as_ref().references_non_lt_error() diff --git a/src/tools/rust-analyzer/crates/hir/src/diagnostics.rs b/src/tools/rust-analyzer/crates/hir/src/diagnostics.rs index f6df18a6fb577..3144db5817272 100644 --- a/src/tools/rust-analyzer/crates/hir/src/diagnostics.rs +++ b/src/tools/rust-analyzer/crates/hir/src/diagnostics.rs @@ -105,6 +105,7 @@ diagnostics![AnyDiagnostic<'db> -> AwaitOutsideOfAsync, BreakOutsideOfLoop, CannotBeDereferenced<'db>, + UnaryOperatorCannotBeApplied<'db>, CannotImplicitlyDerefTraitObject<'db>, CannotIndexInto<'db>, CastToUnsized<'db>, @@ -338,6 +339,13 @@ pub struct CannotBeDereferenced<'db> { pub found: Type<'db>, } +#[derive(Debug)] +pub struct UnaryOperatorCannotBeApplied<'db> { + pub expr: InFile, + pub op: ast::UnaryOp, + pub found: Type<'db>, +} + #[derive(Debug)] pub struct MutRefInImmRefPat { pub pat: InFile, @@ -985,6 +993,10 @@ impl<'db> AnyDiagnostic<'db> { let expr = expr_syntax(*expr)?; CannotBeDereferenced { expr, found: new_ty(found.as_ref()) }.into() } + InferenceDiagnostic::UnaryOperatorCannotBeApplied { expr, op, found } => { + let expr = expr_syntax(*expr)?; + UnaryOperatorCannotBeApplied { expr, op: *op, found: new_ty(found.as_ref()) }.into() + } InferenceDiagnostic::MutRefInImmRefPat { pat } => { let pat = pat_syntax(*pat)?.map(Into::into); MutRefInImmRefPat { pat }.into() diff --git a/src/tools/rust-analyzer/crates/ide-diagnostics/src/handlers/mismatched_arg_count.rs b/src/tools/rust-analyzer/crates/ide-diagnostics/src/handlers/mismatched_arg_count.rs index fb9095e0f4bd2..a577353e27cf0 100644 --- a/src/tools/rust-analyzer/crates/ide-diagnostics/src/handlers/mismatched_arg_count.rs +++ b/src/tools/rust-analyzer/crates/ide-diagnostics/src/handlers/mismatched_arg_count.rs @@ -488,6 +488,7 @@ fn main() { fn legacy_const_generics() { check_diagnostics( r#" +//- minicore: unary_ops, builtin_impls #[rustc_legacy_const_generics(1, 3)] fn mixed( _a: u8, diff --git a/src/tools/rust-analyzer/crates/ide-diagnostics/src/handlers/unary_operator_cannot_be_applied.rs b/src/tools/rust-analyzer/crates/ide-diagnostics/src/handlers/unary_operator_cannot_be_applied.rs new file mode 100644 index 0000000000000..1dc779d43ac0f --- /dev/null +++ b/src/tools/rust-analyzer/crates/ide-diagnostics/src/handlers/unary_operator_cannot_be_applied.rs @@ -0,0 +1,158 @@ +use hir::HirDisplay; +use syntax::ast::UnaryOp; + +use crate::{Diagnostic, DiagnosticCode, DiagnosticsContext}; + +// Diagnostic: unary-operator-cannot-be-applied +// +// This diagnostic is triggered if a unary operator (`!` or `-`) is applied +// to a value whose type does not implement the corresponding trait +// (`Not` or `Neg`). +pub(crate) fn unary_operator_cannot_be_applied( + ctx: &DiagnosticsContext<'_, '_>, + d: &hir::UnaryOperatorCannotBeApplied<'_>, +) -> Diagnostic { + let op = match d.op { + UnaryOp::Not => "!", + UnaryOp::Neg => "-", + // `Deref` uses a different diagnostic (`CannotBeDereferenced`). + UnaryOp::Deref => "*", + }; + Diagnostic::new_with_syntax_node_ptr( + ctx, + DiagnosticCode::RustcHardError("E0600"), + format!( + "cannot apply unary operator `{op}` to type `{}`", + d.found.display(ctx.sema.db, ctx.display_target) + ), + d.expr.map(Into::into), + ) + .stable() +} + +#[cfg(test)] +mod tests { + use crate::tests::check_diagnostics; + + #[test] + fn not_on_enum() { + check_diagnostics( + r#" +//- minicore: unary_ops, builtin_impls +enum Question { Yes, No } + +fn f() { + let _ = !Question::Yes; + //^^^^^^^^^^^^^^ error: cannot apply unary operator `!` to type `Question` +} +"#, + ); + } + + #[test] + fn neg_on_struct() { + check_diagnostics( + r#" +//- minicore: unary_ops, builtin_impls +struct S; + +fn f() { + let _ = -S; + //^^ error: cannot apply unary operator `-` to type `S` +} +"#, + ); + } + + #[test] + fn allows_not_on_bool() { + check_diagnostics( + r#" +//- minicore: unary_ops, builtin_impls +fn f() { + let _ = !true; + let _ = !false; +} +"#, + ); + } + + #[test] + fn allows_not_on_integer() { + check_diagnostics( + r#" +//- minicore: unary_ops, builtin_impls +fn f() { + let _ = !0u32; + let _ = !0i32; +} +"#, + ); + } + + #[test] + fn allows_neg_on_numeric() { + check_diagnostics( + r#" +//- minicore: unary_ops, builtin_impls +fn f() { + let _ = -1i32; + let _ = -1.0f64; +} +"#, + ); + } + + #[test] + fn neg_on_unsigned() { + check_diagnostics( + r#" +//- minicore: unary_ops, builtin_impls +fn f() { + let _ = -1u32; + //^^^^^ error: cannot apply unary operator `-` to type `u32` +} +"#, + ); + } + + #[test] + fn allows_not_with_impl() { + check_diagnostics( + r#" +//- minicore: unary_ops, builtin_impls +struct Bar; +struct Foo; + +impl core::ops::Not for Bar { + type Output = Foo; + fn not(self) -> Foo { Foo } +} + +fn f() { + let _ = !Bar; +} +"#, + ); + } + + #[test] + fn allows_neg_with_impl() { + check_diagnostics( + r#" +//- minicore: unary_ops, builtin_impls +struct Bar; +struct Foo; + +impl core::ops::Neg for Bar { + type Output = Foo; + fn neg(self) -> Foo { Foo } +} + +fn f() { + let _ = -Bar; +} +"#, + ); + } +} diff --git a/src/tools/rust-analyzer/crates/ide-diagnostics/src/lib.rs b/src/tools/rust-analyzer/crates/ide-diagnostics/src/lib.rs index 8ba59edcbc871..5d816a8d41c3c 100644 --- a/src/tools/rust-analyzer/crates/ide-diagnostics/src/lib.rs +++ b/src/tools/rust-analyzer/crates/ide-diagnostics/src/lib.rs @@ -85,6 +85,7 @@ mod handlers { pub(crate) mod type_mismatch; pub(crate) mod type_must_be_known; pub(crate) mod typed_hole; + pub(crate) mod unary_operator_cannot_be_applied; pub(crate) mod undeclared_label; pub(crate) mod unimplemented_builtin_macro; pub(crate) mod unimplemented_trait; @@ -437,6 +438,7 @@ pub fn semantic_diagnostics( let d = match diag { AnyDiagnostic::AwaitOutsideOfAsync(d) => handlers::await_outside_of_async::await_outside_of_async(&ctx, &d), AnyDiagnostic::CannotBeDereferenced(d) => handlers::cannot_be_dereferenced::cannot_be_dereferenced(&ctx, &d), + AnyDiagnostic::UnaryOperatorCannotBeApplied(d) => handlers::unary_operator_cannot_be_applied::unary_operator_cannot_be_applied(&ctx, &d), AnyDiagnostic::CannotImplicitlyDerefTraitObject(d) => handlers::cannot_implicitly_deref_trait_object::cannot_implicitly_deref_trait_object(&ctx, &d), AnyDiagnostic::CannotIndexInto(d) => handlers::cannot_index_into::cannot_index_into(&ctx, &d), AnyDiagnostic::CastToUnsized(d) => handlers::invalid_cast::cast_to_unsized(&ctx, &d), diff --git a/src/tools/rust-analyzer/crates/test-utils/src/minicore.rs b/src/tools/rust-analyzer/crates/test-utils/src/minicore.rs index 2e0994a1885d9..0d9bb4f92bdc0 100644 --- a/src/tools/rust-analyzer/crates/test-utils/src/minicore.rs +++ b/src/tools/rust-analyzer/crates/test-utils/src/minicore.rs @@ -34,9 +34,9 @@ //! discriminant: //! drop: sized //! env: option -//! eq: sized +//! eq: sized, unary_ops, builtin_impls //! error: fmt -//! float_consts: +//! float_consts: unary_ops, builtin_impls //! fmt: option, result, transmute, coerce_unsized, copy, clone, derive //! fn: sized, tuple //! from: sized, result @@ -370,11 +370,13 @@ pub mod clone { } } + // region:index impl Clone for [T; 1] { fn clone(&self) -> Self { [self[0].clone()] } } + // endregion:index // endregion:builtin_impls // region:derive @@ -1213,6 +1215,30 @@ pub mod ops { #[must_use = "this returns the result of the operation, without modifying the original"] fn neg(self) -> Self::Output; } + + // region:builtin_impls + macro_rules! not_impl { + ($($t:ty)*) => ($( + impl const Not for $t { + type Output = $t; + fn not(self) -> $t { !self } + } + )*) + } + + not_impl! { bool usize u8 u16 u32 u64 u128 isize i8 i16 i32 i64 i128 } + + macro_rules! neg_impl { + ($($t:ty)*) => ($( + impl const Neg for $t { + type Output = $t; + fn neg(self) -> $t { -self } + } + )*) + } + + neg_impl! { isize i8 i16 i32 i64 i128 f16 f32 f64 f128 } + // endregion:builtin_impls // endregion:unary_ops // region:coroutine From 55f8db6f9a99a4414c917a715224dfc4fe390d87 Mon Sep 17 00:00:00 2001 From: The rustc-josh-sync Cronjob Bot Date: Mon, 17 Aug 2026 04:21:44 +0000 Subject: [PATCH 06/18] Prepare for merging from rust-lang/rust This updates the rust-version file to 2c39ff499469be916d4e45506d1afed69bbaddb7. --- src/tools/rust-analyzer/rust-version | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/tools/rust-analyzer/rust-version b/src/tools/rust-analyzer/rust-version index f930573748e28..2f175e966812d 100644 --- a/src/tools/rust-analyzer/rust-version +++ b/src/tools/rust-analyzer/rust-version @@ -1 +1 @@ -7fb284d9037fa54f6a9b24261c82b394472cbfd7 +2c39ff499469be916d4e45506d1afed69bbaddb7 From 642c35c0d82ec4fddd8e2c2d388c62d435342baa Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Lauren=C8=9Biu=20Nicola?= Date: Mon, 17 Aug 2026 11:24:06 +0300 Subject: [PATCH 07/18] Download all artifacts in a single step --- .../.github/workflows/metrics.yaml | 30 ++-------------- .../.github/workflows/release.yaml | 35 ++----------------- 2 files changed, 5 insertions(+), 60 deletions(-) diff --git a/src/tools/rust-analyzer/.github/workflows/metrics.yaml b/src/tools/rust-analyzer/.github/workflows/metrics.yaml index a482235105c04..51a12386088ac 100644 --- a/src/tools/rust-analyzer/.github/workflows/metrics.yaml +++ b/src/tools/rust-analyzer/.github/workflows/metrics.yaml @@ -91,35 +91,11 @@ jobs: - name: Checkout repository uses: actions/checkout@v6 - - name: Download build metrics + - name: Download metrics uses: actions/download-artifact@v8 with: - name: build-${{ github.sha }} - - - name: Download self metrics - uses: actions/download-artifact@v8 - with: - name: self-${{ github.sha }} - - - name: Download ripgrep-13.0.0 metrics - uses: actions/download-artifact@v8 - with: - name: ripgrep-13.0.0-${{ github.sha }} - - - name: Download webrender-2022 metrics - uses: actions/download-artifact@v8 - with: - name: webrender-2022-${{ github.sha }} - - - name: Download diesel-1.4.8 metrics - uses: actions/download-artifact@v8 - with: - name: diesel-1.4.8-${{ github.sha }} - - - name: Download hyper-0.14.18 metrics - uses: actions/download-artifact@v8 - with: - name: hyper-0.14.18-${{ github.sha }} + pattern: '*-${{ github.sha }}' + merge-multiple: true - name: Combine json run: | diff --git a/src/tools/rust-analyzer/.github/workflows/release.yaml b/src/tools/rust-analyzer/.github/workflows/release.yaml index 50af766db3cbd..e6e459d7eeb6f 100644 --- a/src/tools/rust-analyzer/.github/workflows/release.yaml +++ b/src/tools/rust-analyzer/.github/workflows/release.yaml @@ -237,40 +237,9 @@ jobs: - uses: actions/download-artifact@v8 with: - name: dist-aarch64-apple-darwin - path: dist - - uses: actions/download-artifact@v8 - with: - name: dist-x86_64-apple-darwin - path: dist - - uses: actions/download-artifact@v8 - with: - name: dist-x86_64-unknown-linux-gnu - path: dist - - uses: actions/download-artifact@v8 - with: - name: dist-x86_64-unknown-linux-musl - path: dist - - uses: actions/download-artifact@v8 - with: - name: dist-aarch64-unknown-linux-gnu - path: dist - - uses: actions/download-artifact@v8 - with: - name: dist-arm-unknown-linux-gnueabihf - path: dist - - uses: actions/download-artifact@v8 - with: - name: dist-x86_64-pc-windows-msvc - path: dist - - uses: actions/download-artifact@v8 - with: - name: dist-i686-pc-windows-msvc - path: dist - - uses: actions/download-artifact@v8 - with: - name: dist-aarch64-pc-windows-msvc + pattern: dist-* path: dist + merge-multiple: true - run: ls -al ./dist - name: Publish Release From 2e32d85ac962fb22df6eaaeaf222f3e62cd7fe29 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Lauren=C8=9Biu=20Nicola?= Date: Mon, 17 Aug 2026 12:14:27 +0300 Subject: [PATCH 08/18] Drop zigbuild support --- .../.github/workflows/release.yaml | 1 - src/tools/rust-analyzer/xtask/src/dist.rs | 33 ++++--------------- src/tools/rust-analyzer/xtask/src/flags.rs | 3 -- src/tools/rust-analyzer/xtask/src/pgo.rs | 9 ++--- 4 files changed, 9 insertions(+), 37 deletions(-) diff --git a/src/tools/rust-analyzer/.github/workflows/release.yaml b/src/tools/rust-analyzer/.github/workflows/release.yaml index 50af766db3cbd..d9904a6dec459 100644 --- a/src/tools/rust-analyzer/.github/workflows/release.yaml +++ b/src/tools/rust-analyzer/.github/workflows/release.yaml @@ -40,7 +40,6 @@ jobs: - os: ubuntu-latest target: x86_64-unknown-linux-gnu # Use a container with glibc 2.28 - # Zig is not used because it doesn't work with PGO container: quay.io/pypa/manylinux_2_28_x86_64 code-target: linux-x64 allocator: system diff --git a/src/tools/rust-analyzer/xtask/src/dist.rs b/src/tools/rust-analyzer/xtask/src/dist.rs index e8bedbe79e56e..1f1254286d1c5 100644 --- a/src/tools/rust-analyzer/xtask/src/dist.rs +++ b/src/tools/rust-analyzer/xtask/src/dist.rs @@ -43,7 +43,6 @@ impl flags::Dist { &format!("{version}-standalone"), &target, allocator, - self.zig, self.pgo, // Profiling requires debug information. self.enable_profiling, @@ -56,7 +55,6 @@ impl flags::Dist { "0.0.0-standalone", &target, allocator, - self.zig, self.pgo, // Profiling requires debug information. self.enable_profiling, @@ -101,7 +99,6 @@ fn dist_server( release: &str, target: &Target, allocator: Malloc, - zig: bool, pgo: Option, dev_rel: bool, ) -> anyhow::Result<()> { @@ -116,33 +113,27 @@ fn dist_server( // * on Linux, this blows up the binary size from 8MB to 43MB, which is unreasonable. // let _e = sh.push_env("CARGO_PROFILE_RELEASE_DEBUG", "1"); - let linux_target = target.is_linux(); - let target_name = match &target.libc_suffix { - Some(libc_suffix) if zig => format!("{}.{libc_suffix}", target.name), - _ => target.name.to_owned(), - }; let features = allocator.to_features(); - let command = if linux_target && zig { "zigbuild" } else { "build" }; let pgo_profile = if let Some(train_crate) = pgo { Some(crate::pgo::gather_pgo_profile( sh, - crate::pgo::build_command(sh, command, &target_name, features), - &target_name, + crate::pgo::build_command(sh, &target.name, features), + &target.name, train_crate, )?) } else { None }; - let mut cmd = build_command(sh, command, &target_name, features, dev_rel); + let mut cmd = build_command(sh, &target.name, features, dev_rel); let mut rustflags = Vec::new(); if let Some(profile) = pgo_profile { rustflags.push(format!("-Cprofile-use={}", profile.to_str().unwrap())); } - if target_name.ends_with("-windows-msvc") { + if target.name.ends_with("-windows-msvc") { // https://github.com/rust-lang/rust-analyzer/issues/20970 rustflags.push("-Ctarget-feature=+crt-static".to_owned()); } @@ -153,7 +144,7 @@ fn dist_server( cmd.run().context("cannot build Rust Analyzer")?; let dst = Path::new("dist").join(&target.artifact_name); - if target_name.contains("-windows-") { + if target.name.contains("-windows-") { zip(&target.server_path, target.symbols_path.as_ref(), &dst.with_extension("zip"))?; } else { gzip(&target.server_path, &dst.with_extension("gz"))?; @@ -164,7 +155,6 @@ fn dist_server( fn build_command<'a>( sh: &'a Shell, - command: &str, target_name: &str, features: &[&str], dev_rel: bool, @@ -172,7 +162,7 @@ fn build_command<'a>( let profile = if dev_rel { "dev-rel" } else { "release" }; cmd!( sh, - "cargo {command} --manifest-path ./crates/rust-analyzer/Cargo.toml --bin rust-analyzer --target {target_name} {features...} --profile {profile}" + "cargo build --manifest-path ./crates/rust-analyzer/Cargo.toml --bin rust-analyzer --target {target_name} {features...} --profile {profile}" ) } @@ -222,7 +212,6 @@ fn zip(src_path: &Path, symbols_path: Option<&PathBuf>, dest_path: &Path) -> any struct Target { name: String, - libc_suffix: Option, server_path: PathBuf, symbols_path: Option, artifact_name: String, @@ -231,10 +220,6 @@ struct Target { impl Target { fn get(project_root: &Path, sh: &Shell) -> Self { let name = detect_target(sh); - let (name, libc_suffix) = match name.split_once('.') { - Some((l, r)) => (l.to_owned(), Some(r.to_owned())), - None => (name, None), - }; let out_path = project_root.join("target").join(&name).join("release"); let (exe_suffix, symbols_path) = if name.contains("-windows-") { (".exe".into(), Some(out_path.join("rust_analyzer.pdb"))) @@ -243,11 +228,7 @@ impl Target { }; let server_path = out_path.join(format!("rust-analyzer{exe_suffix}")); let artifact_name = format!("rust-analyzer-{name}{exe_suffix}"); - Self { name, libc_suffix, server_path, symbols_path, artifact_name } - } - - fn is_linux(&self) -> bool { - self.name.contains("-linux-") + Self { name, server_path, symbols_path, artifact_name } } } diff --git a/src/tools/rust-analyzer/xtask/src/flags.rs b/src/tools/rust-analyzer/xtask/src/flags.rs index e72d8f22e4f0a..ecf38666934a0 100644 --- a/src/tools/rust-analyzer/xtask/src/flags.rs +++ b/src/tools/rust-analyzer/xtask/src/flags.rs @@ -76,8 +76,6 @@ xflags::xflags! { // **Warning:** This will produce a slower build of rust-analyzer, use only for profiling. optional --enable-profiling optional --client-patch-version version: String - /// Use cargo-zigbuild - optional --zig /// Apply PGO optimizations optional --pgo pgo: PgoTrainingCrate } @@ -154,7 +152,6 @@ pub struct Dist { pub jemalloc: bool, pub enable_profiling: bool, pub client_patch_version: Option, - pub zig: bool, pub pgo: Option, } diff --git a/src/tools/rust-analyzer/xtask/src/pgo.rs b/src/tools/rust-analyzer/xtask/src/pgo.rs index ca6dace940b54..c8bb1417df94d 100644 --- a/src/tools/rust-analyzer/xtask/src/pgo.rs +++ b/src/tools/rust-analyzer/xtask/src/pgo.rs @@ -97,15 +97,10 @@ fn download_crate_for_training(sh: &Shell, pgo_dir: &Path, repo: &str) -> anyhow } /// Helper function to create a build command for rust-analyzer -pub(crate) fn build_command<'a>( - sh: &'a Shell, - command: &str, - target_name: &str, - features: &[&str], -) -> Cmd<'a> { +pub(crate) fn build_command<'a>(sh: &'a Shell, target_name: &str, features: &[&str]) -> Cmd<'a> { cmd!( sh, - "cargo {command} --manifest-path ./crates/rust-analyzer/Cargo.toml --bin rust-analyzer --target {target_name} {features...} --release" + "cargo build --manifest-path ./crates/rust-analyzer/Cargo.toml --bin rust-analyzer --target {target_name} {features...} --release" ) } From d458b39fad35d5a5113d6c07f0f9b885e961cd8c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Lauren=C8=9Biu=20Nicola?= Date: Mon, 17 Aug 2026 13:57:59 +0300 Subject: [PATCH 09/18] Drop duplicate function --- src/tools/rust-analyzer/xtask/src/dist.rs | 8 ++------ src/tools/rust-analyzer/xtask/src/pgo.rs | 8 -------- 2 files changed, 2 insertions(+), 14 deletions(-) diff --git a/src/tools/rust-analyzer/xtask/src/dist.rs b/src/tools/rust-analyzer/xtask/src/dist.rs index 1f1254286d1c5..0b8b14cb22717 100644 --- a/src/tools/rust-analyzer/xtask/src/dist.rs +++ b/src/tools/rust-analyzer/xtask/src/dist.rs @@ -115,13 +115,9 @@ fn dist_server( let features = allocator.to_features(); + let cmd = build_command(sh, &target.name, features, dev_rel); let pgo_profile = if let Some(train_crate) = pgo { - Some(crate::pgo::gather_pgo_profile( - sh, - crate::pgo::build_command(sh, &target.name, features), - &target.name, - train_crate, - )?) + Some(crate::pgo::gather_pgo_profile(sh, cmd, &target.name, train_crate)?) } else { None }; diff --git a/src/tools/rust-analyzer/xtask/src/pgo.rs b/src/tools/rust-analyzer/xtask/src/pgo.rs index c8bb1417df94d..9eb41faf26019 100644 --- a/src/tools/rust-analyzer/xtask/src/pgo.rs +++ b/src/tools/rust-analyzer/xtask/src/pgo.rs @@ -96,14 +96,6 @@ fn download_crate_for_training(sh: &Shell, pgo_dir: &Path, repo: &str) -> anyhow Ok(target_path) } -/// Helper function to create a build command for rust-analyzer -pub(crate) fn build_command<'a>(sh: &'a Shell, target_name: &str, features: &[&str]) -> Cmd<'a> { - cmd!( - sh, - "cargo build --manifest-path ./crates/rust-analyzer/Cargo.toml --bin rust-analyzer --target {target_name} {features...} --release" - ) -} - pub(crate) fn apply_pgo_to_cmd<'a>(cmd: Cmd<'a>, profile_path: &Path) -> Cmd<'a> { cmd.env("RUSTFLAGS", format!("-Cprofile-use={}", profile_path.to_str().unwrap())) } From 3e3dd5355c5ed4c76bf69cc53bacd2a45dac4215 Mon Sep 17 00:00:00 2001 From: A4-Tacks Date: Mon, 17 Aug 2026 21:21:21 +0800 Subject: [PATCH 10/18] minor: skip iter excludes 'into_iter' method --- .../crates/ide-completion/src/completions/dot.rs | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/src/tools/rust-analyzer/crates/ide-completion/src/completions/dot.rs b/src/tools/rust-analyzer/crates/ide-completion/src/completions/dot.rs index 774e14df48340..fb6a747a64803 100644 --- a/src/tools/rust-analyzer/crates/ide-completion/src/completions/dot.rs +++ b/src/tools/rust-analyzer/crates/ide-completion/src/completions/dot.rs @@ -118,6 +118,9 @@ pub(crate) fn complete_dot( ctx: dot_access.ctx, }; complete_methods(ctx, &iter, &traits_in_scope, |func| { + if func.name(ctx.db) == hir::sym::into_iter { + return; + } acc.add_method(ctx, &dot_access, func, Some(iter_sym.clone()), None) }); } @@ -1681,7 +1684,6 @@ fn foo() { expect![[r#" me into_iter() (as IntoIterator) fn(self) -> ::IntoIter me into_iter().by_ref() (as Iterator) fn(&mut self) -> &mut Self - me into_iter().into_iter() (as IntoIterator) fn(self) -> ::IntoIter me into_iter().next() (as Iterator) fn(&mut self) -> Option<::Item> me into_iter().nth(…) (as Iterator) fn(&mut self, usize) -> Option<::Item> "#]], @@ -1715,7 +1717,6 @@ fn foo() { me into_iter() (as IntoIterator) fn(self) -> ::IntoIter me iter() fn(&self) -> Iter me iter().by_ref() (as Iterator) fn(&mut self) -> &mut Self - me iter().into_iter() (as IntoIterator) fn(self) -> ::IntoIter me iter().next() (as Iterator) fn(&mut self) -> Option<::Item> me iter().nth(…) (as Iterator) fn(&mut self, usize) -> Option<::Item> "#]], From 8884e78f43a13bfbb6309bc929b1bcfb0086e461 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Lauren=C8=9Biu=20Nicola?= Date: Mon, 17 Aug 2026 16:19:18 +0300 Subject: [PATCH 11/18] Split VSIX publishing into different jobs and skip duplicates --- .../.github/workflows/release.yaml | 55 +++++++++++-------- 1 file changed, 31 insertions(+), 24 deletions(-) diff --git a/src/tools/rust-analyzer/.github/workflows/release.yaml b/src/tools/rust-analyzer/.github/workflows/release.yaml index 70b96a64274b3..7d6e0199abe0b 100644 --- a/src/tools/rust-analyzer/.github/workflows/release.yaml +++ b/src/tools/rust-analyzer/.github/workflows/release.yaml @@ -199,15 +199,9 @@ jobs: publish: if: ${{ github.repository == 'rust-lang/rust-analyzer' || github.event_name == 'workflow_dispatch' }} - name: publish runs-on: ubuntu-latest needs: ["dist", "dist-x86_64-unknown-linux-musl"] steps: - - name: Install Nodejs - uses: actions/setup-node@v6 - with: - node-version: 22 - - name: Checkout repository uses: actions/checkout@v6 with: @@ -248,28 +242,41 @@ jobs: name: ${{ env.TAG }} token: ${{ secrets.GITHUB_TOKEN }} - - run: npm ci - working-directory: ./editors/code + publish-extension: + if: ${{ github.repository == 'rust-lang/rust-analyzer' || github.event_name == 'workflow_dispatch' }} + name: publish-extension (${{ matrix.cmd }}) + runs-on: ubuntu-latest + needs: publish + strategy: + fail-fast: false + matrix: + include: + - cmd: vsce + pat: MARKETPLACE_TOKEN + - cmd: ovsx + pat: OPENVSX_TOKEN + steps: + - name: Install Nodejs + uses: actions/setup-node@v6 + with: + node-version: 22 - - name: Publish Extension (Code Marketplace, release) - if: github.ref == 'refs/heads/release' && github.repository == 'rust-lang/rust-analyzer' - working-directory: ./editors/code - # token from https://dev.azure.com/rust-analyzer/ - run: npx vsce publish --pat ${{ secrets.MARKETPLACE_TOKEN }} --packagePath ../../dist/rust-analyzer-*.vsix + - name: Checkout repository + uses: actions/checkout@v6 + with: + fetch-depth: ${{ env.FETCH_DEPTH }} - - name: Publish Extension (OpenVSX, release) - if: github.ref == 'refs/heads/release' && github.repository == 'rust-lang/rust-analyzer' - working-directory: ./editors/code - run: npx ovsx publish --pat ${{ secrets.OPENVSX_TOKEN }} --packagePath ../../dist/rust-analyzer-*.vsix - timeout-minutes: 2 + - uses: actions/download-artifact@v8 + with: + pattern: dist-* + path: dist + merge-multiple: true - - name: Publish Extension (Code Marketplace, nightly) - if: github.ref != 'refs/heads/release' && github.repository == 'rust-lang/rust-analyzer' + - run: npm ci working-directory: ./editors/code - run: npx vsce publish --pat ${{ secrets.MARKETPLACE_TOKEN }} --packagePath ../../dist/rust-analyzer-*.vsix --pre-release - - name: Publish Extension (OpenVSX, nightly) - if: github.ref != 'refs/heads/release' && github.repository == 'rust-lang/rust-analyzer' + - name: Publish Extension + if: github.repository == 'rust-lang/rust-analyzer' working-directory: ./editors/code - run: npx ovsx publish --pat ${{ secrets.OPENVSX_TOKEN }} --packagePath ../../dist/rust-analyzer-*.vsix + run: npx ${{ matrix.cmd }} publish --skip-duplicate --pat ${{ secrets[matrix.pat] }} --packagePath ../../dist/rust-analyzer-*.vsix ${{ github.ref != 'refs/heads/release' && '--pre-release' || '' }} timeout-minutes: 2 From 778fc4d9e3e9b3c64a9dbee5c75c92364cd73cd2 Mon Sep 17 00:00:00 2001 From: Parman Mohammadalizadeh Date: Tue, 18 Aug 2026 22:06:11 +0200 Subject: [PATCH 12/18] fix: allow `asm!` label blocks to diverge Label operands were inferred with `infer_expr`, which demands the block's type be equal to `()`. A block that diverges has type `!`, so code like `label { break; }` inside a loop reported a false `expected (), found !`. Follow rustc's handling in `check_expr_asm` and only demand a supertype when the block does not diverge, saving and restoring `diverges` around it. --- .../crates/hir-ty/src/infer/expr.rs | 12 +++++++++-- .../crates/hir-ty/src/tests/simple.rs | 21 +++++++++++++++++++ 2 files changed, 31 insertions(+), 2 deletions(-) diff --git a/src/tools/rust-analyzer/crates/hir-ty/src/infer/expr.rs b/src/tools/rust-analyzer/crates/hir-ty/src/infer/expr.rs index 0446978dd996a..f247b517c541f 100644 --- a/src/tools/rust-analyzer/crates/hir-ty/src/infer/expr.rs +++ b/src/tools/rust-analyzer/crates/hir-ty/src/infer/expr.rs @@ -821,7 +821,7 @@ impl<'db> InferenceContext<'db> { } }; - let diverge = asm.options.contains(AsmOptions::NORETURN); + let mut diverge = asm.options.contains(AsmOptions::NORETURN); asm.operands.iter().for_each(|(_, operand)| match *operand { AsmOperand::In { expr, .. } => check_expr_asm_operand(self, expr, true), AsmOperand::Out { expr: Some(expr), .. } | AsmOperand::InOut { expr, .. } => { @@ -835,11 +835,19 @@ impl<'db> InferenceContext<'db> { } } AsmOperand::Label(expr) => { - self.infer_expr( + let previous_diverges = self.diverges; + // The label blocks should have unit return value or diverge. + let ty = self.infer_expr_inner( expr, &Expectation::HasType(self.types.types.unit), ExprIsRead::No, ); + if !ty.is_never() { + _ = self.demand_suptype(expr.into(), self.types.types.unit, ty); + diverge = false; + } + // We need this to avoid false unreachable warning when a label diverges. + self.diverges = previous_diverges; } AsmOperand::Const(expr) => { self.infer_expr(expr, &Expectation::None, ExprIsRead::No); diff --git a/src/tools/rust-analyzer/crates/hir-ty/src/tests/simple.rs b/src/tools/rust-analyzer/crates/hir-ty/src/tests/simple.rs index 3cdfe4edcb908..97921e8ab92d5 100644 --- a/src/tools/rust-analyzer/crates/hir-ty/src/tests/simple.rs +++ b/src/tools/rust-analyzer/crates/hir-ty/src/tests/simple.rs @@ -4162,6 +4162,27 @@ extern "C" fn foo() -> ! { ); } +#[test] +fn asm_label_can_diverge() { + check_no_mismatches( + r#" +//- minicore: asm +fn foo() { + loop { + unsafe { + core::arch::asm!( + "/* {} */", + label { + break; + } + ); + } + } +} + "#, + ); +} + #[test] fn regression_21478() { check_infer( From 88f14afc99eacc923455b679dee2bf2da52879a8 Mon Sep 17 00:00:00 2001 From: The rustc-josh-sync Cronjob Bot Date: Thu, 20 Aug 2026 04:20:54 +0000 Subject: [PATCH 13/18] Prepare for merging from rust-lang/rust This updates the rust-version file to f7d782a3be46d6bb4b9792fe69a61db389ba1769. --- src/tools/rust-analyzer/rust-version | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/tools/rust-analyzer/rust-version b/src/tools/rust-analyzer/rust-version index 2f175e966812d..9ff8b0c27d19c 100644 --- a/src/tools/rust-analyzer/rust-version +++ b/src/tools/rust-analyzer/rust-version @@ -1 +1 @@ -2c39ff499469be916d4e45506d1afed69bbaddb7 +f7d782a3be46d6bb4b9792fe69a61db389ba1769 From 0b8f0072488d6504a8148d7a8791f159ceae9589 Mon Sep 17 00:00:00 2001 From: YUZHEthefool <2804776511@qq.com> Date: Thu, 20 Aug 2026 22:16:44 +0800 Subject: [PATCH 14/18] fix:prevent stack overflow for recursive ADT layouts --- .../crates/hir-ty/src/layout/adt.rs | 4 ++++ .../crates/hir-ty/src/layout/tests.rs | 19 +++++++++++++++++++ 2 files changed, 23 insertions(+) diff --git a/src/tools/rust-analyzer/crates/hir-ty/src/layout/adt.rs b/src/tools/rust-analyzer/crates/hir-ty/src/layout/adt.rs index 47a960e300cc7..2ba6408c45f2c 100644 --- a/src/tools/rust-analyzer/crates/hir-ty/src/layout/adt.rs +++ b/src/tools/rust-analyzer/crates/hir-ty/src/layout/adt.rs @@ -16,6 +16,7 @@ use crate::{ db::HirDatabase, layout::{Layout, LayoutCx, LayoutError, field_ty}, next_solver::StoredGenericArgs, + representability::{Representability, representability}, traits::StoredParamEnvAndCrate, }; @@ -30,6 +31,9 @@ pub fn layout_of_adt_query( let Ok(target) = db.target_data_layout(krate) else { return Err(LayoutError::TargetLayoutNotAvailable); }; + if representability(db, def) == Representability::Infinite { + return Err(LayoutError::RecursiveTypeWithoutIndirection); + } let dl = target; let cx = LayoutCx::new(dl); let handle_variant = |def: VariantId, var: &VariantFields| { diff --git a/src/tools/rust-analyzer/crates/hir-ty/src/layout/tests.rs b/src/tools/rust-analyzer/crates/hir-ty/src/layout/tests.rs index 72befc58a1236..5098b38c4380c 100644 --- a/src/tools/rust-analyzer/crates/hir-ty/src/layout/tests.rs +++ b/src/tools/rust-analyzer/crates/hir-ty/src/layout/tests.rs @@ -275,6 +275,13 @@ fn recursive() { struct BoxLike(*mut T); struct Goal(BoxLike); } + size_and_align! { + struct Foo { + x: *const Foo<[T; 1]>, + y: *const T, + } + struct Goal(Foo); + } check_fail(r#"struct Goal(Goal);"#, LayoutError::RecursiveTypeWithoutIndirection); check_fail( r#" @@ -283,6 +290,18 @@ fn recursive() { "#, LayoutError::RecursiveTypeWithoutIndirection, ); + check_fail( + r#" +struct Foo { + x: Foo<[T; 1]>, + y: T, +} +struct Goal { + x: Foo, +} +"#, + LayoutError::RecursiveTypeWithoutIndirection, + ); } #[test] From 31c34201b87f9b2071bb13e0f47dbddfe643ebc8 Mon Sep 17 00:00:00 2001 From: Chayim Refael Friedman Date: Fri, 21 Aug 2026 02:22:39 +0300 Subject: [PATCH 15/18] Fix 1.98.0 Clippy and rustfmt --- .../rust-analyzer/crates/ide-ssr/src/tests.rs | 15 ++++-------- .../ide/src/annotations/fn_references.rs | 1 + .../src/legacy_protocol/msg/flat.rs | 23 ++++++++++--------- .../crates/profile/src/memory_usage.rs | 13 ++++------- 4 files changed, 23 insertions(+), 29 deletions(-) diff --git a/src/tools/rust-analyzer/crates/ide-ssr/src/tests.rs b/src/tools/rust-analyzer/crates/ide-ssr/src/tests.rs index d8c15cabb2fb8..8d1858ccf2226 100644 --- a/src/tools/rust-analyzer/crates/ide-ssr/src/tests.rs +++ b/src/tools/rust-analyzer/crates/ide-ssr/src/tests.rs @@ -72,18 +72,13 @@ pub(crate) fn single_file(code: &str) -> (ide_db::RootDatabase, FilePosition, Ve let (db, file_id) = ide_db::RootDatabase::with_single_file(code); (db, file_id, RangeOrOffset::Offset(0.into())) }; - let selections; - let position; - match range_or_offset { + + let (position, selections) = match range_or_offset { RangeOrOffset::Range(range) => { - position = FilePosition { file_id, offset: range.start() }; - selections = vec![FileRange { file_id, range }]; - } - RangeOrOffset::Offset(offset) => { - position = FilePosition { file_id, offset }; - selections = vec![]; + (FilePosition { file_id, offset: range.start() }, vec![FileRange { file_id, range }]) } - } + RangeOrOffset::Offset(offset) => (FilePosition { file_id, offset }, vec![]), + }; let mut local_roots = FxHashSet::default(); local_roots.insert(WORKSPACE); LocalRoots::get(&db).set_roots(&mut db).to(local_roots); diff --git a/src/tools/rust-analyzer/crates/ide/src/annotations/fn_references.rs b/src/tools/rust-analyzer/crates/ide/src/annotations/fn_references.rs index 427a2eff82017..b43a23e69f37b 100644 --- a/src/tools/rust-analyzer/crates/ide/src/annotations/fn_references.rs +++ b/src/tools/rust-analyzer/crates/ide/src/annotations/fn_references.rs @@ -87,6 +87,7 @@ mod tests { ); let refs = super::find_all_methods(&analysis.db, pos.file_id); + #[expect(clippy::single_range_in_vec_init, reason = "this is not a mistake")] check_result(&refs, &[28..=34]); } diff --git a/src/tools/rust-analyzer/crates/proc-macro-api/src/legacy_protocol/msg/flat.rs b/src/tools/rust-analyzer/crates/proc-macro-api/src/legacy_protocol/msg/flat.rs index ae03be9aa7a26..b9b6247b54fab 100644 --- a/src/tools/rust-analyzer/crates/proc-macro-api/src/legacy_protocol/msg/flat.rs +++ b/src/tools/rust-analyzer/crates/proc-macro-api/src/legacy_protocol/msg/flat.rs @@ -54,7 +54,7 @@ pub type SpanDataIndexMap = pub fn serialize_span_data_index_map(map: &SpanDataIndexMap) -> Vec { map.iter() - .flat_map(|span| { + .map(|span| { [ span.anchor.file_id.as_u32(), span.anchor.ast_id.into_raw(), @@ -63,14 +63,16 @@ pub fn serialize_span_data_index_map(map: &SpanDataIndexMap) -> Vec { span.ctx.into_u32(), ] }) - .collect() + .collect::>() + .into_flattened() } pub fn deserialize_span_data_index_map(map: &[u32]) -> SpanDataIndexMap { - debug_assert!(map.len().is_multiple_of(5)); - map.chunks_exact(5) - .map(|span| { - let &[file_id, ast_id, start, end, e] = span else { unreachable!() }; + let (chunks, remainder) = map.as_chunks(); + assert!(remainder.is_empty()); + chunks + .iter() + .map(|&[file_id, ast_id, start, end, e]| { Span { anchor: SpanAnchor { file_id: EditionedFileId::from_raw(file_id), @@ -345,14 +347,13 @@ impl FlatTree { } fn read_vec T, const N: usize>(xs: Vec, f: F) -> Vec { - let mut chunks = xs.chunks_exact(N); - let res = chunks.by_ref().map(|chunk| f(chunk.try_into().unwrap())).collect(); - assert!(chunks.remainder().is_empty()); - res + let (chunks, remainder) = xs.as_chunks(); + assert!(remainder.is_empty()); + chunks.iter().map(|chunk| f(*chunk)).collect() } fn write_vec [u32; N], const N: usize>(xs: Vec, f: F) -> Vec { - xs.into_iter().flat_map(f).collect() + xs.into_iter().map(f).collect::>().into_flattened() } impl SubtreeRepr { diff --git a/src/tools/rust-analyzer/crates/profile/src/memory_usage.rs b/src/tools/rust-analyzer/crates/profile/src/memory_usage.rs index a8a409bd4686d..072a01c8298ec 100644 --- a/src/tools/rust-analyzer/crates/profile/src/memory_usage.rs +++ b/src/tools/rust-analyzer/crates/profile/src/memory_usage.rs @@ -30,28 +30,25 @@ impl MemoryUsage { allocated: Bytes(jemalloc_ctl::stats::allocated::read().unwrap() as isize), } } - all(target_os = "linux", target_env = "gnu") => { - memusage_linux() - } + all(target_os = "linux", target_env = "gnu") => memusage_linux(), windows => { // There doesn't seem to be an API for determining heap usage, so we try to // approximate that by using the Commit Charge value. - use windows_sys::Win32::System::{Threading::*, ProcessStatus::*}; use std::mem::MaybeUninit; + use windows_sys::Win32::System::{ProcessStatus::*, Threading::*}; let proc = unsafe { GetCurrentProcess() }; let mut mem_counters = MaybeUninit::uninit(); let cb = size_of::(); - let ret = unsafe { GetProcessMemoryInfo(proc, mem_counters.as_mut_ptr(), cb as u32) }; + let ret = + unsafe { GetProcessMemoryInfo(proc, mem_counters.as_mut_ptr(), cb as u32) }; assert!(ret != 0); let usage = unsafe { mem_counters.assume_init().PagefileUsage }; MemoryUsage { allocated: Bytes(usage as isize) } } - _ => { - MemoryUsage { allocated: Bytes(0) } - } + _ => MemoryUsage { allocated: Bytes(0) }, } } } From cef9fe6d0825eb060823d1835d42dd2cf1bb8c82 Mon Sep 17 00:00:00 2001 From: Chayim Refael Friedman Date: Fri, 21 Aug 2026 13:24:13 +0300 Subject: [PATCH 16/18] Remove two unused public methods I forgot to change this when I changed https://github.com/rust-lang/rust-analyzer/pull/23079. --- src/tools/rust-analyzer/crates/intern/src/symbol.rs | 13 ------------- 1 file changed, 13 deletions(-) diff --git a/src/tools/rust-analyzer/crates/intern/src/symbol.rs b/src/tools/rust-analyzer/crates/intern/src/symbol.rs index cf41db85a163d..91fc11fa0dae9 100644 --- a/src/tools/rust-analyzer/crates/intern/src/symbol.rs +++ b/src/tools/rust-analyzer/crates/intern/src/symbol.rs @@ -174,19 +174,6 @@ impl Symbol { self.repr.as_str() } - #[inline] - pub fn into_raw(self) -> NonNull<*const str> { - ManuallyDrop::new(self).repr.packed - } - - /// # Safety - /// - /// The pointer must have come from [`Symbol::into_raw()`]. - #[inline] - pub unsafe fn from_raw(ptr: NonNull<*const str>) -> Symbol { - Symbol { repr: TaggedArcPtr { packed: ptr } } - } - #[inline] fn select_shard( storage: &'static Map, From 555a43c67a0d62703498cb10d19d29f62e9d67de Mon Sep 17 00:00:00 2001 From: Riccardo Mazzarini Date: Fri, 21 Aug 2026 23:14:15 +0200 Subject: [PATCH 17/18] Use Cargo build directory for flycheck logs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit R-A currently stores flycheck’s captured stdout and stderr under Cargo’s `target_directory`. Cargo 1.91 stabilized `build.build-dir`, which allows intermediate build artifacts to be stored separately and exposes the resolved location as `build_directory` in cargo metadata (see https://github.com/rust-lang/cargo/pull/15833 and https://github.com/rust-lang/cargo/pull/15377). This PR changes the directory selection order for flycheck output to be `rust-analyzer.cargo.targetDir` → `build_directory` → `target_directory`. --- src/tools/rust-analyzer/Cargo.toml | 2 +- .../project-model/src/cargo_workspace.rs | 7 ++++ .../crates/rust-analyzer/src/flycheck.rs | 23 ++++++++----- .../crates/rust-analyzer/src/reload.rs | 9 +++-- .../tests/slow-tests/flycheck.rs | 34 +++++++++++++++++++ 5 files changed, 62 insertions(+), 13 deletions(-) diff --git a/src/tools/rust-analyzer/Cargo.toml b/src/tools/rust-analyzer/Cargo.toml index d043a3aee4f7f..b798e364555df 100644 --- a/src/tools/rust-analyzer/Cargo.toml +++ b/src/tools/rust-analyzer/Cargo.toml @@ -104,7 +104,7 @@ lsp-server = { version = "0.7.9" } anyhow = "1.0.98" arrayvec = "0.7.6" bitflags = "2.9.1" -cargo_metadata = "0.23.0" +cargo_metadata = "0.23.1" camino = "1.2.2" crossbeam-channel = "0.5.15" dissimilar = "1.0.10" diff --git a/src/tools/rust-analyzer/crates/project-model/src/cargo_workspace.rs b/src/tools/rust-analyzer/crates/project-model/src/cargo_workspace.rs index a0a9815d8756f..9d2588380c22c 100644 --- a/src/tools/rust-analyzer/crates/project-model/src/cargo_workspace.rs +++ b/src/tools/rust-analyzer/crates/project-model/src/cargo_workspace.rs @@ -36,6 +36,7 @@ pub struct CargoWorkspace { targets: Arena, workspace_root: AbsPathBuf, target_directory: AbsPathBuf, + build_directory: Option, manifest_path: ManifestPath, is_virtual_workspace: bool, /// Whether this workspace represents the sysroot workspace. @@ -359,6 +360,7 @@ impl CargoWorkspace { let workspace_root = AbsPathBuf::assert(meta.workspace_root); let target_directory = AbsPathBuf::assert(meta.target_directory); + let build_directory = meta.build_directory.map(AbsPathBuf::assert); let mut is_virtual_workspace = true; let mut requires_rustc_private = false; @@ -517,6 +519,7 @@ impl CargoWorkspace { targets, workspace_root, target_directory, + build_directory, manifest_path: ws_manifest_path, is_virtual_workspace, requires_rustc_private, @@ -548,6 +551,10 @@ impl CargoWorkspace { &self.target_directory } + pub fn build_directory(&self) -> Option<&AbsPath> { + self.build_directory.as_deref() + } + pub fn package_flag(&self, package: &PackageData) -> String { if self.is_unique(&package.name) { package.name.clone() diff --git a/src/tools/rust-analyzer/crates/rust-analyzer/src/flycheck.rs b/src/tools/rust-analyzer/crates/rust-analyzer/src/flycheck.rs index 99e640a3cecb9..85edb239e3f57 100644 --- a/src/tools/rust-analyzer/crates/rust-analyzer/src/flycheck.rs +++ b/src/tools/rust-analyzer/crates/rust-analyzer/src/flycheck.rs @@ -226,6 +226,7 @@ impl FlycheckHandle { workspace_root: AbsPathBuf, manifest_path: Option, ws_target_dir: Option, + ws_build_dir: Option, toolchain_version: Option, ) -> FlycheckHandle { let actor = FlycheckActor::new( @@ -238,6 +239,7 @@ impl FlycheckHandle { workspace_root, manifest_path, ws_target_dir, + ws_build_dir, toolchain_version, ); let (sender, receiver) = unbounded::(); @@ -431,6 +433,7 @@ struct FlycheckActor { manifest_path: Option, ws_target_dir: Option, + ws_build_dir: Option, /// Either the workspace root of the workspace we are flychecking, /// or the project root of the project. root: Arc, @@ -533,6 +536,7 @@ impl FlycheckActor { workspace_root: AbsPathBuf, manifest_path: Option, ws_target_dir: Option, + ws_build_dir: Option, toolchain_version: Option, ) -> FlycheckActor { tracing::info!(%id, ?workspace_root, "Spawning flycheck"); @@ -547,6 +551,7 @@ impl FlycheckActor { scope: FlycheckScope::Workspace, manifest_path, ws_target_dir, + ws_build_dir, command_handle: None, command_receiver: None, diagnostics_cleared_for: Default::default(), @@ -633,14 +638,14 @@ impl FlycheckActor { sender, match &self.config { FlycheckConfig::Automatic { cargo_options, .. } => { - let ws_target_dir = - self.ws_target_dir.as_ref().map(Utf8PathBuf::as_path); - let target_dir = - cargo_options.target_dir_config.target_dir(ws_target_dir); + let target_dir = cargo_options + .target_dir_config + .target_dir(self.ws_target_dir.as_deref()); - // If `"rust-analyzer.cargo.targetDir": null`, we should use - // workspace's target dir instead of hard-coded fallback. - let target_dir = target_dir.as_deref().or(ws_target_dir); + let output_dir = target_dir + .as_deref() + .or(self.ws_build_dir.as_deref()) + .or(self.ws_target_dir.as_deref()); Some( // As `CommandHandle::spawn`'s working directory is @@ -648,10 +653,10 @@ impl FlycheckActor { // from the flycheck's working directory, we should canonicalize // the output directory, otherwise we might write it into the // wrong target dir. - // If `target_dir` is an absolute path, it will replace + // If `output_dir` is an absolute path, it will replace // `self.root` and that's an intended behavior. self.root - .join(target_dir.unwrap_or( + .join(output_dir.unwrap_or( Utf8Path::new("target").join("rust-analyzer").as_path(), )) .join(format!("flycheck{}", self.id)) diff --git a/src/tools/rust-analyzer/crates/rust-analyzer/src/reload.rs b/src/tools/rust-analyzer/crates/rust-analyzer/src/reload.rs index 86c954b089e4d..039fbeff828e7 100644 --- a/src/tools/rust-analyzer/crates/rust-analyzer/src/reload.rs +++ b/src/tools/rust-analyzer/crates/rust-analyzer/src/reload.rs @@ -902,6 +902,7 @@ impl GlobalState { None, None, None, + None, )] } crate::flycheck::InvocationStrategy::PerWorkspace => { @@ -921,6 +922,7 @@ impl GlobalState { cargo.workspace_root(), Some(cargo.manifest_path()), Some(cargo.target_directory()), + cargo.build_directory(), ), ProjectWorkspaceKind::Json(project) => { let config_json = crate::flycheck::FlycheckConfigJson { @@ -932,10 +934,10 @@ impl GlobalState { // in the workspace configuration. match config { _ if config_json.any_configured() => { - (config_json, project.path(), None, None) + (config_json, project.path(), None, None, None) } FlycheckConfig::CustomCommand { .. } => { - (config_json, project.path(), None, None) + (config_json, project.path(), None, None, None) } _ => return None, } @@ -949,7 +951,7 @@ impl GlobalState { .map( |( id, - (config_json, root, manifest_path, target_dir), + (config_json, root, manifest_path, target_dir, build_dir), sysroot_root, toolchain, )| { @@ -963,6 +965,7 @@ impl GlobalState { root.to_path_buf(), manifest_path.map(|it| it.to_path_buf()), target_dir.map(|it| AsRef::::as_ref(it).to_path_buf()), + build_dir.map(|it| AsRef::::as_ref(it).to_path_buf()), toolchain, ) }, diff --git a/src/tools/rust-analyzer/crates/rust-analyzer/tests/slow-tests/flycheck.rs b/src/tools/rust-analyzer/crates/rust-analyzer/tests/slow-tests/flycheck.rs index 7700643f03e75..327fa025f490d 100644 --- a/src/tools/rust-analyzer/crates/rust-analyzer/tests/slow-tests/flycheck.rs +++ b/src/tools/rust-analyzer/crates/rust-analyzer/tests/slow-tests/flycheck.rs @@ -42,6 +42,40 @@ fn main() { ); } +#[test] +fn test_flycheck_output_uses_build_directory() { + if skip_slow_tests() { + return; + } + + let server = Project::with_fixture( + r#" +//- /.cargo/config.toml +[build] +build-dir = "build" + +//- /Cargo.toml +[package] +name = "foo" +version = "0.0.0" + +//- /src/main.rs +fn main() { + let x = 1; +} +"#, + ) + .with_config(serde_json::json!({ + "checkOnSave": true, + })) + .server() + .wait_until_workspace_is_loaded(); + + _ = server.wait_for_diagnostics(); + assert!(server.path().join("build/flycheck0/stdout").exists()); + assert!(server.path().join("build/flycheck0/stderr").exists()); +} + #[test] fn test_flycheck_diagnostic_cleared_after_fix() { if skip_slow_tests() { From 78da05d0bfc013a04807d71f3fde69b348a12ca8 Mon Sep 17 00:00:00 2001 From: Suryansh Dey Date: Sun, 23 Aug 2026 12:51:31 +0530 Subject: [PATCH 18/18] fix(hir): Use expression store of parent body if available --- .../crates/hir/src/source_analyzer.rs | 6 ++--- .../crates/ide/src/hover/tests.rs | 22 +++++++++++++++++++ 2 files changed, 25 insertions(+), 3 deletions(-) diff --git a/src/tools/rust-analyzer/crates/hir/src/source_analyzer.rs b/src/tools/rust-analyzer/crates/hir/src/source_analyzer.rs index 209091683a01b..907193fe1d7ff 100644 --- a/src/tools/rust-analyzer/crates/hir/src/source_analyzer.rs +++ b/src/tools/rust-analyzer/crates/hir/src/source_analyzer.rs @@ -464,7 +464,7 @@ impl<'db> SourceAnalyzer<'db> { db, &self.resolver, self.store()?, - generic_def.into(), + self.resolver.expression_store_owner().unwrap_or_else(|| generic_def.into()), generic_def, &generics, // FIXME: Is this correct here? Anyway that should impact mostly diagnostics, which we don't emit here @@ -1890,7 +1890,7 @@ fn resolve_hir_path_<'db>( db, resolver, store?, - def.into(), + resolver.expression_store_owner().unwrap_or_else(|| def.into()), def, &generics, LifetimeElisionKind::Infer, @@ -2094,7 +2094,7 @@ fn resolve_hir_path_qualifier<'db>( db, resolver, store, - def.into(), + resolver.expression_store_owner().unwrap_or_else(|| def.into()), def, &generics, LifetimeElisionKind::Infer, diff --git a/src/tools/rust-analyzer/crates/ide/src/hover/tests.rs b/src/tools/rust-analyzer/crates/ide/src/hover/tests.rs index 4fea4468c0725..fad322aa4f5da 100644 --- a/src/tools/rust-analyzer/crates/ide/src/hover/tests.rs +++ b/src/tools/rust-analyzer/crates/ide/src/hover/tests.rs @@ -11948,3 +11948,25 @@ fn main() {} "#]], ); } + +#[test] +fn resolve_array_type_with_anon_const_panic() { + use syntax::AstNode; + let (analysis, position) = crate::fixture::position( + r#" +fn main() { + let x: [u8; 2 + 2$0] = [0; 4]; +} +"#, + ); + let db = &analysis.db; + hir::attach_db(db, || { + let sema = hir::Semantics::new(db); + let file = sema.parse_guess_edition(position.file_id); + let token = file.syntax().token_at_offset(position.offset).right_biased().unwrap(); + let type_node = token.parent_ancestors().find_map(syntax::ast::Type::cast).unwrap(); + + let resolved = sema.resolve_type(&type_node).unwrap(); + let _ = resolved.as_array(db); + }); +}