From c7dcc50768fbb3a802c26f6eb3c226a4e91f38f5 Mon Sep 17 00:00:00 2001 From: Richard Smith Date: Tue, 25 Aug 2026 21:26:46 +0000 Subject: [PATCH] language-server: Remove an unnecessary vector copy on each source change. (#7669) We track a "next" index into each bucket when we insert instructions. The insert loop effectively shifts each element in the "next" vector left by one place, so if we instead start the bucket counts shifted one place to the right, we can use the same vector for "next" and for the bucket start indexes. Assisted-by: Claude Code --------- Co-authored-by: Geoff Romer --- toolchain/language_server/sem_ir_index.cpp | 39 ++++++++++++++-------- 1 file changed, 26 insertions(+), 13 deletions(-) diff --git a/toolchain/language_server/sem_ir_index.cpp b/toolchain/language_server/sem_ir_index.cpp index 95f6b7d42542..bd71acbf5382 100644 --- a/toolchain/language_server/sem_ir_index.cpp +++ b/toolchain/language_server/sem_ir_index.cpp @@ -54,36 +54,49 @@ static auto GetTokenForInst(const SemIR::File& sem_ir, SemIRIndex::SemIRIndex(const SemIR::File& sem_ir, const Parse::TreeAndSubtrees& tree_and_subtrees) { const auto& tokens = tree_and_subtrees.tree().tokens(); - // Count the instructions per token, leaving a leading zero so that the counts - // can be turned into start offsets in place. - token_starts_.assign(tokens.size() + 1, 0); - int32_t total = 0; + // Populate `token_starts_` in three in-place passes. Note that we need N+1 + // elements to hold the boundaries of N contiguous intervals, plus an + // additional temporary element for reasons discussed below. + // + // First, we count the instructions per token. The array contents are shifted + // by 2: `token_starts[i+2]` will hold the number of insts for the token with + // `.index == i`. + token_starts_.assign(tokens.size() + 2, 0); for (auto [inst_id, inst] : sem_ir.insts().enumerate()) { auto token = GetTokenForInst(sem_ir, tree_and_subtrees, inst_id); if (!token.has_value()) { continue; } - ++token_starts_[token.index + 1]; - ++total; + ++token_starts_[token.index + 2]; } - // Turn the counts into start offsets. + // Perform a prefix sum, so that `token_starts_[i+2]` holds the number of + // insts for tokens with `.index <= i`, i.e. the end of the interval for token + // `i`, and hence `token_starts_[i+1]` is the start of the interval for token + // `i`. for (size_t i = 1; i < token_starts_.size(); ++i) { token_starts_[i] += token_starts_[i - 1]; } - CARBON_CHECK(token_starts_.back() == total); - // Fill each token's group. `next` tracks the next free slot per token, and - // ends up equal to the following token's start, so the offsets stay valid. - insts_.resize(total, SemIR::InstId::None); - llvm::SmallVector next(token_starts_.begin(), token_starts_.end()); + // Pop the final "start" index, which is now the total number of instructions + // that have associated locations. + auto total_insts = token_starts_.pop_back_val(); + + // Populate `insts_`, using `token_starts_[i+1]` as the index to write the + // next inst for token `i`, which is incremented on each write. Thus, at the + // end of the loop, `token_starts_[i+1]` is the past-the-end index for token + // `i`, i.e. the start index for token `i+1`, which is the final state of + // `token_starts_`. + insts_.resize(total_insts, SemIR::InstId::None); for (auto [inst_id, inst] : sem_ir.insts().enumerate()) { auto token = GetTokenForInst(sem_ir, tree_and_subtrees, inst_id); if (!token.has_value()) { continue; } - insts_[next[token.index]++] = inst_id; + insts_[token_starts_[token.index + 1]++] = inst_id; } + + CARBON_CHECK(static_cast(token_starts_.back()) == insts_.size()); } auto SemIRIndex::InstsForToken(Lex::TokenIndex token) const