Speed up type literal lexing and make it more strict. (#4430)

This rejects type literals with more digits than we can lex without
APInt's help, and using a custom diagnostic. This is a pretty arbitrary
implementation limit, I'm wide open to even more strict rules here.

Despite no special casing and a very simplistic approach, by not using
APInt this completely eliminates the lexing overhead for `i32` in the
generated compilation benchmark where that specific type literal is very
common. We see a 10% improvement in lexing there:
```
BM_CompileAPIFileDenseDecls<Phase::Lex>/256        39.0µs ± 4%  34.8µs ± 2%  -10.86%  (p=0.000 n=19+20)
BM_CompileAPIFileDenseDecls<Phase::Lex>/1024        180µs ± 1%   158µs ± 2%  -12.22%  (p=0.000 n=18+20)
BM_CompileAPIFileDenseDecls<Phase::Lex>/4096        731µs ± 2%   641µs ± 1%  -12.31%  (p=0.000 n=18+20)
BM_CompileAPIFileDenseDecls<Phase::Lex>/16384      3.20ms ± 2%  2.86ms ± 2%  -10.47%  (p=0.000 n=18+19)
BM_CompileAPIFileDenseDecls<Phase::Lex>/65536      13.8ms ± 1%  12.4ms ± 2%   -9.78%  (p=0.000 n=18+19)
BM_CompileAPIFileDenseDecls<Phase::Lex>/262144     64.0ms ± 2%  58.4ms ± 2%   -8.70%  (p=0.000 n=19+18)
```

This starts to fix a TODO in the diagnostic for these by giving a
reasonably good diagnostic about a very large type literal. However, in
practice it regresses the diagnostics because error tokens produce noisy
extraneous diagnostics from parse and check currently. Leaving the TODO
there, and I have a follow-up PR to start improving the extraneous
diagnostics.
This commit is contained in:
Chandler Carruth
2024-10-24 21:43:35 +00:00
committed by GitHub
parent b67d03126e
commit 577fda1ca2
6 changed files with 112 additions and 139 deletions
+47 -19
View File
@@ -7,6 +7,7 @@
#include <gmock/gmock.h>
#include <gtest/gtest.h>
#include <cmath>
#include <forward_list>
#include <iterator>
@@ -1023,27 +1024,54 @@ TEST_F(LexerTest, TypeLiterals) {
}
TEST_F(LexerTest, TypeLiteralTooManyDigits) {
std::string code = "i";
constexpr int Count = 10000;
code.append(Count, '9');
// We increase the number of digits until the first one that is to large.
Testing::MockDiagnosticConsumer consumer;
EXPECT_CALL(consumer,
HandleDiagnostic(IsSingleDiagnostic(
DiagnosticKind::TooManyDigits, DiagnosticLevel::Error, 1, 2,
HasSubstr(llvm::formatv(" {0} ", Count)))));
EXPECT_CALL(consumer, HandleDiagnostic(IsSingleDiagnostic(
DiagnosticKind::TooManyTypeBitWidthDigits,
DiagnosticLevel::Error, 1, 2, _)));
std::string code = "i";
// A 128-bit APInt should be plenty large, but if needed in the future it can
// be widened without issue.
llvm::APInt bits = llvm::APInt::getZero(128);
for ([[maybe_unused]] int _ : llvm::seq(1, 30)) {
code.append("9");
bits = bits * 10 + 9;
auto [buffer, value_stores] =
compile_helper_.GetTokenizedBufferWithSharedValueStore(code, &consumer);
if (buffer.has_errors()) {
ASSERT_THAT(buffer, HasTokens(llvm::ArrayRef<ExpectedToken>{
{.kind = TokenKind::FileStart},
{.kind = TokenKind::Error, .text = code},
{.kind = TokenKind::FileEnd},
}));
break;
}
ASSERT_THAT(buffer, HasTokens(llvm::ArrayRef<ExpectedToken>{
{.kind = TokenKind::FileStart},
{.kind = TokenKind::IntTypeLiteral, .text = code},
{.kind = TokenKind::FileEnd},
}));
auto token = buffer.tokens().begin()[1];
EXPECT_TRUE(llvm::APInt::isSameValue(
value_stores.ints().Get(buffer.GetTypeLiteralSize(token)), bits));
}
// Make sure we can also gracefully reject very large number of digits without
// crashing or hanging, and show the correct number.
constexpr int Count = 10000;
EXPECT_CALL(consumer, HandleDiagnostic(IsSingleDiagnostic(
DiagnosticKind::TooManyTypeBitWidthDigits,
DiagnosticLevel::Error, 1, 2,
HasSubstr(llvm::formatv(" {0} ", Count)))));
code = "i";
code.append(Count, '9');
auto& buffer = compile_helper_.GetTokenizedBuffer(code, &consumer);
EXPECT_TRUE(buffer.has_errors());
ASSERT_THAT(buffer,
HasTokens(llvm::ArrayRef<ExpectedToken>{
{.kind = TokenKind::FileStart, .line = 1, .column = 1},
{.kind = TokenKind::Error,
.line = 1,
.column = 1,
.indent_column = 1,
.text = code},
{.kind = TokenKind::FileEnd, .line = 1, .column = Count + 2},
}));
ASSERT_TRUE(buffer.has_errors());
ASSERT_THAT(buffer, HasTokens(llvm::ArrayRef<ExpectedToken>{
{.kind = TokenKind::FileStart},
{.kind = TokenKind::Error, .text = code},
{.kind = TokenKind::FileEnd},
}));
}
TEST_F(LexerTest, DiagnosticTrailingComment) {