Add better algorithm for repairing mismatched brackets (#7574)

Adds an algorithm to compute where to insert brackets to repair
bracketing mismatches during lexing. This takes indentation, as well as
a number of other cues, into account to predict where the brackets
should have gone. Detects when there is ambiguity between solutions and
makes no suggestion in that case. Reduces the problem by splitting on
properly bracketed top-level constructs, then uses a beam search to find
good candidate solutions quickly.

This includes both a fuzzer and an eval tool that can be used to
determine how well the algorithm fares against a given corpus of valid
Carbon code, by damaging it in various ways and seeing whether the
algorithm can correctly fix it. On all the eval modes, this algorithm
can correctly infer the positions for over 80% of lost brackets (and can
correctly restore 95+% of brackets in some modes), with low rates of
incorrect suggestions.

See added documentation for full details.

Assisted-by: Gemini via Antigravity, Claude via Claude Code
This commit is contained in:
Richard Smith
2026-08-19 18:55:32 +00:00
committed by GitHub
parent c06165d3e0
commit 3bbc03f527
168 changed files with 6767 additions and 268 deletions
+20 -12
View File
@@ -591,11 +591,14 @@ TEST_F(LexerTest, MatchingGroups) {
TEST_F(LexerTest, MismatchedGroups) {
auto& buffer1 = compile_helper_.GetTokenizedBuffer("{");
EXPECT_TRUE(buffer1.has_errors());
EXPECT_THAT(buffer1, HasTokens(llvm::ArrayRef<ExpectedToken>{
{.kind = TokenKind::FileStart},
{.kind = TokenKind::Error, .text = "{"},
{.kind = TokenKind::FileEnd},
}));
EXPECT_THAT(
buffer1,
HasTokens(llvm::ArrayRef<ExpectedToken>{
{.kind = TokenKind::FileStart},
{.kind = TokenKind::OpenCurlyBrace, .column = 1},
{.kind = TokenKind::CloseCurlyBrace, .column = 2, .recovery = true},
{.kind = TokenKind::FileEnd},
}));
auto& buffer2 = compile_helper_.GetTokenizedBuffer("}");
EXPECT_TRUE(buffer2.has_errors());
@@ -618,9 +621,7 @@ TEST_F(LexerTest, MismatchedGroups) {
{.kind = TokenKind::FileEnd},
}));
// Two recovery tokens inserted at the same point: each must be flagged at
// its own merged index, not the pre-insertion index of the token it was
// inserted before.
// `{((}` is recovered by closing both parens before the `}`, i.e. `{(())}`.
auto& buffer3b = compile_helper_.GetTokenizedBuffer("{((}");
EXPECT_TRUE(buffer3b.has_errors());
EXPECT_THAT(
@@ -672,9 +673,12 @@ TEST_F(LexerTest, MismatchedGroups) {
}
TEST_F(LexerTest, Whitespace) {
// The trailing `{(` is recovered by inserting `)` and `}` at end of file,
// giving `{()} {()}`.
auto& buffer = compile_helper_.GetTokenizedBuffer("{( } {(");
// Whether there should be whitespace before/after each token.
// Whether there should be whitespace at each boundary, from before the
// first token to after the last.
bool space[] = {false,
// start-of-file
true,
@@ -683,13 +687,17 @@ TEST_F(LexerTest, Whitespace) {
// (
true,
// inserted )
true,
false,
// }
true,
// error {
// {
false,
// error (
// (
true,
// inserted )
false,
// inserted }
false,
// EOF
false};
int pos = 0;