From 3bbc03f52700443131c22bacafb0bc8e24b088aa Mon Sep 17 00:00:00 2001 From: Richard Smith Date: Wed, 19 Aug 2026 18:55:32 +0000 Subject: [PATCH] Add better algorithm for repairing mismatched brackets (#7574) Adds an algorithm to compute where to insert brackets to repair bracketing mismatches during lexing. This takes indentation, as well as a number of other cues, into account to predict where the brackets should have gone. Detects when there is ambiguity between solutions and makes no suggestion in that case. Reduces the problem by splitting on properly bracketed top-level constructs, then uses a beam search to find good candidate solutions quickly. This includes both a fuzzer and an eval tool that can be used to determine how well the algorithm fares against a given corpus of valid Carbon code, by damaging it in various ways and seeing whether the algorithm can correctly fix it. On all the eval modes, this algorithm can correctly infer the positions for over 80% of lost brackets (and can correctly restore 95+% of brackets in some modes), with low rates of incorrect suggestions. See added documentation for full details. Assisted-by: Gemini via Antigravity, Claude via Claude Code --- .codespell_ignore | 1 + toolchain/diagnostics/kind.def | 1 + toolchain/docs/lex.md | 6 +- toolchain/docs/lex/mismatched_brackets.md | 303 ++ .../testdata/fail_errors_in_two_files.carbon | 23 +- .../driver/testdata/fail_errors_sorted.carbon | 25 +- .../testdata/fail_errors_streamed.carbon | 5 +- .../document_symbol/incomplete.carbon | 62 +- toolchain/lex/BUILD | 65 + .../02291a60c819f2cc7ab15a85a7175d5973f19e25 | Bin 0 -> 72 bytes .../0414d6fe824625bc1530146da13aa32374fd46d6 | Bin 0 -> 628 bytes .../0555ecc1ef4513737146822f1fd51ae68a547467 | Bin 0 -> 1188 bytes .../059a5b5c84d754cbb4906f44b63728579e734631 | Bin 0 -> 68 bytes .../065042d64952c910d17ea859d81917576e302342 | Bin 0 -> 24 bytes .../08bcd130daabd96f58ee7b0715a78777560cfb26 | Bin 0 -> 587 bytes .../09e011ff19204249d39b625c3e9260b7c305f0c3 | Bin 0 -> 563 bytes .../0e4fb8ce694d85bf038298e4ad8fa7a0110431ff | Bin 0 -> 574 bytes .../0f6bb168f87053dac09200c3414d9b615f6910c4 | Bin 0 -> 916 bytes .../13ba392feb0e03ec01c993445f0a08b058ca2171 | Bin 0 -> 192 bytes .../18834d4a428bbeaccb8a404fe7fae0dcf93f4270 | Bin 0 -> 37 bytes .../19bed79c054e96809d4caf6bf1af64f4ece47b73 | 1 + .../1c7e413f92ad0e7ab4733319bbac0686b657296a | Bin 0 -> 1200 bytes .../1e25a64582780cb898d1cb4d7a74575cc09c7268 | Bin 0 -> 1194 bytes .../1e4b46a64fdc7696a12913c08dd93e784f4119b3 | Bin 0 -> 40 bytes .../22a82dfe1fbb9d53e9d61a1c1c4952163e361bb6 | Bin 0 -> 508 bytes .../23489ba87f2b31ac3db6673a24b7b469cbabda87 | Bin 0 -> 76 bytes .../2456c6a824c0665c2a5d00f35930ed9c6d347329 | Bin 0 -> 4 bytes .../269032960d96d517ffc75040c8ec95d2e3f0b183 | Bin 0 -> 8 bytes .../276d9f52aebb239dd4ecce745b77c4dd4b1b0578 | Bin 0 -> 237 bytes .../2d51bfcf695c49e0c4be32977c64c7cac505ce2c | Bin 0 -> 60 bytes .../2e606ac14b99b963c5afc7174780ab0d77e3b215 | Bin 0 -> 566 bytes .../300530eebfb3e6ac7f8f9a5f00651501909cc3f4 | Bin 0 -> 36 bytes .../31bc69f2e8affdea979f6e656ee8a574b8130ee7 | Bin 0 -> 607 bytes .../32457f5239eb7d81f6e53c1bc36373dcb1d9882f | Bin 0 -> 562 bytes .../32a8d791f0b805807b8a4041dc97471004329c5e | Bin 0 -> 1189 bytes .../33fb500b866385aaefed956a6ce249316d70d080 | Bin 0 -> 44 bytes .../35a1bea88022d11d4f58d7bd0796d11fa10b6357 | Bin 0 -> 56 bytes .../37a7413bade89fa21b77cff4a7b21d0692bf9f66 | Bin 0 -> 558 bytes .../3800fb3acb3c97ce0d3ce6657a37f757d1c5bc28 | Bin 0 -> 579 bytes .../38905a052321fb17f6f7b7cc4eb075f2391236f5 | Bin 0 -> 589 bytes .../39a8e924d0e1e988b9c1789ab5155b5b7ac30b36 | Bin 0 -> 580 bytes .../3a826d7d1c28b26b260230b7f3951ce05d60e918 | Bin 0 -> 680 bytes .../3efecca68da1229dace3e7d99d9a8b6c7549d6c1 | Bin 0 -> 8 bytes .../423d4c98d4f179bdc1b435106a482b6cf0ff95fe | Bin 0 -> 8 bytes .../42f36ad525f2f39e31cb88ae655a8d9ef7cde06f | Bin 0 -> 80 bytes .../433f63bbb76f9b121d22f9093032b45f4836090c | Bin 0 -> 646 bytes .../44435a420f8901a578c4f4b883999d598ba32dea | Bin 0 -> 28 bytes .../4770d82071135475785b554ca070e88c6e28bc01 | Bin 0 -> 12 bytes .../49c9f2b76a67f8f0f9661251bcadfacefc2a3bbf | Bin 0 -> 136 bytes .../4ac7946083e5366cbbaa4995d72811819131314d | Bin 0 -> 682 bytes .../4b0b3a93496859e1a291be3b23152e9c5c541105 | 2 + .../4c0480528122297ee012028ee0c0f1bfe7a5e6d6 | Bin 0 -> 513 bytes .../4f0577f8b986b9affe260cd84f9c06421707c4b1 | Bin 0 -> 48 bytes .../531cb419e85c387c49e91299f7e1ad82bc36c68d | Bin 0 -> 205 bytes .../53d13a0338eac1acb5c3c37752f98a3a7a4b023c | Bin 0 -> 549 bytes .../541b03a56c7b624a7885c674f517df068d2f3df1 | Bin 0 -> 12 bytes .../570cc51c1f9750bd5ed85e8ba31da8ee93689d51 | Bin 0 -> 73 bytes .../5a119a6eeee02301b24b76ee4bda22811aa2d97e | Bin 0 -> 132 bytes .../5ab28833044348ba5278604dd22ba0567b41b41e | Bin 0 -> 68 bytes .../5af6527f6893491b72d31d2eece1aec4623780de | Bin 0 -> 652 bytes .../5bbcd3f49cf21ae3a57a89d9a263338ec61d510b | Bin 0 -> 657 bytes .../5d73de1428afbe09c069d43ad284a5fe6f495973 | Bin 0 -> 572 bytes .../5f2b5aa0595d7fd3783a2d6bb6c009ce6fb714dc | Bin 0 -> 525 bytes .../628b92eaa822d952e20607173c729914504c5445 | Bin 0 -> 20 bytes .../62a6592b1da8b7e8c1a9699b6fb6ad53b68d9181 | Bin 0 -> 569 bytes .../6898560748989b2280634db5de86c68bc6dcde58 | Bin 0 -> 20 bytes .../69567a015e7a7aedcf174db8e9a1707b69d96ebd | 2 + .../69f71552dbc82df457e247a2f988c46a0184fc14 | Bin 0 -> 569 bytes .../6a323e0daac7358ba8c7f1e3f392877190e4a99c | Bin 0 -> 24 bytes .../6cfa1c02aee7e8c2993901381416b1b46f6afaf2 | Bin 0 -> 619 bytes .../6f2a9d134179f78ef2d14da59c88423bee6ec263 | Bin 0 -> 1236 bytes .../724e9d106919f6bfaf00446115072df31b2652b1 | Bin 0 -> 24 bytes .../76eae89b045b5d336c735f2a96e38a95f95311c3 | Bin 0 -> 2001 bytes .../7c2ea32624e918c4e4bf89e41e40390c46b1f627 | Bin 0 -> 700 bytes .../7cca047c3f7c202ff0871a4272ba1f6879794b79 | Bin 0 -> 560 bytes .../7d0b0fcf5acd4ced16f3d1a8dbf57f4d60bec85c | Bin 0 -> 16 bytes .../7fe8dcf2ae676ca44a59797207a5637776e7f2b1 | Bin 0 -> 20 bytes .../8326ae645e600f2e13ed57db6a221d109a1c8454 | Bin 0 -> 20 bytes .../842ca9afb9e4daa2535a51180423c957ddde3ff1 | Bin 0 -> 136 bytes .../84803ce199f02a6bf9ba5ac0fdce24f928343bdf | Bin 0 -> 16 bytes .../8ad4f3e9bd4cffdf5ef5caa05d405a39f9944901 | Bin 0 -> 32 bytes .../8c651b261db765b2d48f09d0d021b1283ca6c7cc | Bin 0 -> 32 bytes .../8d0cc0e6fb6e5fc9bae742024f42b841d68b7faa | Bin 0 -> 512 bytes .../8eb7b216072c947527540228feab2753e36fea31 | Bin 0 -> 12 bytes .../90f5e0abe9dcd825ff129821b3fe459b45d4fec7 | Bin 0 -> 28 bytes .../911cfb8a1f361083f0300f62796a00ff223e7f4e | Bin 0 -> 967 bytes .../92034fdaa329a6d4b42b74beed99b3ad6c7e9a2a | Bin 0 -> 504 bytes .../92884d0852d1a40b4d9d2a9edda160a08c6de1ca | 2 + .../9565a88574a575258c68a9185b6725c42970134f | Bin 0 -> 584 bytes .../95774d49a60bc80d0327e11a2d0d3bb96c34ef5c | Bin 0 -> 341 bytes .../96363ea89ebd24a2df43591854551bbe8bd613d3 | Bin 0 -> 16 bytes .../9bc418a5119d526ee866e9a47dff66ee3417baca | Bin 0 -> 40 bytes .../a1342d9d00f39be49a7756c4f959462366b0dd8a | Bin 0 -> 169 bytes .../a688400d1ff128a68e9d8cc533fff3f02434ef1d | Bin 0 -> 1110 bytes .../a8f6d9d0870707a5959b8d4cc3a453fb7cea6ddd | Bin 0 -> 8 bytes .../a9d82a1c17c7606f140406cf0d28ea2b9f085d23 | Bin 0 -> 602 bytes .../aa90de943c210fa36f3009dd955a0f679de63f82 | Bin 0 -> 160 bytes .../aa92e70ca6c167c7259af670a52dfd98c8eeb566 | Bin 0 -> 1246 bytes .../ad44c2ee0a0573728e37ec9e2176f87bcc4c8077 | Bin 0 -> 809 bytes .../ae4f281df5a5d0ff3cad6371f76d5c29b6d953ec | 1 + .../ae5f685cecfd3b1e34b1b382a6e48ca94f8f78c6 | Bin 0 -> 12 bytes .../b07d35856888d003b4d157460fcf4e6b13fc2ef3 | Bin 0 -> 1181 bytes .../b09ee244f6afcb09fc98300a976721bdcdd6e83b | Bin 0 -> 68 bytes .../b0b0f529b8fbc71a44ccea7b08e528ecfbf72868 | Bin 0 -> 552 bytes .../b0b12c4a7676e78b62ec5078bd56b9061dcfc290 | Bin 0 -> 600 bytes .../b50a1858c3dc8e84332b73093b99ea2822e252ee | Bin 0 -> 128 bytes .../b53c234830eb46d17db95aa5c50259fc07a94ed3 | Bin 0 -> 40 bytes .../b684c0a65507a9a780310a5bfc2d440c16543e5a | Bin 0 -> 1113 bytes .../b7b271a26b942ca7efb9d26f3ba98970ddc86e2f | Bin 0 -> 562 bytes .../ba6497e3fbb353a9c99cf2726c2f3e23924c1cbf | Bin 0 -> 16 bytes .../bbb9c74f6c873f94f88ea0c651f0c1b388c4133e | Bin 0 -> 584 bytes .../bbf2ce06fc350f4c96366eea7d884a63fbe6fee0 | Bin 0 -> 152 bytes .../bd61b3d8f7935b75efc8de6d06141fbefcce5b01 | Bin 0 -> 642 bytes .../bd844ee1a1aa7210f16f161f94261f48ba025d22 | Bin 0 -> 176 bytes .../bfe7cf0386b32de999e6df7dc21b12697a6a37d1 | 2 + .../c021157170822334d9a0d61484e397da4bdd2a9d | Bin 0 -> 14 bytes .../c145b4235b15e63df7d6e2416a5be5d2833c9b2e | Bin 0 -> 53 bytes .../c44576da3a941248af87bc80e73c9b53c6b5b03c | Bin 0 -> 689 bytes .../c9d317a48760e299e87110bb568d3cc61a0dbd6f | Bin 0 -> 1153 bytes .../cb52eff5b914e5ce27bb12e9c84db8a197f55458 | Bin 0 -> 36 bytes .../cbf41ea731049ebf1108a27af5b67daa35439052 | Bin 0 -> 582 bytes .../cd56e6c481065717030fb754daf2b4a9c8684677 | Bin 0 -> 12 bytes .../cdbdfdd6934b9992c6b39cde391fd0203b51f50c | Bin 0 -> 516 bytes .../cfdb377027ea66b748dbde05f3123b9be7cc5ad6 | Bin 0 -> 724 bytes .../d0b3c063c83c3e2974bd1f6fc0c7e4bd830df77b | Bin 0 -> 611 bytes .../d1ec0788b87512574e7aee146742a47eb040b17e | Bin 0 -> 704 bytes .../d2b2295c9d5982db079ffe02850d21d965b6ddc2 | Bin 0 -> 551 bytes .../d455faaab92a6a739c2e1c863dcaf9c5fd112401 | Bin 0 -> 4 bytes .../d58e04391ec50b86e89eaa72cee255c0b0206935 | Bin 0 -> 32 bytes .../d5f34c9240ddb934c3a2eafa51cd34b94bb7ea5c | Bin 0 -> 64 bytes .../d6acff1671e452c53a6c85e3b0971fbf704d9d44 | Bin 0 -> 264 bytes .../d737bb16db6497859637d61be876cb4ad0d5da11 | Bin 0 -> 44 bytes .../d7629f122ce603bda8ecd588f9f20e6b938046a8 | Bin 0 -> 565 bytes .../d89f546e0556b5690f105dd5d31fdf7fc97a1c94 | Bin 0 -> 84 bytes .../d937e3e7989f4cdb2e9e96653e581ad9515280a6 | Bin 0 -> 124 bytes .../de3d0e9bf501a8eac7ecf730ba95ac050f426760 | Bin 0 -> 100 bytes .../dee33c587880622173a253cf90a50df677f1be3f | Bin 0 -> 1242 bytes .../e4e8bf0a6b9f3842dcb4414914456eb056c1286f | Bin 0 -> 630 bytes .../e531ca673cb1539c6ec88007288e3f0ee7bca076 | Bin 0 -> 16 bytes .../e61fac0d819dc59655318ab3fbc97b2c5f8036c4 | Bin 0 -> 1196 bytes .../ea32a24630718593936ff0283ec073626a7712e0 | Bin 0 -> 553 bytes .../eab033cad3db5fdcefc3933bb037c3ad33f72210 | Bin 0 -> 512 bytes .../ec295ddebf8fdd6bb466d11aaa259271f7b67e50 | 1 + .../ede772c4d5f471c9b5d0116c40436e90486a2d85 | Bin 0 -> 613 bytes .../eeebad550d517e1093c9f7d19e98775e56a758f3 | Bin 0 -> 8 bytes .../f0be983c6a82d192728766326471d6247fee95cb | Bin 0 -> 8 bytes .../f3f29b7a46f0c534890c76378c00942248be6edb | Bin 0 -> 565 bytes .../f68e794689809ad81a60922c6a20cfc845aecd1b | Bin 0 -> 611 bytes .../f6e6185a8f57674faeb1af4865995bc778ef1b84 | Bin 0 -> 120 bytes .../f85ec3589a345f5bb51517c1759ffb8792a357c2 | Bin 0 -> 24 bytes .../ff18fde4618c218b3a8ef00607467cad5126cd8c | Bin 0 -> 1682 bytes toolchain/lex/lex.cpp | 614 +++- toolchain/lex/lex.h | 7 + toolchain/lex/mismatched_brackets.cpp | 2656 +++++++++++++++++ toolchain/lex/mismatched_brackets.h | 223 ++ toolchain/lex/mismatched_brackets_eval.cpp | 1547 ++++++++++ toolchain/lex/mismatched_brackets_fuzzer.cpp | 196 ++ toolchain/lex/mismatched_brackets_test.cpp | 330 ++ .../testdata/fail_mismatched_brackets.carbon | 614 +++- .../fail_mismatched_brackets_2.carbon | 39 - ...ail_mismatched_brackets_close_brace.carbon | 174 ++ toolchain/lex/token_info.h | 8 + toolchain/lex/tokenized_buffer_test.cpp | 32 +- .../parse/testdata/array/fail_syntax.carbon | 29 +- .../basics/fail_bracket_recovery.carbon | 20 +- .../testdata/function/declaration.carbon | 15 +- .../fail_missing_guard_close_paren.carbon | 5 +- .../fail_missing_guard_open_paren.carbon | 24 +- 168 files changed, 6767 insertions(+), 268 deletions(-) create mode 100644 toolchain/docs/lex/mismatched_brackets.md create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/02291a60c819f2cc7ab15a85a7175d5973f19e25 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/0414d6fe824625bc1530146da13aa32374fd46d6 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/0555ecc1ef4513737146822f1fd51ae68a547467 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/059a5b5c84d754cbb4906f44b63728579e734631 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/065042d64952c910d17ea859d81917576e302342 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/08bcd130daabd96f58ee7b0715a78777560cfb26 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/09e011ff19204249d39b625c3e9260b7c305f0c3 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/0e4fb8ce694d85bf038298e4ad8fa7a0110431ff create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/0f6bb168f87053dac09200c3414d9b615f6910c4 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/13ba392feb0e03ec01c993445f0a08b058ca2171 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/18834d4a428bbeaccb8a404fe7fae0dcf93f4270 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/19bed79c054e96809d4caf6bf1af64f4ece47b73 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/1c7e413f92ad0e7ab4733319bbac0686b657296a create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/1e25a64582780cb898d1cb4d7a74575cc09c7268 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/1e4b46a64fdc7696a12913c08dd93e784f4119b3 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/22a82dfe1fbb9d53e9d61a1c1c4952163e361bb6 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/23489ba87f2b31ac3db6673a24b7b469cbabda87 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/2456c6a824c0665c2a5d00f35930ed9c6d347329 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/269032960d96d517ffc75040c8ec95d2e3f0b183 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/276d9f52aebb239dd4ecce745b77c4dd4b1b0578 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/2d51bfcf695c49e0c4be32977c64c7cac505ce2c create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/2e606ac14b99b963c5afc7174780ab0d77e3b215 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/300530eebfb3e6ac7f8f9a5f00651501909cc3f4 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/31bc69f2e8affdea979f6e656ee8a574b8130ee7 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/32457f5239eb7d81f6e53c1bc36373dcb1d9882f create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/32a8d791f0b805807b8a4041dc97471004329c5e create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/33fb500b866385aaefed956a6ce249316d70d080 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/35a1bea88022d11d4f58d7bd0796d11fa10b6357 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/37a7413bade89fa21b77cff4a7b21d0692bf9f66 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/3800fb3acb3c97ce0d3ce6657a37f757d1c5bc28 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/38905a052321fb17f6f7b7cc4eb075f2391236f5 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/39a8e924d0e1e988b9c1789ab5155b5b7ac30b36 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/3a826d7d1c28b26b260230b7f3951ce05d60e918 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/3efecca68da1229dace3e7d99d9a8b6c7549d6c1 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/423d4c98d4f179bdc1b435106a482b6cf0ff95fe create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/42f36ad525f2f39e31cb88ae655a8d9ef7cde06f create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/433f63bbb76f9b121d22f9093032b45f4836090c create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/44435a420f8901a578c4f4b883999d598ba32dea create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/4770d82071135475785b554ca070e88c6e28bc01 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/49c9f2b76a67f8f0f9661251bcadfacefc2a3bbf create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/4ac7946083e5366cbbaa4995d72811819131314d create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/4b0b3a93496859e1a291be3b23152e9c5c541105 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/4c0480528122297ee012028ee0c0f1bfe7a5e6d6 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/4f0577f8b986b9affe260cd84f9c06421707c4b1 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/531cb419e85c387c49e91299f7e1ad82bc36c68d create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/53d13a0338eac1acb5c3c37752f98a3a7a4b023c create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/541b03a56c7b624a7885c674f517df068d2f3df1 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/570cc51c1f9750bd5ed85e8ba31da8ee93689d51 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/5a119a6eeee02301b24b76ee4bda22811aa2d97e create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/5ab28833044348ba5278604dd22ba0567b41b41e create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/5af6527f6893491b72d31d2eece1aec4623780de create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/5bbcd3f49cf21ae3a57a89d9a263338ec61d510b create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/5d73de1428afbe09c069d43ad284a5fe6f495973 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/5f2b5aa0595d7fd3783a2d6bb6c009ce6fb714dc create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/628b92eaa822d952e20607173c729914504c5445 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/62a6592b1da8b7e8c1a9699b6fb6ad53b68d9181 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/6898560748989b2280634db5de86c68bc6dcde58 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/69567a015e7a7aedcf174db8e9a1707b69d96ebd create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/69f71552dbc82df457e247a2f988c46a0184fc14 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/6a323e0daac7358ba8c7f1e3f392877190e4a99c create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/6cfa1c02aee7e8c2993901381416b1b46f6afaf2 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/6f2a9d134179f78ef2d14da59c88423bee6ec263 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/724e9d106919f6bfaf00446115072df31b2652b1 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/76eae89b045b5d336c735f2a96e38a95f95311c3 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/7c2ea32624e918c4e4bf89e41e40390c46b1f627 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/7cca047c3f7c202ff0871a4272ba1f6879794b79 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/7d0b0fcf5acd4ced16f3d1a8dbf57f4d60bec85c create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/7fe8dcf2ae676ca44a59797207a5637776e7f2b1 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/8326ae645e600f2e13ed57db6a221d109a1c8454 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/842ca9afb9e4daa2535a51180423c957ddde3ff1 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/84803ce199f02a6bf9ba5ac0fdce24f928343bdf create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/8ad4f3e9bd4cffdf5ef5caa05d405a39f9944901 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/8c651b261db765b2d48f09d0d021b1283ca6c7cc create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/8d0cc0e6fb6e5fc9bae742024f42b841d68b7faa create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/8eb7b216072c947527540228feab2753e36fea31 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/90f5e0abe9dcd825ff129821b3fe459b45d4fec7 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/911cfb8a1f361083f0300f62796a00ff223e7f4e create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/92034fdaa329a6d4b42b74beed99b3ad6c7e9a2a create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/92884d0852d1a40b4d9d2a9edda160a08c6de1ca create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/9565a88574a575258c68a9185b6725c42970134f create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/95774d49a60bc80d0327e11a2d0d3bb96c34ef5c create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/96363ea89ebd24a2df43591854551bbe8bd613d3 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/9bc418a5119d526ee866e9a47dff66ee3417baca create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/a1342d9d00f39be49a7756c4f959462366b0dd8a create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/a688400d1ff128a68e9d8cc533fff3f02434ef1d create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/a8f6d9d0870707a5959b8d4cc3a453fb7cea6ddd create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/a9d82a1c17c7606f140406cf0d28ea2b9f085d23 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/aa90de943c210fa36f3009dd955a0f679de63f82 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/aa92e70ca6c167c7259af670a52dfd98c8eeb566 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/ad44c2ee0a0573728e37ec9e2176f87bcc4c8077 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/ae4f281df5a5d0ff3cad6371f76d5c29b6d953ec create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/ae5f685cecfd3b1e34b1b382a6e48ca94f8f78c6 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/b07d35856888d003b4d157460fcf4e6b13fc2ef3 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/b09ee244f6afcb09fc98300a976721bdcdd6e83b create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/b0b0f529b8fbc71a44ccea7b08e528ecfbf72868 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/b0b12c4a7676e78b62ec5078bd56b9061dcfc290 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/b50a1858c3dc8e84332b73093b99ea2822e252ee create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/b53c234830eb46d17db95aa5c50259fc07a94ed3 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/b684c0a65507a9a780310a5bfc2d440c16543e5a create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/b7b271a26b942ca7efb9d26f3ba98970ddc86e2f create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/ba6497e3fbb353a9c99cf2726c2f3e23924c1cbf create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/bbb9c74f6c873f94f88ea0c651f0c1b388c4133e create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/bbf2ce06fc350f4c96366eea7d884a63fbe6fee0 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/bd61b3d8f7935b75efc8de6d06141fbefcce5b01 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/bd844ee1a1aa7210f16f161f94261f48ba025d22 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/bfe7cf0386b32de999e6df7dc21b12697a6a37d1 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/c021157170822334d9a0d61484e397da4bdd2a9d create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/c145b4235b15e63df7d6e2416a5be5d2833c9b2e create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/c44576da3a941248af87bc80e73c9b53c6b5b03c create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/c9d317a48760e299e87110bb568d3cc61a0dbd6f create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/cb52eff5b914e5ce27bb12e9c84db8a197f55458 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/cbf41ea731049ebf1108a27af5b67daa35439052 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/cd56e6c481065717030fb754daf2b4a9c8684677 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/cdbdfdd6934b9992c6b39cde391fd0203b51f50c create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/cfdb377027ea66b748dbde05f3123b9be7cc5ad6 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/d0b3c063c83c3e2974bd1f6fc0c7e4bd830df77b create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/d1ec0788b87512574e7aee146742a47eb040b17e create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/d2b2295c9d5982db079ffe02850d21d965b6ddc2 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/d455faaab92a6a739c2e1c863dcaf9c5fd112401 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/d58e04391ec50b86e89eaa72cee255c0b0206935 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/d5f34c9240ddb934c3a2eafa51cd34b94bb7ea5c create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/d6acff1671e452c53a6c85e3b0971fbf704d9d44 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/d737bb16db6497859637d61be876cb4ad0d5da11 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/d7629f122ce603bda8ecd588f9f20e6b938046a8 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/d89f546e0556b5690f105dd5d31fdf7fc97a1c94 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/d937e3e7989f4cdb2e9e96653e581ad9515280a6 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/de3d0e9bf501a8eac7ecf730ba95ac050f426760 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/dee33c587880622173a253cf90a50df677f1be3f create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/e4e8bf0a6b9f3842dcb4414914456eb056c1286f create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/e531ca673cb1539c6ec88007288e3f0ee7bca076 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/e61fac0d819dc59655318ab3fbc97b2c5f8036c4 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/ea32a24630718593936ff0283ec073626a7712e0 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/eab033cad3db5fdcefc3933bb037c3ad33f72210 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/ec295ddebf8fdd6bb466d11aaa259271f7b67e50 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/ede772c4d5f471c9b5d0116c40436e90486a2d85 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/eeebad550d517e1093c9f7d19e98775e56a758f3 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/f0be983c6a82d192728766326471d6247fee95cb create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/f3f29b7a46f0c534890c76378c00942248be6edb create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/f68e794689809ad81a60922c6a20cfc845aecd1b create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/f6e6185a8f57674faeb1af4865995bc778ef1b84 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/f85ec3589a345f5bb51517c1759ffb8792a357c2 create mode 100644 toolchain/lex/fuzzer_corpus/mismatched_brackets/ff18fde4618c218b3a8ef00607467cad5126cd8c create mode 100644 toolchain/lex/mismatched_brackets.cpp create mode 100644 toolchain/lex/mismatched_brackets.h create mode 100644 toolchain/lex/mismatched_brackets_eval.cpp create mode 100644 toolchain/lex/mismatched_brackets_fuzzer.cpp create mode 100644 toolchain/lex/mismatched_brackets_test.cpp delete mode 100644 toolchain/lex/testdata/fail_mismatched_brackets_2.carbon create mode 100644 toolchain/lex/testdata/fail_mismatched_brackets_close_brace.carbon diff --git a/.codespell_ignore b/.codespell_ignore index 67eebc7dd207..c612f438f3ff 100644 --- a/.codespell_ignore +++ b/.codespell_ignore @@ -3,6 +3,7 @@ # SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception AggregateT +AnyOther ArchType atleast circularly diff --git a/toolchain/diagnostics/kind.def b/toolchain/diagnostics/kind.def index 1068035f3f09..11d2704c891c 100644 --- a/toolchain/diagnostics/kind.def +++ b/toolchain/diagnostics/kind.def @@ -87,6 +87,7 @@ CARBON_DIAGNOSTIC_KIND(UnsupportedCrLineEnding) CARBON_DIAGNOSTIC_KIND(UnsupportedLfCrLineEnding) CARBON_DIAGNOSTIC_KIND(UnmatchedOpening) CARBON_DIAGNOSTIC_KIND(UnmatchedClosing) +CARBON_DIAGNOSTIC_KIND(PossiblyMissingBracketHere) CARBON_DIAGNOSTIC_KIND(UnrecognizedCharacters) CARBON_DIAGNOSTIC_KIND(UnterminatedString) CARBON_DIAGNOSTIC_KIND(WrongRealLiteralExponent) diff --git a/toolchain/docs/lex.md b/toolchain/docs/lex.md index 94a6aeaa6a3e..22f61e2f93a3 100644 --- a/toolchain/docs/lex.md +++ b/toolchain/docs/lex.md @@ -27,8 +27,10 @@ The lexer handles matching for `()`, `[]`, and `{}`. When a bracket lacks a match, it will insert a "recovery" token to produce a match. As a consequence, the lexer's output should always have matched brackets, even with invalid code. -While bracket matching could use hints such as contextual clues from -indentation, that is not yet implemented. +Where the missing bracket belongs is decided by a search over candidate repairs, +using indentation, structural facts about the language, and formatting +conventions as evidence. See +[mismatched bracket recovery](lex/mismatched_brackets.md). ## Alternatives considered diff --git a/toolchain/docs/lex/mismatched_brackets.md b/toolchain/docs/lex/mismatched_brackets.md new file mode 100644 index 000000000000..6eef3f2485db --- /dev/null +++ b/toolchain/docs/lex/mismatched_brackets.md @@ -0,0 +1,303 @@ +# Mismatched bracket recovery + + + + + +## Table of contents + +- [Overview](#overview) +- [Repairs](#repairs) +- [Reducing the problem](#reducing-the-problem) + - [Token abstraction](#token-abstraction) + - [Trusting matched pairs](#trusting-matched-pairs) + - [Regions](#regions) +- [Searching](#searching) +- [The cost model](#the-cost-model) +- [Cues](#cues) +- [Ambiguity](#ambiguity) +- [Reporting and applying repairs](#reporting-and-applying-repairs) +- [Evaluation](#evaluation) +- [Limitations](#limitations) + + + +## Overview + +Everything downstream of lexing assumes the token stream is well bracketed: each +`(`, `[`, and `{` has a matching closer, and the two cross-reference each other. +When the source doesn't satisfy that, the lexer repairs the stream before +handing it on, inserting recovery tokens and turning brackets it can't place +into error tokens. + +The interesting question is not _whether_ a bracket is unmatched — a stack +matcher answers that — but _where_ the missing one belongs. Recovery has to pick +an insertion point, and that choice matters twice over: it is what the +diagnostic tells the developer, and it decides how much of the rest of the file +is misinterpreted. A `}` missing from the middle of a class definition, closed +at end of file instead of where it belongs, turns every following declaration +into a member of that class. + +So we treat recovery as a search: enumerate ways to make a damaged part of the +file well bracketed, price each by how plausible the mistake it implies is, and +take the cheapest. The prices form a cost model calibrated by measurement +against real Carbon code, and the evidence it reasons from is indentation, +structural facts about the language, and formatting conventions. + +`FixMismatchedBrackets` is the whole of that: it takes the token sequence and +returns a list of `BracketCorrection`s, each pairing a diagnostic to report with +a fix to apply to the token stream. Nothing else in the lexer knows how the +decision was made. + +## Repairs + +A repair is built from four moves: + +- Insert a closing bracket directly before a token: the group ended here. +- Insert an opening bracket directly before a token. A synthetic opener only + becomes a suggestion once a real closer matches it, since that is the point + at which we know which closer it explains; a repair that reaches the end of + the region with a synthetic opener still unmatched is rejected rather than + inventing an empty pair. +- Close whatever is still open when the region or the file ends. +- Replace a bracket with an error token: give up on it. The token stays in the + stream but carries no bracket structure, so nothing downstream tries to pair + it. + +Giving up is what lets the rest of the model be assertive. It is always +available at a fixed price, so it acts as a ceiling: a repair less plausible +than admitting defeat is not chosen. + +## Reducing the problem + +### Token abstraction + +Recovery does not see tokens. It sees a `MismatchedBracketToken` per token: a +projection onto the properties bracket structure could depend on. That is a +coarse `BracketTokenKind` (the six brackets, `;`, `,`, `.`, statement +introducers, _leaf_ tokens that form a complete primary expression on their own, +three flavors of operator, end of file, and everything else), the line and +indentation, and a few flags — whether a `{` looks like a struct literal, +whether a keyword demands a bracket after it, whether whitespace precedes the +token. Nothing else about the source is available, which keeps the search cheap +and the rules easy to reason about. + +### Trusting matched pairs + +An ordinary stack match runs first. Most of the pairs it finds are genuine, and +reconsidering them would be both slow and risky, so every pair that passes a +cleanliness test collapses into a single `Item` that the search steps over as a +unit. `Item` is the search's unit of input from here on: one per token, except +where a trusted pair spans several. + +Cleanliness is deliberately conservative. A brace pair is trusted when its `}` +lines up with the indentation of the statement that owns the `{` — not with the +column of the `{`, which for a header wrapped across lines is somewhere else +entirely — and when the separators directly inside it suit the brace's flavor: a +`;` directly inside a struct literal, or a `,` directly inside a block, means +the pairing has captured too much. A paren or square pair is suspect when an +unmatched opener of the same kind appears earlier in the same statement, or an +unmatched closer of the same kind later, because either could be the real +partner. Finally, the interior must be clean itself, with nothing inside it +pairing to something outside. + +### Regions + +The resulting item sequence is cut at top-level declaration boundaries: a +statement introducer at column 1 whose predecessor ended a statement. Each +region is solved independently. That bounds how far one mistake can smear, and +it gives unclosed brackets a natural place to be closed — before the next +declaration, rather than at end of file. + +Regions whose remaining loose brackets already balance are skipped without +searching at all. That is the large majority of them, since a region typically +holds pairs that merely failed the cleanliness test, and skipping them is worth +an order of magnitude in running time. + +## Searching + +`RegionSearch` solves one region. A search state — a `BeamNode` — is a position +in the item sequence plus the stack of brackets open at that point, together +with which closer, if any, was inserted at this position already. + +The search runs in layers, one per item. Within a layer, insertion moves are +applied: these don't consume the item, so they stay in the layer and can chain, +which is how several groups get closed at one point. Then every surviving state +advances over the item into the next layer. Both phases prune to a fixed beam +width, keeping the cheapest states. + +Two states in a layer whose stacks agree are interchangeable, so they merge. A +merged state retains _every_ cheapest way it was reached rather than one, which +is what makes ambiguity visible at the end. + +The search is bounded in each direction: beam width, stack depth, items per +region, and paths enumerated. Exceeding a bound costs quality, never correctness +— the fallback for an oversized region is a naive greedy matcher, and the output +is well bracketed either way. Beam search offers no guarantee of finding the +cheapest repair, but an intended repair is a cheap one by construction, so it +survives pruning. + +## The cost model + +Costs are ordinal; only their ordering carries meaning. They live in three +tables of `BracketRule`s: + +- `CloserRules` prices inserting a closing bracket before the current token. +- `OpenerRules` prices inserting an opening bracket before it. +- `AdvanceRules` prices stepping over the token with the current group still + open. + +The first two are first-match tables, ordered from the most specific cue to the +least, and a row may decline outright, meaning the move is not worth considering +there. The third is additive: every matching row contributes and the penalties +sum. That table is the other half of the same judgment as the first: a penalty +for swallowing a token that has no business inside the current group is what +makes "close the group before it" win. + +A lookup buckets on two categorical facts — the context category (which bracket +is innermost, or which one would be inserted, with block and struct braces +distinguished) and the token's kind — and a compile-time index maps each bucket +to the rows that could apply in it, so only a handful are ever tested. Rows +condition further on `Cue`s, as sets of properties that must all hold, must not +hold, must not all hold, or must hold in part. + +The model is calibrated on a single principle: the intended repair must be +_strictly_ cheapest. A tie is not resolved but reported, as described below, so +a rule that gets the right answer only about half the time is worse than no rule +at all. + +## Cues + +- **Indentation.** A line dedented to or past the indentation of the statement + that opened a block ends that block. The statement's indentation is what + counts, not the column of the `{`: a `}` lines up with its `if` even when + the `{` sits at the end of a wrapped header. A first-on-line `else` is + treated the same way, by walking back over complete blocks to the branch it + continues. +- **Facts about the language.** A `;` cannot appear inside `(` or `[` at all. + A block `{` cannot be the content of a keyword's parentheses or of a struct + literal. A binary operator cannot start a group. A leaf cannot directly + follow a value-ending token, so finding that pair is evidence a bracket is + missing between them. `if`, `while`, `for`, and `match` require a following + `(`, and `forall` a following `[`. +- **Cascades.** Closing a group at a point where an enclosing group is already + being closed is cheap, because a run of missing closers is usually a single + mistake. +- **Formatting conventions.** Formatted Carbon writes no space before `,`, + `)`, or `.`, and none before a call's `(`. A space in one of those positions + means the code is unformatted or something was deleted from the gap, and + either way is weak evidence that a bracket belongs there. Such cues can only + fire on input that isn't formatted, which is what makes them safe to state + liberally. +- **Position in the file.** Closing at the end of a region or of the file has + a fixed price, set above every precise cue, so that a real cue always wins + and an unclosed group only runs to the region end when nothing local + explains it. + +## Ambiguity + +Once the cheapest cost is known, every repair achieving it is enumerated, up to +a cap. A suggestion present in one cheapest repair but absent from another is +ambiguous: the model has no basis for preferring either placement. Rather than +guess, we discard the suggestion and replace the bracket with an error token. +The developer is told the bracket is unmatched and is not pointed anywhere +misleading, and the parser receives a stream with no invented structure. + +This is the reason the tables are ordered and priced as deliberately as they +are. Two competing repairs at distinguishable prices produce one answer; the +same two at equal prices produce none. + +## Reporting and applying repairs + +Each correction produces an error at the bracket that has no partner, plus a +note at the position the missing bracket would occupy. That position lies +between two tokens rather than at one, so the note is emitted against a source +pointer: immediately past the end of the token the bracket would follow, then +adjusted for how the bracket is conventionally written. A `}` at end of line +moves past the newline and in to its opener's indentation, since it belongs on a +line of its own, and an opening bracket moves past any spaces, since it binds to +what follows. + +Insertions are buffered in an `ErrorRecoveryBuffer` and applied in one pass that +renumbers the token stream, after which bracket cross-references are recomputed. +Insertions requested at the same anchor are ordered closers before openers, so +that the result is well nested. + +## Evaluation + +`mismatched_brackets_eval` measures recovery by corrupting known-good code and +asking whether recovery reconstructs it. It deletes brackets from the Carbon +files in this repository, re-lexes, and compares the suggestions against what it +deleted: + +```shell +bazel run -c opt //toolchain/lex:mismatched_brackets_eval -- \ + --trials=4587 --d-values=1 --mode=gapless +``` + +There are four corruption modes, because how a bracket goes missing changes what +evidence survives: + +- **blank** replaces the bracket with a space, preserving byte offsets. Cheap + to reason about, but the space it leaves is itself a cue, so this mode + flatters the formatting rules. +- **gapless** deletes the bracket and closes up the text, leaving no space + behind and so no artifact. This is the mode to trust for formatted input. +- **truncate** cuts the file at a random token, so that everything still open + is closed at end of file. +- **truncate-region** deletes from a declaration boundary through the close of + a pair, modeling code typed into an existing class. Over-extending is + expensive here: swallowing the following declaration is the failure recovery + exists to prevent. + +Scoring is by structure, not position. A suggestion is matched to the deletion +it restores through the first surviving token it precedes, so closing anywhere +in trailing whitespace, or anywhere within a run of identical brackets, counts +as the same repair, while closing somewhere that changes the structure — +re-pairing a surviving bracket, swallowing the next declaration — does not. Each +trial is then Correct, Partial, None, or Incorrect. + +Those outcomes are not weighted equally. An Incorrect suggestion misparses the +file and points the developer at the wrong place, whereas a trial with no +suggestion falls back to an error token and costs only the missing note. The +model is therefore tuned to minimize Incorrect first and maximize Correct +second. + +Every rule carries a name, which the eval aggregates into per-rule firing counts +and precision. That is how an over-firing rule is found, and it is also what +makes the cost column tunable mechanically, by coordinate descent against the +eval rather than by hand. + +`--d-values` sets how many brackets a trial deletes, which is worth raising to +see how gracefully recovery degrades when mistakes overlap. With one deletion +per trial, as of this writing: + +| Mode | Correct | Incorrect | +| --------------- | ------- | --------- | +| blank | 95.6% | 0.0% | +| gapless | 82.3% | 4.3% | +| truncate | 98.9% | 0.9% | +| truncate-region | 82.6% | 16.5% | + +The `fail_mismatched_brackets` tests in [testdata](/toolchain/lex/testdata) are +a showcase rather than a corpus: each case is a shape recovery handles, with a +comment naming the cue that decides it. + +## Limitations + +- Recovery has no grammar. Its cues are local and categorical, so a bracket + missing from deep inside an expression, with nothing unusual around it, is + reported but not placed. +- Region splitting relies on a top-level introducer at column 1. A file that + never returns to top level is a single region, in which a mistake can smear + further. +- Costs are calibrated against this repository's own code in this repository's + formatting. Substantially different style may want retuning. +- The largest remaining source of wrong suggestions is a deleted tail: when + the text that vanished held both the closer and the tokens that would have + said where it goes, an unclosed `(` or `[` has nothing local to work from + and over-extends. diff --git a/toolchain/driver/testdata/fail_errors_in_two_files.carbon b/toolchain/driver/testdata/fail_errors_in_two_files.carbon index a23d52cd18f9..05c95a7075d9 100644 --- a/toolchain/driver/testdata/fail_errors_in_two_files.carbon +++ b/toolchain/driver/testdata/fail_errors_in_two_files.carbon @@ -10,19 +10,30 @@ // TIP: To dump output, run: // TIP: bazel run //toolchain/testing:file_test -- --dump_output --file_tests=toolchain/driver/testdata/fail_errors_in_two_files.carbon +// The `}` that recovery suggests goes at the start of the line after +// `return True;`, so the `CHECK`s must live in their own split: autoupdate would +// otherwise write them at that very position, and the note would quote its own +// `CHECK` line. + // --- fail_file1.carbon -// CHECK:STDERR: fail_file1.carbon:[[@LINE+4]]:24: error: opening symbol without a corresponding closing symbol -// CHECK:STDERR: fn run(String program) { -// CHECK:STDERR: ^ -// CHECK:STDERR: fn run(String program) { return True; // --- fail_file2.carbon -// CHECK:STDERR: fail_file2.carbon:[[@LINE+4]]:10: error: invalid digit 'a' in decimal numeric literal +var x = 3a; + +// --- AUTOUPDATE-SPLIT + +// CHECK:STDERR: fail_file1.carbon:2:24: error: opening symbol without a corresponding closing symbol +// CHECK:STDERR: fn run(String program) { +// CHECK:STDERR: ^ +// CHECK:STDERR: fail_file1.carbon:4:1: note: possibly missing `}` here +// CHECK:STDERR: +// CHECK:STDERR: ^ +// CHECK:STDERR: +// CHECK:STDERR: fail_file2.carbon:2:10: error: invalid digit 'a' in decimal numeric literal // CHECK:STDERR: var x = 3a; // CHECK:STDERR: ^ // CHECK:STDERR: -var x = 3a; diff --git a/toolchain/driver/testdata/fail_errors_sorted.carbon b/toolchain/driver/testdata/fail_errors_sorted.carbon index a4514661806a..380c3b7ab053 100644 --- a/toolchain/driver/testdata/fail_errors_sorted.carbon +++ b/toolchain/driver/testdata/fail_errors_sorted.carbon @@ -10,15 +10,28 @@ // TIP: To dump output, run: // TIP: bazel run //toolchain/testing:file_test -- --dump_output --file_tests=toolchain/driver/testdata/fail_errors_sorted.carbon -// CHECK:STDERR: fail_errors_sorted.carbon:[[@LINE+4]]:24: error: opening symbol without a corresponding closing symbol -// CHECK:STDERR: fn run(String program) { -// CHECK:STDERR: ^ -// CHECK:STDERR: +// The `}` that recovery suggests goes at the start of the line after +// `return True;`, so the `CHECK`s must live in their own split: autoupdate would +// otherwise write them at that very position, and the note would quote its own +// `CHECK` line. + +// --- fail_errors_sorted.carbon + fn run(String program) { return True; -// CHECK:STDERR: fail_errors_sorted.carbon:[[@LINE+4]]:10: error: invalid digit 'a' in decimal numeric literal +var x = 3a; + +// --- AUTOUPDATE-SPLIT + +// CHECK:STDERR: fail_errors_sorted.carbon:2:24: error: opening symbol without a corresponding closing symbol +// CHECK:STDERR: fn run(String program) { +// CHECK:STDERR: ^ +// CHECK:STDERR: fail_errors_sorted.carbon:4:1: note: possibly missing `}` here +// CHECK:STDERR: +// CHECK:STDERR: ^ +// CHECK:STDERR: +// CHECK:STDERR: fail_errors_sorted.carbon:5:10: error: invalid digit 'a' in decimal numeric literal // CHECK:STDERR: var x = 3a; // CHECK:STDERR: ^ // CHECK:STDERR: -var x = 3a; diff --git a/toolchain/driver/testdata/fail_errors_streamed.carbon b/toolchain/driver/testdata/fail_errors_streamed.carbon index 789edd7a2f7f..0bad7410446a 100644 --- a/toolchain/driver/testdata/fail_errors_streamed.carbon +++ b/toolchain/driver/testdata/fail_errors_streamed.carbon @@ -13,12 +13,15 @@ fn run(String program) { return True; -// CHECK:STDERR: fail_errors_streamed.carbon:[[@LINE+8]]:10: error: invalid digit 'a' in decimal numeric literal +// CHECK:STDERR: fail_errors_streamed.carbon:[[@LINE+11]]:10: error: invalid digit 'a' in decimal numeric literal // CHECK:STDERR: var x = 3a; // CHECK:STDERR: ^ // CHECK:STDERR: // CHECK:STDERR: fail_errors_streamed.carbon:[[@LINE-7]]:24: error: opening symbol without a corresponding closing symbol // CHECK:STDERR: fn run(String program) { // CHECK:STDERR: ^ +// CHECK:STDERR: fail_errors_streamed.carbon:[[@LINE-8]]:1: note: possibly missing `}` here +// CHECK:STDERR: +// CHECK:STDERR: ^ // CHECK:STDERR: var x = 3a; diff --git a/toolchain/language_server/testdata/document_symbol/incomplete.carbon b/toolchain/language_server/testdata/document_symbol/incomplete.carbon index 3825ca74ca34..7a7eb64868d9 100644 --- a/toolchain/language_server/testdata/document_symbol/incomplete.carbon +++ b/toolchain/language_server/testdata/document_symbol/incomplete.carbon @@ -44,7 +44,7 @@ class Incomplete { // CHECK:STDOUT: "typeDefinitionProvider": true // CHECK:STDOUT: } // CHECK:STDOUT: } -// CHECK:STDOUT: }Content-Length: 1552{{\r}} +// CHECK:STDOUT: }Content-Length: 834{{\r}} // CHECK:STDOUT: {{\r}} // CHECK:STDOUT: { // CHECK:STDOUT: "jsonrpc": "2.0", @@ -67,21 +67,6 @@ class Incomplete { // CHECK:STDOUT: "source": "carbon" // CHECK:STDOUT: }, // CHECK:STDOUT: { -// CHECK:STDOUT: "message": "`class` declarations must either end with a `;` or have a `{ ... }` block for a definition", -// CHECK:STDOUT: "range": { -// CHECK:STDOUT: "end": { -// CHECK:STDOUT: "character": 18, -// CHECK:STDOUT: "line": 0 -// CHECK:STDOUT: }, -// CHECK:STDOUT: "start": { -// CHECK:STDOUT: "character": 17, -// CHECK:STDOUT: "line": 0 -// CHECK:STDOUT: } -// CHECK:STDOUT: }, -// CHECK:STDOUT: "severity": 1, -// CHECK:STDOUT: "source": "carbon" -// CHECK:STDOUT: }, -// CHECK:STDOUT: { // CHECK:STDOUT: "message": "opening symbol without a corresponding closing symbol", // CHECK:STDOUT: "range": { // CHECK:STDOUT: "end": { @@ -95,37 +80,48 @@ class Incomplete { // CHECK:STDOUT: }, // CHECK:STDOUT: "severity": 1, // CHECK:STDOUT: "source": "carbon" -// CHECK:STDOUT: }, -// CHECK:STDOUT: { -// CHECK:STDOUT: "message": "semantics TODO: `handle invalid parse trees in `check``", -// CHECK:STDOUT: "range": { -// CHECK:STDOUT: "end": { -// CHECK:STDOUT: "character": 18, -// CHECK:STDOUT: "line": 0 -// CHECK:STDOUT: }, -// CHECK:STDOUT: "start": { -// CHECK:STDOUT: "character": 0, -// CHECK:STDOUT: "line": 0 -// CHECK:STDOUT: } -// CHECK:STDOUT: }, -// CHECK:STDOUT: "severity": 1, -// CHECK:STDOUT: "source": "carbon" // CHECK:STDOUT: } // CHECK:STDOUT: ], // CHECK:STDOUT: "uri": "file:///incomplete.carbon" // CHECK:STDOUT: } -// CHECK:STDOUT: }Content-Length: 469{{\r}} +// CHECK:STDOUT: }Content-Length: 1004{{\r}} // CHECK:STDOUT: {{\r}} // CHECK:STDOUT: { // CHECK:STDOUT: "id": 2, // CHECK:STDOUT: "jsonrpc": "2.0", // CHECK:STDOUT: "result": [ // CHECK:STDOUT: { +// CHECK:STDOUT: "children": [ +// CHECK:STDOUT: { +// CHECK:STDOUT: "kind": 12, +// CHECK:STDOUT: "name": "Foo", +// CHECK:STDOUT: "range": { +// CHECK:STDOUT: "end": { +// CHECK:STDOUT: "character": 13, +// CHECK:STDOUT: "line": 1 +// CHECK:STDOUT: }, +// CHECK:STDOUT: "start": { +// CHECK:STDOUT: "character": 2, +// CHECK:STDOUT: "line": 1 +// CHECK:STDOUT: } +// CHECK:STDOUT: }, +// CHECK:STDOUT: "selectionRange": { +// CHECK:STDOUT: "end": { +// CHECK:STDOUT: "character": 8, +// CHECK:STDOUT: "line": 1 +// CHECK:STDOUT: }, +// CHECK:STDOUT: "start": { +// CHECK:STDOUT: "character": 5, +// CHECK:STDOUT: "line": 1 +// CHECK:STDOUT: } +// CHECK:STDOUT: } +// CHECK:STDOUT: } +// CHECK:STDOUT: ], // CHECK:STDOUT: "kind": 5, // CHECK:STDOUT: "name": "Incomplete", // CHECK:STDOUT: "range": { // CHECK:STDOUT: "end": { -// CHECK:STDOUT: "character": 12, +// CHECK:STDOUT: "character": 13, // CHECK:STDOUT: "line": 1 // CHECK:STDOUT: }, // CHECK:STDOUT: "start": { diff --git a/toolchain/lex/BUILD b/toolchain/lex/BUILD index 00baa0d0cb8e..393e8c9de2f9 100644 --- a/toolchain/lex/BUILD +++ b/toolchain/lex/BUILD @@ -187,6 +187,7 @@ cc_library( deps = [ ":character_set", ":helpers", + ":mismatched_brackets", ":numeric_literal", ":string_literal", ":token_index", @@ -204,6 +205,70 @@ cc_library( ], ) +cc_library( + name = "mismatched_brackets", + srcs = ["mismatched_brackets.cpp"], + hdrs = ["mismatched_brackets.h"], + deps = [ + ":token_index", + ":token_kind", + "//common:check", + "//common:hashing", + "//common:hashtable_key_context", + "//common:map", + "//common:set", + "@llvm-project//llvm:Support", + ], +) + +cc_test( + name = "mismatched_brackets_test", + size = "small", + srcs = ["mismatched_brackets_test.cpp"], + deps = [ + ":mismatched_brackets", + ":token_index", + ":token_kind", + "//testing/base:gtest_main", + "@googletest//:gtest", + ], +) + +cc_binary( + name = "mismatched_brackets_eval", + testonly = 1, + srcs = ["mismatched_brackets_eval.cpp"], + deps = [ + ":lex", + ":mismatched_brackets", + ":token_kind", + ":tokenized_buffer", + "//common:bazel_working_dir", + "//common:check", + "//common:command_line", + "//common:init_llvm", + "//common:ostream", + "//toolchain/base:shared_value_stores", + "//toolchain/diagnostics:emitter", + "//toolchain/diagnostics:null_diagnostics", + "//toolchain/source:source_buffer", + "@llvm-project//llvm:Support", + ], +) + +cc_fuzz_test( + name = "mismatched_brackets_fuzzer", + size = "small", + srcs = ["mismatched_brackets_fuzzer.cpp"], + corpus = glob(["fuzzer_corpus/mismatched_brackets/*"]), + deps = [ + ":mismatched_brackets", + "//common:check", + "//testing/fuzzing:libfuzzer_header", + "@llvm-project//llvm:Support", + ], +) + cc_library( name = "dump", srcs = ["dump.cpp"], diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/02291a60c819f2cc7ab15a85a7175d5973f19e25 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/02291a60c819f2cc7ab15a85a7175d5973f19e25 new file mode 100644 index 0000000000000000000000000000000000000000..8418937b7dca7c110f39534e1c9ee0a32cb044af GIT binary patch literal 72 zcmZQz`Og3Zj35F6{`~*(|G$Z>r|G`q+U?C9kAE+1r DA#E0% literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/0414d6fe824625bc1530146da13aa32374fd46d6 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/0414d6fe824625bc1530146da13aa32374fd46d6 new file mode 100644 index 0000000000000000000000000000000000000000..248d53763a2f8a59aa3d5e5c4c4e5ad4eca2beef GIT binary patch literal 628 zcmZQ#U|?j3f`esCmo9~KsL7aGD9Qi^8bER<0|OHSlC5YK%>eSVumiBvvUNxrxC_yY zX2mobVl$~E&=Lz&05Ji{axfEB4K;Wtj)TEI>}Kg%SYQtvBvEuXMny6F`41A*!IU9R U1mt(ZDzQc#G>Rax&Y&U!0F{oi=>Px# literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/0555ecc1ef4513737146822f1fd51ae68a547467 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/0555ecc1ef4513737146822f1fd51ae68a547467 new file mode 100644 index 0000000000000000000000000000000000000000..f92ffe56fbf1c9db50ac44b057868e26fe6ae936 GIT binary patch literal 1188 zcmZQ#U|?j3f`esCmo9~KXwBd*WB>sTAl}Kqz{G&$4m7K00Qp%kFl_>ig-OHofi=K5 zV4AydYN06DKB$GPL|cey4MYUle5e{^AzE3B>>LdHfC9+!SlC|3V%YTK6Z=mj_y<&1 zN2FSWvcCu>n1#zFYPclui9-{>PTZ*stL_4@sC`tF0UCg$r9c0%>OvAE%dtrMk(nro hSQF+GYG+!^yoj7+U~Cs;hf$SX4#}#JGJ!!w1OPG++LQnQ literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/059a5b5c84d754cbb4906f44b63728579e734631 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/059a5b5c84d754cbb4906f44b63728579e734631 new file mode 100644 index 0000000000000000000000000000000000000000..79fc88385c961ce3ff7c4427937880c29718e3b8 GIT binary patch literal 68 XcmZQ#U|;vhBve9g01 literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/0e4fb8ce694d85bf038298e4ad8fa7a0110431ff b/toolchain/lex/fuzzer_corpus/mismatched_brackets/0e4fb8ce694d85bf038298e4ad8fa7a0110431ff new file mode 100644 index 0000000000000000000000000000000000000000..b1e7a91c490ac55bc45c9794dbcc70846b8f63b1 GIT binary patch literal 574 zcmZQ#U|?j3f`esCmo9~Hq~Ky$8PQlp@d<`R8Nj6-fGkD^27Raick$FhQ3f!`0)xL0 z4wA#7xJ%(`G=Q=@8CaPZAo5aBQ)aM2bwSKPA@fjpP!3cdn39#1jlxvpg)D@uTmzYd y$_BaWKdKO%hbz=#;VOt@Na0Xs3>C*^4@8R@ literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/0f6bb168f87053dac09200c3414d9b615f6910c4 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/0f6bb168f87053dac09200c3414d9b615f6910c4 new file mode 100644 index 0000000000000000000000000000000000000000..3a921702fb49b02e9afe6fca0d0b2f5f36a2059f GIT binary patch literal 916 zcmZQ#U|?LL9fbs}kl0}6Bs2jH2tSYk1pfSwM3Lq$oLVRfmSNBUQac$Km>3vPRLnr( z5mt|^d`1=$07>jZmfefYL1#mpi{=*mE=4y2g^w8ufe3l-LQoK(gapG*gfy513npL? zqR2D=b)pDFg|dQ70|AbOc*6@Ra4?;RY9?MD=>dsm94^1Xlr;eT0%Ma%6AA;Uad`cY ZHxM!0LrH?gnMpvg2?WTNEd&BE9{?)<3ikj2 literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/13ba392feb0e03ec01c993445f0a08b058ca2171 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/13ba392feb0e03ec01c993445f0a08b058ca2171 new file mode 100644 index 0000000000000000000000000000000000000000..cb3b1dd34466c535854ac058d3d92565753eb239 GIT binary patch literal 192 zcmZQzj=}>>86W_nL>WZQ!=s$LaB86_M27~D+R4Dm#J~Vo4`;8umZ{d{|*1&0}25dQDDHp5Cs79_6?%| literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/19bed79c054e96809d4caf6bf1af64f4ece47b73 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/19bed79c054e96809d4caf6bf1af64f4ece47b73 new file mode 100644 index 000000000000..91c86cc79e01 --- /dev/null +++ b/toolchain/lex/fuzzer_corpus/mismatched_brackets/19bed79c054e96809d4caf6bf1af64f4ece47b73 @@ -0,0 +1 @@ +4ÿ¿ÿiÿ‘z \ No newline at end of file diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/1c7e413f92ad0e7ab4733319bbac0686b657296a b/toolchain/lex/fuzzer_corpus/mismatched_brackets/1c7e413f92ad0e7ab4733319bbac0686b657296a new file mode 100644 index 0000000000000000000000000000000000000000..28a3e149bfd0b2e2123ceca6607ebbd78646189c GIT binary patch literal 1200 zcmZQ#U|?j3A{k7=rGtTii2=8HGl1H%NCCL4fr?Ypyf}2b0vR9xT^yOu$iR3WnIFYn zIJHm|s`1Z%6e)^0gUT@^_(Ux#3TIluZwxU$hM0g9<{H4D-$`s5e1Vb@(OeUT%E#iG z{}8YcsE}M2pf~^<2h|b8@vt}*Q==+Kh+Jo*1{^kbL&T7*1(pi9%b92-h1i%#DF^^6 CDwPQU literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/1e25a64582780cb898d1cb4d7a74575cc09c7268 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/1e25a64582780cb898d1cb4d7a74575cc09c7268 new file mode 100644 index 0000000000000000000000000000000000000000..7b6edf6c4f5681f65f1776676da57e1e30140354 GIT binary patch literal 1194 zcmZQ#U|?j3Vu%6)CNMK9>IjH~14LtrPr?+UmIyARvrxexaxu*3Xbyx418G#5tU=^< zB10R^16f|k0!SVu#Pk9x&?lq_O`f}OYN03t7-%qPGBE69U;w6CG(}NS-00%SP6V;N zkOeTb;}#&o%|OTPW?=a9AGaZ7%lu!0O)D@coP{Jz#u9L02FEq&j0`b literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/22a82dfe1fbb9d53e9d61a1c1c4952163e361bb6 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/22a82dfe1fbb9d53e9d61a1c1c4952163e361bb6 new file mode 100644 index 0000000000000000000000000000000000000000..eaf04398bc97caeca322841d8f1755b15f667fff GIT binary patch literal 508 zcmZQ#U|?j>jv@f8(B&tg3lQLIK;#1%K;X~+NEGGVg;NVf85lIck_-$x85o!t7*JFo zb7mkoJENkYdO=!2`aybO@-%g0R8#}dI+#_&(V?tBEnvX05R1DJ#$_P@TEM|A0)+xp T6;upnHUyaAZ literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/2456c6a824c0665c2a5d00f35930ed9c6d347329 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/2456c6a824c0665c2a5d00f35930ed9c6d347329 new file mode 100644 index 0000000000000000000000000000000000000000..d81cfc8877e886ffb47602920c05d19b69211320 GIT binary patch literal 4 LcmX@>$Y1~f1Sa+ya26`#C@R+$#AW#NA4KVZ2$H}I zplTu<3RSXf8Jgi1Xned}?!v2uq6}c50VH=aFfcLTjs((N26P}^yYWlvp-OWXP6as> NZhfSFK&fdR<S(tV>U5W|d za>xursEd%rplt5KsfD5tTQwLMb^=*U2ybFhKLaR+a1k*qFJwhg$tY$Tpzx^1!5=+f jYXXS1Y&tMVh*b+zrvpI^8bE3% z*!?&ym;sbwV93G_OraKFm&Iu&7KuOx1Bh8zan%C@{eQfd&xoWMBmb1&WGcm9h=sz{Jel#Kged*xbO< z)YQ<-&=ed741Yk085IOUxXMrf;(>D#k}9N7RR5<53PuL>V1&etB+wQlefXLGaRVG0 Mq3(cK&!8d#06?*%oB#j- literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/33fb500b866385aaefed956a6ce249316d70d080 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/33fb500b866385aaefed956a6ce249316d70d080 new file mode 100644 index 0000000000000000000000000000000000000000..e4d8ccc446d35e1a0fbf10b3f5d6f201623ab213 GIT binary patch literal 44 fcmdP8{~ru;|NsBZ00;m7M}Yu@jEZ6au^AWu3NsU| literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/35a1bea88022d11d4f58d7bd0796d11fa10b6357 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/35a1bea88022d11d4f58d7bd0796d11fa10b6357 new file mode 100644 index 0000000000000000000000000000000000000000..40fcf436495e79cd30632341c3ba1ea8363cd44c GIT binary patch literal 56 fcmdP8U;iHg{&Aym85kH;6c`wy;DEs$$N>@nggYPY literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/37a7413bade89fa21b77cff4a7b21d0692bf9f66 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/37a7413bade89fa21b77cff4a7b21d0692bf9f66 new file mode 100644 index 0000000000000000000000000000000000000000..ea0709bc43cb7b8bc9b9598d8bbd16f4089da6ab GIT binary patch literal 558 zcmZQ#U|?j3q9T|?mRVUu0J01QYT_=OS}4i@0e}9JW=b59%83g?B$sFa-LsQ{fr$aP e&KW?vi3k>i5+dD7gfgf@Ar55N&H@QM6%ha~XuL51 literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/3800fb3acb3c97ce0d3ce6657a37f757d1c5bc28 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/3800fb3acb3c97ce0d3ce6657a37f757d1c5bc28 new file mode 100644 index 0000000000000000000000000000000000000000..0063f19bd308ebb6f5e81b5e6664dc78958e4e8c GIT binary patch literal 579 zcmZQ#U|?j3f`esCmo9~K2s5%^0Gm?o!l{L#3?QJO0VH-Zure_qS%pmtx>y5HAG#RX z{5qiDmPR35#Q<>?13p*H0NMugBa8;R3C7II!f*`(5ypZ{MKTs>APis_N0tD@ry#8$ b-vZ74^PenpAex{d7Nw1B22_xdfl&wm$t=4k literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/38905a052321fb17f6f7b7cc4eb075f2391236f5 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/38905a052321fb17f6f7b7cc4eb075f2391236f5 new file mode 100644 index 0000000000000000000000000000000000000000..1cc08afdcb24821b553780e4b8ee607d262ce00d GIT binary patch literal 589 zcmZQ#U|?j3f`esCmo9~K5R7If1~5Q~U}sgK3rnL5fce~oQwv2QCTakwoeT^>^HEhY zGBD@^MP>kT7H*J{#mK;T9wZObO+W*tQ8109({Qs8T9G`)@aI25oX)J2C?tny@IgYF dK^g7}cxb^zFd1=}LKq@oTmAz9DE=5!L;!5m#K`~v literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/39a8e924d0e1e988b9c1789ab5155b5b7ac30b36 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/39a8e924d0e1e988b9c1789ab5155b5b7ac30b36 new file mode 100644 index 0000000000000000000000000000000000000000..1227247ba53fb62d6ff67791d28194a322fd26a5 GIT binary patch literal 580 zcmZQ#U|?j3A{k5~RR;qYpz7i-oLVRfkr40Tjq01)v*%j~@uJ z9&ZRh!iQlw&>WzPp$=yF^Pdr@7oXV>IV0?VutpRWNP$pAR9K+sCXthXUp@CzNW4Mf Ii$O&M07VtOng9R* literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/3a826d7d1c28b26b260230b7f3951ce05d60e918 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/3a826d7d1c28b26b260230b7f3951ce05d60e918 new file mode 100644 index 0000000000000000000000000000000000000000..8cae2d7a86d4d0eb2c38f639d5e93bb43b599b0a GIT binary patch literal 680 zcmZQ#U|?j3A{k5~SqE4{AOi%T>gFz-S||#U&;U|985o!taJwms6o6_GCJ*YMpO`WT z5k>~a^I+E@s{#r9`HxUUAkh}>1{sQ2L literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/3efecca68da1229dace3e7d99d9a8b6c7549d6c1 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/3efecca68da1229dace3e7d99d9a8b6c7549d6c1 new file mode 100644 index 0000000000000000000000000000000000000000..c0f2b6b5c2b7440d766b881bc2e964a337105c98 GIT binary patch literal 8 Ocmey*@Sg#Q{sRCJ_5=L@ literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/423d4c98d4f179bdc1b435106a482b6cf0ff95fe b/toolchain/lex/fuzzer_corpus/mismatched_brackets/423d4c98d4f179bdc1b435106a482b6cf0ff95fe new file mode 100644 index 0000000000000000000000000000000000000000..9f45d93fac27e8b53ad2e6da4238f05630bcf47d GIT binary patch literal 8 PcmX@jz`&sRpPK;y3HJf_ literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/42f36ad525f2f39e31cb88ae655a8d9ef7cde06f b/toolchain/lex/fuzzer_corpus/mismatched_brackets/42f36ad525f2f39e31cb88ae655a8d9ef7cde06f new file mode 100644 index 0000000000000000000000000000000000000000..445bec207d51b0d54279eddf99e522e89fd4e935 GIT binary patch literal 80 bcmZQzjv@@gfl5~V!=mm#D-#0)viN@hS4130 literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/433f63bbb76f9b121d22f9093032b45f4836090c b/toolchain/lex/fuzzer_corpus/mismatched_brackets/433f63bbb76f9b121d22f9093032b45f4836090c new file mode 100644 index 0000000000000000000000000000000000000000..88544544e398f170f27f509bd23211c2bb9253b5 GIT binary patch literal 646 zcmZRI0)a_jk|1FOSD^u61u}rZpZ}3jQI<=WE`>^S7fvk{1G4cI?oD>P7QQgEdebXb@OG7x#Z~AU^~LGR(4t+Q=4N Jz6=&`0syA9fuH~Y literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/4f0577f8b986b9affe260cd84f9c06421707c4b1 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/4f0577f8b986b9affe260cd84f9c06421707c4b1 new file mode 100644 index 0000000000000000000000000000000000000000..189b11974b6c1481e6354f5f0e73921c73ebdb8f GIT binary patch literal 48 fcmeyb@c;k+fB*k8z(EuUR5C=l0;x1G2bll>*4hza literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/531cb419e85c387c49e91299f7e1ad82bc36c68d b/toolchain/lex/fuzzer_corpus/mismatched_brackets/531cb419e85c387c49e91299f7e1ad82bc36c68d new file mode 100644 index 0000000000000000000000000000000000000000..7469bc3c2249d0c7b324a3b0949e917a5ef600f1 GIT binary patch literal 205 zcmZQ#U|?j>jzR(sFt$ci)M*9=1`sz8NLfPJQBmB5Qwv2I7*+!XG=LZcb}}$9L6m}& lqX;Pe=LU&k13+WJ07Whu#Dhw}DWC{M(F|;IQBe!QvH(|*Eye%< literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/53d13a0338eac1acb5c3c37752f98a3a7a4b023c b/toolchain/lex/fuzzer_corpus/mismatched_brackets/53d13a0338eac1acb5c3c37752f98a3a7a4b023c new file mode 100644 index 0000000000000000000000000000000000000000..0f2ce93a790dc1ef1906d8e836e8a9ed5d7f73e6 GIT binary patch literal 549 zcmZQ#hysEtApQ>q3<4lN1aKElEf(d2i2vtf*vY`a!~j+fQxKI!2*6eDh>FUJdI#r5 zMF9;j6lDMd4IsJm&#%A#Q4I$wCS*K{eAKd~V6XtUMg@q43=EDi20j|<^WXT?faIzO wDPUj#IkXoHyt1;QqFl)UY9#6jb#(;=HzbHaEMVj@B0_lSvZ$z~Ku8P#00(WQa{vGU literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/541b03a56c7b624a7885c674f517df068d2f3df1 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/541b03a56c7b624a7885c674f517df068d2f3df1 new file mode 100644 index 0000000000000000000000000000000000000000..bdb5cca6737d0e17dce02fb94c0ec2a0ae74b94a GIT binary patch literal 12 RcmZQzWN`cs0=NJF2LL6|2p0eV literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/570cc51c1f9750bd5ed85e8ba31da8ee93689d51 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/570cc51c1f9750bd5ed85e8ba31da8ee93689d51 new file mode 100644 index 0000000000000000000000000000000000000000..18ccb0281e52d8ee321a5cc45b22404fc484ae83 GIT binary patch literal 73 kcmd;P0D)8p&;wH{U=k$C0AYgpVh|+=G94i)CML!J00&tXP5=M^ literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/5a119a6eeee02301b24b76ee4bda22811aa2d97e b/toolchain/lex/fuzzer_corpus/mismatched_brackets/5a119a6eeee02301b24b76ee4bda22811aa2d97e new file mode 100644 index 0000000000000000000000000000000000000000..6afb749c5f9546da56a70de6b45e819d3d69bfaa GIT binary patch literal 132 UcmZSJ((zEHnU<2Ft-XQBio! z0V%^~7Mjth+$fN5fnWi`5)8+n>d5j!<|Ao?Fgbvp#SnxGc)?k?7~l~5^BvfS(d literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/5bbcd3f49cf21ae3a57a89d9a263338ec61d510b b/toolchain/lex/fuzzer_corpus/mismatched_brackets/5bbcd3f49cf21ae3a57a89d9a263338ec61d510b new file mode 100644 index 0000000000000000000000000000000000000000..a26ac36dc5f231efd42ecc9a4e9d12b0a5d399b9 GIT binary patch literal 657 zcmezW9}YAa7#N@&ZUzQM25tr>Aj!a>4dgMf05KeBa_M+7G(kaKZS7T{3=7;GoQy1F zfKUZ89t02~AeIT3iz$w(ioQIEkAc2GSUboa-;W%sK<9%48iRmEC5Zk{$Cw4iIg(rc I{|DI*0E1oYdH?_b literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/5d73de1428afbe09c069d43ad284a5fe6f495973 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/5d73de1428afbe09c069d43ad284a5fe6f495973 new file mode 100644 index 0000000000000000000000000000000000000000..0a90c5718af00142529ac2d31c011fc04f6f14fb GIT binary patch literal 572 zcmZQ#U|?j>j=~QZlR@&6@GHWQ)qscuGJwFJ|B)yPxC^Hiih^YrG=S7j1_mYu1{4*j z98~oJ1k}#}YREMghhsJZD8zEX-*bQXasg`fPg;?JAn)p5day80;m80 literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/6cfa1c02aee7e8c2993901381416b1b46f6afaf2 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/6cfa1c02aee7e8c2993901381416b1b46f6afaf2 new file mode 100644 index 0000000000000000000000000000000000000000..847471a7a5cd1c8acb136e727678bdd76fc47fb1 GIT binary patch literal 619 zcmZQ#U|?j>j-njv^IIB0+-&L~S4g1fc5VE}U8@3X#wNQac$Km>3vPl_2qEAhDyONY|T% z0n{)=-~uRi5#V6=9I7Tp2FCL!B1rCM`12n{iYlB?R%nnja4bZ!i^y<>hXa1b!i7*v z7J>mpF`3SXITt7G_y0c*Ieaz}5%B;2|8KyZ>@W;BH#Rr0G&MCeGc<*`2Y2-OFkdJtz;M73kJzFMY}T^n2qW>Zu<0il$-wZV8o=W=EK*6BVgR5yv`7E| literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/724e9d106919f6bfaf00446115072df31b2652b1 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/724e9d106919f6bfaf00446115072df31b2652b1 new file mode 100644 index 0000000000000000000000000000000000000000..e3f494d0cb0e26c99b7b1faf20595154f322dc89 GIT binary patch literal 24 UcmezWpZh-qFhIe7J|Le50KA|L82|tP literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/76eae89b045b5d336c735f2a96e38a95f95311c3 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/76eae89b045b5d336c735f2a96e38a95f95311c3 new file mode 100644 index 0000000000000000000000000000000000000000..80e33dcc49732413d94564928f6baad258f41016 GIT binary patch literal 2001 zcmd6mK@Na02n2iZq(5o=fBc=AXn@3ZOj}Lmw1EW{X!BZV=@lDNihY}F#$te*`y2&2 zsfWRCLen|02v@8^REaKV7HNiwpN(aJ=tgP<(B!(xrJvE96_-^hogTT8QsMhreUL=A t*kkadWG3|qYU4;Mu literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/7c2ea32624e918c4e4bf89e41e40390c46b1f627 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/7c2ea32624e918c4e4bf89e41e40390c46b1f627 new file mode 100644 index 0000000000000000000000000000000000000000..71c544b345491bfe01beeea414ac4f9997e07244 GIT binary patch literal 700 zcmZQ#U|?j3f`esCmo9~KsKi)B$Qo%K1I>jR&WdR`ND>TyQpll!Vh{$$2_Z5G!NkH!!XgG13WQpVDv#k}@`D2$ gOsJ+~w-Z$jBLm}k6cHpRG5k^ak0Lif9AHWT07g#VG5`Po literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/7cca047c3f7c202ff0871a4272ba1f6879794b79 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/7cca047c3f7c202ff0871a4272ba1f6879794b79 new file mode 100644 index 0000000000000000000000000000000000000000..e9b5df8d091df34221b3dd453d39b8278a76434b GIT binary patch literal 560 zcmZQ#U|?j3q95R9V9;P-VA#pPz{CKyw;8M5Gk~I5*gz5%L98ZFBFJ4hwNMo3N)Y(- tA7WArRs*pH-Tx(6)glCu%ts9juo9ep$KrLc5W-;Uv!G6e#215#2mnm=yt)7Y literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/7d0b0fcf5acd4ced16f3d1a8dbf57f4d60bec85c b/toolchain/lex/fuzzer_corpus/mismatched_brackets/7d0b0fcf5acd4ced16f3d1a8dbf57f4d60bec85c new file mode 100644 index 0000000000000000000000000000000000000000..dfe9edad30e3bea82eb5a908d84573335ad2e6c3 GIT binary patch literal 16 ScmZQzkYa#vDW#D44W?)caU|`4uVgL?b0bBq8 literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/8326ae645e600f2e13ed57db6a221d109a1c8454 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/8326ae645e600f2e13ed57db6a221d109a1c8454 new file mode 100644 index 0000000000000000000000000000000000000000..b2f043c92fdd027878a7c44309583244a57130a1 GIT binary patch literal 20 bcmZQzV31)b{Qrl6fsuiML6w1FC+9{0CnE$R literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/842ca9afb9e4daa2535a51180423c957ddde3ff1 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/842ca9afb9e4daa2535a51180423c957ddde3ff1 new file mode 100644 index 0000000000000000000000000000000000000000..522306b8bb00e644956052be3bb8b0242c5b6b41 GIT binary patch literal 136 rcmdP;j|~3(|IYy9gV<026k=jvU}T8G0^o8)sEvw(t422IKac|eWb{1C literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/84803ce199f02a6bf9ba5ac0fdce24f928343bdf b/toolchain/lex/fuzzer_corpus/mismatched_brackets/84803ce199f02a6bf9ba5ac0fdce24f928343bdf new file mode 100644 index 0000000000000000000000000000000000000000..3f5a6cf16641b5481d97c0ec3d47d57a126aec74 GIT binary patch literal 16 Tcmd<;;&EbN_zwmQ?hO9{J8uXh literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/8ad4f3e9bd4cffdf5ef5caa05d405a39f9944901 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/8ad4f3e9bd4cffdf5ef5caa05d405a39f9944901 new file mode 100644 index 0000000000000000000000000000000000000000..1659f5f88f7b4e634514a8cb7e8d18bb84d8575f GIT binary patch literal 32 WcmWd=6MJ4I1_u+sECv{#fdK$l0|YJr literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/8c651b261db765b2d48f09d0d021b1283ca6c7cc b/toolchain/lex/fuzzer_corpus/mismatched_brackets/8c651b261db765b2d48f09d0d021b1283ca6c7cc new file mode 100644 index 0000000000000000000000000000000000000000..6b573b8117543328c33e22ca91d192d3a03b15a0 GIT binary patch literal 32 VcmWG#Wq<+>2)z&jz?^ah1^_}t1wjA+ literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/8d0cc0e6fb6e5fc9bae742024f42b841d68b7faa b/toolchain/lex/fuzzer_corpus/mismatched_brackets/8d0cc0e6fb6e5fc9bae742024f42b841d68b7faa new file mode 100644 index 0000000000000000000000000000000000000000..f2af2e1705bfef1a5494468e9ecf4ed0008292bd GIT binary patch literal 512 zcmdP;4+0F>03!qA`D{4Y$>0H0#AQnYcu6)uv4op}lYv2-8|p-ZuP6ZSJ{SW>C*X2&w^Z literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/90f5e0abe9dcd825ff129821b3fe459b45d4fec7 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/90f5e0abe9dcd825ff129821b3fe459b45d4fec7 new file mode 100644 index 0000000000000000000000000000000000000000..e277499f2c28df8b54b5990124e34523329bb132 GIT binary patch literal 28 VcmZQz0E0v@X#n8>NiiVc0sssj0Q3L= literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/911cfb8a1f361083f0300f62796a00ff223e7f4e b/toolchain/lex/fuzzer_corpus/mismatched_brackets/911cfb8a1f361083f0300f62796a00ff223e7f4e new file mode 100644 index 0000000000000000000000000000000000000000..581b80e679e00ecc255835ef97a84553d6938125 GIT binary patch literal 967 zcmd;P00IpLhTR~VXu!z8AOKXX0Mek%&A{+$Cy2$s0Ayk@>F0m2BnrXE;K0BHG=_yi zlS{{wp@}r0^`9Jo;sA0uL|TesGz$X*vNjYsayZC_QI!pfI}pHTD3PGPhCEmRmiRDa;rhS~TXf)vD&Pd<{s*SiY&-yL1<173sAiMSLw78a3FuO2 PeD003!qA`D{4Y$>0H0#AQnYcu6)uSVEeO2z?kVd$34$wj$g|NDNRI903Ba zeOtZg1wi(p0X}FnAPa+>h7WKvFfcN3GcW;31_o^)kAVeh3XCcQf=>GCB(t3_Ta4x<4z+3t51K(DMSS)yJ=%0R;a1 z$FGPi+5bzhX#=K^WNhN-V%&vO3q`?>W6)sGWMJ6Iz`(=+N`9z-8(kbZq(E#hWC2vY nczMJH-R{8_c9_A2=1YtKLzBng;>u(g3W*bdIuH_V3@Rc3ml%Cu literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/a8f6d9d0870707a5959b8d4cc3a453fb7cea6ddd b/toolchain/lex/fuzzer_corpus/mismatched_brackets/a8f6d9d0870707a5959b8d4cc3a453fb7cea6ddd new file mode 100644 index 0000000000000000000000000000000000000000..079f7bfb77da79522737a5eef906afa62b68b8b1 GIT binary patch literal 8 PcmZQzV2Fy+jEVvP1iJx} literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/a9d82a1c17c7606f140406cf0d28ea2b9f085d23 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/a9d82a1c17c7606f140406cf0d28ea2b9f085d23 new file mode 100644 index 0000000000000000000000000000000000000000..6f4e2569a03bb28b6d76c2181eabd3917c502dea GIT binary patch literal 602 zcmZQ#U|?j3f`esCmo9~KurmU%3&TZ&z$_36G;A$g7zcy9aB86_0~lxk$(;-gObkfw zLDMxOsxB%E403=ZToS_&R)irCrH3?j8BiAPA6yG+3b+Q&_n4bfxYnL+{t>Po;-Od{#jjW(e#Wn?NvvC5c%tZkl)-$0j{UHJic*CDp|e&uy7YXATM literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/aa92e70ca6c167c7259af670a52dfd98c8eeb566 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/aa92e70ca6c167c7259af670a52dfd98c8eeb566 new file mode 100644 index 0000000000000000000000000000000000000000..b6fb09fc71688204e41cb522c4ab6d020804765c GIT binary patch literal 1246 zcmZQ#U|?j3f`esCmo9~K=*8gv4+e!(3q={gKm$nbWME)oKyn$}F;6%;!lYX^5Ufntf0z#%rdVWtK&dj~FsBRdT(C{P>)mP1wyW20%J PDpwPhq#s%e+LAh zDTXNc3zkDtjKPFD21se|1hfAFl^_I>Sc*t&I1|G>xBy%UP6k#3ajHg@fLjA%ph`h` zC@P>FSSS+_^jOU#M-bH=z&HdM@*khySx_S$oF4Joj2ZxF+EH{?q42PCxM0Gwv=L^F#wamCOE?@+RdAzn5j;?$0P_H>vVlhc literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/b09ee244f6afcb09fc98300a976721bdcdd6e83b b/toolchain/lex/fuzzer_corpus/mismatched_brackets/b09ee244f6afcb09fc98300a976721bdcdd6e83b new file mode 100644 index 0000000000000000000000000000000000000000..4c834d3253a20d423b06360b00e1c0148a3a6adf GIT binary patch literal 68 scmdP8{~rnb<7QxBPyu3w|5`AKfB*l3_&~tm4x&LM2!ORkMFDjH0O61(Y5)KL literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/b0b0f529b8fbc71a44ccea7b08e528ecfbf72868 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/b0b0f529b8fbc71a44ccea7b08e528ecfbf72868 new file mode 100644 index 0000000000000000000000000000000000000000..abfe6816f198457d6f76526c6b493c77132eab31 GIT binary patch literal 552 zcmZQ#U|?j3A{k5~SqE4{AOi%zbu(btodp7Lc_J7XW@6F9$iR3WMVPyAYN069F@OG} xND1Tpq&g1Adjht005E$v6KJ+ literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/b0b12c4a7676e78b62ec5078bd56b9061dcfc290 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/b0b12c4a7676e78b62ec5078bd56b9061dcfc290 new file mode 100644 index 0000000000000000000000000000000000000000..8650fd1ad63bb86bd95f05660578d1e23d3803a2 GIT binary patch literal 600 zcmZQ#U|?j3f`esCmo9~K@G+wCNn(`?i86qTI{;aX3=H~E0q)|dg`x~#paCRzGB7YP zFhpe`1GrWYgS!+>Ju65(2uOj+sHho0)4_lhN@rzhAsYu`=b;OL_)z^2N)`;DLRncc y*fhY!yx=S#1H~BJ9FVL3>b-r!@ literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/b50a1858c3dc8e84332b73093b99ea2822e252ee b/toolchain/lex/fuzzer_corpus/mismatched_brackets/b50a1858c3dc8e84332b73093b99ea2822e252ee new file mode 100644 index 0000000000000000000000000000000000000000..4fce1e2a6e76fdae8532ea5164b793457e9ca798 GIT binary patch literal 128 YcmdPW;cod)BH(6#01~wUm1F7z0O><)kN^Mx literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/b53c234830eb46d17db95aa5c50259fc07a94ed3 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/b53c234830eb46d17db95aa5c50259fc07a94ed3 new file mode 100644 index 0000000000000000000000000000000000000000..4153b61b060a3571b7a63edc210912822bb8473c GIT binary patch literal 40 bcmWG#Wq<+>7z0QzWZb`yK?5$p$iM&qG4KM8 literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/b684c0a65507a9a780310a5bfc2d440c16543e5a b/toolchain/lex/fuzzer_corpus/mismatched_brackets/b684c0a65507a9a780310a5bfc2d440c16543e5a new file mode 100644 index 0000000000000000000000000000000000000000..ddb3eb7f7f09cc83661e581a39be74e3f31d98e0 GIT binary patch literal 1113 zcmZQ#U|?j>j=}>189?CA|40-y+=WvMMZq!*8bE3%0|OHS1Bwb{&J1Ku6gIv1%=-WT z|8^7u;2ekp7#KJfVlxxjJzh~TwMaDF9IQ@4){amaB^4C~@=g{6KqY{N0$m9L>PY5e zV}dj=e1ph9HDjd+2N}o=P}sZw|Nr;@|HW8M0SaOo1XT%&*zf-_Vl9*vWH1QOGgP4t xrFy7JGe#{0f~Y9c5-?3-m`t~zM=WxLRzV~od6aNeLZTLHhNUEfV1`yy6aaFs7xVxC literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/b7b271a26b942ca7efb9d26f3ba98970ddc86e2f b/toolchain/lex/fuzzer_corpus/mismatched_brackets/b7b271a26b942ca7efb9d26f3ba98970ddc86e2f new file mode 100644 index 0000000000000000000000000000000000000000..4af33bd089ae72c8266c555575ba6e37bb7755c8 GIT binary patch literal 562 zcmZQ#U|?j3q95R9V9;P-VA#pPz{CKyw;8M5Gk~I5*gz5%L98ZFBFJ4hwNMo3N)Y(- hA7WArB?gkBm;Bg3ayDuhlVUxFTBxaz7-LWo0RSp4xp4ph literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/ba6497e3fbb353a9c99cf2726c2f3e23924c1cbf b/toolchain/lex/fuzzer_corpus/mismatched_brackets/ba6497e3fbb353a9c99cf2726c2f3e23924c1cbf new file mode 100644 index 0000000000000000000000000000000000000000..1b8dd9bc2b701554bffc92169c286c6a6fdb597d GIT binary patch literal 16 XcmXRbV@OS9U|=Y#)z&Wiwqp|jCy)Ps2ZpW8z9gD3UL52 mH;}yuEd3t>1Q-}>Y@PxQfrB}^bLKEGM`41Xs3@pKU@>Lm literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/bd61b3d8f7935b75efc8de6d06141fbefcce5b01 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/bd61b3d8f7935b75efc8de6d06141fbefcce5b01 new file mode 100644 index 0000000000000000000000000000000000000000..0d9f831871026f8cad0d817e1fc0ecf95f60e3bd GIT binary patch literal 642 zcmZQzVqjooh$0D0B1tt|aUcT(zy-hz?qyR8MIjOzKx!ug1JFgNN|1Onfb1+10FqYx zOpxR7sbFMaJdYyJT?lkHBgEk~|52nU;6#(I7wix!1z;6s(4skrp$kobR4(^ZP{6?= Inn6Vb0M2UDn*aa+ literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/bd844ee1a1aa7210f16f161f94261f48ba025d22 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/bd844ee1a1aa7210f16f161f94261f48ba025d22 new file mode 100644 index 0000000000000000000000000000000000000000..19dfb31d4b090cee4147a8491be3dc47b0530509 GIT binary patch literal 176 zcmZQzj=}?s7$5*ui86>g4^<3>$6YwJkb#wnfdNGn!pXu4uqcc|)(tXIib3bbe*iJZ BI(Yy9 literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/bfe7cf0386b32de999e6df7dc21b12697a6a37d1 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/bfe7cf0386b32de999e6df7dc21b12697a6a37d1 new file mode 100644 index 000000000000..63c203574591 --- /dev/null +++ b/toolchain/lex/fuzzer_corpus/mismatched_brackets/bfe7cf0386b32de999e6df7dc21b12697a6a37d1 @@ -0,0 +1,2 @@ +*, +„ \ No newline at end of file diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/c021157170822334d9a0d61484e397da4bdd2a9d b/toolchain/lex/fuzzer_corpus/mismatched_brackets/c021157170822334d9a0d61484e397da4bdd2a9d new file mode 100644 index 0000000000000000000000000000000000000000..478216f5700f2c3bcd7badda6a9d55519e3ff108 GIT binary patch literal 14 Pcmccr9|WSnfSUmTPPqsI literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/c145b4235b15e63df7d6e2416a5be5d2833c9b2e b/toolchain/lex/fuzzer_corpus/mismatched_brackets/c145b4235b15e63df7d6e2416a5be5d2833c9b2e new file mode 100644 index 0000000000000000000000000000000000000000..7270b7f03381983ada2529b2b7e6076fea1844e1 GIT binary patch literal 53 fcmdP;j{po@It&c<3{gm+B?>6b#K6FaEV>^6{N5Ny literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/c44576da3a941248af87bc80e73c9b53c6b5b03c b/toolchain/lex/fuzzer_corpus/mismatched_brackets/c44576da3a941248af87bc80e73c9b53c6b5b03c new file mode 100644 index 0000000000000000000000000000000000000000..d2b6590a5fe23727894f0fb39c889839cc6cd276 GIT binary patch literal 689 zcmZQ#U|?j3f`esCmo9~KsKHpjbZIEiOk9R81KGh{IJHoefnhlVg9Wt!0~i3!#%W*{ z9FXiyFC-o2a9{>EFbheZI3}`**mW>4`~iCDKNd+xEMg=HapMUkoWTc?h_c0M)Bh!? s!l1~2i2OsAfW|vq5-!KRlYxN=m?A(FgEj+$qT>HVMMgzM1_l)o0DiyIGynhq literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/c9d317a48760e299e87110bb568d3cc61a0dbd6f b/toolchain/lex/fuzzer_corpus/mismatched_brackets/c9d317a48760e299e87110bb568d3cc61a0dbd6f new file mode 100644 index 0000000000000000000000000000000000000000..c4b7566c5890bd7b1e9dde7027a95603bdb38e58 GIT binary patch literal 1153 zcmZQ#U|?j3A{k5~QHKUZeINq_pz7u>oLVRfk_VP`;r@ WZbr6~785tPQ!!>{yNbYC&qw*g`4#8Q4VA70*t<8 literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/cd56e6c481065717030fb754daf2b4a9c8684677 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/cd56e6c481065717030fb754daf2b4a9c8684677 new file mode 100644 index 0000000000000000000000000000000000000000..70d4e1061c47c918b487dbe667486da1a0786994 GIT binary patch literal 12 TcmZQzDBKwp6~(|%c%K0P6>0;C literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/cdbdfdd6934b9992c6b39cde391fd0203b51f50c b/toolchain/lex/fuzzer_corpus/mismatched_brackets/cdbdfdd6934b9992c6b39cde391fd0203b51f50c new file mode 100644 index 0000000000000000000000000000000000000000..c89ee610e5d40ed100465046d2cd450f040bb955 GIT binary patch literal 516 zcmey*_n$PtKp|j)>qoX1E=U~XKOaz?BAQxc^Qg(@hT4N+CYt@kaN%Y!Fu<7z1~Dce W)ByG2bQ(}j3t=vpg+r|lmlgo`{`T|$ literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/cfdb377027ea66b748dbde05f3123b9be7cc5ad6 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/cfdb377027ea66b748dbde05f3123b9be7cc5ad6 new file mode 100644 index 0000000000000000000000000000000000000000..4366f0a7e6b8c256db364d1aaa8d5769a8554811 GIT binary patch literal 724 zcmZQ#U|?j3f`esCmo9~Kurjy{rxuDbfB=gIkl4w7*X;e+@H77@r&KsGlFpoxRI_??RCQgogdvH&*25n@0GA~>{V0RWF! B`w0L5 literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/d0b3c063c83c3e2974bd1f6fc0c7e4bd830df77b b/toolchain/lex/fuzzer_corpus/mismatched_brackets/d0b3c063c83c3e2974bd1f6fc0c7e4bd830df77b new file mode 100644 index 0000000000000000000000000000000000000000..bd52e97a914d4d00912d55f25b2b3ef82b6487a1 GIT binary patch literal 611 zcmZQ#U|?j3f`esCmo9~K$Ysrm8UoFsc5eI$rZWo89p{UPt(L)SbSE9%1>OFXj^?+y`g*oQ_m=x literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/d2b2295c9d5982db079ffe02850d21d965b6ddc2 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/d2b2295c9d5982db079ffe02850d21d965b6ddc2 new file mode 100644 index 0000000000000000000000000000000000000000..6c075c5a38756ce63d2c6fd10a60aa3755bb5f4e GIT binary patch literal 551 zcmZQ#V2Gj{;4Y-CRZ|N^8NfgTNbY1{V8Y|M89=9Gkqj1~>%jDoA*K*nA{$Tun!RM1 UNR+0jg;QZ+$FLm|nhYu;0GDpTYXATM literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/d455faaab92a6a739c2e1c863dcaf9c5fd112401 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/d455faaab92a6a739c2e1c863dcaf9c5fd112401 new file mode 100644 index 0000000000000000000000000000000000000000..8d410dcb3f4a94f5bb4cc9277b8259259aede598 GIT binary patch literal 4 LcmZQzzR>~z0zm;_ literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/d58e04391ec50b86e89eaa72cee255c0b0206935 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/d58e04391ec50b86e89eaa72cee255c0b0206935 new file mode 100644 index 0000000000000000000000000000000000000000..e552e8fadefcb8ed506affa518a0d74906909206 GIT binary patch literal 32 WcmYdIO-)HnU7v~p(tsii3?cy4lnc-R literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/d5f34c9240ddb934c3a2eafa51cd34b94bb7ea5c b/toolchain/lex/fuzzer_corpus/mismatched_brackets/d5f34c9240ddb934c3a2eafa51cd34b94bb7ea5c new file mode 100644 index 0000000000000000000000000000000000000000..67b3c3a96acf721bf7b38d00cbb7f4fe8ef07c79 GIT binary patch literal 64 kcmd<;(sNq4a3KQ(z!?k-jQfE?pBYks0LVb)K~;k&0GMkEumAu6 literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/d6acff1671e452c53a6c85e3b0971fbf704d9d44 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/d6acff1671e452c53a6c85e3b0971fbf704d9d44 new file mode 100644 index 0000000000000000000000000000000000000000..57e62ec1c176ed3a5dde3cc00bbaf5275649f43a GIT binary patch literal 264 zcmdP;j|~3(|IYvg2;nHA!1AR)WxI(|8U-|)fgNTMkj@|k8lZ{{wf<`ZnFzQ2|BuDV WQBi2-X@mG4Ah&@SU^k*UHv<5EVs|D0 literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/d737bb16db6497859637d61be876cb4ad0d5da11 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/d737bb16db6497859637d61be876cb4ad0d5da11 new file mode 100644 index 0000000000000000000000000000000000000000..fbe342a8f3f3c43c721832697d222d1ac1614262 GIT binary patch literal 44 mcmZQzU}9w8W?*7qWME*><^~dA0AjE(6cz%h0FVGkRt*3eg#oYt literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/d7629f122ce603bda8ecd588f9f20e6b938046a8 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/d7629f122ce603bda8ecd588f9f20e6b938046a8 new file mode 100644 index 0000000000000000000000000000000000000000..e8577d2a5e994d66ff707f437f5477855c13c161 GIT binary patch literal 565 zcmZQ#U|?j3q8A8cfB?9CSx^8MpaFxsaB86_)Z#z?Xt Vkk}+Kxt}sH>||hI0*0fC2moA;zhnRa literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/d89f546e0556b5690f105dd5d31fdf7fc97a1c94 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/d89f546e0556b5690f105dd5d31fdf7fc97a1c94 new file mode 100644 index 0000000000000000000000000000000000000000..f32cb2378c156d36bd41822feec27774fca22e12 GIT binary patch literal 84 vcmZSh^&bKlK!6)eGW=&qW?GAsh_!AyjxcNRSHva9=K= literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/d937e3e7989f4cdb2e9e96653e581ad9515280a6 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/d937e3e7989f4cdb2e9e96653e581ad9515280a6 new file mode 100644 index 0000000000000000000000000000000000000000..8855754a76aea1d9142e23f1751b6e94db710990 GIT binary patch literal 124 zcmd<;;&ECC0ZW0D1C*@@qTRq?;ljmG7MN>q2jqhQl5#~L7gdql@~_M7?CeZ{7z=O& Hi!lHI=5sQM literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/de3d0e9bf501a8eac7ecf730ba95ac050f426760 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/de3d0e9bf501a8eac7ecf730ba95ac050f426760 new file mode 100644 index 0000000000000000000000000000000000000000..ae414b479a8e0ce2678b3291bce6d2a05f82df85 GIT binary patch literal 100 qcmdP;4+V2T^uPcA8K3|r9)${aBXgsXIZ=uT9#As_JJ6s-V8Z|p9WEdM literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/dee33c587880622173a253cf90a50df677f1be3f b/toolchain/lex/fuzzer_corpus/mismatched_brackets/dee33c587880622173a253cf90a50df677f1be3f new file mode 100644 index 0000000000000000000000000000000000000000..fe8a7e8dc4415e104bc2cf3b546a10ffb0c91617 GIT binary patch literal 1242 zcmZQ#U|?j>jv^IIB2j||M13Fw1fb|H6om+C0I8h}3``6RDALHB8OWR{?!u`g>dhhr zpx8x#gVkebdXSvU@aI38Bt5yItk7s+;8+N86bDFv$nYmJ2p2AdTCxxfAiBu(CCs_R z(?E-o5S9_?WYXP8INCLqLIA`u1DXWxA^;+2qM|TU12~wenHs2>7Eogr*&`4(D7ArE E0K}(}MgRZ+ literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/e4e8bf0a6b9f3842dcb4414914456eb056c1286f b/toolchain/lex/fuzzer_corpus/mismatched_brackets/e4e8bf0a6b9f3842dcb4414914456eb056c1286f new file mode 100644 index 0000000000000000000000000000000000000000..cb82ab2da6941849fcb88b189ee5a7bbfb07f366 GIT binary patch literal 630 zcmb_Y%L#xm41J0s=tWRB@S=3N?%+}$-AjnkwxKo`6$4FPUcOBObQ#bL{NSYW_JN^9 z;u0TfWA4J6FiG_XS6T%y1#q`J2e}W!300$UPOaK4? literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/e531ca673cb1539c6ec88007288e3f0ee7bca076 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/e531ca673cb1539c6ec88007288e3f0ee7bca076 new file mode 100644 index 0000000000000000000000000000000000000000..4b8ad231c559cf18f621f3f71f3daf036b528773 GIT binary patch literal 16 Vcma!I5EBziWMB|80AjH!E&veC0rUU> literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/e61fac0d819dc59655318ab3fbc97b2c5f8036c4 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/e61fac0d819dc59655318ab3fbc97b2c5f8036c4 new file mode 100644 index 0000000000000000000000000000000000000000..b2ced611ecb90dffbd4af1a90ef95e009bc7ae53 GIT binary patch literal 1196 zcmZQ#U|?j3f`esCmo9~K=)>Ut4+e!(3q={gKm$nbWME)oKynw{DOp)r89(ii-(xKz8jyws0>p2b~S^9@gN3D`jM0JP+r9 z7)ZWn`12n{iY^>PJcY7?q6cK@;EyMYg9a(s38iz4M1o>DBK&~DB&Ty==AwSK4?=NZ VR1~)Si=u>Zo`Qw}P*+w|761(!fl2@X literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/ea32a24630718593936ff0283ec073626a7712e0 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/ea32a24630718593936ff0283ec073626a7712e0 new file mode 100644 index 0000000000000000000000000000000000000000..13dcdffbc32802a295bce5f7de01f09a8bc64414 GIT binary patch literal 553 zcmZQ#U|?j3f`esCmo9~K2A;uPIJHoe0Sq*Njv@f8(B&tg3lQLIK;#1%K;X~+NEGGVg;NVf85lIck_-$x85o!t7*JFo zb7mkoJENkYdO=!2`U&VoXpMq61&=*IwIEl4!Jq#~IuXnUpd^AvEGv{1VkiRx$3nc$ d&q4r1_#rt36cSJ+P`hCUW6r%>hu6l0M%U?;!` zBym(`D=L>r-VY+xp(v|?@ZbRhb{mQcqBz`8KYlf i3mG(k)J_Hl)IdSiNm3A@YNdde1k?+4C&YaWDk1L(RP!RzD D=ykyo literal 0 HcmV?d00001 diff --git a/toolchain/lex/fuzzer_corpus/mismatched_brackets/f6e6185a8f57674faeb1af4865995bc778ef1b84 b/toolchain/lex/fuzzer_corpus/mismatched_brackets/f6e6185a8f57674faeb1af4865995bc778ef1b84 new file mode 100644 index 0000000000000000000000000000000000000000..e6547a1502089e93f3289696e7abac3bf04086d3 GIT binary patch literal 120 tcmd;PKmcB51RKWs4rFMgq5u^*j{(MHNC5(v76t~00zGSWp9VxlAOi%Ts^>18S||#U&;U|985o!taJyv&ie~OYsF4f| zMC#4L2maF#pxcY$53C&Q!G)@nk%94iRFoEyAX0EL{P~Y0K@`&rJ0MCgaf(7&VNu4h z5Xr5e=pz{UP#ZuhiL(r*3Y<`tVLak#>`{Q^Y~qbVs3ho3^fXRt{y<4;B`7>h4#Hiy zSl9yrRW-HKA*$)9JW%xG1E{KT^Y9fjxOI^y!(E83FAhxr6zQ0;Lr7!+ngTRK4bk}I iayI}0R4<!rYybd`08Hip literal 0 HcmV?d00001 diff --git a/toolchain/lex/lex.cpp b/toolchain/lex/lex.cpp index 57f1c21a5f29..a44eb9a1076c 100644 --- a/toolchain/lex/lex.cpp +++ b/toolchain/lex/lex.cpp @@ -4,6 +4,7 @@ #include "toolchain/lex/lex.h" +#include #include #include #include @@ -19,6 +20,7 @@ #include "toolchain/diagnostics/format_providers.h" #include "toolchain/lex/character_set.h" #include "toolchain/lex/helpers.h" +#include "toolchain/lex/mismatched_brackets.h" #include "toolchain/lex/numeric_literal.h" #include "toolchain/lex/string_literal.h" #include "toolchain/lex/token_index.h" @@ -82,9 +84,10 @@ class [[clang::internal_linkage]] Lexer { bool formed_token_; }; - Lexer(SharedValueStores& value_stores, SourceBuffer& source, - Diagnostics::Consumer& consumer) - : buffer_(value_stores, source), + Lexer(const LexOptions& options, SharedValueStores& value_stores, + SourceBuffer& source, Diagnostics::Consumer& consumer) + : options_(options), + buffer_(value_stores, source), consumer_(consumer), emitter_(&consumer_, &buffer_), token_emitter_(&consumer_, &buffer_) {} @@ -227,6 +230,8 @@ class [[clang::internal_linkage]] Lexer { // Handles `//@dump-sem-ir-end` for a `DumpSemIRRange`. auto EndDumpSemIRRange(const char* diag_loc) -> void; + LexOptions options_; + TokenizedBuffer buffer_; LineIndex line_index_ = LineIndex::None; @@ -1630,36 +1635,22 @@ class Lexer::ErrorRecoveryBuffer { explicit ErrorRecoveryBuffer(TokenizedBuffer* buffer) : buffer_(buffer) {} auto empty() const -> bool { - return new_tokens_.empty() && !any_error_tokens_; + return insertions_.empty() && !any_error_tokens_; } - // Insert a recovery token of kind `kind` before `insert_before`. Note that we - // currently require insertions to be specified in source order, but this - // restriction would be easy to relax. - auto InsertBefore(TokenIndex insert_before, TokenKind kind) -> void { - CARBON_CHECK(insert_before.index > 0, - "Cannot insert before the start of file token."); - CARBON_CHECK( - insert_before.index < static_cast(buffer_->token_infos_.size()), - "Cannot insert after the end of file token."); - CARBON_CHECK( - new_tokens_.empty() || new_tokens_.back().first <= insert_before, - "Insertions performed out of order."); + // Insert a recovery token of kind `kind` before `insert_before`. Multiple + // insertions before the same token are applied in reverse of the order they + // were requested (LIFO). Returns an id that `GetInsertedTokenIndex` maps to + // the inserted token, once `Apply` has run. + auto InsertBefore(TokenIndex insert_before, TokenKind kind) -> int { + return AddInsertion(insert_before, kind, /*is_after=*/false); + } - // If the `insert_before` token has leading whitespace, mark the - // inserted token as also having leading whitespace. This avoids changing - // whether the prior tokens had leading or trailing whitespace when - // inserting. - bool insert_leading_space = buffer_->HasLeadingWhitespace(insert_before); - - // Find the end of the token before the target token, and add the new token - // there. - TokenIndex insert_after(insert_before.index - 1); - const auto& prev_info = buffer_->token_infos_.Get(insert_after); - int32_t byte_offset = - prev_info.byte_offset() + buffer_->GetTokenText(insert_after).size(); - new_tokens_.push_back( - {insert_before, TokenInfo(kind, insert_leading_space, byte_offset)}); + // Insert a recovery token of kind `kind` after `insert_after`. Multiple + // insertions after the same token are applied in the order they were + // requested (FIFO). Returns an id as `InsertBefore` does. + auto InsertAfter(TokenIndex insert_after, TokenKind kind) -> int { + return AddInsertion(insert_after, kind, /*is_after=*/true); } // Replace the given token with an error token. We do this immediately, @@ -1673,27 +1664,65 @@ class Lexer::ErrorRecoveryBuffer { // Merge the recovery tokens into the token list of the tokenized buffer. auto Apply() -> void { + llvm::sort(insertions_); + ValueStore old_tokens = std::exchange(buffer_->token_infos_, {}); - int new_size = old_tokens.size() + new_tokens_.size(); + int new_size = old_tokens.size() + insertions_.size(); buffer_->token_infos_.Reserve(new_size); buffer_->recovery_tokens_.resize(new_size); + inserted_token_index_.assign(insertions_.size(), TokenIndex::None); + new_token_index_.assign(old_tokens.size(), TokenIndex::None); - auto old_tokens_range = old_tokens.enumerate(); - auto old_tokens_it = old_tokens_range.begin(); - for (auto [next_offset, info] : new_tokens_) { - for (; old_tokens_it->first < next_offset; ++old_tokens_it) { - buffer_->token_infos_.Add(old_tokens_it->second); + size_t ins_idx = 0; + for (TokenIndex old_idx(0); + old_idx.index < static_cast(old_tokens.size()); ++old_idx.index) { + if (ins_idx == insertions_.size() || + insertions_[ins_idx].target() != old_idx) { + // Nothing is inserted before this token: it keeps its leading + // whitespace and simply moves across. + new_token_index_[old_idx.index] = + buffer_->token_infos_.Add(old_tokens.Get(old_idx)); + continue; } - // Flag the token just added at its index in the merged list, which - // shifts past `next_offset` (the pre-insertion index) once any earlier - // insertion has been applied. - TokenIndex added = buffer_->AddToken(info); - buffer_->recovery_tokens_.set(added.index); + + // At least one insertion goes before this token. `insertions_` is sorted + // by target, so they are exactly the next run of entries. The first of + // them takes over the token's leading whitespace, so that the whitespace + // stays at the start of the run. + bool orig_leading_space = old_tokens.Get(old_idx).has_leading_space(); + bool is_first = true; + for (; ins_idx < insertions_.size() && + insertions_[ins_idx].target() == old_idx; + ++ins_idx) { + TokenInfo info = insertions_[ins_idx].info.WithLeadingSpace( + is_first ? orig_leading_space : false); + is_first = false; + TokenIndex added = buffer_->AddToken(info); + buffer_->recovery_tokens_.set(added.index); + inserted_token_index_[insertions_[ins_idx].insertion_order] = added; + } + new_token_index_[old_idx.index] = buffer_->token_infos_.Add( + old_tokens.Get(old_idx).WithLeadingSpace(false)); } - for (; old_tokens_it != old_tokens_range.end(); ++old_tokens_it) { - buffer_->token_infos_.Add(old_tokens_it->second); + } + + // Maps a token index from before `Apply` to the same token's index after it. + // `Apply` must have run, except that with no insertions to apply this is the + // identity either way. + auto GetNewTokenIndex(TokenIndex old_index) const -> TokenIndex { + if (insertions_.empty()) { + return old_index; } + CARBON_CHECK(!new_token_index_.empty(), + "Token indexes are only renumbered by `Apply`."); + return new_token_index_[old_index.index]; + } + + // Maps an id returned by `InsertBefore` or `InsertAfter` to the token that + // insertion added. `Apply` must have run. + auto GetInsertedTokenIndex(int insertion_id) const -> TokenIndex { + return inserted_token_index_[insertion_id]; } // Perform bracket matching to fix cross-references between tokens. This must @@ -1722,23 +1751,371 @@ class Lexer::ErrorRecoveryBuffer { } private: + struct Insertion { + TokenIndex anchor; + bool is_after; + // Where this insertion came in the sequence of requests, which is both the + // id handed back to the caller and the tiebreak between insertions that go + // in the same place. + int insertion_order; + TokenInfo info; + + // The token this insertion goes before, indexed in the stream as it was + // before `Apply`. `Apply` sorts insertions by this. + auto target() const -> TokenIndex { + return is_after ? TokenIndex(anchor.index + 1) : anchor; + } + + // Orders insertions as `Apply` emits them, so that sorting produces the + // final token order. Insertions are grouped by the token they go before, + // because `Apply` walks the old tokens in order, and within a group: + // + // - An insertion requested as "after the previous token" comes before one + // requested as "before this token", so that each stays on the side of + // the gap it was anchored to. + // - A closing bracket comes before an opening one, so that a group ending + // in the gap is closed before a new group is opened in it. + // - Otherwise the request order breaks the tie, in the direction that puts + // the earliest request nearest its anchor: first-requested first for + // insertions after a token, last-requested first for insertions before + // one. + friend auto operator<(const Insertion& lhs, const Insertion& rhs) -> bool { + if (lhs.target() != rhs.target()) { + return lhs.target() < rhs.target(); + } + if (lhs.is_after != rhs.is_after) { + return lhs.is_after; + } + bool lhs_is_closing = lhs.info.kind().is_closing_symbol(); + bool rhs_is_closing = rhs.info.kind().is_closing_symbol(); + if (lhs_is_closing != rhs_is_closing) { + return lhs_is_closing; + } + return lhs.is_after ? lhs.insertion_order < rhs.insertion_order + : lhs.insertion_order > rhs.insertion_order; + } + }; + + auto AddInsertion(TokenIndex anchor, TokenKind kind, bool is_after) -> int { + CARBON_CHECK(anchor.index >= 0, "Invalid anchor token index."); + CARBON_CHECK(anchor.index < static_cast(buffer_->token_infos_.size()), + "Cannot insert past the end of file token."); + if (!is_after) { + CARBON_CHECK(anchor.index > 0, + "Cannot insert before the start of file token."); + } + + bool insert_leading_space = false; + int32_t byte_offset = 0; + + if (!is_after) { + insert_leading_space = buffer_->HasLeadingWhitespace(anchor); + TokenIndex insert_after_idx(anchor.index - 1); + const auto& prev_info = buffer_->token_infos_.Get(insert_after_idx); + byte_offset = prev_info.byte_offset() + + buffer_->GetTokenText(insert_after_idx).size(); + } else { + const auto& anchor_info = buffer_->token_infos_.Get(anchor); + byte_offset = + anchor_info.byte_offset() + buffer_->GetTokenText(anchor).size(); + TokenIndex next_tok(anchor.index + 1); + if (next_tok.index < static_cast(buffer_->token_infos_.size())) { + insert_leading_space = buffer_->HasLeadingWhitespace(next_tok); + } + } + + int insertion_order = static_cast(insertions_.size()); + insertions_.push_back({ + .anchor = anchor, + .is_after = is_after, + .insertion_order = insertion_order, + .info = TokenInfo(kind, insert_leading_space, byte_offset), + }); + return insertion_order; + } + TokenizedBuffer* buffer_; - - // A list of tokens to insert into the token stream to fix mismatched - // brackets. The first element in each pair is the original token index to - // insert the new token before. - llvm::SmallVector> new_tokens_; - - // Whether we have changed any tokens into error tokens. + llvm::SmallVector insertions_; bool any_error_tokens_ = false; + + // Filled in by `Apply`: which token each insertion added, indexed by the id + // `AddInsertion` handed out. + llvm::SmallVector inserted_token_index_; + + // Filled in by `Apply`: the new index of each token that existed before it, + // indexed by that token's old index. + llvm::SmallVector new_token_index_; }; -// Issue an UnmatchedOpening diagnostic. -static auto DiagnoseUnmatchedOpening(Diagnostics::Emitter& emitter, - TokenIndex opening_token) -> void { - CARBON_DIAGNOSTIC(UnmatchedOpening, Error, - "opening symbol without a corresponding closing symbol"); - emitter.Emit(opening_token, UnmatchedOpening); +// Returns true if the token kind forms a complete primary expression on its +// own: an identifier, a literal, `self`, a type keyword, and so on. +static auto IsLeafTokenKind(TokenKind kind) -> bool { + switch (kind) { + case TokenKind::Identifier: + case TokenKind::IntLiteral: + case TokenKind::RealLiteral: + case TokenKind::StringLiteral: + case TokenKind::CharLiteral: + case TokenKind::IntTypeLiteral: + case TokenKind::UnsignedIntTypeLiteral: + case TokenKind::FloatTypeLiteral: + case TokenKind::True: + case TokenKind::False: + case TokenKind::SelfValueIdentifier: + case TokenKind::SelfTypeIdentifier: + case TokenKind::Underscore: + case TokenKind::Bool: + case TokenKind::Type: + case TokenKind::Auto: + case TokenKind::Array: + case TokenKind::Str: + case TokenKind::Char: + case TokenKind::Core: + case TokenKind::Cpp: + return true; + default: + return false; + } +} + +static auto CollectMismatchedBracketTokens(const TokenizedBuffer& buffer) + -> llvm::SmallVector { + llvm::SmallVector input_tokens; + input_tokens.reserve(buffer.size()); + + // Where the previous token in the full stream ended (comments are not + // tokens, so they don't interrupt this). + int32_t prev_end_byte = -1; + int32_t prev_line_index = -1; + + for (auto it = buffer.tokens().begin(); it != buffer.tokens().end(); ++it) { + TokenIndex token = *it; + auto kind = buffer.GetKind(token); + int32_t byte_offset = buffer.GetByteOffset(token); + auto token_line = buffer.GetLine(token); + bool has_wide_leading_space = prev_end_byte >= 0 && + token_line.index == prev_line_index && + byte_offset - prev_end_byte >= 2; + prev_end_byte = + byte_offset + static_cast(buffer.GetTokenText(token).size()); + prev_line_index = token_line.index; + + bool is_paren_keyword = false; + bool is_else_keyword = false; + BracketTokenKind bracket_kind; + switch (kind) { + case TokenKind::OpenParen: + bracket_kind = BracketTokenKind::OpenParen; + break; + case TokenKind::OpenCurlyBrace: + bracket_kind = BracketTokenKind::OpenCurlyBrace; + break; + case TokenKind::OpenSquareBracket: + bracket_kind = BracketTokenKind::OpenSquareBracket; + break; + case TokenKind::CloseParen: + bracket_kind = BracketTokenKind::CloseParen; + break; + case TokenKind::CloseCurlyBrace: + bracket_kind = BracketTokenKind::CloseCurlyBrace; + break; + case TokenKind::CloseSquareBracket: + bracket_kind = BracketTokenKind::CloseSquareBracket; + break; + case TokenKind::Semi: + bracket_kind = BracketTokenKind::Semi; + break; + case TokenKind::Comma: + bracket_kind = BracketTokenKind::Comma; + break; + case TokenKind::Period: + bracket_kind = BracketTokenKind::Period; + break; + case TokenKind::If: + case TokenKind::While: + case TokenKind::For: + case TokenKind::Match: + bracket_kind = BracketTokenKind::StatementIntroducer; + is_paren_keyword = true; + break; + case TokenKind::Else: { + bracket_kind = BracketTokenKind::StatementIntroducer; + // Only a statement `else` (followed by `{` or `if`) normally follows + // a `}`; a ternary `if..then..else` is followed by an expression. + auto else_next = std::next(it); + if (else_next != buffer.tokens().end()) { + auto next_kind = buffer.GetKind(*else_next); + is_else_keyword = next_kind == TokenKind::OpenCurlyBrace || + next_kind == TokenKind::If; + } + break; + } +#define CARBON_DECL_INTRODUCER_TOKEN(kind, name) case TokenKind::kind: +#include "toolchain/lex/token_kind.def" + case TokenKind::Abstract: + case TokenKind::Case: + case TokenKind::Continue: + case TokenKind::Default: + case TokenKind::Eval: + case TokenKind::Extend: + case TokenKind::Final: + case TokenKind::Friend: + case TokenKind::Inline: + case TokenKind::MustEval: + case TokenKind::Observe: + case TokenKind::Override: + case TokenKind::Private: + case TokenKind::Protected: + case TokenKind::Return: + case TokenKind::Returned: + case TokenKind::Static: + case TokenKind::Virtual: + bracket_kind = BracketTokenKind::StatementIntroducer; + break; + case TokenKind::Forall: + bracket_kind = BracketTokenKind::Other; + is_paren_keyword = true; + break; + case TokenKind::Equal: + bracket_kind = BracketTokenKind::Assignment; + break; + case TokenKind::As: + bracket_kind = BracketTokenKind::As; + break; + case TokenKind::MinusGreater: + case TokenKind::Where: + bracket_kind = BracketTokenKind::StructuralOp; + break; + case TokenKind::Ref: + case TokenKind::Unused: + case TokenKind::Template: + case TokenKind::Const: + bracket_kind = BracketTokenKind::ModifierKeyword; + break; + case TokenKind::EqualEqual: + case TokenKind::ExclaimEqual: + case TokenKind::Less: + case TokenKind::LessEqual: + case TokenKind::Greater: + case TokenKind::GreaterEqual: + case TokenKind::And: + case TokenKind::Or: + bracket_kind = BracketTokenKind::ComparisonOp; + break; + case TokenKind::FileEnd: + bracket_kind = BracketTokenKind::FileEnd; + break; + default: + bracket_kind = IsLeafTokenKind(kind) ? BracketTokenKind::Leaf + : BracketTokenKind::Other; + break; + } + + auto line = token_line; + int32_t line_indent = + (kind == TokenKind::FileEnd) ? 0 : buffer.GetIndentColumnNumber(line); + + auto next_it = std::next(it); + bool is_at_end_of_line = + (next_it == buffer.tokens().end() || buffer.GetLine(*next_it) != line); + + bool is_struct_brace = false; + if (kind == TokenKind::OpenCurlyBrace) { + if (next_it != buffer.tokens().end()) { + auto next_kind = buffer.GetKind(*next_it); + if (next_kind == TokenKind::Period || + next_kind == TokenKind::CloseCurlyBrace) { + is_struct_brace = true; + } else if (next_kind == TokenKind::Identifier) { + auto next2_it = std::next(next_it); + if (next2_it != buffer.tokens().end() && + buffer.GetKind(*next2_it) == TokenKind::Colon) { + is_struct_brace = true; + } + } + } + } + + input_tokens.push_back(MismatchedBracketToken{ + .token_index = token, + .kind = bracket_kind, + .line = line.index, + .line_indent = line_indent, + .is_at_end_of_line = is_at_end_of_line, + .is_struct_brace = is_struct_brace, + .is_paren_keyword = is_paren_keyword, + .is_else_keyword = is_else_keyword, + .has_leading_space = buffer.HasLeadingWhitespace(token), + .has_wide_leading_space = has_wide_leading_space, + }); + } + + return input_tokens; +} + +// The source position where `token` starts. Note that this can't go through +// `GetTokenText`, which returns the kind's fixed spelling rather than a pointer +// into the source for most token kinds. +static auto TokenStartPosition(const TokenizedBuffer& buffer, TokenIndex token) + -> const char* { + return buffer.source().text().begin() + buffer.GetByteOffset(token); +} + +// The source position just past the end of `token`. +static auto TokenEndPosition(const TokenizedBuffer& buffer, TokenIndex token) + -> const char* { + return TokenStartPosition(buffer, token) + buffer.GetTokenText(token).size(); +} + +// Where in the source a bracket that `correction` proposes would be written: +// just past the end of the token it goes after, rather than at the start of the +// token it goes before. For `f(x` on one line and `;` on the next, that points +// the suggestion at the position directly after the `x`, where the `)` belongs, +// instead of down at the `;`. Where the two tokens are separated, the bracket +// is then placed on the side that matches how it would be written. +static auto BracketInsertionPosition(const TokenizedBuffer& buffer, + const BracketCorrection& correction) + -> const char* { + TokenIndex insert_after = + correction.fix_action == BracketFixAction::InsertAfter + ? correction.fix_token_index + : TokenIndex(correction.fix_token_index.index - 1); + // Nothing precedes the first token, so an insertion before it goes at the + // very start of the file. + if (!insert_after.has_value()) { + return TokenStartPosition(buffer, correction.fix_token_index); + } + + const char* pos = TokenEndPosition(buffer, insert_after); + const char* end = buffer.source().text().end(); + + // Advances over the spaces starting at `from`, up to `limit` of them. + auto skip_spaces = [end](const char* from, int32_t limit) { + for (; limit > 0 && from != end && *from == ' '; --limit) { + ++from; + } + return from; + }; + constexpr int32_t NoLimit = std::numeric_limits::max(); + + if (correction.fix_token_kind == TokenKind::CloseCurlyBrace) { + // A `}` closing a multi-line scope goes on a line of its own, indented to + // line up with the `{` it closes. Point at that column when the next line + // is indented far enough to have a position there, and otherwise at the + // start of it: a diagnostic can't name the virtual space to the right of + // the end of a line, so this is as close as the source gets. + if (const char* line_end = skip_spaces(pos, NoLimit); + line_end != end && *line_end == '\n') { + auto open_line = buffer.GetLine(correction.diagnostic_token_index); + pos = skip_spaces(line_end + 1, + buffer.GetIndentColumnNumber(open_line) - 1); + } + } else if (correction.fix_token_kind.is_opening_symbol()) { + // An opening bracket binds to what comes after it, so it belongs on the far + // side of any space separating it from the token it follows. + pos = skip_spaces(pos, NoLimit); + } + return pos; } // If brackets didn't pair or nest properly, find a set of places to insert @@ -1746,83 +2123,80 @@ static auto DiagnoseUnmatchedOpening(Diagnostics::Emitter& emitter, // token list to describe the fixes. auto Lexer::DiagnoseAndFixMismatchedBrackets() -> void { ErrorRecoveryBuffer fixes(&buffer_); + auto input_tokens = CollectMismatchedBracketTokens(buffer_); + auto corrections = FixMismatchedBrackets(input_tokens); - // Look for mismatched brackets and decide where to add tokens to fix them. - // - // TODO: For now, we use a greedy algorithm for this. We could do better by - // taking indentation into account. For example: - // - // 1 fn F() { - // 2 if (thing1) - // 3 thing2; - // 4 } - // 5 } - // - // Here, we'll match the `{` on line 1 with the `}` on line 4, and then - // report that the `}` on line 5 is unmatched. Instead, we should notice that - // line 1 matches better with line 5 due to indentation, and work out that - // the missing `{` was on line 2, also based on indentation. - open_groups_.clear(); - for (auto token : buffer_.tokens()) { - auto kind = buffer_.GetKind(token); - if (kind.is_opening_symbol()) { - open_groups_.push_back(token); - continue; + // For each correction, the insertion it requested, or -1 if it made none. + // Applying the fixes renumbers the token stream, so this is what lets the + // corrections be reported against the indexes the caller will see. + llvm::SmallVector insertion_ids(corrections.size(), -1); + + for (auto [correction_index, correction] : llvm::enumerate(corrections)) { + CARBON_DIAGNOSTIC(UnmatchedOpening, Error, + "opening symbol without a corresponding closing symbol"); + CARBON_DIAGNOSTIC(UnmatchedClosing, Error, + "closing symbol without a corresponding opening symbol"); + CARBON_DIAGNOSTIC(PossiblyMissingBracketHere, Note, + "possibly missing `{0}` here", Lex::TokenKind); + + // The note names a position between two tokens, which only the source + // pointer emitter can express. + auto builder = emitter_.Build( + TokenStartPosition(buffer_, correction.diagnostic_token_index), + correction.diagnostic_kind == BracketDiagnosticKind::UnmatchedOpening + ? UnmatchedOpening + : UnmatchedClosing); + + if (correction.fix_action == BracketFixAction::ReplaceWithError) { + fixes.ReplaceWithError(correction.fix_token_index); + } else if (correction.is_tied) { + // The cheapest repairs disagree about where this bracket goes, so give up + // on the bracket rather than suggest one of them. + fixes.ReplaceWithError(correction.diagnostic_token_index); + } else { + builder.Note(BracketInsertionPosition(buffer_, correction), + PossiblyMissingBracketHere, correction.fix_token_kind); + insertion_ids[correction_index] = + correction.fix_action == BracketFixAction::InsertBefore + ? fixes.InsertBefore(correction.fix_token_index, + correction.fix_token_kind) + : fixes.InsertAfter(correction.fix_token_index, + correction.fix_token_kind); } - if (!kind.is_closing_symbol()) { - continue; - } - - // Find the innermost matching opening symbol. - auto opening_it = llvm::find_if( - llvm::reverse(open_groups_), [&](TokenIndex opening_token) { - return buffer_.token_infos_.Get(opening_token) - .kind() - .closing_symbol() == kind; - }); - if (opening_it == open_groups_.rend()) { - CARBON_DIAGNOSTIC( - UnmatchedClosing, Error, - "closing symbol without a corresponding opening symbol"); - token_emitter_.Emit(token, UnmatchedClosing); - fixes.ReplaceWithError(token); - continue; - } - - // All intermediate open tokens have no matching close token. - for (auto it = open_groups_.rbegin(); it != opening_it; ++it) { - DiagnoseUnmatchedOpening(token_emitter_, *it); - - // Add a closing bracket for the unclosed group here. - // - // TODO: Indicate in the diagnostic that we did this, perhaps by - // annotating the snippet. - auto opening_kind = buffer_.GetKind(*it); - fixes.InsertBefore(token, opening_kind.closing_symbol()); - } - - open_groups_.erase(opening_it.base() - 1, open_groups_.end()); + builder.Emit(); } - // Diagnose any remaining unmatched opening symbols. - for (auto token : open_groups_) { - // We don't have a good location to insert a close bracket. Convert the - // opening token from a bracket to an error. - DiagnoseUnmatchedOpening(token_emitter_, token); - fixes.ReplaceWithError(token); + buffer_.has_errors_ = true; + + if (!fixes.empty()) { + fixes.Apply(); + fixes.FixTokenCrossReferences(); } - CARBON_CHECK(!fixes.empty(), "Didn't find anything to fix"); - fixes.Apply(); - fixes.FixTokenCrossReferences(); + if (options_.bracket_corrections) { + // Report the corrections against the token stream the caller sees, rather + // than the one recovery started from, which no longer exists. A correction + // that inserted a bracket names the inserted token itself; one that only + // proposed a bracket, or replaced one with an error, names the token it + // still refers to. + for (auto [correction_index, correction] : llvm::enumerate(corrections)) { + correction.diagnostic_token_index = + fixes.GetNewTokenIndex(correction.diagnostic_token_index); + correction.fix_token_index = + insertion_ids[correction_index] >= 0 + ? fixes.GetInsertedTokenIndex(insertion_ids[correction_index]) + : fixes.GetNewTokenIndex(correction.fix_token_index); + } + *options_.bracket_corrections = std::move(corrections); + } } auto Lex(SharedValueStores& value_stores, SourceBuffer& source, LexOptions options) -> TokenizedBuffer { auto* consumer = options.consumer ? options.consumer : &Diagnostics::ConsoleConsumer(); - auto tokens = Lexer(value_stores, source, *consumer).Lex(); + auto tokens = Lexer(options, value_stores, source, *consumer).Lex(); if (options.vlog_stream || options.dump_stream) { // Flush diagnostics before printing. diff --git a/toolchain/lex/lex.h b/toolchain/lex/lex.h index fdfefe567ed7..663168f7d141 100644 --- a/toolchain/lex/lex.h +++ b/toolchain/lex/lex.h @@ -7,6 +7,7 @@ #include "toolchain/base/shared_value_stores.h" #include "toolchain/diagnostics/emitter.h" +#include "toolchain/lex/mismatched_brackets.h" #include "toolchain/lex/tokenized_buffer.h" #include "toolchain/source/source_buffer.h" @@ -27,6 +28,12 @@ struct LexOptions { // When dumping, whether to omit `FileStart` and `FileEnd` in output. bool omit_file_boundary_tokens = false; + + // If set, points to a caller-owned vector that `Lex` overwrites with the + // bracket corrections recovery made; it need only outlive the `Lex` call. + // Their token indexes refer to the returned buffer, so a correction that + // inserted a bracket names the inserted token itself. + llvm::SmallVector* bracket_corrections = nullptr; }; // Lexes a buffer of source code into a tokenized buffer. diff --git a/toolchain/lex/mismatched_brackets.cpp b/toolchain/lex/mismatched_brackets.cpp new file mode 100644 index 000000000000..19dcf59fea33 --- /dev/null +++ b/toolchain/lex/mismatched_brackets.cpp @@ -0,0 +1,2656 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#include "toolchain/lex/mismatched_brackets.h" + +#include +#include +#include +#include +#include +#include +#include + +#include "common/check.h" +#include "common/hashing.h" +#include "common/hashtable_key_context.h" +#include "common/map.h" +#include "common/set.h" +#include "llvm/ADT/BitVector.h" +#include "llvm/ADT/BitmaskEnum.h" +#include "llvm/ADT/STLExtras.h" +#include "llvm/ADT/STLFunctionalExtras.h" +#include "llvm/ADT/Sequence.h" +#include "llvm/ADT/StringRef.h" + +namespace Carbon::Lex { + +LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE(); + +// Maximum number of collapsed items in a damaged region before falling back to +// naive greedy recovery. The beam search below is linear in the region size, +// so this is mostly a defense against pathological inputs. +constexpr int32_t MaxRegionItemsForSearch = 1500; + +// Layered beam search width limit. +constexpr size_t MaxBeamWidth = 16; + +// Maximum stack depth allowed during search before capping. +constexpr size_t MaxSearchStackDepth = 12; + +// Maximum number of distinct optimal repair paths enumerated when checking +// whether the optimal repairs agree about each correction. Paths beyond this +// are not examined, so a correction they alone would dispute stays untied. +constexpr size_t MaxOptimalPaths = 100; + +// The cost model. All costs are relative; the search finds the cheapest way +// to make the whole region well-bracketed, and the resulting insertions and +// replacements become the suggested corrections. When several cheapest ways +// exist, corrections they disagree about are marked tied and downgraded to +// error replacements, so it's important that the intended repair for a common +// mistake is *strictly* cheaper than the alternatives. +// +// The individual numbers here and in the rule tables below carry no meaning +// beyond how they order the repairs they price. They were not derived; they +// were hill-climbed by coordinate descent against `mismatched_brackets_eval`, +// which deletes brackets from real Carbon source and scores what recovery puts +// back, including a per-rule precision breakdown. So a cost is best changed the +// same way it was chosen: adjust it, rerun the evaluation, and keep the change +// only if the scores improve. Rebalancing is worth doing when the evaluation +// reports poor precision for a rule, and is required when adding a rule, whose +// cost only means something relative to the rules it competes with. See +// /toolchain/docs/lex/mismatched_brackets.md for how to run it. +// +// Costs of replacing a real bracket with an error token. These are the +// "give up on this bracket" fallbacks; a good targeted repair should beat +// them, and a dubious one should lose to them. +constexpr int32_t CostReplaceClosing = 30; +constexpr int32_t CostReplaceOpening = 50; + +// Costs of inserting a synthetic closing bracket in front of the current +// token live in the `CloserRules` table below, which is keyed by how strongly +// the context suggests the group ends here. These few are shared with the +// region-end handling, which isn't part of that table. +// Closing anything at the end of the file or region. +constexpr int32_t CostCloseAtEnd = 12; +constexpr int32_t CostCloseParenAtEnd = 22; +constexpr int32_t CostCloseStructAtEnd = 20; +// Closing a paren/square bracket before a mid-line `.` that has whitespace +// before it: member access is normally written without spaces. Priced below the +// `Adv_SpacedPeriodInParen` rule in `AdvanceRules` below, which penalizes +// stepping over such a `.` instead, so that closing here beats closing earlier +// and leaving the spaced `.` unexplained. +constexpr int32_t CostCloseParenBeforeSpacedPeriod = 6; +// Closing a group at a wide mid-line whitespace gap, which suggests the +// closer was deleted in the gap. +constexpr int32_t CostCloseAtWideGap = 6; + +// Costs of inserting a synthetic opening bracket in front of the current +// token live in the `OpenerRules` table below, and penalties for advancing +// over a token in a context where it doesn't belong in `AdvanceRules`. + +// Penalty for matching a scope `}` whose indentation doesn't match its +// opener's header, when they're on different lines. +constexpr int32_t CostBraceIndentMismatchBase = 30; +constexpr int32_t CostBraceIndentMismatchPerColumn = 20; + +namespace { + +// A contextual cue a rule can test, beyond the bucket's context category and +// token kind. These are bit indexes into a `CueSet`; no cue's value is ever +// significant, so cues can be added, removed, and reordered freely. +enum class Cue : uint8_t { + // Properties of the current token. + // + // The current token is the first on its line. + FirstOnLine, + // There is whitespace (or a comment) directly before the current token. + LeadingSpace, + // The current token is mid-line with two or more bytes of space before it. + WideGap, + // The current token is a `{` with struct-literal cues. + StructBrace, + // The current token is `else`. + ElseKeyword, + + // Properties of the token directly before the current one. All are false at + // the start of the input. + // + // The previous token is value-ending, an adjacency that is illegal before a + // leaf. + PrevValueEnding, + // As `PrevValueEnding`, but also counting `]` and `}`: the previous token + // ends a value, whether or not a leaf may follow it. + PrevValueLike, + // The previous token itself has whitespace before it. + PrevHasLeadingSpace, + // The previous token is a keyword that must be followed by `(` (`if`, + // `while`, `for`, `match`), or one that must be followed by `[` (`forall`). + PrevKeywordWantsParen, + PrevKeywordWantsSquare, + // The previous token is `as`, `->`, or `where`, so what follows it is a type + // rather than something callable. (Not `=`, which is followed by a value.) + PrevIntroducesType, + // The previous token's kind, for the kinds any rule distinguishes. At most + // one of these holds at a time. + PrevIsPeriod, + PrevIsComma, + // `PrevIsOpenBracket` covers `(` and `[` together, which no rule separates. + PrevIsOpenBracket, + PrevIsOpenCurly, + PrevIsCloseParen, + PrevIsCloseSquare, + PrevIsCloseCurly, + + // Properties of the item directly before the current one. Unlike the `Prev` + // cues above, which look at the last token of that item, these look at the + // item as a whole, so they say nothing about a collapsed `(...)` block's + // contents. + // + // The previous item is a leaf. + PrevItemIsLeaf, + // The previous item is a name directly following `as` or `->`, so it is a + // type rather than something callable. + PrevItemIsTypeName, + + // Properties of the item, from the analysis of the surrounding tokens. + // + // This item is a collapsed well-bracketed block rather than a single token. + CollapsedBlock, + // This collapsed block contains a scope `{`. + ContainsScopeBrace, + // The current token starts the body of a statement or declaration whose + // header is complete, and whose header didn't contain a `{`. + FollowsStatementHeader, + HeaderHasOpenCurly, + + // Properties of the innermost open bracket, and of its relationship to the + // current token. All are false when no bracket is open. + // + // The innermost open bracket is a call or index `(`/`[`, rather than one + // following a keyword such as `if` or `forall`. + CallParenTop, + // The innermost open bracket is the token directly before the current one, so + // closing here would make an empty group. + AfterOpenTop, + // The current token is indented no further than the header of the statement + // or declaration containing the innermost open bracket. + DedentToHeader, + // The current token is indented no further than the line of the innermost + // open bracket itself, as opposed to that of its statement header. + DedentToOpenerLine, + // The current token is on a later line than the innermost open bracket. + NewLineFromTop, + // The current token is a closer matching an opener further out on the stack, + // so the innermost group has to close first. + Cascade, + + // Properties of the repair this search path has made so far. + // + // This path inserted a closer directly before the current token. + CloserInserted, + // The closer this path inserted can be directly followed by a leaf (`]` or + // `}`), so it repairs an otherwise illegal adjacency. + CloserFixesAdjacency, + // This path synthesized an opener directly before the current token. + OpenerHere, + // This path inserted some bracket directly before the current token, so + // suspicious whitespace before it is already explained. + BracketInsertedHere, + + // Must stay last: it counts the cues. + Count, +}; + +// A set of cues. +enum class CueSet : uint64_t { + None = 0, + LLVM_MARK_AS_BITMASK_ENUM(uint64_t{1} + << (static_cast(Cue::Count) - 1)) +}; + +} // namespace + +// The set containing exactly `cues`. +template ... CueT> +static constexpr auto CueSetOf(CueT... cues) -> CueSet { + return (CueSet::None | ... | + static_cast(uint64_t{1} << static_cast(cues))); +} + +// Convenience synonym for our token kinds. +using Kind = BracketTokenKind; + +namespace { + +// Internal representation of an item after clean subrange collapsing. +struct Item { + int32_t token_start_index; + int32_t token_end_index; + MismatchedBracketToken token; + int32_t effective_header_indent = 0; + // Every boolean property of this item and its neighbours, computed once by + // `ComputeItemCues`. This is the single record of them: the rule tables match + // against these (combined with the few cues that depend on the search state), + // and the search itself tests them through `Has`. + CueSet cues = CueSet::None; + + // Whether every one of `wanted` holds for this item. + template + constexpr auto HasAll(CueT... wanted) const -> bool { + CueSet wanted_set = CueSetOf(wanted...); + return (cues & wanted_set) == wanted_set; + } +}; + +// The parts of an unclosed opening bracket that make a search state distinct. +// Compared and hashed as a whole, so everything here must matter to the search: +// two stacks that agree on these are interchangeable. +// +// Fields are ordered so the type has no padding, which +// `has_unique_object_representations` requires and `Carbon::Map` relies on to +// hash and compare every byte. +struct OpenBracketKey { + // The real opener token, or None for a synthetic opener. + TokenIndex token_index = TokenIndex::None; + // For a synthetic opener, where it would be inserted. + TokenIndex insertion_token_index = TokenIndex::None; + // Index of the real opener in the input token array, or -1 if synthetic. + int32_t token_pos = -1; + int32_t line = -1; + // Indentation of the line containing the opener. + int32_t line_indent = 0; + // Indentation of the start of the statement or declaration containing the + // opener. + int32_t effective_header_indent = 0; + BracketTokenKind kind; + bool is_synthetic = false; + bool is_struct_brace = false; + // Whether this is a paren/square bracket directly following a value-ending + // token (a call or index), rather than following a keyword like `if`. This is + // a function of the opener token, so comparing it is redundant with + // `token_index` for a real opener, and it is always false for a synthetic + // one; it's included only because excluding it would cost more than it saves. + bool is_call_paren = false; + + friend auto operator==(const OpenBracketKey& lhs, const OpenBracketKey& rhs) + -> bool = default; +}; + +static_assert(std::has_unique_object_representations_v, + "Padding would leave indeterminate bytes for hashing."); + +// An unclosed opening bracket on the search stack, plus the bookkeeping that +// says nothing about which state this is. +struct OpenBracketInfo : OpenBracketKey { + // For a synthetic opener, the name of the rule that proposed it, for + // debugging. Two openers that differ only in which rule proposed them are the + // same state, so this is deliberately outside the key. + llvm::StringLiteral rule_name = ""; +}; + +} // namespace + +// The `{` opening the branch that the first-on-line `else` at `else_index` +// continues. Such an `else` should have been preceded by the `}` closing that +// branch, so with the `}` missing the `else` is still inside it: the brace is +// the innermost one that nothing has closed yet. Returns -1 if there is no +// enclosing brace. +static auto FindBranchBrace(llvm::ArrayRef tokens, + int32_t else_index) -> int32_t { + int32_t depth = 0; + for (int32_t j : llvm::reverse(llvm::seq(0, else_index))) { + if (tokens[j].kind == Kind::CloseCurlyBrace) { + ++depth; + } else if (tokens[j].kind == Kind::OpenCurlyBrace) { + if (depth == 0) { + return j; + } + --depth; + } + } + return -1; +} + +// Computes the associated line indentation for a token by scanning backwards, +// skipping matched parens/brackets, looking for a statement introducer. +static auto GetOuterStatementIntroducerIndent( + llvm::ArrayRef tokens, + llvm::ArrayRef match_partner, int32_t j) -> int32_t { + int32_t result_indent = tokens[j].line_indent; + while (j > 0) { + // An `else` continues the statement its `if` introduced, and the `}` + // closing its block lines up with that `if`, so keep walking out from the + // brace of the branch before it. A first-on-line `else`'s own column says + // nothing about where the statement starts: the `}` that should precede it + // is missing, so the `else` sits wherever the author left it. + if (tokens[j].is_else_keyword && tokens[j].line != tokens[j - 1].line) { + int32_t brace = FindBranchBrace(tokens, j); + if (brace > 0) { + j = brace; + result_indent = tokens[j].line_indent; + continue; + } + } + int32_t p = j - 1; + if ((tokens[p].kind == Kind::CloseParen || + tokens[p].kind == Kind::CloseSquareBracket) && + match_partner[p] != -1 && match_partner[p] < p) { + p = match_partner[p]; + if (p <= 0) { + break; + } + --p; + } + if (tokens[p].kind == Kind::StatementIntroducer) { + result_indent = tokens[p].line_indent; + // Keep walking out only while the introducers stack up on one line or + // dedent; an indented token belongs to this introducer's body. + if (tokens[j].line == tokens[p].line || + tokens[j].line_indent <= tokens[p].line_indent) { + j = p; + continue; + } + } + break; + } + return result_indent; +} + +// Computes the indentation that a token's statement or declaration starts at, +// by scanning back over the current statement to its introducer. This is what +// a `{` opened by that statement should line its `}` up with, which is often +// not the indentation of the `{` itself. +static auto ComputeAssociatedLineIndent( + llvm::ArrayRef tokens, + llvm::ArrayRef match_partner, int32_t token_index) -> int32_t { + if (token_index < 0 || token_index >= static_cast(tokens.size())) { + return 0; + } + if (tokens[token_index].kind == Kind::FileEnd) { + return 0; + } + + int32_t earliest_indent = tokens[token_index].line_indent; + + for (int32_t j = token_index - 1; j >= 0; --j) { + auto kind = tokens[j].kind; + + if (IsClosingBracket(kind) && match_partner[j] != -1 && + match_partner[j] < j) { + j = match_partner[j]; + earliest_indent = tokens[j].line_indent; + continue; + } + + if (kind == Kind::Semi || kind == Kind::OpenCurlyBrace || + kind == Kind::CloseCurlyBrace) { + break; + } + + earliest_indent = tokens[j].line_indent; + + if (kind == Kind::StatementIntroducer) { + return GetOuterStatementIntroducerIndent(tokens, match_partner, j); + } + } + + return earliest_indent; +} + +// Determines if a token follows a statement/declaration header, so that a +// scope `{` could naturally be inserted directly before it. Only tokens that +// could start a body are considered: the first token on a line, a statement +// introducer (e.g. `return` in `if (c) return;`), or a token directly +// following a `)`/`]` that ends a header. +static auto ComputeFollowsStatementHeader( + llvm::ArrayRef tokens, + llvm::ArrayRef match_partner, int32_t token_index) -> bool { + if (token_index <= 0 || token_index >= static_cast(tokens.size())) { + return false; + } + auto curr_kind = tokens[token_index].kind; + if (curr_kind == Kind::Semi || curr_kind == Kind::OpenCurlyBrace || + IsClosingBracket(curr_kind) || curr_kind == Kind::FileEnd) { + return false; + } + // Only tokens that could start a body are considered: the first token on a + // line, a statement introducer (`if (c) return;`), or a token directly + // after a `)` ending a header. (Not after `]`: declaration headers always + // continue after implicit parameter lists and `forall` clauses.) + bool is_first_on_line = + tokens[token_index].line != tokens[token_index - 1].line; + auto prev_kind = tokens[token_index - 1].kind; + if (!is_first_on_line && curr_kind != Kind::StatementIntroducer && + prev_kind != Kind::CloseParen) { + return false; + } + // A body can't start with an operator like `as` or `==`, or with a `.` + // designator (a `where`-clause continuation line). + if (IsStructuralOpKind(curr_kind) || curr_kind == Kind::ComparisonOp || + curr_kind == Kind::Period) { + return false; + } + for (int32_t j = token_index - 1; j >= 0; --j) { + auto kind = tokens[j].kind; + if ((kind == Kind::CloseParen || kind == Kind::CloseSquareBracket) && + match_partner[j] != -1 && match_partner[j] < j) { + j = match_partner[j]; + continue; + } + if (kind == Kind::Semi || kind == Kind::OpenCurlyBrace || + kind == Kind::CloseCurlyBrace || kind == Kind::OpenParen || + kind == Kind::OpenSquareBracket) { + return false; + } + if (kind == Kind::StatementIntroducer) { + if (curr_kind == Kind::StatementIntroducer) { + if (tokens[token_index].line == tokens[j].line) { + // On the same line, a chain of adjacent introducers (`private fn`) + // is a single header; but with other tokens in between, this is a + // body after a single-line header (`if (c) return x;`). + bool all_introducers = true; + for (int32_t k = j + 1; k < token_index; ++k) { + if (tokens[k].kind != Kind::StatementIntroducer) { + all_introducers = false; + break; + } + } + if (all_introducers) { + return false; + } + } else if (tokens[token_index].line_indent <= tokens[j].line_indent) { + return false; + } + } + return true; + } + } + return false; +} + +// Determines if the statement/declaration header starting at token_index +// contains an OpenCurlyBrace. +static auto ComputeHeaderHasOpenCurlyBrace( + llvm::ArrayRef tokens, + llvm::ArrayRef match_partner, int32_t token_index) -> bool { + if (token_index <= 0 || token_index >= static_cast(tokens.size())) { + return false; + } + for (int32_t j = token_index; j < static_cast(tokens.size()); ++j) { + auto kind = tokens[j].kind; + if (kind == Kind::OpenCurlyBrace) { + return true; + } + if ((kind == Kind::OpenParen || kind == Kind::OpenSquareBracket) && + match_partner[j] != -1 && match_partner[j] > j) { + j = match_partner[j]; + continue; + } + if (kind == Kind::Semi || kind == Kind::CloseCurlyBrace || + kind == Kind::FileEnd || kind == Kind::StatementIntroducer) { + return false; + } + } + return false; +} + +namespace { + +struct ParentEdge { + int32_t parent_node_index; + BracketCorrection correction; + bool has_correction = false; +}; + +// Node in the beam search tree. +struct BeamNode { + int32_t item_index; + llvm::SmallVector stack; + int32_t cost; + // The kind of synthetic closer inserted directly before the current item, + // or Other if none; an inserted closer repairs illegal adjacency with the + // preceding token. + Kind closer_inserted = Kind::Other; + llvm::SmallVector parent_edges; +}; + +// The parts of a BeamNode the search reads while expanding it. Snapshotting +// just these (rather than copying the whole node, whose `parent_edges` the +// expansion never reads) avoids copying that vector per node. A copy — not a +// reference — is needed because expanding a node pushes onto the node list, +// which may reallocate. +struct SearchState { + llvm::SmallVector stack; + int32_t cost; + Kind closer_inserted; +}; + +// What makes a search state distinct: the open-bracket stack, plus which +// closer, if any, was just inserted before the current token. Two nodes in a +// layer that agree on this are interchangeable and get merged, which is what +// `RegionSearch::layer_dedup_` looks up. This is a view of a node's state, or +// of a prospective one that has no node yet, so it must not outlive what it +// names. +// +// Only the `OpenBracketKey` part of each stack entry participates, since that's +// all `OpenBracketInfo`'s `operator==` compares and all the hash below reads. +struct StateKey { + llvm::ArrayRef stack; + Kind closer_inserted; + + friend auto operator==(const StateKey& lhs, const StateKey& rhs) -> bool { + return lhs.closer_inserted == rhs.closer_inserted && lhs.stack == rhs.stack; + } + + // Hashes the whole of each stack entry's key, and nothing outside it, so that + // equal states always hash equal. + friend auto CarbonHashValue(const StateKey& key, uint64_t seed) -> HashCode { + Hasher hasher(seed); + hasher.Hash(key.closer_inserted, key.stack.size()); + for (const OpenBracketKey& entry : key.stack) { + hasher.HashRaw(entry); + } + return static_cast(hasher); + } +}; + +// Lets the dedup table key on the search state itself while storing only a node +// index, so no stack is copied into the table. Lookups pass a `StateKey` +// directly; a stored index is translated back into one through the node list. +// +// Holds the node list by pointer, not by `ArrayRef`, because inserting a node +// reallocates it. +class LayerDedupKeyContext + : public TranslatingKeyContext { + public: + explicit LayerDedupKeyContext(const llvm::SmallVector* nodes + [[clang::lifetimebound]]) + : nodes_(nodes) {} + + auto TranslateKey(int32_t node_index) const -> StateKey { + const BeamNode& node = (*nodes_)[node_index]; + return {.stack = node.stack, .closer_inserted = node.closer_inserted}; + } + + private: + const llvm::SmallVector* nodes_; +}; + +} // namespace + +static auto Snapshot(const BeamNode& node) -> SearchState { + return {node.stack, node.cost, node.closer_inserted}; +} + +// A correction that replaces a bracket token with an error token (the "give +// up on this bracket" repair). +static auto ReplaceWithError(const MismatchedBracketToken& token, + BracketDiagnosticKind diagnostic_kind, + llvm::StringLiteral rule_name) + -> BracketCorrection { + return BracketCorrection{ + .diagnostic_kind = diagnostic_kind, + .diagnostic_token_index = token.token_index, + .fix_action = BracketFixAction::ReplaceWithError, + .fix_token_index = token.token_index, + .fix_token_kind = ToTokenKind(token.kind), + .rule_name = rule_name, + }; +} + +// Solve a damaged region using the simple greedy fallback algorithm. +static auto SolveNaive(llvm::ArrayRef items, + llvm::SmallVectorImpl& corrections) + -> void { + llvm::SmallVector open_stack; + for (const auto& item : items) { + if (item.HasAll(Cue::CollapsedBlock)) { + continue; + } + auto kind = item.token.kind; + if (kind == Kind::Semi || kind == Kind::StatementIntroducer || + kind == Kind::OpenCurlyBrace || kind == Kind::FileEnd) { + while (!open_stack.empty() && + (open_stack.back().kind == Kind::OpenParen || + open_stack.back().kind == Kind::OpenSquareBracket)) { + corrections.push_back(ReplaceWithError( + open_stack.pop_back_val(), BracketDiagnosticKind::UnmatchedOpening, + "Naive_UnclosedParenBracket")); + } + } + + if (IsOpeningBracket(kind)) { + open_stack.push_back(item.token); + } else if (IsClosingBracket(kind)) { + auto search_range = llvm::reverse(open_stack); + size_t lookback = 0; + auto match_it = search_range.end(); + for (auto it = search_range.begin(); + it != search_range.end() && lookback < 16; ++it, ++lookback) { + if (MatchingClosingKind(it->kind) == kind) { + match_it = it; + break; + } + } + + if (match_it == search_range.end()) { + corrections.push_back(ReplaceWithError( + item.token, BracketDiagnosticKind::UnmatchedClosing, + "Naive_UnmatchedClosing")); + } else { + for (auto it = search_range.begin(); it != match_it; ++it) { + corrections.push_back( + ReplaceWithError(*it, BracketDiagnosticKind::UnmatchedOpening, + "Naive_PoppedOpener")); + } + open_stack.erase(match_it.base() - 1, open_stack.end()); + } + } + } + + for (const auto& open : llvm::reverse(open_stack)) { + corrections.push_back(ReplaceWithError( + open, BracketDiagnosticKind::UnmatchedOpening, "Naive_UnclosedAtEnd")); + } +} + +// Whether two parent edges represent the same predecessor and the same repair, +// so that keeping both would be a redundant duplicate. +static auto EdgesEqual(const ParentEdge& a, const ParentEdge& b) -> bool { + if (a.parent_node_index != b.parent_node_index || + a.has_correction != b.has_correction) { + return false; + } + return !a.has_correction || + (a.correction.diagnostic_token_index == + b.correction.diagnostic_token_index && + a.correction.fix_action == b.correction.fix_action && + a.correction.fix_token_index == b.correction.fix_token_index && + a.correction.fix_token_kind == b.correction.fix_token_kind); +} + +// Returns true if `kind`'s matching opener appears in `stack` below the top. +static auto MatchesDeeperOpener(llvm::ArrayRef stack, + Kind closing_kind) -> bool { + if (!IsClosingBracket(closing_kind) || stack.size() < 2) { + return false; + } + auto req = MatchingOpeningKind(closing_kind); + for (const OpenBracketKey& entry : stack.drop_back()) { + if (entry.kind == req) { + return true; + } + } + return false; +} + +// Whether the token before the current one ends a value. Unlike +// `IsValueEndingKind`, which the leaf-adjacency rules use, this also counts `]` +// and `}`, which end a value but can be followed by a leaf. `prev_token` is +// null at the start of the input. +static auto PrevIsValueLike(const MismatchedBracketToken* prev_token) -> bool { + return prev_token != nullptr && + (IsValueEndingKind(prev_token->kind) || + prev_token->kind == Kind::CloseSquareBracket || + prev_token->kind == Kind::CloseCurlyBrace); +} + +// Whether the top of `stack` is a synthetic opener inserted directly before +// `item` — i.e. this path just opened a bracket at this position. +static auto OpenerSynthesizedHere(llvm::ArrayRef stack, + const Item& item) -> bool { + return !stack.empty() && stack.back().is_synthetic && + stack.back().insertion_token_index == item.token.token_index; +} + +// Whether a `]` or `}` was inserted directly before the current token. Such a +// closer repairs an illegal leaf adjacency (unlike `)`, these don't end a +// value). `Other` means no closer was inserted here. +static auto CloserFixesLeafAdjacency(Kind closer_inserted) -> bool { + return closer_inserted == Kind::CloseSquareBracket || + closer_inserted == Kind::CloseCurlyBrace; +} + +// The bracket-insertion rules below are expressed as data: each rule states +// the context it applies in and what that context costs, and the rules are +// tried in order so that the first (most specific) match wins. Two small +// categorical facts — a context category and the kind of the current token — +// form a bucket index, and a constexpr-built table maps each bucket to the +// bit-set of rules that can possibly apply there, so a lookup only tests the +// handful of rules relevant to the situation. + +// Context categories for the closer and advance tables: the category of the +// innermost open bracket. `Struct` and `Scope` distinguish the two kinds of +// `{`. Each category has an index, which selects its row of bucket entries, and +// a bit, which is how a rule names it — so a rule can apply to several +// categories at once. +namespace { + +namespace Top { +enum Index : int32_t { + ParenIndex, + SquareIndex, + StructIndex, + ScopeIndex, + // No open bracket at all. Only the advance table, which also scores tokens + // at the top level, uses this. + NoneIndex, + Count, +}; +constexpr uint8_t Paren = 1 << ParenIndex; +constexpr uint8_t Square = 1 << SquareIndex; +constexpr uint8_t Struct = 1 << StructIndex; +constexpr uint8_t Scope = 1 << ScopeIndex; +constexpr uint8_t None = 1 << NoneIndex; +constexpr uint8_t ParenLike = Paren | Square; +constexpr uint8_t Any = ParenLike | Struct | Scope | None; +} // namespace Top + +// Context categories for the opener table: which bracket would be inserted. +namespace Ins { +enum Index : int32_t { + ParenIndex, + SquareIndex, + ScopeBraceIndex, + StructBraceIndex, + Count, +}; +constexpr uint8_t Paren = 1 << ParenIndex; +constexpr uint8_t Square = 1 << SquareIndex; +constexpr uint8_t ScopeBrace = 1 << ScopeBraceIndex; +constexpr uint8_t StructBrace = 1 << StructBraceIndex; +constexpr uint8_t ParenLike = Paren | Square; +} // namespace Ins + +} // namespace + +// Bucket rows are shared by all the tables, so there must be room for whichever +// set of context categories is larger. +constexpr int32_t NumContextCategories = + std::max(Top::Count, Ins::Count); + +// The category of the innermost open bracket, which is always a real bracket. +static auto TopCategoryOf(const OpenBracketInfo& top) -> int32_t { + switch (top.kind) { + case Kind::OpenParen: + return Top::ParenIndex; + case Kind::OpenSquareBracket: + return Top::SquareIndex; + default: + return top.is_struct_brace ? Top::StructIndex : Top::ScopeIndex; + } +} + +// As `TopCategoryOf`, but for a stack that may be empty. +static auto TopCategoryOfStack(llvm::ArrayRef stack) + -> int32_t { + return stack.empty() ? Top::NoneIndex : TopCategoryOf(stack.back()); +} + +// The category of a synthetic opener of kind `kind`. +static auto InsCategoryOf(Kind kind, bool is_struct_brace) -> int32_t { + switch (kind) { + case Kind::OpenParen: + return Ins::ParenIndex; + case Kind::OpenSquareBracket: + return Ins::SquareIndex; + default: + return is_struct_brace ? Ins::StructBraceIndex : Ins::ScopeBraceIndex; + } +} + +// Number of distinct `Kind` values. +constexpr int32_t NumKinds = static_cast(Kind::Other) + 1; + +// A set of token kinds a rule can apply to. `Kind` values are the +// bit indexes, so there are no per-kind enumerators here to keep in sync with +// them; `Rule` takes the kinds themselves. +namespace { + +enum class KindSet : uint32_t { + None = 0, + LLVM_MARK_AS_BITMASK_ENUM(uint32_t{1} << (NumKinds - 1)) +}; + +} // namespace + +// The set containing exactly `kinds`. +template ... KindT> +static constexpr auto KindSetOf(KindT... kinds) -> KindSet { + return (KindSet::None | ... | + static_cast(uint32_t{1} << static_cast(kinds))); +} + +// Sets of kinds that name a concept more than one rule shares. A rule wanting +// just one or two particular kinds names them directly instead. +namespace Kinds { + +// Every kind: the default, for a rule that doesn't care. +constexpr KindSet Any = ~KindSet::None; + +// A `(` or `[`, opening or closing: the bracket kinds written without a +// preceding space in formatted code. +constexpr KindSet OpenGroup = + KindSetOf(Kind::OpenParen, Kind::OpenSquareBracket); +constexpr KindSet CloseGroup = + KindSetOf(Kind::CloseParen, Kind::CloseSquareBracket); + +// Any opening bracket, and everything that isn't one. +constexpr KindSet Opener = OpenGroup | KindSetOf(Kind::OpenCurlyBrace); +constexpr KindSet NonOpener = ~Opener; + +// The statement-structuring operators together. +constexpr KindSet AnyStructural = + KindSetOf(Kind::Assignment, Kind::As, Kind::StructuralOp); + +// A binary connector, which needs a left operand and so can't start a group. +constexpr KindSet BinaryConnector = + AnyStructural | KindSetOf(Kind::ComparisonOp); + +// Everything that carries no bracket structure of its own: `Other` plus the +// operator and keyword classifications split out of it. +constexpr KindSet AnyOther = + AnyStructural | + KindSetOf(Kind::Other, Kind::ComparisonOp, Kind::ModifierKeyword); + +// A token that could start a bracketed group. +constexpr KindSet GroupStarter = + Opener | KindSetOf(Kind::Leaf, Kind::Period, Kind::ModifierKeyword); + +// A leaf, or a binding modifier keyword, which like a leaf can't directly +// follow a value-ending token. +constexpr KindSet LeafLike = KindSetOf(Kind::Leaf, Kind::ModifierKeyword); + +// The kinds a dedent penalty can apply to: a closer is expected to dedent, and +// `FileEnd` isn't content. +constexpr KindSet Dedentable = + ~(CloseGroup | KindSetOf(Kind::CloseCurlyBrace, Kind::FileEnd)); + +} // namespace Kinds + +// A rule's `cost` when the move should not be considered at all. +constexpr int32_t DeclineCost = -1; + +// A rule in a bracket-insertion table. The rule applies when the context +// category is in `ctx`, the current token's kind is in `kinds`, and all four +// cue conditions hold. Rules are tried in order and the first match wins, so +// earlier rules express stronger, more specific cues. +namespace { + +struct BracketRule { + // Bit-set of context categories: `Top::` for the closer table (the category + // of the innermost open bracket), `Ins::` for the opener table (which bracket + // would be inserted). + uint8_t ctx; + // The token kinds this rule applies to. + KindSet kinds = Kinds::Any; + // Every cue listed here must hold. + CueSet when = CueSet::None; + // No cue listed here may hold. + CueSet unless = CueSet::None; + // The cues listed here must not all hold together. + CueSet not_all = CueSet::None; + // At least one cue listed here must hold, if any are listed. + CueSet any_of = CueSet::None; + // Cost of the insertion this rule proposes, or `DeclineCost`. + int32_t cost; + // See `BracketCorrection::rule_name`. + llvm::StringLiteral rule_name = ""; + + // Builders, so a rule reads as one sentence. Each returns an updated copy, + // and `Cost` or `Decline` finishes the rule. + template + constexpr auto When(CueT... cues) const -> BracketRule { + auto result = *this; + result.when = CueSetOf(cues...); + return result; + } + template + constexpr auto Unless(CueT... cues) const -> BracketRule { + auto result = *this; + result.unless = CueSetOf(cues...); + return result; + } + template + constexpr auto NotAll(CueT... cues) const -> BracketRule { + auto result = *this; + result.not_all = CueSetOf(cues...); + return result; + } + template + constexpr auto AnyOf(CueT... cues) const -> BracketRule { + auto result = *this; + result.any_of = CueSetOf(cues...); + return result; + } + constexpr auto Cost(int32_t rule_cost, llvm::StringLiteral name) const + -> BracketRule { + auto result = *this; + result.cost = rule_cost; + result.rule_name = name; + return result; + } + constexpr auto Decline() const -> BracketRule { + auto result = *this; + result.cost = DeclineCost; + return result; + } +}; + +} // namespace + +// Starts a rule that applies to context categories `ctx` and token kinds +// `kinds`. +static constexpr auto Rule(uint8_t ctx, KindSet kinds = Kinds::Any) + -> BracketRule { + return BracketRule{.ctx = ctx, .kinds = kinds, .cost = DeclineCost}; +} + +// As above, for a rule that applies to a few particular kinds. Taking the first +// kind separately keeps the no-kinds call unambiguous. +template ... RestT> +static constexpr auto Rule(uint8_t ctx, Kind kind, RestT... rest) + -> BracketRule { + return Rule(ctx, KindSetOf(kind, rest...)); +} + +// Whether `rule`'s cue conditions hold for the cue bit-set `cues`. +static constexpr auto Matches(const BracketRule& rule, CueSet cues) -> bool { + return (cues & rule.when) == rule.when && + (cues & rule.unless) == CueSet::None && + (rule.not_all == CueSet::None || + (cues & rule.not_all) != rule.not_all) && + (rule.any_of == CueSet::None || (cues & rule.any_of) != CueSet::None); +} + +// Where to insert a synthetic closing bracket, and what that costs. All costs +// are relative; see the cost model comment at the top of this file for what +// they mean and how to change one. +constexpr BracketRule CloserRules[] = { + // `(` and `[`. + // + // An empty group: the opener is directly followed by a token that can't + // start group content, so the group's closer must have been right after + // the opener. Such a token is a `,` (`f(x(), y)` becoming `f(x(, y)`); a + // binary connector like `as`/`->`/`==`, which needs a left operand + // (`f() as T` becoming `f( as T`); or a spaced `.`, which as group content + // would be written unspaced (`f(.a = 1)`). + Rule(Top::ParenLike, Kind::Comma) + .When(Cue::AfterOpenTop) + .Cost(4, "Close_EmptyGroup"), + Rule(Top::ParenLike, Kinds::BinaryConnector) + .When(Cue::AfterOpenTop) + .Cost(4, "Close_EmptyGroup"), + Rule(Top::ParenLike, Kind::Period) + .When(Cue::AfterOpenTop, Cue::LeadingSpace) + .Cost(4, "Close_EmptyGroup"), + // A `;` can't appear inside parens or square brackets at all. + Rule(Top::ParenLike, Kind::Semi).Cost(6, "Close_ParenBeforeSemi"), + Rule(Top::ParenLike).When(Cue::Cascade).Cost(3, "Close_ParenCascade"), + // A `{` starting a block means the paren should have closed: `if (c) {`, + // `while (c) {`. A struct-literal `{...}` can legitimately sit inside a + // *call* paren (`f({.x = 1})`), but not inside a keyword or grouping paren + // (whose `{` — even an empty `{}` misread as a struct — is a block). This + // is not a cue for `[`: a `]` is essentially never immediately followed by + // `{` (only in `fn [captures] {...}`, which is unimplemented), so a `[` + // should close at an earlier cue, or not here at all. + Rule(Top::Paren, Kind::OpenCurlyBrace) + .NotAll(Cue::StructBrace, Cue::CallParenTop) + .Cost(11, "Close_ParenBeforeBrace"), + // `=` can directly follow both `)` and `]`, but `->` and `as` only + // plausibly follow `)`. An unspaced structural operator is not a cue: + // formatted code spaces these operators, and an unspaced `->` is a + // pointer member access (`p->x`). + Rule(Top::Paren, Kinds::AnyStructural) + .When(Cue::LeadingSpace) + .Cost(8, "Close_ParenBeforeStructuralOp"), + Rule(Top::Square, Kind::Assignment) + .When(Cue::LeadingSpace) + .Cost(8, "Close_ParenBeforeStructuralOp"), + // A leaf directly following a value-ending token is illegal, and a `]` + // between them fixes the adjacency (unlike `)`, `]` can be directly + // followed by a leaf, as in `impl forall [...] T as ...`). + Rule(Top::Square, Kind::Leaf) + .When(Cue::PrevValueEnding) + .Cost(4, "Close_SquareAtLeafAdjacency"), + // A `.` with whitespace before it mid-line suggests a closer was deleted + // right before it: member access is written without spaces, `x.y`. + Rule(Top::ParenLike, Kind::Period) + .When(Cue::LeadingSpace, Cue::PrevValueLike) + .Unless(Cue::FirstOnLine) + .Cost(6, "Close_ParenBeforeSpacedPeriod"), + // Similarly, a `(` or `[` with whitespace before it directly following a + // value-ending token: calls and indexing are written without spaces. + Rule(Top::ParenLike, Kinds::OpenGroup) + .When(Cue::LeadingSpace, Cue::PrevValueLike) + .Unless(Cue::FirstOnLine) + .Cost(6, "Close_ParenBeforeSpacedOpen"), + // A comparison or logical operator is unlikely inside square brackets or + // call/index argument lists (but common in `if (...)` etc.). + Rule(Top::Square, Kind::ComparisonOp) + .Cost(11, "Close_ParenBeforeComparison"), + Rule(Top::Paren, Kind::ComparisonOp) + .When(Cue::CallParenTop) + .Cost(11, "Close_ParenBeforeComparison"), + // A `,` with whitespace before it: formatted code has no space before a + // comma, so a closer was likely deleted in the gap. + Rule(Top::ParenLike, Kind::Comma) + .When(Cue::LeadingSpace, Cue::PrevValueLike) + .Unless(Cue::FirstOnLine) + .Cost(6, "Close_BeforeSpacedComma"), + // Likewise a `)` or `]` with whitespace before it: formatted code has no + // space before closers either. + Rule(Top::ParenLike, Kinds::CloseGroup) + .When(Cue::LeadingSpace, Cue::PrevValueLike) + .Unless(Cue::FirstOnLine) + .Cost(6, "Close_BeforeSpacedCloser"), + Rule(Top::ParenLike, Kind::FileEnd) + .Cost(CostCloseAtEnd, "Close_ParenAtFileEnd"), + // A wide whitespace gap mid-line suggests a deleted token in the gap. + Rule(Top::ParenLike) + .When(Cue::WideGap) + .Unless(Cue::FirstOnLine) + .Cost(CostCloseAtWideGap, "Close_ParenAtWideGap"), + // A `[` group rarely spans lines except in wrapped declaration headers + // (`impl forall [...]` etc.), where the line break follows the `]`. + Rule(Top::Square) + .When(Cue::FirstOnLine, Cue::NewLineFromTop) + .Cost(10, "Close_SquareAtContinuation"), + // A block `{` can't be content of a header/grouping `[` (only an index + // `arr[...]` could hold a lambda block, but a `[` after a keyword can't): + // the `]` must close before it. A last-resort bound, below the precise + // cues above, that stops the `[` from swallowing the block. Priced above + // the precise cues (so they win) but below closing at the region end, so + // an unclosed group can't swallow a whole block. + Rule(Top::Square, Kind::OpenCurlyBrace) + .Unless(Cue::StructBrace, Cue::CallParenTop) + .Cost(14, "Close_SquareBeforeBlock"), + // No positive cue that a `(`/`[` closes here. Closing before a bare + // dedent, statement introducer, or arbitrary token was never a correct + // guess in practice (it just closes too early), so decline: the search + // will close at a real cue, at the region end, or, failing both, replace + // the unmatched opener with an error token. + Rule(Top::ParenLike).Decline(), + + // Struct `{`. + Rule(Top::Struct, Kind::Semi).Cost(6, "Close_StructBeforeSemi"), + Rule(Top::Struct).When(Cue::Cascade).Cost(6, "Close_StructCascade"), + // A block `{` can't be content of a struct literal/type `{...}` (a struct + // field is `.name = value`, and a bare `{` isn't a value here): the struct + // must close before it, as in `-> {.x: i32} { body }`. + Rule(Top::Struct, Kind::OpenCurlyBrace) + .Unless(Cue::StructBrace) + .Cost(14, "Close_StructBeforeBlock"), + Rule(Top::Struct) + .When(Cue::WideGap) + .Unless(Cue::FirstOnLine) + .Cost(CostCloseAtWideGap, "Close_StructAtWideGap"), + Rule(Top::Struct, Kind::FileEnd) + .Cost(CostCloseAtEnd, "Close_StructAtFileEnd"), + Rule(Top::Struct) + .When(Cue::FirstOnLine, Cue::DedentToHeader) + .Cost(12, "Close_StructAtDedent"), + Rule(Top::Struct).Cost(40, "Close_StructBaseline"), + + // Scope `{`. + // A first-on-line `else` must have been preceded by the `}` closing the + // branch before it. Ordered ahead of Close_ScopeAtDedent, which would + // otherwise match first and charge more for the same insertion. + Rule(Top::Scope) + .When(Cue::ElseKeyword, Cue::FirstOnLine) + .Cost(4, "Close_ScopeBeforeElse"), + Rule(Top::Scope) + .When(Cue::FirstOnLine, Cue::DedentToHeader) + .Cost(6, "Close_ScopeAtDedent"), + Rule(Top::Scope).When(Cue::Cascade).Cost(6, "Close_ScopeCascade"), + Rule(Top::Scope, Kind::FileEnd) + .Cost(CostCloseAtEnd, "Close_ScopeAtFileEnd"), + Rule(Top::Scope).Cost(45, "Close_ScopeBaseline"), +}; + +// A bucket index: rules are identified by a bit in a `uint64_t`. +static_assert(std::size(CloserRules) <= 64); + +// Maps each (context category, token kind) bucket to the bit-set of rules that +// can apply in it, so a lookup tests only those rules, in table order. +using RuleIndex = + std::array, NumContextCategories>; + +template +static constexpr auto BuildRuleIndex(const BracketRule (&rules)[N]) + -> RuleIndex { + RuleIndex index = {}; + for (size_t r = 0; r != N; ++r) { + for (int32_t t = 0; t != NumContextCategories; ++t) { + if ((rules[r].ctx & (1 << t)) == 0) { + continue; + } + for (int32_t k = 0; k != NumKinds; ++k) { + if ((rules[r].kinds & KindSetOf(static_cast(k))) == + KindSet::None) { + continue; + } + index[t][k] |= uint64_t{1} << r; + } + } + } + return index; +} + +// The bit-set of rules that could apply in a bucket. A rule is identified by +// its bit position, which is its position in the table, so scanning the bits +// from lowest to highest visits the rules in table order. +static constexpr auto CandidateRules(const RuleIndex& index, + int32_t ctx_category, Kind kind) + -> uint64_t { + return index[ctx_category][static_cast(kind)]; +} + +// Returns the first rule that applies, for a first-match table, or null if +// none does. +template +static auto FindMatchingRule(const BracketRule (&rules)[N], + const RuleIndex& index, int32_t ctx_category, + Kind kind, CueSet cues) -> const BracketRule* { + uint64_t candidates = CandidateRules(index, ctx_category, kind); + while (candidates != 0) { + const auto& rule = rules[std::countr_zero(candidates)]; + candidates &= candidates - 1; + if (Matches(rule, cues)) { + return &rule; + } + } + return nullptr; +} + +// Returns the total cost of every rule that applies, for an additive table. +template +static auto SumMatchingRules(const BracketRule (&rules)[N], + const RuleIndex& index, int32_t ctx_category, + Kind kind, CueSet cues) -> int32_t { + int32_t total = 0; + uint64_t candidates = CandidateRules(index, ctx_category, kind); + while (candidates != 0) { + const auto& rule = rules[std::countr_zero(candidates)]; + candidates &= candidates - 1; + if (Matches(rule, cues)) { + total += rule.cost; + } + } + return total; +} + +static constexpr auto CloserRuleIndex = BuildRuleIndex(CloserRules); + +// The cue for the previous token having kind `kind`, if any. +static constexpr auto PrevKindCue(Kind kind) -> CueSet { + switch (kind) { + case Kind::Period: + return CueSetOf(Cue::PrevIsPeriod); + case Kind::OpenParen: + case Kind::OpenSquareBracket: + return CueSetOf(Cue::PrevIsOpenBracket); + case Kind::OpenCurlyBrace: + return CueSetOf(Cue::PrevIsOpenCurly); + case Kind::CloseParen: + return CueSetOf(Cue::PrevIsCloseParen); + case Kind::CloseCurlyBrace: + return CueSetOf(Cue::PrevIsCloseCurly); + case Kind::CloseSquareBracket: + return CueSetOf(Cue::PrevIsCloseSquare); + case Kind::Comma: + return CueSetOf(Cue::PrevIsComma); + default: + return CueSet::None; + } +} + +// Returns the set holding `cue` if `holds`, and the empty set otherwise. +static constexpr auto CueIf(bool holds, Cue cue) -> CueSet { + return holds ? CueSetOf(cue) : CueSet::None; +} + +// Computes every cue that depends only on an item and its neighbours. +// `prev_token` is the token directly before the item and `prev_item` the item +// directly before it, both null at the start of the input. `context_cues` holds +// the cues the caller has already determined from the surrounding token arrays +// (`Cue::CollapsedBlock`, `Cue::FirstOnLine`, and so on). +static auto ComputeItemCues(const MismatchedBracketToken& token, + const MismatchedBracketToken* prev_token, + const Item* prev_item, CueSet context_cues) + -> CueSet { + CueSet cues = context_cues | + CueIf(token.has_leading_space, Cue::LeadingSpace) | + CueIf(token.has_wide_leading_space, Cue::WideGap) | + CueIf(token.is_else_keyword, Cue::ElseKeyword) | + CueIf(token.is_struct_brace, Cue::StructBrace) | + CueIf(PrevIsValueLike(prev_token), Cue::PrevValueLike); + if (prev_token != nullptr) { + // `forall` requires a following `[`; the other paren keywords (`if`, + // `while`, `for`, `match`) are statement introducers and require a `(`. + bool wants_paren = prev_token->kind == Kind::StatementIntroducer; + cues |= CueIf(prev_token->is_paren_keyword && wants_paren, + Cue::PrevKeywordWantsParen) | + CueIf(prev_token->is_paren_keyword && !wants_paren, + Cue::PrevKeywordWantsSquare) | + CueIf(prev_token->has_leading_space, Cue::PrevHasLeadingSpace) | + CueIf(IsValueEndingKind(prev_token->kind), Cue::PrevValueEnding) | + CueIf(prev_token->kind == Kind::As || + prev_token->kind == Kind::StructuralOp, + Cue::PrevIntroducesType) | + PrevKindCue(prev_token->kind); + } + if (prev_item != nullptr) { + // A name directly following `as` or `->` is a type, not something + // callable, so empty parens after it are implausible. + cues |= CueIf(prev_item->token.kind == Kind::Leaf, Cue::PrevItemIsLeaf) | + CueIf(prev_item->HasAll(Cue::PrevIntroducesType), + Cue::PrevItemIsTypeName); + } + return cues; +} + +// Computes the cues that depend on the innermost open bracket `top` and the +// search state, to combine with `item.cues`. +static auto ComputeTopCues(const OpenBracketInfo& top, const Item& item, + llvm::ArrayRef stack) -> CueSet { + const auto& token = item.token; + return CueIf(top.is_call_paren, Cue::CallParenTop) | + CueIf(MatchesDeeperOpener(stack, token.kind), Cue::Cascade) | + CueIf(top.token_pos == item.token_start_index - 1, Cue::AfterOpenTop) | + CueIf(token.line_indent <= top.effective_header_indent, + Cue::DedentToHeader) | + CueIf(token.line != top.line, Cue::NewLineFromTop); +} + +// Computes the cost of inserting a synthetic closer for `top` directly before +// `item`, or nullopt if this insertion isn't worth exploring. Sets `rule_name` +// to the name of the rule that fired. +static auto ClassifyCloserInsertion(const OpenBracketInfo& top, + const Item& item, + llvm::ArrayRef stack, + llvm::StringLiteral& rule_name) + -> std::optional { + CueSet cues = item.cues | ComputeTopCues(top, item, stack); + const auto* rule = FindMatchingRule( + CloserRules, CloserRuleIndex, TopCategoryOf(top), item.token.kind, cues); + if (rule == nullptr || rule->cost == DeclineCost) { + return std::nullopt; + } + rule_name = rule->rule_name; + return rule->cost; +} + +// Where to insert a synthetic opening bracket, and what that costs. A `(` or +// `[` can always be synthesized, just expensively without a cue, so those end +// in a baseline; a brace is only proposed where a cue supports it. +constexpr BracketRule OpenerRules[] = { + // `if`/`while`/`for`/`match` (statement introducers) require a following + // `(`; `forall` (an Other token) requires a following `[`. + Rule(Ins::Paren, Kinds::NonOpener) + .When(Cue::PrevKeywordWantsParen) + .Cost(3, "Open_AfterParenKeyword"), + Rule(Ins::Square, Kinds::NonOpener) + .When(Cue::PrevKeywordWantsSquare) + .Cost(3, "Open_AfterParenKeyword"), + // A leaf or binding modifier directly following a value-ending token is + // illegal; an opener here fixes the adjacency. + Rule(Ins::ParenLike, Kinds::LeafLike) + .When(Cue::PrevValueEnding) + .Cost(3, "Open_AtLeafAdjacency"), + // A `.` with whitespace before it directly following a value-ending + // token: likely a designator argument that lost its `(`, as in + // `ImplicitAs(.Self)`. + Rule(Ins::Paren, Kind::Period) + .When(Cue::LeadingSpace, Cue::PrevValueEnding) + .Unless(Cue::FirstOnLine) + .Cost(CostCloseParenBeforeSpacedPeriod, "Open_BeforeSpacedPeriod"), + // A mid-line leaf with whitespace before it directly following an + // unspaced `.`: member access is written without spaces, `x.y`, so a + // bracket was likely deleted in the gap. + Rule(Ins::ParenLike, Kind::Leaf) + .When(Cue::LeadingSpace, Cue::PrevIsPeriod) + .Unless(Cue::FirstOnLine, Cue::PrevHasLeadingSpace) + .Cost(4, "Open_AfterPeriodGap"), + // A mid-line leaf with whitespace before it directly following an + // opener: formatted code has no space after `(` or `[`, so a bracket was + // likely deleted in the gap. + Rule(Ins::ParenLike, Kind::Leaf) + .When(Cue::LeadingSpace, Cue::PrevIsOpenBracket) + .Unless(Cue::FirstOnLine) + .Cost(4, "Open_AfterOpenGap"), + // A wide whitespace gap before a token that could start a group suggests + // an opener was deleted in the gap. + Rule(Ins::ParenLike, Kinds::GroupStarter) + .When(Cue::WideGap) + .Unless(Cue::FirstOnLine) + .Cost(CostCloseAtWideGap, "Open_AtWideGap"), + // An empty group directly after a name: `Op()`. Only applies after a + // leaf (a call of a just-computed value, `f(x)()`, is much rarer than a + // call of a name), and not when the name is a type after `as` or `->`, + // where a parenthesized group is more plausible than an empty call. + Rule(Ins::Paren, Kind::CloseParen) + .When(Cue::PrevItemIsLeaf, Cue::LeadingSpace) + .Unless(Cue::PrevItemIsTypeName) + .Cost(5, "Open_EmptyParens"), + // An empty `[]` is rarer than empty parens. + Rule(Ins::Square, Kind::CloseSquareBracket) + .When(Cue::PrevItemIsLeaf, Cue::LeadingSpace) + .Unless(Cue::PrevItemIsTypeName) + .Cost(12, "Open_EmptySquares"), + // An empty call of a just-computed value: `T.(Default.Op)()`. Only + // trusted when the `)` is spaced, marking the deletion gap. This looks at + // the last token of the previous item, so it fires after a collapsed + // `(...)` block too. + Rule(Ins::Paren, Kind::CloseParen) + .When(Cue::PrevIsCloseParen, Cue::LeadingSpace) + .Unless(Cue::FirstOnLine) + .Cost(8, "Open_EmptyParensAfterClose"), + // A `(` or `[` anywhere else an expression could start. + Rule(Ins::ParenLike).Cost(70, "Open_ParenBaseline"), + // A scope `{` is never inserted directly before another opener: the body + // it would open starts with that opener, so the `{` belongs before it only + // via one of the rules above. + Rule(Ins::ScopeBrace, Kinds::Opener).Decline(), + // A scope `{` between an unbraced declaration or statement header and its + // body, as in `if (c) return;`. + Rule(Ins::ScopeBrace) + .When(Cue::FollowsStatementHeader) + .Unless(Cue::HeaderHasOpenCurly) + .Cost(8, "Open_ScopeAfterHeader"), + Rule(Ins::ScopeBrace).Cost(30, "Open_ScopeBaseline"), + // A struct `{` before a `.` designator that isn't a member access. + Rule(Ins::StructBrace, Kind::Period) + .Unless(Cue::PrevValueEnding) + .Cost(5, "Open_StructBeforeDesignator"), + // A struct literal `{...}` that lost its `{`, leaving content directly + // before the `}`. Real content is required, so a stray `}` is reported as + // an error instead. Priced above Open_ScopeAfterHeader: when a single-line + // body lost its `{`, inserting it before the body beats making empty + // braces at the `}`. + Rule(Ins::StructBrace, Kind::CloseCurlyBrace) + .Unless(Cue::FirstOnLine) + .AnyOf(Cue::PrevValueEnding, Cue::PrevIsCloseCurly, + Cue::PrevIsCloseSquare, Cue::PrevIsComma) + .Cost(10, "Open_StructEmptyBraces"), + // Any other struct brace has no cue at all, and isn't worth proposing. + Rule(Ins::StructBrace).Decline(), +}; + +static_assert(std::size(OpenerRules) <= 64); + +static constexpr auto OpenerRuleIndex = BuildRuleIndex(OpenerRules); + +// Computes the cost of inserting a synthetic opener directly before `item`, or +// nullopt if this insertion isn't worth exploring. Sets `rule_name` to the name +// of the rule that fired. +static auto ClassifyOpenerInsertion(Kind kind, bool is_struct_brace, + const Item& item, + llvm::StringLiteral& rule_name) + -> std::optional { + const auto* rule = FindMatchingRule(OpenerRules, OpenerRuleIndex, + InsCategoryOf(kind, is_struct_brace), + item.token.kind, item.cues); + if (rule == nullptr || rule->cost == DeclineCost) { + return std::nullopt; + } + rule_name = rule->rule_name; + return rule->cost; +} + +// Penalties for advancing over an item in a context where it doesn't belong. +// These make "close the group before this token" win over "swallow this token +// into the group". Unlike the tables above, this one is additive: every +// matching rule contributes, and the penalties sum. +constexpr BracketRule AdvanceRules[] = { + // A line that dedents to at-or-before the indentation of the enclosing + // brace's statement header, or of the enclosing paren's own line, while + // the bracket is still open. Closing brackets are excluded, since a `}` + // that closes its group is expected to dedent, and `FileEnd` isn't + // content at all. + Rule(Top::Struct | Top::Scope, Kinds::Dedentable) + .When(Cue::FirstOnLine, Cue::DedentToHeader) + .Cost(40, "Adv_DedentInScope"), + Rule(Top::ParenLike, Kinds::Dedentable) + .When(Cue::FirstOnLine, Cue::DedentToOpenerLine) + .Cost(25, "Adv_DedentInParen"), + // A scope `{` inside parens or a struct brace is a lambda, which is rare; + // a struct `{` there is a struct literal argument, which is common. The + // first rule covers a collapsed block that contains a scope brace. + Rule(Top::ParenLike | Top::Struct, Kinds::Opener) + .When(Cue::CollapsedBlock, Cue::ContainsScopeBrace) + .Cost(40, "Adv_ScopeBlockInParen"), + Rule(Top::ParenLike | Top::Struct, Kind::OpenCurlyBrace) + .Unless(Cue::CollapsedBlock, Cue::StructBrace) + .Cost(40, "Adv_ScopeBraceInParen"), + Rule(Top::ParenLike | Top::Struct, Kind::OpenCurlyBrace) + .When(Cue::StructBrace) + .Unless(Cue::CollapsedBlock) + .Cost(5, "Adv_StructBraceInParen"), + // A spaced `(` or `[` directly after a value: calls and indexing are + // written without spaces, so a bracket was likely deleted in the gap — + // unless this path already inserted a closer there. + Rule(Top::Any, Kinds::OpenGroup) + .When(Cue::LeadingSpace, Cue::PrevValueLike) + .Unless(Cue::FirstOnLine, Cue::CollapsedBlock, Cue::CloserInserted) + .Cost(10, "Adv_SpacedOpenAfterValue"), + // Formatted code has no space before a closer either. + Rule(Top::Any, Kinds::CloseGroup) + .When(Cue::LeadingSpace, Cue::PrevValueLike) + .Unless(Cue::FirstOnLine, Cue::BracketInsertedHere) + .Cost(8, "Adv_SpacedCloserUnexplained"), + // A `;` can't appear inside parens, square brackets, or a struct brace. + Rule(Top::ParenLike | Top::Struct, Kind::Semi).Cost(100, "Adv_SemiInParen"), + // A `,` at statement level: directly in a scope brace or at the top level. + Rule(Top::None | Top::Scope, Kind::Comma) + .Cost(50, "Adv_CommaAtStatementLevel"), + // A `,` directly following a still-open `(`/`[` is illegal. + Rule(Top::ParenLike | Top::Struct, Kind::Comma) + .When(Cue::AfterOpenTop) + .Cost(50, "Adv_CommaAfterOpen"), + // Formatted code has no space before a `,`, so a bracket was likely + // deleted in the gap. + Rule(Top::ParenLike | Top::Struct, Kind::Comma) + .When(Cue::LeadingSpace, Cue::PrevValueLike) + .Unless(Cue::FirstOnLine, Cue::CloserInserted, Cue::AfterOpenTop) + .Cost(8, "Adv_SpacedCommaInParen"), + // A statement introducer keyword inside parens or a struct brace. + Rule(Top::ParenLike | Top::Struct, Kind::StatementIntroducer) + .Cost(60, "Adv_IntroducerInParen"), + // A leaf, or a binding modifier keyword, directly following a value-ending + // token is an illegal adjacency — unless an opener synthesized here, or a + // `]`/`}` inserted here, repairs it. + Rule(Top::Any, Kinds::LeafLike) + .When(Cue::PrevValueEnding) + .Unless(Cue::OpenerHere, Cue::CloserFixesAdjacency) + .Cost(60, "Adv_LeafAdjacency"), + // `=`, `->`, or `as` inside parens or square brackets. These *can* occur + // there (default arguments, function types, casts), so this is mild; it + // serves to prefer the earliest sensible close point. Casts `(x as T)` are + // common enough that `as` keeps only a nominal preference. + Rule(Top::ParenLike, Kind::Assignment, Kind::StructuralOp) + .When(Cue::LeadingSpace) + .Cost(5, "Adv_StructuralOpInParen"), + Rule(Top::ParenLike, Kind::As) + .When(Cue::LeadingSpace) + .Cost(1, "Adv_AsOpInParen"), + // A comparison or logical operator inside square brackets or a call. + Rule(Top::Square, Kind::ComparisonOp).Cost(8, "Adv_ComparisonInSquare"), + Rule(Top::Paren, Kind::ComparisonOp) + .When(Cue::CallParenTop) + .Cost(8, "Adv_ComparisonInCall"), + // A wide mid-line whitespace gap suggests a deleted bracket that this path + // hasn't repaired. + Rule(Top::Any, Kinds::AnyOther) + .When(Cue::WideGap) + .Unless(Cue::FirstOnLine, Cue::BracketInsertedHere) + .Cost(10, "Adv_WideGapUnexplained"), + // A mid-line `.` with whitespace before it suggests a deleted bracket: + // member access is written without spaces. Prefer closing an open group + // before it, or opening one. + Rule(Top::Any, Kind::Period) + .When(Cue::LeadingSpace) + .Unless(Cue::FirstOnLine, Cue::BracketInsertedHere) + .AnyOf(Cue::PrevValueLike, Cue::PrevIsOpenBracket, Cue::PrevIsOpenCurly) + .Cost(10, "Adv_SpacedPeriodInParen"), +}; + +static_assert(std::size(AdvanceRules) <= 64); + +// Declining is meaningless in an additive table: every matching rule +// contributes its cost, so a declining rule would subtract from the penalty. +// This also catches a rule that forgot its `Cost`. +static_assert([] { + for (const BracketRule& rule : AdvanceRules) { + if (rule.cost == DeclineCost) { + return false; + } + } + return true; +}()); + +static constexpr auto AdvanceRuleIndex = BuildRuleIndex(AdvanceRules); + +// Computes the cues for advancing over `item` in search state `node`. +static auto ComputeAdvanceCues(const SearchState& node, const Item& item) + -> CueSet { + bool opener_here = OpenerSynthesizedHere(node.stack, item); + bool closer_here = node.closer_inserted != Kind::Other; + CueSet cues = item.cues | CueIf(closer_here, Cue::CloserInserted) | + CueIf(CloserFixesLeafAdjacency(node.closer_inserted), + Cue::CloserFixesAdjacency) | + CueIf(opener_here, Cue::OpenerHere) | + CueIf(closer_here || opener_here, Cue::BracketInsertedHere); + if (node.stack.empty()) { + return cues; + } + const auto& top = node.stack.back(); + return cues | CueIf(top.is_call_paren, Cue::CallParenTop) | + CueIf(top.token_pos == item.token_start_index - 1, Cue::AfterOpenTop) | + CueIf(item.token.line_indent <= top.effective_header_indent, + Cue::DedentToHeader) | + CueIf(item.token.line_indent <= top.line_indent, + Cue::DedentToOpenerLine); +} + +// Computes the total penalty for advancing over `item` in search state `node`, +// summing every `AdvanceRules` entry that matches. +static auto AdvancePenalty(const SearchState& node, const Item& item) + -> int32_t { + return SumMatchingRules(AdvanceRules, AdvanceRuleIndex, + TopCategoryOfStack(node.stack), item.token.kind, + ComputeAdvanceCues(node, item)); +} + +// Enumerates the distinct optimal repairs, by walking parent edges from each +// goal node back to the root and collecting the corrections along the way. At +// most `MaxOptimalPaths` are returned, so a correction that only a path beyond +// that cap would dispute stays untied. +static auto EnumerateOptimalPaths(llvm::ArrayRef nodes, + llvm::ArrayRef goal_node_indices) + -> llvm::SmallVector> { + llvm::SmallVector> all_paths; + llvm::SmallVector current_path; + struct StackFrame { + int32_t node_index; + int32_t edge_index = 0; + }; + llvm::SmallVector stack; + for (int32_t goal_idx : goal_node_indices) { + stack.push_back({.node_index = goal_idx, .edge_index = 0}); + } + while (!stack.empty()) { + auto& frame = stack.back(); + const auto& node = nodes[frame.node_index]; + if (node.parent_edges.empty()) { + // The root, so `current_path` is now a complete repair, in reverse. + all_paths.push_back(current_path); + std::reverse(all_paths.back().begin(), all_paths.back().end()); + stack.pop_back(); + if (all_paths.size() >= MaxOptimalPaths) { + break; + } + continue; + } + if (frame.edge_index > 0) { + const auto& prev_edge = node.parent_edges[frame.edge_index - 1]; + if (prev_edge.has_correction) { + current_path.pop_back(); + } + } + if (static_cast(frame.edge_index) < node.parent_edges.size()) { + const auto& edge = node.parent_edges[frame.edge_index]; + ++frame.edge_index; + if (edge.has_correction) { + current_path.push_back(edge.correction); + } + stack.push_back({.node_index = edge.parent_node_index, .edge_index = 0}); + } else { + stack.pop_back(); + } + } + return all_paths; +} + +// Finds which of the first optimal path's corrections the optimal repairs +// disagree about. Each of its corrections is matched against an `equivalent` +// one in every other path; one with no counterpart in some path is tied. +static auto FindTiedCorrections( + llvm::ArrayRef> all_paths, + llvm::function_ref + equivalent) -> llvm::BitVector { + llvm::ArrayRef baseline_path = all_paths.front(); + llvm::BitVector tied(baseline_path.size()); + llvm::BitVector used; + for (const auto& path : all_paths) { + used.reset(); + used.resize(path.size()); + for (auto [corr_idx, corr] : llvm::enumerate(baseline_path)) { + bool found = false; + for (auto [path_idx, path_corr] : llvm::enumerate(path)) { + if (!used[path_idx] && equivalent(path_corr, corr)) { + used.set(path_idx); + found = true; + break; + } + } + if (!found) { + tied.set(corr_idx); + } + } + } + return tied; +} + +// From the optimal goal nodes, reconstructs the repair corrections and appends +// them to `corrections`. Every optimal repair is enumerated (up to a cap); a +// correction that the optimal repairs disagree about is marked tied, so the +// caller downgrades it to an error token rather than guessing. Falls back to +// naive recovery if no path can be reconstructed. +static auto ReconstructCorrections( + llvm::ArrayRef nodes, llvm::ArrayRef goal_node_indices, + llvm::ArrayRef items, TokenIndex region_end_token, + llvm::SmallVectorImpl& corrections) -> void { + auto all_paths = EnumerateOptimalPaths(nodes, goal_node_indices); + if (all_paths.empty()) { + SolveNaive(items, corrections); + return; + } + + // Two insertions of the same bracket kind are equivalent if every token + // between their insertion points is that same kind: inserting a `)` on + // either side of an existing `)` produces the same token sequence. + Map token_to_item; + for (auto [idx, region_item] : llvm::enumerate(items)) { + token_to_item.Update(region_item.token.token_index.index, + static_cast(idx)); + } + token_to_item.Update(region_end_token.index, + static_cast(items.size())); + // Compares only the fixes, not the diagnosed brackets: two paths that + // blame different brackets but repair the token stream identically don't + // disagree about the repair. + auto corrections_equivalent = [&](const BracketCorrection& a, + const BracketCorrection& b) -> bool { + if (a.fix_action != b.fix_action || a.fix_token_kind != b.fix_token_kind) { + return false; + } + if (a.fix_token_index == b.fix_token_index) { + return true; + } + if (a.fix_action != BracketFixAction::InsertBefore) { + return false; + } + int32_t* a_item = token_to_item[a.fix_token_index.index]; + int32_t* b_item = token_to_item[b.fix_token_index.index]; + if (a_item == nullptr || b_item == nullptr) { + return false; + } + auto [lo, hi] = std::minmax(*a_item, *b_item); + for (const Item& between : items.slice(lo, hi - lo)) { + if (between.HasAll(Cue::CollapsedBlock) || + ToTokenKind(between.token.kind) != a.fix_token_kind) { + return false; + } + } + return true; + }; + + llvm::BitVector tied = FindTiedCorrections(all_paths, corrections_equivalent); + for (auto [corr_idx, corr] : llvm::enumerate(all_paths.front())) { + corrections.push_back(corr); + corrections.back().is_tied = tied[corr_idx]; + } +} + +// Whether matching the real closer `item` directly against `top` is strictly +// better than synthesizing a duplicate closer in front of it. That's normally +// the case, but not when the closer isn't allowed to match `top` directly, and +// not when it has suspicious whitespace before it suggesting a closer was +// deleted in the gap. +static auto DirectMatchPreferred(const OpenBracketInfo& top, const Item& item) + -> bool { + auto kind = item.token.kind; + bool direct_match_ok = kind != Kind::CloseCurlyBrace || top.is_struct_brace || + item.token.line == top.line || + item.token.line_indent >= top.effective_header_indent; + bool spaced_suspicious = kind != Kind::CloseCurlyBrace && + !item.HasAll(Cue::FirstOnLine) && + item.HasAll(Cue::LeadingSpace, Cue::PrevValueLike); + return direct_match_ok && !spaced_suspicious; +} + +namespace { + +// A flavor of synthetic opening bracket the search can insert. +struct SyntheticOpener { + Kind kind; + bool is_struct_brace; +}; + +} // namespace + +// The synthetic openers the search considers inserting before a token, in the +// order they're proposed. Each is offered to `ClassifyOpenerInsertion`, which +// decides whether the context supports it. +constexpr SyntheticOpener SyntheticOpeners[] = { + {Kind::OpenParen, /*is_struct_brace=*/false}, + {Kind::OpenSquareBracket, /*is_struct_brace=*/false}, + {Kind::OpenCurlyBrace, /*is_struct_brace=*/false}, + {Kind::OpenCurlyBrace, /*is_struct_brace=*/true}, +}; + +// The stack entry for the real opening bracket `item`. +static auto RealOpener(const Item& item) -> OpenBracketInfo { + return OpenBracketInfo{OpenBracketKey{ + .token_index = item.token.token_index, + .token_pos = item.token_start_index, + .line = item.token.line, + .line_indent = item.token.line_indent, + .effective_header_indent = item.effective_header_indent, + .kind = item.token.kind, + .is_struct_brace = item.token.is_struct_brace, + .is_call_paren = item.HasAll(Cue::PrevValueEnding), + }}; +} + +// The extra cost of matching the real closer `item` against `top`, or nullopt +// if the match isn't allowed. A multi-line scope close must not be dedented +// past its header, and pays for indentation disagreement with it. +static auto MatchClosePenalty(const OpenBracketInfo& top, const Item& item) + -> std::optional { + if (item.token.kind != Kind::CloseCurlyBrace || top.is_struct_brace || + item.token.line == top.line) { + return 0; + } + if (item.token.line_indent < top.effective_header_indent) { + return std::nullopt; + } + if (item.HasAll(Cue::FirstOnLine) && + item.token.line_indent != top.effective_header_indent) { + return CostBraceIndentMismatchBase + + CostBraceIndentMismatchPerColumn * + std::abs(top.effective_header_indent - item.token.line_indent); + } + return 0; +} + +// The correction that records where the synthetic opener `opener`, which the +// real closer `closer` has just matched, would be inserted. +static auto InsertOpenerCorrection(const OpenBracketInfo& opener, + const MismatchedBracketToken& closer) + -> BracketCorrection { + return BracketCorrection{ + .diagnostic_kind = BracketDiagnosticKind::UnmatchedClosing, + .diagnostic_token_index = closer.token_index, + .fix_action = BracketFixAction::InsertBefore, + .fix_token_index = opener.insertion_token_index, + .fix_token_kind = ToTokenKind(opener.kind), + .rule_name = opener.rule_name, + }; +} + +// The cost of closing an open bracket at the end of the file or region. +static auto CostToCloseAtEnd(const OpenBracketInfo& open) -> int32_t { + if (open.kind != Kind::OpenCurlyBrace) { + return CostCloseParenAtEnd; + } + return open.is_struct_brace ? CostCloseStructAtEnd : CostCloseAtEnd; +} + +// A layered beam search over one damaged region. +// +// Each layer holds the states that survive just before one item. Expanding a +// layer first proposes bracket insertions before the item — epsilon moves, +// which stay within the layer — and then advances over the item into the next +// layer, pruning each layer back to `MaxBeamWidth`. States in a layer that +// agree on their open-bracket stack merge into one node, which keeps every +// cheapest way of reaching it so that ties can be found afterwards. +namespace { + +class RegionSearch { + public: + // `region_end_token` is the token directly after the region, where any + // still-unclosed brackets are closed. + RegionSearch(llvm::ArrayRef items, TokenIndex region_end_token) + : items_(items), region_end_token_(region_end_token) { + nodes_.reserve(256); + } + + // Runs the search and appends the corrections for the cheapest repair to + // `corrections`. + auto Solve(llvm::SmallVectorImpl& corrections) -> void; + + private: + // Adds the state reached by `edge` at total cost `next_cost` to `layer`, + // merging it into an equal state already there if there is one. + // `next_item_idx` is the item the state sits before. If `worklist` is given, + // any node that was added or became cheaper is appended to it, so that a + // further epsilon move can be applied to it. + auto AddToLayer(llvm::SmallVectorImpl& layer, int32_t next_item_idx, + llvm::SmallVector next_stack, + Kind closer_inserted, int32_t next_cost, ParentEdge edge, + llvm::SmallVectorImpl* worklist = nullptr) -> void; + + // Merges a newly-found way of reaching the state already in `nodes_[idx]`: a + // cheaper cost replaces what's recorded there, and an equal cost adds another + // parent edge, so that every cheapest path stays available. + auto MergeIntoNode(int32_t idx, int32_t cost, const ParentEdge& edge, + llvm::SmallVectorImpl* worklist) -> void; + + // Keeps a layer within the beam width by discarding the costliest states. + auto PruneBeam(llvm::SmallVectorImpl& layer) -> void; + + // Adds to the current layer every state reachable by inserting brackets + // directly before item `item_idx`. + auto InsertBracketsBefore(int32_t item_idx) -> void; + auto InsertSyntheticClosers(int32_t item_idx) -> void; + auto InsertSyntheticOpeners(int32_t item_idx) -> void; + + // Advances the current layer over item `item_idx`, returning the layer that + // results. + auto AdvanceOverItem(int32_t item_idx) -> llvm::SmallVector; + auto AdvanceNode(int32_t node_idx, int32_t item_idx, + llvm::SmallVectorImpl& next_layer) -> void; + + // Closes whatever is still open at the end of the region, returning the goal + // nodes of the cheapest complete repairs. + auto CloseAtRegionEnd() -> llvm::SmallVector; + + llvm::ArrayRef items_; + TokenIndex region_end_token_; + + // Every state the search has reached, in creation order; a layer names its + // states by their index here. Nodes are never removed, so pruning a layer + // leaves its states behind, unreachable. + llvm::SmallVector nodes_; + // The cost of the cheapest complete repair found so far, which bounds the + // cost of any state still worth expanding. + int32_t min_goal_cost_ = std::numeric_limits::max(); + llvm::SmallVector current_layer_; + // The states in the layer currently being built, named by their index in + // `nodes_` and keyed on the state itself, so that reaching a state already in + // the layer merges into it. Kept across layers only to reuse its allocation. + Set layer_dedup_; +}; + +} // namespace + +auto RegionSearch::AddToLayer(llvm::SmallVectorImpl& layer, + int32_t next_item_idx, + llvm::SmallVector next_stack, + Kind closer_inserted, int32_t next_cost, + ParentEdge edge, + llvm::SmallVectorImpl* worklist) + -> void { + if (next_cost > min_goal_cost_) { + return; + } + LayerDedupKeyContext key_context(&nodes_); + StateKey key = {.stack = next_stack, .closer_inserted = closer_inserted}; + if (auto existing = layer_dedup_.Lookup(key, key_context)) { + MergeIntoNode(existing.key(), next_cost, edge, worklist); + return; + } + auto new_idx = static_cast(nodes_.size()); + nodes_.push_back(BeamNode{ + .item_index = next_item_idx, + .stack = std::move(next_stack), + .cost = next_cost, + .closer_inserted = closer_inserted, + .parent_edges = {edge}, + }); + layer.push_back(new_idx); + layer_dedup_.Insert(new_idx, key_context); + if (worklist) { + worklist->push_back(new_idx); + } +} + +auto RegionSearch::MergeIntoNode(int32_t idx, int32_t cost, + const ParentEdge& edge, + llvm::SmallVectorImpl* worklist) + -> void { + BeamNode& node = nodes_[idx]; + if (cost < node.cost) { + node.cost = cost; + node.parent_edges.clear(); + node.parent_edges.push_back(edge); + if (worklist) { + worklist->push_back(idx); + } + } else if (cost == node.cost && + llvm::none_of(node.parent_edges, [&](const ParentEdge& e) { + return EdgesEqual(e, edge); + })) { + node.parent_edges.push_back(edge); + } +} + +auto RegionSearch::PruneBeam(llvm::SmallVectorImpl& layer) -> void { + if (layer.size() > MaxBeamWidth) { + llvm::stable_sort(layer, [&](int32_t a, int32_t b) { + return nodes_[a].cost < nodes_[b].cost; + }); + layer.resize(MaxBeamWidth); + } +} + +auto RegionSearch::InsertBracketsBefore(int32_t item_idx) -> void { + // Seed the dedup table with the states already in the layer, so that an + // insertion reaching one of them merges into it instead of duplicating it. + for (int32_t idx : current_layer_) { + layer_dedup_.Insert(idx, LayerDedupKeyContext(&nodes_)); + } + InsertSyntheticClosers(item_idx); + InsertSyntheticOpeners(item_idx); + layer_dedup_.Clear(); + PruneBeam(current_layer_); +} + +auto RegionSearch::InsertSyntheticClosers(int32_t item_idx) -> void { + const Item& item = items_[item_idx]; + // A worklist rather than one pass over the layer, so that several groups can + // be closed at the same point. + llvm::SmallVector worklist = current_layer_; + for (size_t head = 0; head < worklist.size(); ++head) { + const SearchState current = Snapshot(nodes_[worklist[head]]); + if (current.cost > min_goal_cost_ || current.stack.empty()) { + continue; + } + const auto& top = current.stack.back(); + // Synthetic openers exist only to consume real closers; closing one + // synthetically would insert a pointless empty pair. + if (top.is_synthetic) { + continue; + } + if (item.token.kind == MatchingClosingKind(top.kind) && + DirectMatchPreferred(top, item)) { + continue; + } + llvm::StringLiteral rule_name = ""; + auto cost = ClassifyCloserInsertion(top, item, current.stack, rule_name); + if (!cost) { + continue; + } + + auto next_stack = current.stack; + auto popped = next_stack.pop_back_val(); + auto closer_kind = MatchingClosingKind(popped.kind); + AddToLayer( + current_layer_, item_idx, std::move(next_stack), closer_kind, + current.cost + *cost, + ParentEdge{ + .parent_node_index = worklist[head], + .correction = + BracketCorrection{ + .diagnostic_kind = BracketDiagnosticKind::UnmatchedOpening, + .diagnostic_token_index = popped.token_index, + .fix_action = BracketFixAction::InsertBefore, + .fix_token_index = item.token.token_index, + .fix_token_kind = ToTokenKind(closer_kind), + .rule_name = rule_name, + }, + .has_correction = true, + }, + &worklist); + } +} + +auto RegionSearch::InsertSyntheticOpeners(int32_t item_idx) -> void { + const Item& item = items_[item_idx]; + // Iterate only over the states present after the closer phase, so that + // synthetic openers don't chain onto each other. The bound is taken before + // the loop because the loop appends to the layer. + for (size_t idx : llvm::seq(0, current_layer_.size())) { + int32_t node_idx = current_layer_[idx]; + const SearchState current = Snapshot(nodes_[node_idx]); + if (current.cost > min_goal_cost_ || + current.stack.size() >= MaxSearchStackDepth) { + continue; + } + for (auto [open_kind, is_struct_brace] : SyntheticOpeners) { + llvm::StringLiteral rule_name = ""; + auto cost = + ClassifyOpenerInsertion(open_kind, is_struct_brace, item, rule_name); + if (!cost) { + continue; + } + auto next_stack = current.stack; + next_stack.push_back(OpenBracketInfo{ + OpenBracketKey{ + .insertion_token_index = item.token.token_index, + .line = item.token.line, + .line_indent = item.token.line_indent, + .effective_header_indent = item.effective_header_indent, + .kind = open_kind, + .is_synthetic = true, + .is_struct_brace = is_struct_brace, + }, + rule_name, + }); + AddToLayer(current_layer_, item_idx, std::move(next_stack), + current.closer_inserted, current.cost + *cost, + ParentEdge{.parent_node_index = node_idx}); + } + } +} + +auto RegionSearch::AdvanceOverItem(int32_t item_idx) + -> llvm::SmallVector { + llvm::SmallVector next_layer; + for (int32_t node_idx : current_layer_) { + AdvanceNode(node_idx, item_idx, next_layer); + } + layer_dedup_.Clear(); + PruneBeam(next_layer); + return next_layer; +} + +auto RegionSearch::AdvanceNode(int32_t node_idx, int32_t item_idx, + llvm::SmallVectorImpl& next_layer) + -> void { + const SearchState current = Snapshot(nodes_[node_idx]); + if (current.cost > min_goal_cost_) { + return; + } + const Item& item = items_[item_idx]; + auto kind = item.token.kind; + + auto advance = [&](llvm::SmallVector next_stack, + int32_t add_cost, BracketCorrection correction = {}, + bool has_correction = false) { + AddToLayer(next_layer, item_idx + 1, std::move(next_stack), Kind::Other, + current.cost + add_cost, + ParentEdge{.parent_node_index = node_idx, + .correction = correction, + .has_correction = has_correction}); + }; + + // What it costs to advance over this item in this state, per the + // `AdvanceRules` table. + int32_t penalty = AdvancePenalty(current, item); + + if (item.HasAll(Cue::CollapsedBlock)) { + advance(current.stack, penalty); + return; + } + + if (IsOpeningBracket(kind)) { + // Advance, pushing the opener onto the stack. + if (current.stack.size() < MaxSearchStackDepth) { + auto next_stack = current.stack; + next_stack.push_back(RealOpener(item)); + advance(std::move(next_stack), penalty); + } + // Advance without pushing: replace the unmatched opener with an error + // token. + advance( + current.stack, CostReplaceOpening, + ReplaceWithError(item.token, BracketDiagnosticKind::UnmatchedOpening, + "Adv_ReplaceOpener"), + /*has_correction=*/true); + return; + } + + if (IsClosingBracket(kind)) { + // Advance, popping the opener this closer matches. + if (!current.stack.empty() && + current.stack.back().kind == MatchingOpeningKind(kind)) { + if (auto match_cost = MatchClosePenalty(current.stack.back(), item)) { + auto next_stack = current.stack; + auto popped = next_stack.pop_back_val(); + // Matching a synthetic opener finally pins down where it goes, so this + // is where it becomes a correction. + advance(std::move(next_stack), penalty + *match_cost, + popped.is_synthetic ? InsertOpenerCorrection(popped, item.token) + : BracketCorrection{}, + popped.is_synthetic); + } + } + // Advance without matching: replace the unmatched closer with an error + // token. + advance( + current.stack, CostReplaceClosing, + ReplaceWithError(item.token, BracketDiagnosticKind::UnmatchedClosing, + "Adv_ReplaceCloser"), + /*has_correction=*/true); + return; + } + + // Any other token. + advance(current.stack, penalty); +} + +auto RegionSearch::CloseAtRegionEnd() -> llvm::SmallVector { + llvm::SmallVector goal_node_indices; + for (int32_t node_idx : current_layer_) { + const SearchState current = Snapshot(nodes_[node_idx]); + if (current.cost > min_goal_cost_) { + continue; + } + // A synthetic opener that never matched a real closer is a meaningless + // insertion; reject such states rather than dropping it silently. + if (llvm::any_of(current.stack, [](const OpenBracketInfo& open) { + return open.is_synthetic; + })) { + continue; + } + // Close what's left innermost-first, one node per closer, so that each is + // reconstructed as its own correction. + int32_t finish_cost = current.cost; + int32_t parent = node_idx; + for (const auto& open : llvm::reverse(current.stack)) { + finish_cost += CostToCloseAtEnd(open); + nodes_.push_back(BeamNode{ + .item_index = static_cast(items_.size()), + .stack = {}, + .cost = finish_cost, + .parent_edges = {{ + .parent_node_index = parent, + .correction = + BracketCorrection{ + .diagnostic_kind = + BracketDiagnosticKind::UnmatchedOpening, + .diagnostic_token_index = open.token_index, + .fix_action = BracketFixAction::InsertBefore, + .fix_token_index = region_end_token_, + .fix_token_kind = + ToTokenKind(MatchingClosingKind(open.kind)), + .rule_name = "Close_RegionEnd"}, + .has_correction = true, + }}, + }); + parent = static_cast(nodes_.size()) - 1; + } + + if (finish_cost < min_goal_cost_) { + min_goal_cost_ = finish_cost; + goal_node_indices.clear(); + } + if (finish_cost == min_goal_cost_) { + goal_node_indices.push_back(parent); + } + } + return goal_node_indices; +} + +auto RegionSearch::Solve(llvm::SmallVectorImpl& corrections) + -> void { + nodes_.push_back(BeamNode{ + .item_index = 0, + .stack = {}, + .cost = 0, + .parent_edges = {}, + }); + current_layer_ = {0}; + + for (auto [item_index, item] : llvm::enumerate(items_)) { + auto i = static_cast(item_index); + // Nothing is inserted before the end of the file; brackets still open there + // are closed by `CloseAtRegionEnd` instead. + if (item.token.kind != Kind::FileEnd) { + InsertBracketsBefore(i); + } + current_layer_ = AdvanceOverItem(i); + } + + llvm::SmallVector goal_node_indices = CloseAtRegionEnd(); + if (goal_node_indices.empty()) { + SolveNaive(items_, corrections); + return; + } + ReconstructCorrections(nodes_, goal_node_indices, items_, region_end_token_, + corrections); +} + +// Solve a damaged region using layered beam search with tie detection. +// `region_end_token` is the token directly after the region, where any +// still-unclosed brackets are closed. +static auto SolveRegionCostBased( + llvm::ArrayRef items, TokenIndex region_end_token, + llvm::SmallVectorImpl& corrections) -> void { + if (items.size() > static_cast(MaxRegionItemsForSearch)) { + SolveNaive(items, corrections); + return; + } + RegionSearch(items, region_end_token).Solve(corrections); +} + +// The passes below run in order over the whole token sequence, each consuming +// the results of the ones before it: analyze the tokens, decide which matched +// bracket pairs can be trusted and collapsed, build the item sequence from +// that, then split it into regions and search the damaged ones. + +// Finds the stack position of the opener in `open_stack` that the closing +// bracket `tokens[closer_index]` should pair with, or -1 if none is plausible. +// A `}` matches a `{` only when they're on the same line, the `{` is a struct +// brace, or their line indentation agrees. +static auto FindMatchingOpener(llvm::ArrayRef tokens, + llvm::ArrayRef open_stack, + int32_t closer_index) -> int32_t { + const auto& closer = tokens[closer_index]; + auto num_open = static_cast(open_stack.size()); + for (int32_t s : llvm::reverse(llvm::seq(0, num_open))) { + const auto& opener = tokens[open_stack[s]]; + if (closer.kind != Kind::CloseCurlyBrace) { + if (opener.kind == MatchingOpeningKind(closer.kind)) { + return s; + } + } else if (opener.kind == Kind::OpenCurlyBrace && + (opener.line == closer.line || opener.is_struct_brace || + opener.line_indent == closer.line_indent)) { + return s; + } + } + return -1; +} + +// Pairs up brackets by a stack walk, returning for each token the index of the +// bracket it pairs with, or -1 if it has none. A closer that doesn't match the +// top of the stack pops through to a plausible match if one exists (leaving the +// popped brackets unmatched), and is otherwise left unmatched without +// disturbing the stack. +static auto MatchBracketPairs(llvm::ArrayRef tokens) + -> llvm::SmallVector { + auto num_tokens = static_cast(tokens.size()); + llvm::SmallVector match_partner(num_tokens, -1); + llvm::SmallVector open_stack; + for (int32_t i : llvm::seq(0, num_tokens)) { + auto kind = tokens[i].kind; + if (IsOpeningBracket(kind)) { + open_stack.push_back(i); + } else if (IsClosingBracket(kind)) { + int32_t match_s = FindMatchingOpener(tokens, open_stack, i); + if (match_s != -1) { + match_partner[open_stack[match_s]] = i; + match_partner[i] = open_stack[match_s]; + open_stack.resize(match_s); + } + } + } + return match_partner; +} + +namespace { + +// The statement segments of the token sequence, which are separated by `;`, +// `{`, and `}`. +struct Segments { + // The id of each token's segment. Segments are numbered in order. + llvm::SmallVector id; + // The index of the first token of each token's segment. + llvm::SmallVector first; +}; + +// The indexes of the unmatched `(`, `[`, `)`, and `]` tokens, each list in +// increasing order so it can be binary-searched. +struct UnmatchedGroupBrackets { + llvm::SmallVector open_parens; + llvm::SmallVector open_squares; + llvm::SmallVector close_parens; + llvm::SmallVector close_squares; + + // The openers and closers of the bracket kind that `open_kind` opens. + auto OpenersOfKind(Kind open_kind) const -> llvm::ArrayRef { + return open_kind == Kind::OpenParen ? open_parens : open_squares; + } + auto ClosersOfKind(Kind open_kind) const -> llvm::ArrayRef { + return open_kind == Kind::OpenParen ? close_parens : close_squares; + } +}; + +// Everything the later passes need to know about the token sequence as a whole, +// computed once up front by `AnalyzeTokens`. The vectors are all indexed by +// token index. +struct TokenAnalysis { + // The index of the bracket each token pairs with, or -1. See + // `MatchBracketPairs`. + llvm::SmallVector match_partner; + Segments segments; + UnmatchedGroupBrackets unmatched; + // See `ComputeAssociatedLineIndent`. + llvm::SmallVector effective_header_indent; + llvm::BitVector is_first_on_line; + // See `ComputeFollowsStatementHeader` and `ComputeHeaderHasOpenCurlyBrace`. + // The latter is only computed where the former holds. + llvm::BitVector follows_statement_header; + llvm::BitVector header_has_open_curly_brace; +}; + +} // namespace + +static auto ComputeSegments(llvm::ArrayRef tokens) + -> Segments { + auto num_tokens = static_cast(tokens.size()); + Segments segments = {.id = llvm::SmallVector(num_tokens, 0), + .first = llvm::SmallVector(num_tokens, 0)}; + for (int32_t i : llvm::seq(1, num_tokens)) { + auto prev_kind = tokens[i - 1].kind; + bool new_seg = prev_kind == Kind::Semi || + prev_kind == Kind::OpenCurlyBrace || + prev_kind == Kind::CloseCurlyBrace; + segments.id[i] = segments.id[i - 1] + (new_seg ? 1 : 0); + segments.first[i] = new_seg ? i : segments.first[i - 1]; + } + return segments; +} + +static auto FindUnmatchedGroupBrackets( + llvm::ArrayRef tokens, + llvm::ArrayRef match_partner) -> UnmatchedGroupBrackets { + UnmatchedGroupBrackets unmatched; + for (auto [i, token] : llvm::enumerate(tokens)) { + if (match_partner[i] != -1) { + continue; + } + auto index = static_cast(i); + switch (token.kind) { + case Kind::OpenParen: + unmatched.open_parens.push_back(index); + break; + case Kind::OpenSquareBracket: + unmatched.open_squares.push_back(index); + break; + case Kind::CloseParen: + unmatched.close_parens.push_back(index); + break; + case Kind::CloseSquareBracket: + unmatched.close_squares.push_back(index); + break; + default: + break; + } + } + return unmatched; +} + +static auto AnalyzeTokens(llvm::ArrayRef tokens) + -> TokenAnalysis { + auto num_tokens = static_cast(tokens.size()); + auto match_partner = MatchBracketPairs(tokens); + auto segments = ComputeSegments(tokens); + auto unmatched = FindUnmatchedGroupBrackets(tokens, match_partner); + + llvm::SmallVector effective_header_indent(num_tokens, 0); + llvm::BitVector is_first_on_line(num_tokens); + llvm::BitVector follows_statement_header(num_tokens); + llvm::BitVector header_has_open_curly_brace(num_tokens); + for (int32_t i : llvm::seq(0, num_tokens)) { + is_first_on_line[i] = (i == 0 || tokens[i].line != tokens[i - 1].line); + effective_header_indent[i] = + ComputeAssociatedLineIndent(tokens, match_partner, i); + follows_statement_header[i] = + ComputeFollowsStatementHeader(tokens, match_partner, i); + if (follows_statement_header[i]) { + header_has_open_curly_brace[i] = + ComputeHeaderHasOpenCurlyBrace(tokens, match_partner, i); + } + } + + return TokenAnalysis{ + .match_partner = std::move(match_partner), + .segments = std::move(segments), + .unmatched = std::move(unmatched), + .effective_header_indent = std::move(effective_header_indent), + .is_first_on_line = std::move(is_first_on_line), + .follows_statement_header = std::move(follows_statement_header), + .header_has_open_curly_brace = std::move(header_has_open_curly_brace), + }; +} + +// Whether the sorted list `list` contains an element in [lo, hi]. +static auto ContainsInRange(llvm::ArrayRef list, int32_t lo, + int32_t hi) -> bool { + const auto* it = std::lower_bound(list.begin(), list.end(), lo); + return it != list.end() && *it <= hi; +} + +// Whether a matched `{`...`}` pair is trustworthy, judging by how the `}` lines +// up and by the separators directly inside the pair. +static auto BracePairIsClean(llvm::ArrayRef tokens, + const TokenAnalysis& analysis, int32_t open_idx, + int32_t close_idx) -> bool { + const auto& open = tokens[open_idx]; + int32_t header_indent = analysis.effective_header_indent[open_idx]; + if (open.line != tokens[close_idx].line) { + if (!open.is_struct_brace && + (!open.is_at_end_of_line || + header_indent != tokens[close_idx].line_indent)) { + return false; + } + if (open.is_struct_brace && tokens[close_idx].line_indent < header_indent) { + return false; + } + } + // A `;` directly inside a struct brace, or a `,` directly inside a scope + // brace, is illegal; the brace pairing has likely captured too much. + auto bad_kind = open.is_struct_brace ? Kind::Semi : Kind::Comma; + int32_t depth = 0; + for (int32_t j : llvm::seq(open_idx + 1, close_idx)) { + if (IsOpeningBracket(tokens[j].kind)) { + ++depth; + } else if (IsClosingBracket(tokens[j].kind)) { + --depth; + } else if (tokens[j].kind == bad_kind && depth == 0) { + return false; + } + } + return true; +} + +// Whether a matched `(`...`)` or `[`...`]` pair is trustworthy. An unmatched +// opener of the same kind earlier in the same statement segment could really +// own our closer, and an unmatched closer of the same kind later in the same +// segment could really own our opener; both make the pairing suspect. +static auto GroupPairIsClean(llvm::ArrayRef tokens, + const TokenAnalysis& analysis, int32_t open_idx, + int32_t close_idx) -> bool { + auto kind = tokens[open_idx].kind; + if (ContainsInRange(analysis.unmatched.OpenersOfKind(kind), + analysis.segments.first[open_idx], open_idx - 1)) { + return false; + } + // A scan for an unmatched closer would be bounded by the end of the closer's + // segment, so it's enough to check whether the first one after `close_idx` is + // still in that segment. + llvm::ArrayRef closers = analysis.unmatched.ClosersOfKind(kind); + const auto* it = std::upper_bound(closers.begin(), closers.end(), close_idx); + return it == closers.end() || + analysis.segments.id[*it] != analysis.segments.id[close_idx]; +} + +// Whether everything between a matched pair is itself safe to collapse: every +// bracket inside pairs up within the pair and is clean, and for a scope brace, +// no line inside is dedented to or past the header. +static auto PairInteriorIsClean(llvm::ArrayRef tokens, + const TokenAnalysis& analysis, + const llvm::BitVector& is_clean_range, + int32_t open_idx, int32_t close_idx) -> bool { + const auto& open = tokens[open_idx]; + for (int32_t j : llvm::seq(open_idx + 1, close_idx)) { + int32_t partner = analysis.match_partner[j]; + if (partner == -1) { + // An unmatched bracket inside means the pair can't be trusted. + if (IsOpeningBracket(tokens[j].kind) || + IsClosingBracket(tokens[j].kind)) { + return false; + } + } else if (partner < open_idx || partner > close_idx) { + return false; + } + if (IsOpeningBracket(tokens[j].kind) && !is_clean_range[j]) { + return false; + } + if (open.kind == Kind::OpenCurlyBrace && !open.is_struct_brace && + analysis.is_first_on_line[j] && + tokens[j].line != tokens[close_idx].line && + tokens[j].line_indent <= analysis.effective_header_indent[open_idx]) { + return false; + } + } + return true; +} + +// Marks the matched pairs that can be trusted, so that the search can collapse +// each into a single item instead of reconsidering brackets that clearly pair +// up. Pairs are visited in reverse order, so an inner pair is decided before +// the pairs enclosing it. +// +// Note: an illegal leaf adjacency inside a *matched* pair is treated as invalid +// user code, not a bracket error; the pair is still trusted and collapsed. The +// adjacency cue only guides where to insert brackets in regions that are +// already unbalanced. +static auto MarkCleanRanges(llvm::ArrayRef tokens, + const TokenAnalysis& analysis) -> llvm::BitVector { + auto num_tokens = static_cast(tokens.size()); + llvm::BitVector is_clean_range(num_tokens); + for (int32_t i : llvm::reverse(llvm::seq(0, num_tokens))) { + // Only consider a pair from its opener; this also skips the unmatched + // tokens, whose partner is -1. + int32_t close_idx = analysis.match_partner[i]; + if (close_idx <= i) { + continue; + } + bool clean = tokens[i].kind == Kind::OpenCurlyBrace + ? BracePairIsClean(tokens, analysis, i, close_idx) + : GroupPairIsClean(tokens, analysis, i, close_idx); + is_clean_range[i] = + clean && + PairInteriorIsClean(tokens, analysis, is_clean_range, i, close_idx); + } + return is_clean_range; +} + +// Builds the item sequence the search runs over: one item per token, except +// that each clean matched pair collapses into a single item spanning it. +static auto BuildItems(llvm::ArrayRef tokens, + const TokenAnalysis& analysis, + const llvm::BitVector& is_clean_range) + -> llvm::SmallVector { + auto num_tokens = static_cast(tokens.size()); + llvm::SmallVector items; + auto make_item = [&](int32_t start, int32_t end, bool collapsed, + bool has_scope) { + // Cues the surrounding analysis has already determined; the rest follow + // from the tokens themselves. + CueSet context_cues = + CueIf(collapsed, Cue::CollapsedBlock) | + CueIf(has_scope, Cue::ContainsScopeBrace) | + CueIf(analysis.is_first_on_line[start], Cue::FirstOnLine) | + CueIf(analysis.follows_statement_header[start], + Cue::FollowsStatementHeader) | + CueIf(analysis.header_has_open_curly_brace[start], + Cue::HeaderHasOpenCurly); + items.push_back(Item{ + .token_start_index = start, + .token_end_index = end, + .token = tokens[start], + .effective_header_indent = analysis.effective_header_indent[start], + .cues = ComputeItemCues( + tokens[start], start > 0 ? &tokens[start - 1] : nullptr, + items.empty() ? nullptr : &items.back(), context_cues), + }); + }; + for (int32_t i = 0; i < num_tokens;) { + if (!is_clean_range[i]) { + make_item(i, i, /*collapsed=*/false, + tokens[i].kind == Kind::OpenCurlyBrace || + tokens[i].kind == Kind::CloseCurlyBrace); + ++i; + continue; + } + int32_t close_idx = analysis.match_partner[i]; + bool has_scope = llvm::any_of(llvm::seq(i, close_idx + 1), [&](int32_t j) { + return tokens[j].kind == Kind::OpenCurlyBrace && + !tokens[j].is_struct_brace; + }); + make_item(i, close_idx, /*collapsed=*/true, has_scope); + i = close_idx + 1; + } + return items; +} + +// Finds the item indexes where each independently-solved region starts, +// beginning with 0 and ending with the number of items, so that consecutive +// boundaries delimit the regions. A region ends at a top-level declaration +// boundary: a statement introducer at zero indentation whose predecessor ended +// a statement. This bounds how far a single mistake can smear, and gives +// unclosed brackets a natural place to be closed (the region end). +static auto FindRegionBoundaries(llvm::ArrayRef tokens, + llvm::ArrayRef items) + -> llvm::SmallVector { + llvm::SmallVector boundaries = {0}; + for (auto [item_index, item] : llvm::enumerate(items)) { + auto i = static_cast(item_index); + // Note: line_indent is a 1-based column number, so top-level tokens have + // line_indent 1. + if (i == 0 || item.HasAll(Cue::CollapsedBlock) || + item.token.kind != Kind::StatementIntroducer || + item.token.is_else_keyword || item.token.line_indent > 1 || + !item.HasAll(Cue::FirstOnLine)) { + continue; + } + auto prev_end_kind = tokens[items[i - 1].token_end_index].kind; + if ((prev_end_kind == Kind::Semi || + prev_end_kind == Kind::CloseCurlyBrace) && + boundaries.back() != i) { + boundaries.push_back(i); + } + } + if (boundaries.back() != static_cast(items.size())) { + boundaries.push_back(static_cast(items.size())); + } + return boundaries; +} + +// Whether a region's loose (non-collapsed) brackets already form a balanced, +// well-nested sequence. Such a region needs no solving: it has no unmatched +// bracket, so the search would simply match everything and emit no corrections. +// Skipping it avoids running the beam search over the many regions whose +// matched pairs merely weren't collapsed. +static auto RegionIsBalanced(llvm::ArrayRef region) -> bool { + llvm::SmallVector open_kinds; + for (const Item& item : region) { + if (item.HasAll(Cue::CollapsedBlock)) { + continue; + } + auto kind = item.token.kind; + if (IsOpeningBracket(kind)) { + open_kinds.push_back(kind); + } else if (IsClosingBracket(kind)) { + if (open_kinds.empty() || + MatchingClosingKind(open_kinds.back()) != kind) { + return false; + } + open_kinds.pop_back(); + } + } + return open_kinds.empty(); +} + +auto FixMismatchedBrackets(llvm::ArrayRef tokens) + -> llvm::SmallVector { + llvm::SmallVector corrections; + if (tokens.empty()) { + return corrections; + } + + TokenAnalysis analysis = AnalyzeTokens(tokens); + llvm::BitVector is_clean_range = MarkCleanRanges(tokens, analysis); + llvm::SmallVector items = BuildItems(tokens, analysis, is_clean_range); + llvm::SmallVector region_boundaries = + FindRegionBoundaries(tokens, items); + + // Each region runs between consecutive boundaries. + for (auto [start, end] : + llvm::zip(region_boundaries, llvm::drop_begin(region_boundaries))) { + if (start >= end) { + continue; + } + auto region = llvm::ArrayRef(items).slice(start, end - start); + if (RegionIsBalanced(region)) { + continue; + } + // Any bracket still open at the end of the region is closed before the + // token the next region starts at, or before the final `FileEnd`. + TokenIndex region_end_token = end < static_cast(items.size()) + ? items[end].token.token_index + : tokens.back().token_index; + SolveRegionCostBased(region, region_end_token, corrections); + } + + llvm::stable_sort( + corrections, [](const BracketCorrection& a, const BracketCorrection& b) { + if (a.diagnostic_token_index != b.diagnostic_token_index) { + return a.diagnostic_token_index < b.diagnostic_token_index; + } + return a.fix_token_index < b.fix_token_index; + }); + + return corrections; +} + +} // namespace Carbon::Lex diff --git a/toolchain/lex/mismatched_brackets.h b/toolchain/lex/mismatched_brackets.h new file mode 100644 index 000000000000..d743d6252646 --- /dev/null +++ b/toolchain/lex/mismatched_brackets.h @@ -0,0 +1,223 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#ifndef CARBON_TOOLCHAIN_LEX_MISMATCHED_BRACKETS_H_ +#define CARBON_TOOLCHAIN_LEX_MISMATCHED_BRACKETS_H_ + +#include + +#include "llvm/ADT/ArrayRef.h" +#include "llvm/ADT/SmallVector.h" +#include "llvm/ADT/StringRef.h" +#include "toolchain/lex/token_index.h" +#include "toolchain/lex/token_kind.h" + +namespace Carbon::Lex { + +// Represents a category of token significant for bracket matching recovery. +enum class BracketTokenKind : int8_t { + OpenParen, + CloseParen, + OpenCurlyBrace, + CloseCurlyBrace, + OpenSquareBracket, + CloseSquareBracket, + Semi, + Comma, + Period, + StatementIntroducer, // fn, class, var, if, while, etc. + + // A token that is a complete primary expression on its own: an identifier, a + // literal, `self`, a type keyword, and so on. A leaf token never directly + // follows another leaf token or a close paren or close square bracket, so + // such an adjacent pair is evidence that a bracket is missing between them. + Leaf, + + // The statement-structuring operators, which usually appear outside parens + // and square brackets, so are a cue that an unclosed one should close before + // them. `Assignment` (`=`) can also directly follow a `]` (`a[i] = v;`) and + // `As` commonly appears inside parens as a cast (`(x as T)`), unlike + // `StructuralOp` (`->` and `where`), so they are distinguished. + Assignment, + As, + StructuralOp, + + // A comparison or logical operator (`==`, `<`, `and`, ...), which is unlikely + // to appear inside square brackets. + ComparisonOp, + + // A binding modifier keyword (`ref`, `unused`, `template`), which like a leaf + // cannot directly follow a value-ending token. + ModifierKeyword, + + FileEnd, + + // Anything else. Must stay last: it bounds the kinds. + Other, +}; + +// Returns true if a token of this kind can be the last token of a primary +// expression. A leaf directly following a value-ending token is an illegal +// adjacency in a well-formed program, and so is a strong cue that an opening +// bracket is missing between them. Note that `]` is not value-ending: a type +// can directly follow one, as in `impl forall [T: Copy] T as ...`. +constexpr auto IsValueEndingKind(BracketTokenKind kind) -> bool { + return kind == BracketTokenKind::Leaf || kind == BracketTokenKind::CloseParen; +} + +// Returns true if this kind is one of the statement-structuring operators. +constexpr auto IsStructuralOpKind(BracketTokenKind kind) -> bool { + return kind == BracketTokenKind::Assignment || kind == BracketTokenKind::As || + kind == BracketTokenKind::StructuralOp; +} + +// Returns true if the token kind is an opening bracket. +constexpr auto IsOpeningBracket(BracketTokenKind kind) -> bool { + return kind == BracketTokenKind::OpenParen || + kind == BracketTokenKind::OpenCurlyBrace || + kind == BracketTokenKind::OpenSquareBracket; +} + +// Returns true if the token kind is a closing bracket. +constexpr auto IsClosingBracket(BracketTokenKind kind) -> bool { + return kind == BracketTokenKind::CloseParen || + kind == BracketTokenKind::CloseCurlyBrace || + kind == BracketTokenKind::CloseSquareBracket; +} + +// Returns the matching closing bracket kind for an opening bracket. +constexpr auto MatchingClosingKind(BracketTokenKind kind) -> BracketTokenKind { + switch (kind) { + case BracketTokenKind::OpenParen: + return BracketTokenKind::CloseParen; + case BracketTokenKind::OpenCurlyBrace: + return BracketTokenKind::CloseCurlyBrace; + case BracketTokenKind::OpenSquareBracket: + return BracketTokenKind::CloseSquareBracket; + default: + return BracketTokenKind::Other; + } +} + +// Returns the matching opening bracket kind for a closing bracket. +constexpr auto MatchingOpeningKind(BracketTokenKind kind) -> BracketTokenKind { + switch (kind) { + case BracketTokenKind::CloseParen: + return BracketTokenKind::OpenParen; + case BracketTokenKind::CloseCurlyBrace: + return BracketTokenKind::OpenCurlyBrace; + case BracketTokenKind::CloseSquareBracket: + return BracketTokenKind::OpenSquareBracket; + default: + return BracketTokenKind::Other; + } +} + +// Converts a BracketTokenKind to standard TokenKind. +constexpr auto ToTokenKind(BracketTokenKind kind) -> TokenKind { + switch (kind) { + case BracketTokenKind::OpenParen: + return TokenKind::OpenParen; + case BracketTokenKind::CloseParen: + return TokenKind::CloseParen; + case BracketTokenKind::OpenCurlyBrace: + return TokenKind::OpenCurlyBrace; + case BracketTokenKind::CloseCurlyBrace: + return TokenKind::CloseCurlyBrace; + case BracketTokenKind::OpenSquareBracket: + return TokenKind::OpenSquareBracket; + case BracketTokenKind::CloseSquareBracket: + return TokenKind::CloseSquareBracket; + case BracketTokenKind::Semi: + return TokenKind::Semi; + default: + return TokenKind::Error; + } +} + +// Lightweight token description passed into the bracket matching algorithm. +struct MismatchedBracketToken { + TokenIndex token_index = TokenIndex::None; + BracketTokenKind kind; + int32_t line; + int32_t line_indent; + + // Whether this token is the last non-comment token on its line. + bool is_at_end_of_line = false; + + // For OpenCurlyBrace, whether it has struct-like cues (e.g. followed by '.', + // '}', or ':'). + bool is_struct_brace = false; + + // For StatementIntroducer, whether this is a keyword that must be directly + // followed by an opening bracket: `if`, `while`, `for`, `match` (which + // require `(`), or `forall` (which requires `[`). + bool is_paren_keyword = false; + + // For StatementIntroducer, whether this is the `else` keyword, which + // normally directly follows a `}` on the same line. + bool is_else_keyword = false; + + // Whether this token has whitespace (or a comment) directly before it. + bool has_leading_space = false; + + // Whether this token is mid-line with two or more bytes of whitespace + // before it. Formatted code separates mid-line tokens by at most one + // space, so a wide gap suggests something was deleted in it. + bool has_wide_leading_space = false; +}; + +// An action to fix mismatched brackets in the token stream. +enum class BracketFixAction : int8_t { + // Insert a missing bracket before the specified token. + InsertBefore, + + // Insert a missing bracket after the specified token. + InsertAfter, + + // Replace an unmatched bracket with an error token. + ReplaceWithError, +}; + +// A diagnostic to issue for an unmatched or repaired bracket. +enum class BracketDiagnosticKind : int8_t { + UnmatchedOpening, + UnmatchedClosing, +}; + +// Represents a single correction made to recover from a mismatched bracket, +// pairing the diagnostic to report with the token-stream fix to apply. +// +// The token indexes are those of the stream `FixMismatchedBrackets` was given. +// A caller that applies the fixes renumbers that stream, so it is responsible +// for updating them before handing corrections on; see `LexOptions:: +// bracket_corrections`. +struct BracketCorrection { + // The diagnostic to report for this bracket error. + BracketDiagnosticKind diagnostic_kind; + TokenIndex diagnostic_token_index = TokenIndex::None; + + // The fix action to apply to the token stream. + BracketFixAction fix_action; + TokenIndex fix_token_index = TokenIndex::None; + TokenKind fix_token_kind; + + // Set to true if multiple optimal paths tie/disagree on the repair. + bool is_tied = false; + + // The name of the rule that chose this correction. This never reaches a + // user-facing diagnostic; it exists so that debug dumps and the evaluation + // tool's per-rule precision table can say which rule to blame. + llvm::StringLiteral rule_name = ""; +}; + +// Analyzes the input token stream, finds the optimal set of bracket insertions +// and error replacements based on indentation and structural cues, and returns +// the corresponding list of corrections. +auto FixMismatchedBrackets(llvm::ArrayRef tokens) + -> llvm::SmallVector; + +} // namespace Carbon::Lex + +#endif // CARBON_TOOLCHAIN_LEX_MISMATCHED_BRACKETS_H_ diff --git a/toolchain/lex/mismatched_brackets_eval.cpp b/toolchain/lex/mismatched_brackets_eval.cpp new file mode 100644 index 000000000000..10e9d7cd23a3 --- /dev/null +++ b/toolchain/lex/mismatched_brackets_eval.cpp @@ -0,0 +1,1547 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "common/bazel_working_dir.h" +#include "common/check.h" +#include "common/command_line.h" +#include "common/init_llvm.h" +#include "common/ostream.h" +#include "llvm/ADT/ArrayRef.h" +#include "llvm/ADT/DenseMap.h" +#include "llvm/ADT/DenseSet.h" +#include "llvm/ADT/Hashing.h" +#include "llvm/ADT/STLExtras.h" +#include "llvm/ADT/Sequence.h" +#include "llvm/ADT/SmallVector.h" +#include "llvm/ADT/StringExtras.h" +#include "llvm/ADT/StringRef.h" +#include "llvm/Support/FileSystem.h" +#include "llvm/Support/FormatVariadic.h" +#include "llvm/Support/VirtualFileSystem.h" +#include "toolchain/base/shared_value_stores.h" +#include "toolchain/diagnostics/consumer.h" +#include "toolchain/diagnostics/null_diagnostics.h" +#include "toolchain/lex/lex.h" +#include "toolchain/lex/mismatched_brackets.h" +#include "toolchain/lex/token_kind.h" +#include "toolchain/lex/tokenized_buffer.h" +#include "toolchain/source/source_buffer.h" + +namespace Carbon::Lex { + +// How well recovery did on one trial. See `PrintMetricDefinitions` for what +// each means. +namespace { +enum class TestClassification { + Correct, + Partial, + None, + Incorrect, +}; +} // namespace + +// Lex options that discard diagnostics, since a trial only cares about the +// tokens and the corrections. +static auto QuietLexOptions() -> LexOptions { + LexOptions options; + options.consumer = &Diagnostics::NullConsumer(); + return options; +} + +namespace { + +// A deletion level from `--d-values`: how many brackets each trial deletes, +// either as an absolute count or as a percentage of the file's clean pairs. +struct DSpec { + // The level as written on the command line, used to label it in the report. + std::string label; + bool is_percent = false; + double percent_val = 0.0; + int count_val = 0; + + // How many of a file's `num_clean_pairs` pairs a trial deletes, at least one + // and never more than the file has. + auto DeletionCount(int num_clean_pairs) const -> int { + int count = + is_percent + ? std::max(1, static_cast(num_clean_pairs * percent_val)) + : count_val; + return std::min(count, num_clean_pairs); + } +}; + +// Trial counts by classification, for one file and deletion level or for the +// whole run. +struct TrialStats { + int total = 0; + int correct = 0; + int partial = 0; + int none = 0; + int incorrect = 0; + + auto Add(TestClassification classification) -> void { + ++total; + switch (classification) { + case TestClassification::Correct: + ++correct; + break; + case TestClassification::Partial: + ++partial; + break; + case TestClassification::None: + ++none; + break; + case TestClassification::Incorrect: + ++incorrect; + break; + } + } + + auto CorrectPct() const -> double { return Pct(correct); } + auto PartialPct() const -> double { return Pct(partial); } + auto NonePct() const -> double { return Pct(none); } + auto IncorrectPct() const -> double { return Pct(incorrect); } + + // Percentage of trials with no incorrect suggestion. + auto SafetyPct() const -> double { return Pct(correct + partial + none); } + + // Precision of the suggestions that were made. + auto AccuracyPct() const -> double { + int decisive = correct + incorrect; + return decisive == 0 ? 100.0 : (100.0 * correct) / decisive; + } + + private: + auto Pct(int count) const -> double { + return total == 0 ? 0.0 : (100.0 * count) / total; + } +}; + +struct BracketPair { + TokenIndex open_token; + TokenIndex close_token; +}; + +// One bracket the corruption removed, and where recovery would have to put it +// back for that to count as correct. Offsets are in the corrupted text. +struct DeletedToken { + TokenKind kind; + int32_t byte_offset; + int32_t length; + int32_t line; + int32_t column; + // The offset of the first token that survived after this one. + int32_t next_token_byte_offset; + // Every offset an insertion could name and still be the same repair: the + // surviving members of the run of identical brackets this one belonged to, + // plus `next_token_byte_offset`. + llvm::SmallVector valid_next_token_byte_offsets; +}; + +// One bracket recovery inserted, located by the first surviving token it +// precedes. +struct Suggestion { + TokenKind kind; + int32_t byte_offset; + int32_t line; + int32_t column; + // The name of the rule that inserted this bracket. See + // `BracketCorrection::rule_name`. + llvm::StringRef rule_name; +}; + +} // namespace + +static auto MatchesDeletedToken(const DeletedToken& del, const Suggestion& sugg) + -> bool { + if (del.kind != sugg.kind) { + return false; + } + return llvm::is_contained(del.valid_next_token_byte_offsets, + sugg.byte_offset); +} + +// Whether any deleted bracket is the one `sugg` puts back, and the same +// question from the other side. +static auto AnyDeletionMatches(llvm::ArrayRef deleted, + const Suggestion& sugg) -> bool { + return llvm::any_of(deleted, [&](const DeletedToken& del) { + return MatchesDeletedToken(del, sugg); + }); +} +static auto AnySuggestionMatches(llvm::ArrayRef suggestions, + const DeletedToken& del) -> bool { + return llvm::any_of(suggestions, [&](const Suggestion& sugg) { + return MatchesDeletedToken(del, sugg); + }); +} + +// Parses the `--d-values` list. Entries that don't parse are dropped; the +// caller errors out if nothing is left. +static auto ParseDSpecs(llvm::StringRef str) -> llvm::SmallVector { + llvm::SmallVector specs; + llvm::SmallVector parts; + str.split(parts, ','); + for (llvm::StringRef part : parts) { + part = part.trim(); + if (part.empty()) { + continue; + } + DSpec spec; + spec.label = part.str(); + if (part.ends_with("%")) { + spec.is_percent = true; + llvm::StringRef num = part.drop_back(1); + double val = 0.0; + if (!num.getAsDouble(val)) { + spec.percent_val = val / 100.0; + specs.push_back(spec); + } + } else { + spec.is_percent = false; + int val = 0; + if (!part.getAsInteger(10, val)) { + spec.count_val = val; + specs.push_back(spec); + } + } + } + return specs; +} + +// The bracket pairs a clean file matched up, which are the pairs a trial can +// delete an endpoint of. +static auto GetCleanBracketPairs(const TokenizedBuffer& buffer) + -> llvm::SmallVector { + llvm::SmallVector pairs; + for (TokenIndex t : buffer.tokens()) { + if (!buffer.GetKind(t).is_opening_symbol()) { + continue; + } + TokenIndex close = buffer.GetMatchedClosingToken(t); + if (close != TokenIndex::None) { + pairs.push_back({.open_token = t, .close_token = close}); + } + } + return pairs; +} + +// Scores one trial: every suggestion must restore a distinct deleted bracket, +// and a suggestion that restores none makes the whole trial incorrect. +static auto ClassifyTrial(llvm::ArrayRef deleted_tokens, + llvm::ArrayRef suggestions) + -> TestClassification { + if (deleted_tokens.empty()) { + return suggestions.empty() ? TestClassification::Correct + : TestClassification::Incorrect; + } + + std::vector deleted_matched(deleted_tokens.size(), false); + int correct_suggestions = 0; + + for (const auto& sugg : suggestions) { + bool found_match = false; + for (size_t i = 0; i < deleted_tokens.size(); ++i) { + if (!deleted_matched[i] && MatchesDeletedToken(deleted_tokens[i], sugg)) { + deleted_matched[i] = true; + found_match = true; + ++correct_suggestions; + break; + } + } + if (!found_match) { + return TestClassification::Incorrect; + } + } + + if (correct_suggestions == static_cast(deleted_tokens.size())) { + return TestClassification::Correct; + } + if (correct_suggestions > 0) { + return TestClassification::Partial; + } + return TestClassification::None; +} + +namespace { + +// How a clean file is corrupted for a trial. See `--mode` help for details. +enum class CorruptionMode { + // Blank each deleted bracket with a space (byte offsets preserved). + Blank, + // Delete each bracket character, closing the gap so that no space is left + // behind. More realistic, and doesn't leave the whitespace artifacts the + // algorithm can key on. + Gapless, + // Truncate the file at a random token (models incomplete, in-development + // code); recovery should close all still-open brackets at the new EOF. + Truncate, + // Delete from a random statement/element boundary inside one bracketed + // region through that region's closing bracket (models typing new code + // inside an existing class/scope, whose tail and close aren't there yet); + // recovery must infer where the region should have ended. + TruncateRegion, +}; + +// A corrupted source plus the ground-truth insertions recovery should make. +struct CorruptedCase { + std::string text; + std::vector expected; +}; + +} // namespace + +// Whether `mode` corrupts by cutting the file short rather than by deleting +// individual brackets. Such a mode ignores the deletion level, and its +// ground-truth insertions land at the new end of the file. +static auto IsTruncateMode(CorruptionMode mode) -> bool { + return mode == CorruptionMode::Truncate || + mode == CorruptionMode::TruncateRegion; +} + +// Remaps a byte offset from original to corrupted coordinates, given sorted, +// disjoint deleted ranges [begin, end). An offset inside a deleted range maps +// to where the gap closes. +static auto RemapOffset(int32_t off, + llvm::ArrayRef> deleted) + -> int32_t { + int32_t shift = 0; + for (auto [begin, end] : deleted) { + if (end <= off) { + shift += end - begin; + } else if (begin <= off) { + return begin - shift; + } else { + break; + } + } + return off - shift; +} + +// Returns `text` with the given sorted, disjoint byte ranges removed. +static auto RemoveRanges(llvm::StringRef text, + llvm::ArrayRef> ranges) + -> std::string { + std::string out; + int32_t pos = 0; + for (auto [begin, end] : ranges) { + out += text.substr(pos, begin - pos); + pos = end; + } + out += text.substr(pos); + return out; +} + +// Builds a trial that deletes one endpoint of each of `d_count` random clean +// pairs, either blanking them or closing the gap. +static auto MakeDeletionCase(const TokenizedBuffer& buffer, + llvm::StringRef source_text, + llvm::ArrayRef pairs, int d_count, + bool close_gap, std::mt19937_64& rng) + -> std::optional { + CARBON_CHECK(d_count <= static_cast(pairs.size()), + "Asked to delete more pairs than the file has."); + std::vector pair_indices(pairs.size()); + std::iota(pair_indices.begin(), pair_indices.end(), 0); + std::shuffle(pair_indices.begin(), pair_indices.end(), rng); + pair_indices.resize(d_count); + + std::vector is_deleted(buffer.size(), false); + std::vector sampled; + sampled.reserve(d_count); + for (int p_idx : pair_indices) { + const auto& pair = pairs[p_idx]; + TokenIndex tok = (rng() % 2 == 0) ? pair.open_token : pair.close_token; + is_deleted[tok.index] = true; + sampled.push_back(tok); + } + + std::vector deleted; + deleted.reserve(d_count); + for (TokenIndex tok : sampled) { + TokenKind tok_kind = buffer.GetKind(tok); + + // Any surviving bracket in a run of identical ones is an equally good place + // to reinsert this one, so find the whole run. + int32_t run_start = tok.index; + while (run_start > 0 && + buffer.GetKind(TokenIndex(run_start - 1)) == tok_kind) { + --run_start; + } + int32_t run_end = tok.index; + while (run_end + 1 < static_cast(buffer.size()) && + buffer.GetKind(TokenIndex(run_end + 1)) == tok_kind) { + ++run_end; + } + + llvm::SmallVector valid_offsets; + for (int32_t idx = run_start; idx <= run_end; ++idx) { + if (!is_deleted[idx]) { + valid_offsets.push_back(buffer.GetByteOffset(TokenIndex(idx))); + } + } + auto succ = TokenIndex(run_end + 1); + while (succ.index < buffer.size() && is_deleted[succ.index]) { + succ = TokenIndex(succ.index + 1); + } + int32_t succ_byte = (succ.index < buffer.size()) + ? buffer.GetByteOffset(succ) + : static_cast(source_text.size()); + valid_offsets.push_back(succ_byte); + + deleted.push_back(DeletedToken{ + .kind = tok_kind, + .byte_offset = buffer.GetByteOffset(tok), + .length = static_cast(buffer.GetTokenText(tok).size()), + .line = buffer.GetLineNumber(tok), + .column = buffer.GetColumnNumber(tok), + .next_token_byte_offset = succ_byte, + .valid_next_token_byte_offsets = std::move(valid_offsets), + }); + } + + if (!close_gap) { + std::string corrupted = source_text.str(); + for (const auto& del : deleted) { + for (int i = 0; i < del.length; ++i) { + corrupted[del.byte_offset + i] = ' '; + } + } + return CorruptedCase{.text = std::move(corrupted), + .expected = std::move(deleted)}; + } + + llvm::SmallVector> ranges; + for (TokenIndex tok : sampled) { + int32_t off = buffer.GetByteOffset(tok); + ranges.push_back( + {off, off + static_cast(buffer.GetTokenText(tok).size())}); + } + llvm::sort(ranges); + std::string corrupted = RemoveRanges(source_text, ranges); + for (auto& del : deleted) { + del.byte_offset = RemapOffset(del.byte_offset, ranges); + del.next_token_byte_offset = + RemapOffset(del.next_token_byte_offset, ranges); + for (auto& off : del.valid_next_token_byte_offsets) { + off = RemapOffset(off, ranges); + } + } + return CorruptedCase{.text = std::move(corrupted), + .expected = std::move(deleted)}; +} + +// Builds a trial that truncates the file at a random token boundary; recovery +// should close every bracket still open there, at the new EOF. +static auto MakeTruncateCase(const TokenizedBuffer& buffer, + llvm::StringRef source_text, std::mt19937_64& rng) + -> std::optional { + int32_t size = buffer.size(); + if (size <= 2) { + return std::nullopt; + } + // Cut before a random token in (FileStart, FileEnd), keeping [0, cut). + int32_t cut = 1 + static_cast(rng() % (size - 2)); + int32_t cut_byte = buffer.GetByteOffset(TokenIndex(cut)); + std::string corrupted = source_text.substr(0, cut_byte).str(); + + llvm::SmallVector stack; + for (int32_t i : llvm::seq(0, cut)) { + auto kind = buffer.GetKind(TokenIndex(i)); + if (kind.is_opening_symbol()) { + stack.push_back(kind); + } else if (kind.is_closing_symbol() && !stack.empty() && + stack.back().closing_symbol() == kind) { + stack.pop_back(); + } + } + + auto eof = static_cast(corrupted.size()); + std::vector expected; + for (TokenKind open_kind : llvm::reverse(stack)) { + expected.push_back(DeletedToken{ + .kind = open_kind.closing_symbol(), + .byte_offset = eof, + .length = 1, + .line = -1, + .column = -1, + .next_token_byte_offset = eof, + // Every open bracket must close at EOF: the surviving token each + // closer precedes is the FileEnd token. (The real FileEnd offset, + // which excludes trailing whitespace, is added after lexing.) + .valid_next_token_byte_offsets = {eof}, + }); + } + return CorruptedCase{.text = std::move(corrupted), + .expected = std::move(expected)}; +} + +// Builds a trial that deletes from a region-top-level boundary inside a random +// pair through that pair's closing bracket. Recovery must reinsert the one +// closing bracket at the join, where the region should have ended. +static auto MakeTruncateRegionCase(const TokenizedBuffer& buffer, + llvm::StringRef source_text, + llvm::ArrayRef pairs, + std::mt19937_64& rng) + -> std::optional { + if (pairs.empty()) { + return std::nullopt; + } + const auto& pair = pairs[rng() % pairs.size()]; + int32_t open = pair.open_token.index; + int32_t close = pair.close_token.index; + + // Collect cut points at the region's own nesting level (not inside a nested + // pair), so that deleting through the close orphans only this one bracket. + llvm::SmallVector candidates; + int32_t depth = 0; + for (int32_t i : llvm::seq(open + 1, close + 1)) { + if (depth == 0) { + candidates.push_back(i); + } + if (i < close) { + auto kind = buffer.GetKind(TokenIndex(i)); + if (kind.is_opening_symbol()) { + ++depth; + } else if (kind.is_closing_symbol()) { + --depth; + } + } + } + if (candidates.empty()) { + return std::nullopt; + } + int32_t cut = candidates[rng() % candidates.size()]; + + int32_t del_begin = buffer.GetByteOffset(TokenIndex(cut)); + int32_t del_end = + buffer.GetByteOffset(pair.close_token) + + static_cast(buffer.GetTokenText(pair.close_token).size()); + llvm::SmallVector> ranges = { + {del_begin, del_end}}; + std::string corrupted = RemoveRanges(source_text, ranges); + + int32_t succ = close + 1; + int32_t succ_byte = (succ < buffer.size()) + ? buffer.GetByteOffset(TokenIndex(succ)) + : static_cast(source_text.size()); + int32_t succ_corrupted = RemapOffset(succ_byte, ranges); + + TokenKind close_kind = buffer.GetKind(pair.open_token).closing_symbol(); + std::vector expected = {DeletedToken{ + .kind = close_kind, + .byte_offset = del_begin, + .length = 1, + .line = buffer.GetLineNumber(pair.close_token), + .column = buffer.GetColumnNumber(pair.close_token), + .next_token_byte_offset = succ_corrupted, + .valid_next_token_byte_offsets = {succ_corrupted}, + }}; + return CorruptedCase{.text = std::move(corrupted), + .expected = std::move(expected)}; +} + +// Corrupts a clean file for one trial, per `mode`. Returns nullopt if the file +// has nothing this mode can corrupt. +static auto MakeCorruptedCase(CorruptionMode mode, + const TokenizedBuffer& buffer, + llvm::StringRef source_text, + llvm::ArrayRef pairs, int d_count, + std::mt19937_64& rng) + -> std::optional { + switch (mode) { + case CorruptionMode::Blank: + case CorruptionMode::Gapless: + return MakeDeletionCase(buffer, source_text, pairs, d_count, + /*close_gap=*/mode == CorruptionMode::Gapless, + rng); + case CorruptionMode::Truncate: + return MakeTruncateCase(buffer, source_text, rng); + case CorruptionMode::TruncateRegion: + return MakeTruncateRegionCase(buffer, source_text, pairs, rng); + } +} + +// Accepts the real `FileEnd` offset for any ground-truth insertion at the end +// of the file. Recovery inserts before `FileEnd`, whose offset excludes +// trailing whitespace, whereas the ground truth was computed as the text size. +static auto AcceptFileEndOffset(const TokenizedBuffer& buffer, + llvm::StringRef text, + std::vector& deleted) -> void { + int32_t eof = buffer.GetByteOffset(TokenIndex(buffer.size() - 1)); + auto text_size = static_cast(text.size()); + for (auto& del : deleted) { + if (llvm::is_contained(del.valid_next_token_byte_offsets, text_size)) { + del.valid_next_token_byte_offsets.push_back(eof); + } + } +} + +// Whether closing the gap fused two tokens into one (e.g. `f(x)` -> `fx)`), +// leaving some ground-truth insertion with no token boundary to land on. That +// both is unrealistic and makes the trial unscoreable, so the caller skips it. +static auto TokensWereFused(const TokenizedBuffer& buffer, llvm::StringRef text, + llvm::ArrayRef deleted) -> bool { + llvm::DenseSet token_offsets; + for (TokenIndex t : buffer.tokens()) { + if (!buffer.IsRecoveryToken(t)) { + token_offsets.insert(buffer.GetByteOffset(t)); + } + } + token_offsets.insert(static_cast(text.size())); + return llvm::any_of(deleted, [&](const DeletedToken& del) { + return llvm::none_of(del.valid_next_token_byte_offsets, [&](int32_t off) { + return token_offsets.contains(off); + }); + }); +} + +// Checks the invariant the rule-name lookup relies on: corrections name tokens +// of the buffer recovery produced, so an insertion names the recovery token it +// inserted. A wrong index would silently lose rule names rather than fail. +static auto CheckCorrectionsNameRealTokens( + const TokenizedBuffer& buffer, + llvm::ArrayRef corrections) -> void { + for (const auto& c : corrections) { + CARBON_CHECK( + c.fix_token_index.index >= 0 && c.fix_token_index.index < buffer.size(), + "Correction names a token outside the buffer."); + if (c.fix_action != BracketFixAction::ReplaceWithError && !c.is_tied) { + CARBON_CHECK(buffer.IsRecoveryToken(c.fix_token_index), + "Insertion doesn't name the token it inserted."); + CARBON_CHECK(buffer.GetKind(c.fix_token_index) == c.fix_token_kind, + "Inserted token has the wrong kind."); + } + } +} + +// Names the rule that inserted `token`. Corrections name the tokens of this +// buffer, so the one that inserted it names it directly. A tied correction was +// downgraded to an error token and has no rule to report. +static auto RuleNameOfInsertion(llvm::ArrayRef corrections, + TokenIndex token) -> llvm::StringRef { + for (const auto& c : corrections) { + if (c.fix_action != BracketFixAction::ReplaceWithError && !c.is_tied && + c.fix_token_index == token) { + return c.rule_name; + } + } + return "Unknown"; +} + +// Turns the brackets recovery inserted into scoreable suggestions. +// +// Structure-equality: a fix is identified by the first *surviving* token it +// precedes, not its raw offset. Other inserted (recovery) tokens are skipped, +// so a cascade of closers all point at the same real anchor, and closing among +// trailing whitespace or a deleted span still resolves to the token that +// structurally follows. +static auto CollectSuggestions(const TokenizedBuffer& buffer, + llvm::StringRef text, + llvm::ArrayRef corrections) + -> llvm::SmallVector { + llvm::SmallVector suggestions; + for (TokenIndex t : buffer.tokens()) { + auto kind = buffer.GetKind(t); + if (!buffer.IsRecoveryToken(t) || + !(kind.is_opening_symbol() || kind.is_closing_symbol())) { + continue; + } + auto succ = TokenIndex(t.index + 1); + while (succ.index < buffer.size() && buffer.IsRecoveryToken(succ)) { + succ = TokenIndex(succ.index + 1); + } + bool has_succ = succ.index < buffer.size(); + suggestions.push_back(Suggestion{ + .kind = kind, + .byte_offset = has_succ ? buffer.GetByteOffset(succ) + : static_cast(text.size()), + .line = has_succ ? buffer.GetLineNumber(succ) : -1, + .column = has_succ ? buffer.GetColumnNumber(succ) : -1, + .rule_name = RuleNameOfInsertion(corrections, t), + }); + } + return suggestions; +} + +namespace { + +// A file worth testing: it lexes cleanly and has at least one matched pair, so +// a trial can delete a bracket from it. +struct CandidateFile { + std::string filename; + int clean_pairs_count = 0; +}; + +// The corpus a run evaluates. +struct Corpus { + llvm::SmallVector files; + int total_clean_pairs = 0; +}; + +} // namespace + +static auto CollectCarbonFiles(llvm::ArrayRef input_paths) + -> llvm::SmallVector { + llvm::SmallVector files; + + auto scan_directory = [&](llvm::StringRef dir_path) { + std::error_code ec; + for (llvm::sys::fs::recursive_directory_iterator it(dir_path, ec), end; + it != end && !ec; it.increment(ec)) { + if (it->path().ends_with(".carbon")) { + files.push_back(it->path()); + } + } + }; + + if (input_paths.empty()) { + scan_directory("core"); + scan_directory("examples"); + } else { + for (llvm::StringRef path : input_paths) { + if (llvm::sys::fs::is_directory(path)) { + scan_directory(path); + } else if (path.ends_with(".carbon")) { + files.push_back(path.str()); + } + } + } + + llvm::sort(files); + files.erase(std::unique(files.begin(), files.end()), files.end()); + return files; +} + +// Lexes each of `files` and keeps the ones a trial can be built from, which is +// what makes the corpus. A file that fails to lex cleanly can't serve as ground +// truth, and one with no matched pairs has no bracket to delete. +static auto FindCandidateFiles(llvm::ArrayRef files) -> Corpus { + Corpus corpus; + for (const auto& filepath : files) { + auto source = SourceBuffer::MakeFromFile( + *llvm::vfs::getRealFileSystem(), filepath, Diagnostics::NullConsumer()); + if (!source) { + continue; + } + + SharedValueStores value_stores; + auto clean_buffer = Lex::Lex(value_stores, *source, QuietLexOptions()); + if (clean_buffer.has_errors()) { + continue; + } + + auto clean_pairs = GetCleanBracketPairs(clean_buffer); + if (clean_pairs.empty()) { + continue; + } + + auto num_pairs = static_cast(clean_pairs.size()); + corpus.total_clean_pairs += num_pairs; + corpus.files.push_back( + {.filename = filepath, .clean_pairs_count = num_pairs}); + } + return corpus; +} + +// Apportions `total_trials` across the corpus for one deletion level, returning +// each file's share. +// +// A percentage level tests every file equally; an absolute level weights each +// file by its clean pair count, so that every pair in the corpus is equally +// likely to be picked. Fractional quotas are handed out +// largest-remainder-first, with ties broken by filename to keep the allocation +// deterministic. +static auto AllocateTrials(const Corpus& corpus, const DSpec& spec, + int total_trials) -> std::vector { + std::vector allocation(corpus.files.size(), 0); + if (total_trials <= 0) { + return allocation; + } + + double total_weight = spec.is_percent + ? static_cast(corpus.files.size()) + : static_cast(corpus.total_clean_pairs); + int allocated = 0; + std::vector> remainders; + remainders.reserve(corpus.files.size()); + for (auto [i, file] : llvm::enumerate(corpus.files)) { + double weight = + spec.is_percent ? 1.0 : static_cast(file.clean_pairs_count); + double exact_quota = + static_cast(total_trials) * weight / total_weight; + auto base_count = static_cast(exact_quota); + allocation[i] = base_count; + allocated += base_count; + remainders.push_back({exact_quota - base_count, i}); + } + + std::sort(remainders.begin(), remainders.end(), + [&](const auto& a, const auto& b) { + if (a.first != b.first) { + return a.first > b.first; + } + return corpus.files[a.second].filename < + corpus.files[b.second].filename; + }); + int remainder_trials = total_trials - allocated; + for (int i = 0; + i < remainder_trials && i < static_cast(remainders.size()); ++i) { + ++allocation[remainders[i].second]; + } + return allocation; +} + +namespace { + +struct FileResult { + std::string filename; + int clean_pairs_count = 0; + // Indexed as the deletion levels. + std::vector stats_by_level; +}; + +// How often a named rule was right. +struct RuleStat { + int correct = 0; + int incorrect = 0; +}; + +// For incorrect trials: how far (in tokens, signed; + = closed later / +// swallowing following code) the wrong close is from the correct anchor. +struct DistStat { + int later = 0; + int earlier = 0; + int no_close = 0; + std::map token_dist_hist; +}; + +// The shape of a distance histogram, for the report. +struct DistanceSummary { + int median = 0; + int p90 = 0; + int max_abs = 0; +}; + +// Everything the trials measured. +struct Report { + // Indexed as the deletion levels. + std::vector stats_by_level; + std::vector results_by_file; + std::map stats_by_rule; + std::map wrong_close_by_kind; + // Trials skipped because closing the gap fused two tokens. + int merged_skips = 0; +}; + +// The evaluation configuration. `d_values` and `mode_name` hold what the +// command line said; `d_specs` and `mode` are the parsed forms that `Resolve` +// fills in. +struct EvalOptions { + llvm::SmallVector input_files; + llvm::StringRef d_values = "1,2,5,10%,25%"; + llvm::StringRef mode_name = "blank"; + int total_trials = 1000; + int base_seed = 42; + bool verbose = false; + bool json_output = false; + int dump_incorrect = 0; + int dump_none = 0; + + llvm::SmallVector d_specs; + CorruptionMode mode = CorruptionMode::Blank; + + // Parses `d_values` and `mode_name`, reporting to stderr and returning false + // if either is invalid. + auto Resolve() -> bool; + + // The one deletion level the rule table reports on, so that its precisions + // aren't a blend of easy and hard configurations. The truncate modes have + // only the single level; otherwise it's D=1, if that was asked for. + auto RuleLevelLabel() const -> llvm::StringRef { + return IsTruncateMode(mode) ? llvm::StringRef(d_specs.front().label) : "1"; + } +}; + +} // namespace + +auto EvalOptions::Resolve() -> bool { + d_specs = ParseDSpecs(d_values); + if (d_specs.empty()) { + llvm::errs() << "error: No valid D deletion specifications provided.\n"; + return false; + } + + if (mode_name == "blank") { + mode = CorruptionMode::Blank; + } else if (mode_name == "gapless") { + mode = CorruptionMode::Gapless; + } else if (mode_name == "truncate") { + mode = CorruptionMode::Truncate; + } else if (mode_name == "truncate-region") { + mode = CorruptionMode::TruncateRegion; + } else { + llvm::errs() << "error: Unknown --mode '" << mode_name << "'.\n"; + return false; + } + + // The truncate modes don't delete a set number of brackets, so collapse the + // D configurations to a single pass. + if (IsTruncateMode(mode)) { + d_specs.resize(1); + d_specs[0].label = mode_name.str(); + } + return true; +} + +// The seed for one trial, mixed from the file, level, and trial number so that +// any single trial can be reproduced on its own. +static auto TrialSeed(int base_seed, const std::string& filename, + const std::string& spec_label, int trial) -> uint64_t { + uint64_t file_hash = llvm::hash_value(filename); + uint64_t spec_hash = llvm::hash_value(spec_label); + return static_cast(base_seed) ^ file_hash ^ (spec_hash << 16) ^ + (static_cast(trial) * 0x9e3779b97f4a7c15ULL); +} + +namespace { + +// Runs the trials over a corpus and accumulates the measurements the report is +// built from. +class Evaluator { + public: + Evaluator(const EvalOptions& options, const Corpus& corpus) + : options_(options), + corpus_(corpus), + dump_incorrect_(options.dump_incorrect), + dump_none_(options.dump_none) {} + + // Runs every trial and returns what they measured. + auto Run() -> Report; + + private: + // The clean file a trial corrupts, lexed once and shared by all its trials. + struct FileContext { + const CandidateFile& candidate; + const TokenizedBuffer& clean_buffer; + llvm::StringRef source_text; + llvm::ArrayRef clean_pairs; + }; + + // Runs every trial allocated to one file, appending its statistics to the + // report. + auto RunFile(size_t file_index, const CandidateFile& candidate) -> void; + + // Runs one trial, returning how it scored, or nullopt if it had to be + // skipped. + auto RunTrial(const FileContext& file, const DSpec& spec, int d_count, + std::mt19937_64& rng) -> std::optional; + + // Credits or blames the rule behind each suggestion, for the rule table. + auto RecordRuleNames(llvm::ArrayRef deleted_tokens, + llvm::ArrayRef suggestions) -> void; + + // Records, for each deleted bracket recovery failed to restore, how far the + // nearest same-kind close it did suggest is from where the bracket belonged. + auto RecordWrongCloseDistances(const TokenizedBuffer& buffer, + llvm::ArrayRef deleted_tokens, + llvm::ArrayRef suggestions) + -> void; + + // Prints a trial in full if the `--dump-*` budget for its classification + // allows, spending one from that budget. + auto MaybeDumpTrial(const FileContext& file, const DSpec& spec, + TestClassification classification, + llvm::StringRef corrupted_text, + llvm::ArrayRef deleted_tokens, + llvm::ArrayRef suggestions, + llvm::ArrayRef corrections) -> void; + + const EvalOptions& options_; + const Corpus& corpus_; + // Remaining `--dump-incorrect` and `--dump-none` budget. + int dump_incorrect_; + int dump_none_; + // Trial counts indexed by [deletion level][file]. + std::vector> trials_; + Report report_; +}; + +} // namespace + +auto Evaluator::Run() -> Report { + report_.stats_by_level.resize(options_.d_specs.size()); + for (const DSpec& spec : options_.d_specs) { + trials_.push_back(AllocateTrials(corpus_, spec, options_.total_trials)); + } + for (auto [file_index, candidate] : llvm::enumerate(corpus_.files)) { + RunFile(file_index, candidate); + } + return std::move(report_); +} + +auto Evaluator::RunFile(size_t file_index, const CandidateFile& candidate) + -> void { + // Without `--verbose` a file with no trials has nothing to report, so don't + // spend the time lexing it. + bool has_any_trials = + llvm::any_of(trials_, [&](const std::vector& level_trials) { + return level_trials[file_index] > 0; + }); + if (!options_.verbose && !has_any_trials) { + return; + } + + auto source = SourceBuffer::MakeFromFile(*llvm::vfs::getRealFileSystem(), + candidate.filename, + Diagnostics::NullConsumer()); + if (!source) { + return; + } + SharedValueStores value_stores; + auto clean_buffer = Lex::Lex(value_stores, *source, QuietLexOptions()); + auto clean_pairs = GetCleanBracketPairs(clean_buffer); + FileContext file = {.candidate = candidate, + .clean_buffer = clean_buffer, + .source_text = source->text(), + .clean_pairs = clean_pairs}; + + FileResult result = { + .filename = candidate.filename, + .clean_pairs_count = candidate.clean_pairs_count, + .stats_by_level = std::vector(options_.d_specs.size())}; + + for (auto [level, spec] : llvm::enumerate(options_.d_specs)) { + int d_count = spec.DeletionCount(static_cast(clean_pairs.size())); + for (int trial : llvm::seq(0, trials_[level][file_index])) { + std::mt19937_64 rng( + TrialSeed(options_.base_seed, candidate.filename, spec.label, trial)); + auto classification = RunTrial(file, spec, d_count, rng); + if (!classification) { + continue; + } + result.stats_by_level[level].Add(*classification); + report_.stats_by_level[level].Add(*classification); + } + } + + report_.results_by_file.push_back(std::move(result)); +} + +auto Evaluator::RunTrial(const FileContext& file, const DSpec& spec, + int d_count, std::mt19937_64& rng) + -> std::optional { + auto corrupted_case = + MakeCorruptedCase(options_.mode, file.clean_buffer, file.source_text, + file.clean_pairs, d_count, rng); + if (!corrupted_case) { + return std::nullopt; + } + std::string corrupted_text = std::move(corrupted_case->text); + std::vector deleted_tokens = + std::move(corrupted_case->expected); + + auto corrupted_source = SourceBuffer::MakeFromStringCopy( + file.candidate.filename, corrupted_text, Diagnostics::NullConsumer()); + if (!corrupted_source) { + return std::nullopt; + } + + SharedValueStores value_stores; + LexOptions lex_options = QuietLexOptions(); + llvm::SmallVector corrections; + lex_options.bracket_corrections = &corrections; + auto buffer = Lex::Lex(value_stores, *corrupted_source, lex_options); + + if (IsTruncateMode(options_.mode)) { + AcceptFileEndOffset(buffer, corrupted_text, deleted_tokens); + } + if ((options_.mode == CorruptionMode::Gapless || + options_.mode == CorruptionMode::TruncateRegion) && + TokensWereFused(buffer, corrupted_text, deleted_tokens)) { + ++report_.merged_skips; + return std::nullopt; + } + + CheckCorrectionsNameRealTokens(buffer, corrections); + auto suggestions = CollectSuggestions(buffer, corrupted_text, corrections); + + if (spec.label == options_.RuleLevelLabel()) { + RecordRuleNames(deleted_tokens, suggestions); + } + + TestClassification classification = + ClassifyTrial(deleted_tokens, suggestions); + if (classification == TestClassification::Incorrect) { + RecordWrongCloseDistances(buffer, deleted_tokens, suggestions); + } + MaybeDumpTrial(file, spec, classification, corrupted_text, deleted_tokens, + suggestions, corrections); + return classification; +} + +auto Evaluator::RecordRuleNames(llvm::ArrayRef deleted_tokens, + llvm::ArrayRef suggestions) + -> void { + for (const auto& sugg : suggestions) { + auto& stat = report_.stats_by_rule[sugg.rule_name]; + ++(AnyDeletionMatches(deleted_tokens, sugg) ? stat.correct + : stat.incorrect); + } +} + +auto Evaluator::RecordWrongCloseDistances( + const TokenizedBuffer& buffer, llvm::ArrayRef deleted_tokens, + llvm::ArrayRef suggestions) -> void { + // Ground truth and suggestions meet in byte offsets, but distances are + // measured in tokens, so map back. An offset that isn't a token boundary is + // treated as the end of the file. + llvm::DenseMap offset_to_token; + for (TokenIndex t : buffer.tokens()) { + offset_to_token[buffer.GetByteOffset(t)] = t.index; + } + int32_t eof_index = buffer.size() - 1; + auto to_token = [&](int32_t offset) -> int32_t { + auto it = offset_to_token.find(offset); + return it != offset_to_token.end() ? it->second : eof_index; + }; + + for (const auto& del : deleted_tokens) { + if (AnySuggestionMatches(suggestions, del)) { + continue; + } + auto& stat = report_.wrong_close_by_kind[del.kind.name().str()]; + // Blame the nearest same-kind close, as the suggestion that most likely + // ended the group in the wrong place. + int32_t expected = to_token(del.next_token_byte_offset); + const Suggestion* best = nullptr; + for (const auto& sugg : suggestions) { + if (sugg.kind != del.kind) { + continue; + } + if (best == nullptr || + std::abs(to_token(sugg.byte_offset) - expected) < + std::abs(to_token(best->byte_offset) - expected)) { + best = &sugg; + } + } + if (best == nullptr) { + ++stat.no_close; + continue; + } + int32_t token_dist = to_token(best->byte_offset) - expected; + ++(token_dist > 0 ? stat.later : stat.earlier); + ++stat.token_dist_hist[token_dist]; + } +} + +auto Evaluator::MaybeDumpTrial(const FileContext& file, const DSpec& spec, + TestClassification classification, + llvm::StringRef corrupted_text, + llvm::ArrayRef deleted_tokens, + llvm::ArrayRef suggestions, + llvm::ArrayRef corrections) + -> void { + const char* dump_label = nullptr; + if (classification == TestClassification::Incorrect && dump_incorrect_ > 0) { + --dump_incorrect_; + dump_label = "INCORRECT"; + } else if ((classification == TestClassification::None || + classification == TestClassification::Partial) && + dump_none_ > 0) { + --dump_none_; + dump_label = + classification == TestClassification::None ? "NONE" : "PARTIAL"; + } + if (dump_label == nullptr) { + return; + } + + llvm::errs() << llvm::formatv( + R"( +=== {0} TRIAL in {1} (D={2}) === +)", + dump_label, file.candidate.filename, spec.label); + for (const auto& del : deleted_tokens) { + llvm::errs() << llvm::formatv( + " Deleted token: kind={0} at byte={1} (line={2}, col={3})\n", + del.kind.name(), del.byte_offset, del.line, del.column); + } + llvm::errs() << llvm::formatv(" Suggestions ({0}):\n", suggestions.size()); + for (const auto& s : suggestions) { + llvm::errs() << llvm::formatv( + " Suggestion ({0}): kind={1} at byte={2} (line={3}, col={4})\n", + s.rule_name, s.kind.name(), s.byte_offset, s.line, s.column); + } + llvm::errs() << llvm::formatv(" Raw corrections ({0}):\n", + corrections.size()); + for (const auto& c : corrections) { + llvm::StringRef action = + c.fix_action == BracketFixAction::InsertBefore ? "InsertBefore" + : c.fix_action == BracketFixAction::InsertAfter ? "InsertAfter" + : "ReplaceWithError"; + llvm::errs() << llvm::formatv(" {0} kind={1} tok={2}{3} rule={4}\n", + action, c.fix_token_kind.name(), + c.fix_token_index.index, + c.is_tied ? " TIED" : "", c.rule_name); + } + // Center the excerpt on the first deletion, or on the first suggestion when + // there was nothing to delete (a truncation with nothing left open) yet + // recovery suggested something anyway. + int32_t center = 0; + if (!deleted_tokens.empty()) { + center = deleted_tokens.front().byte_offset; + } else if (!suggestions.empty()) { + center = suggestions.front().byte_offset; + } + int32_t print_start = std::max(0, center - 100); + int32_t print_end = + std::min(static_cast(corrupted_text.size()), center + 100); + llvm::errs() << llvm::formatv( + R"(--- Corrupted Text Sample --- +{0} +=============================== + +)", + corrupted_text.substr(print_start, print_end - print_start)); +} + +static auto PrintJsonReport(const EvalOptions& options, const Corpus& corpus, + const Report& report) -> void { + llvm::outs() << llvm::formatv( + R"({{ + "seed": {0}, + "total_trials": {1}, + "files_tested": {2}, + "total_bracket_pairs": {3}, + "scenarios": [)", + options.base_seed, options.total_trials, corpus.files.size(), + corpus.total_clean_pairs); + llvm::ListSeparator sep(","); + for (auto [i, spec] : llvm::enumerate(options.d_specs)) { + const auto& stats = report.stats_by_level[i]; + llvm::outs() << llvm::formatv( + R"({0} + {{ + "d_spec": "{1}", + "total": {2}, + "correct": {3}, + "partial": {4}, + "none": {5}, + "incorrect": {6}, + "correct_pct": {7:F1}, + "partial_pct": {8:F1}, + "none_pct": {9:F1}, + "incorrect_pct": {10:F1}, + "safety_pct": {11:F1}, + "accuracy_pct": {12:F1} + })", + llvm::StringRef(sep), spec.label, stats.total, stats.correct, + stats.partial, stats.none, stats.incorrect, stats.CorrectPct(), + stats.PartialPct(), stats.NonePct(), stats.IncorrectPct(), + stats.SafetyPct(), stats.AccuracyPct()); + } + llvm::outs() << "\n ]\n}\n"; +} + +static auto PrintLevelTable(const EvalOptions& options, const Report& report) + -> void { + llvm::outs() << R"(## Overall Performance by Deletion Level (D) + +| Deletion Level (D) | Total Trials | Correct | Partial | None | Incorrect | Safety (%) | Accuracy (%) | +|:---|---:|---:|---:|---:|---:|---:|---:| +)"; + for (auto [i, spec] : llvm::enumerate(options.d_specs)) { + const auto& stats = report.stats_by_level[i]; + llvm::outs() << llvm::formatv( + "| D = {0,-6} | {1,12} | {2,5} ({3,4:F1}%) | {4,5} ({5,4:F1}%) | {6,5} " + "({7,4:F1}%) | {8,5} ({9,4:F1}%) | {10,9:F1}% | {11,11:F1}% |\n", + spec.label, stats.total, stats.correct, stats.CorrectPct(), + stats.partial, stats.PartialPct(), stats.none, stats.NonePct(), + stats.incorrect, stats.IncorrectPct(), stats.SafetyPct(), + stats.AccuracyPct()); + } + llvm::outs() << "\n"; +} + +static auto PrintRuleTable(const EvalOptions& options, const Report& report) + -> void { + llvm::outs() << llvm::formatv( + R"(## Suggestion Rule Breakdown (D = {0}) + +| Rule | Total | Correct | Incorrect | Precision (%) | +|:---|---:|---:|---:|---:| +)", + options.RuleLevelLabel()); + for (const auto& [name, stat] : report.stats_by_rule) { + int total = stat.correct + stat.incorrect; + double prec = total == 0 ? 100.0 : (100.0 * stat.correct) / total; + llvm::outs() << llvm::formatv( + "| {0,-32} | {1,5} | {2,5} | {3,5} | {4,8:F1}% |\n", name, total, + stat.correct, stat.incorrect, prec); + } + llvm::outs() << "\n"; +} + +static auto PrintPerFileTables(const EvalOptions& options, const Report& report) + -> void { + llvm::outs() << "## Per-File Breakdown\n\n"; + for (const auto& file : report.results_by_file) { + llvm::outs() << llvm::formatv( + R"(### `{0}` ({1} pairs) + +| D | Total | Correct | Partial | None | Incorrect | Safety | Accuracy | +|:---|---:|---:|---:|---:|---:|---:|---:| +)", + file.filename, file.clean_pairs_count); + for (auto [i, spec] : llvm::enumerate(options.d_specs)) { + const auto& stats = file.stats_by_level[i]; + llvm::outs() << llvm::formatv( + "| {0} | {1} | {2} ({3:F1}%) | {4} ({5:F1}%) | {6} ({7:F1}%) | {8} " + "({9:F1}%) | {10:F1}% | {11:F1}% |\n", + spec.label, stats.total, stats.correct, stats.CorrectPct(), + stats.partial, stats.PartialPct(), stats.none, stats.NonePct(), + stats.incorrect, stats.IncorrectPct(), stats.SafetyPct(), + stats.AccuracyPct()); + } + llvm::outs() << "\n"; + } +} + +// Summarizes a distance histogram. The median is the first distance past the +// halfway point and the 90th percentile the last one at or below the 90% mark, +// so both name an observed distance rather than an interpolated one. +static auto SummarizeDistances(const std::map& hist) + -> DistanceSummary { + int count = 0; + DistanceSummary summary; + for (const auto& [dist, n] : hist) { + count += n; + summary.max_abs = std::max(summary.max_abs, std::abs(dist)); + } + bool have_median = false; + int seen = 0; + for (const auto& [dist, n] : hist) { + seen += n; + if (!have_median && seen > count / 2) { + summary.median = dist; + have_median = true; + } + if (seen <= count * 9 / 10) { + summary.p90 = dist; + } + } + return summary; +} + +static auto PrintWrongCloseTable(const Report& report) -> void { + llvm::outs() << R"(## Wrong-Close Distance (incorrect trials) + +Signed token distance from the correct anchor to the nearest same-kind close (+ = closed later / swallowing). + +| Deleted kind | Wrong | Later | Earlier | No close | Median | P90 | Max | +|:---|---:|---:|---:|---:|---:|---:|---:| +)"; + for (const auto& [kind, stat] : report.wrong_close_by_kind) { + int total = stat.later + stat.earlier + stat.no_close; + DistanceSummary dist = SummarizeDistances(stat.token_dist_hist); + llvm::outs() << llvm::formatv( + "| {0,-18} | {1,5} | {2,5} | {3,7} | {4,8} | {5,6} | {6,4} | {7,4} " + "|\n", + kind, total, stat.later, stat.earlier, stat.no_close, dist.median, + dist.p90, dist.max_abs); + } + llvm::outs() << "\n"; +} + +static auto PrintMetricDefinitions() -> void { + llvm::outs() << R"(## Metric Definitions + +- **Correct**: Suggested correct locations for all removed tokens. +- **Partial**: Suggested correct locations for some removed tokens, and gave no suggestions for others. +- **None**: Gave no suggestions for any removed tokens (e.g., recovered cleanly with errors and no hallucinated notes). +- **Incorrect**: Suggested a location for any removed token that was not where the token was removed from. +- **Safety**: Percentage of trials with no incorrect suggestions `(Correct + Partial + None) / Total`. +- **Accuracy**: Precision of suggestions when suggestions were made `Correct / (Correct + Incorrect)`. +)"; +} + +static auto PrintMarkdownReport(const EvalOptions& options, + const Corpus& corpus, const Report& report) + -> void { + llvm::outs() << llvm::formatv( + R"(# Bracket Recovery Measurement Report + +- **Corruption mode**: {0} +- **Files tested**: {1} files ({2} clean matched bracket pairs) +- **Total trials per configuration**: {3} +- **Random seed**: {4} +)", + options.mode_name, corpus.files.size(), corpus.total_clean_pairs, + options.total_trials, options.base_seed); + if (report.merged_skips > 0) { + llvm::outs() << llvm::formatv("- **Trials skipped (token fusion)**: {0}\n", + report.merged_skips); + } + llvm::outs() << "\n"; + + PrintLevelTable(options, report); + PrintRuleTable(options, report); + if (options.verbose) { + PrintPerFileTables(options, report); + } + if (!report.wrong_close_by_kind.empty()) { + PrintWrongCloseTable(report); + } + PrintMetricDefinitions(); +} + +constexpr CommandLine::CommandInfo CommandInfo = { + .name = "mismatched_brackets_eval", + .help = R"""( +A measurement and benchmarking tool for Carbon bracket recovery. + +Evaluates how accurately and safely the bracket error recovery algorithm +recovers deleted subsets of brackets across Carbon source files. +)""", +}; + +static auto AddOptions(CommandLine::CommandBuilder& b, EvalOptions& options) + -> void { + b.AddStringPositionalArg( + { + .name = "FILE", + .help = "Input Carbon source file(s) or directories to test.", + }, + [&](auto& arg_b) { arg_b.Append(&options.input_files); }); + + b.AddStringOption( + { + .name = "d-values", + .value_name = "LIST", + .help = "Comma-separated deletion levels (e.g. '1,2,5,10%,25%'). " + "Ignored by the truncate modes.", + }, + [&](auto& arg_b) { arg_b.Set(&options.d_values); }); + + b.AddStringOption( + { + .name = "mode", + .value_name = "MODE", + .help = "How to corrupt each file: 'blank' (replace brackets with " + "spaces; the default), 'gapless' (delete bracket characters, " + "leaving no space behind), 'truncate' (cut the file at a " + "random token; recovery should close all open brackets at " + "EOF), or 'truncate-region' (delete from inside a random " + "pair through its close, as when typing new code in an " + "existing class).", + }, + [&](auto& arg_b) { arg_b.Set(&options.mode_name); }); + + b.AddIntegerOption( + { + .name = "trials", + .value_name = "N", + .help = "Total number of trials per D configuration.", + }, + [&](auto& arg_b) { arg_b.Set(&options.total_trials); }); + + b.AddIntegerOption( + { + .name = "seed", + .value_name = "N", + .help = "Random seed for deterministic sampling.", + }, + [&](auto& arg_b) { arg_b.Set(&options.base_seed); }); + + b.AddFlag( + { + .name = "verbose", + .help = "Print detailed per-file results.", + }, + [&](auto& arg_b) { arg_b.Set(&options.verbose); }); + + b.AddFlag( + { + .name = "json", + .help = "Output results in JSON format.", + }, + [&](auto& arg_b) { arg_b.Set(&options.json_output); }); + + b.AddIntegerOption( + { + .name = "dump-incorrect", + .value_name = "N", + .help = "Print details for up to N incorrect trials.", + }, + [&](auto& arg_b) { arg_b.Set(&options.dump_incorrect); }); + + b.AddIntegerOption( + { + .name = "dump-none", + .value_name = "N", + .help = "Print details for up to N trials classified None.", + }, + [&](auto& arg_b) { arg_b.Set(&options.dump_none); }); + + b.Do([] {}); +} + +static auto Run(llvm::ArrayRef args) -> bool { + EvalOptions options; + auto parse_result = CommandLine::Parse( + args, llvm::outs(), CommandInfo, + [&](CommandLine::CommandBuilder& b) { AddOptions(b, options); }); + if (!parse_result.ok()) { + llvm::errs() << "error: " << *parse_result << "\n"; + return false; + } + if (*parse_result == CommandLine::ParseResult::MetaSuccess) { + return true; + } + if (!options.Resolve()) { + return false; + } + + auto files = CollectCarbonFiles(options.input_files); + if (files.empty()) { + llvm::errs() << "error: No Carbon source files found to test.\n"; + return false; + } + Corpus corpus = FindCandidateFiles(files); + if (corpus.files.empty() || corpus.total_clean_pairs == 0) { + llvm::errs() + << "error: No Carbon source files with bracket pairs found to test.\n"; + return false; + } + + Report report = Evaluator(options, corpus).Run(); + if (options.json_output) { + PrintJsonReport(options, corpus, report); + } else { + PrintMarkdownReport(options, corpus, report); + } + return true; +} + +} // namespace Carbon::Lex + +auto main(int argc, char** argv) -> int { + Carbon::InitLLVM init_llvm(argc, argv); + Carbon::SetWorkingDirForBazelRun(); + llvm::SmallVector args(argv + 1, argv + argc); + bool success = Carbon::Lex::Run(args); + return success ? EXIT_SUCCESS : EXIT_FAILURE; +} diff --git a/toolchain/lex/mismatched_brackets_fuzzer.cpp b/toolchain/lex/mismatched_brackets_fuzzer.cpp new file mode 100644 index 000000000000..7d872748247a --- /dev/null +++ b/toolchain/lex/mismatched_brackets_fuzzer.cpp @@ -0,0 +1,196 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#include + +#include "common/check.h" +#include "testing/fuzzing/libfuzzer.h" +#include "toolchain/lex/mismatched_brackets.h" + +namespace Carbon::Lex::Testing { +namespace { + +struct Insertion { + TokenIndex anchor; + bool is_after; + size_t order; + TokenKind kind; +}; + +// Verifies that applying all corrections to the input token sequence results in +// a correctly bracket-balanced stream. +auto VerifyBracketBalance(llvm::ArrayRef tokens, + llvm::ArrayRef corrections) + -> void { + llvm::SmallVector is_replaced_with_error(tokens.size(), false); + llvm::SmallVector insertions; + + for (size_t order = 0; order < corrections.size(); ++order) { + const auto& corr = corrections[order]; + if (corr.fix_action == BracketFixAction::ReplaceWithError) { + is_replaced_with_error[corr.fix_token_index.index] = true; + } else if (corr.fix_action == BracketFixAction::InsertBefore) { + insertions.push_back({ + .anchor = corr.fix_token_index, + .is_after = false, + .order = order, + .kind = corr.fix_token_kind, + }); + } else if (corr.fix_action == BracketFixAction::InsertAfter) { + insertions.push_back({ + .anchor = corr.fix_token_index, + .is_after = true, + .order = order, + .kind = corr.fix_token_kind, + }); + } + } + + // This must match `ErrorRecoveryBuffer::Apply` in lex.cpp, which is how the + // corrections are actually applied to the token stream. In particular, at a + // shared insertion point, closing brackets are inserted before opening + // brackets, so that closing an outer group and opening a new one land in + // the correct order. + llvm::stable_sort(insertions, [](const Insertion& a, const Insertion& b) { + TokenIndex a_target = + a.is_after ? TokenIndex(a.anchor.index + 1) : a.anchor; + TokenIndex b_target = + b.is_after ? TokenIndex(b.anchor.index + 1) : b.anchor; + if (a_target != b_target) { + return a_target < b_target; + } + if (a.is_after != b.is_after) { + return a.is_after; + } + bool a_is_closing = a.kind.is_closing_symbol(); + bool b_is_closing = b.kind.is_closing_symbol(); + if (a_is_closing != b_is_closing) { + return a_is_closing; + } + if (a.is_after) { + return a.order < b.order; + } else { + return a.order > b.order; + } + }); + + llvm::SmallVector resulting_stream; + size_t ins_idx = 0; + + for (int32_t i = 0; i <= static_cast(tokens.size()); ++i) { + while (ins_idx < insertions.size()) { + TokenIndex target = insertions[ins_idx].is_after + ? TokenIndex(insertions[ins_idx].anchor.index + 1) + : insertions[ins_idx].anchor; + if (target.index != i) { + break; + } + resulting_stream.push_back(insertions[ins_idx].kind); + ++ins_idx; + } + + if (i < static_cast(tokens.size())) { + if (!is_replaced_with_error[i]) { + resulting_stream.push_back(ToTokenKind(tokens[i].kind)); + } + } + } + + llvm::SmallVector stack; + for (TokenKind kind : resulting_stream) { + if (kind.is_opening_symbol()) { + stack.push_back(kind); + } else if (kind.is_closing_symbol()) { + CARBON_CHECK(!stack.empty(), + "Unmatched closing bracket in fixed stream!"); + TokenKind top = stack.pop_back_val(); + CARBON_CHECK(top.closing_symbol() == kind, + "Mismatched bracket pair in fixed stream!"); + } + } + CARBON_CHECK(stack.empty(), + "Unclosed opening brackets remaining in fixed stream!"); +} + +} // namespace + +// Fuzz tester for mismatched bracket recovery. +// NOLINTNEXTLINE: Match the documented fuzzer entry point declaration style. +extern "C" int LLVMFuzzerTestOneInput(const unsigned char* data, size_t size) { + if (size > 2000) { + return 0; + } + + // Every kind except `FileEnd`, which only ever appears as the final token. + constexpr auto NumGeneratedKinds = + static_cast(BracketTokenKind::Other); + static_assert(static_cast(BracketTokenKind::FileEnd) == + NumGeneratedKinds - 1); + + // Bytes per generated token: kind, indentation, line advance, and flags. Most + // recovery cues come from the kind and the flags, so leaving either coarse + // would make most of the rules unreachable. + constexpr size_t BytesPerToken = 4; + + llvm::SmallVector tokens; + tokens.reserve(size / BytesPerToken); + + size_t i = 0; + int32_t token_idx = 0; + int32_t current_line = 1; + + while (i + BytesPerToken <= size) { + uint8_t kind_byte = data[i++]; + uint8_t indent_byte = data[i++]; + uint8_t line_delta = data[i++]; + uint8_t flags_byte = data[i++]; + + auto kind = static_cast(kind_byte % NumGeneratedKinds); + if (kind == BracketTokenKind::FileEnd) { + kind = BracketTokenKind::Other; + } + int32_t indent = (indent_byte % 32) * 2; + current_line += line_delta % 3; + + tokens.push_back(MismatchedBracketToken{ + .token_index = TokenIndex(token_idx++), + .kind = kind, + .line = current_line, + .line_indent = indent, + .is_at_end_of_line = (flags_byte & 1) != 0, + .is_struct_brace = (flags_byte & 2) != 0, + .is_paren_keyword = (flags_byte & 8) != 0, + .is_else_keyword = (flags_byte & 16) != 0, + .has_leading_space = (flags_byte & 32) != 0, + .has_wide_leading_space = (flags_byte & 64) != 0, + }); + } + + tokens.push_back(MismatchedBracketToken{ + .token_index = TokenIndex(token_idx++), + .kind = BracketTokenKind::FileEnd, + .line = current_line, + .line_indent = 0, + .is_at_end_of_line = true, + }); + + auto corrections = FixMismatchedBrackets(tokens); + + // Invariant verification: all indices must be valid. + for (const auto& corr : corrections) { + CARBON_CHECK(corr.diagnostic_token_index.index >= 0 && + corr.diagnostic_token_index.index < token_idx, + "Invalid diag token index!"); + CARBON_CHECK(corr.fix_token_index.index >= 0 && + corr.fix_token_index.index < token_idx, + "Invalid fix token index!"); + } + + // Verification: applying fixes must result in a balanced bracket sequence. + VerifyBracketBalance(tokens, corrections); + + return 0; +} + +} // namespace Carbon::Lex::Testing diff --git a/toolchain/lex/mismatched_brackets_test.cpp b/toolchain/lex/mismatched_brackets_test.cpp new file mode 100644 index 000000000000..25860ba8d1b6 --- /dev/null +++ b/toolchain/lex/mismatched_brackets_test.cpp @@ -0,0 +1,330 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#include "toolchain/lex/mismatched_brackets.h" + +#include +#include + +namespace Carbon::Lex { +namespace { + +using ::testing::SizeIs; + +class MismatchedBracketsTest : public ::testing::Test { + protected: + auto MakeToken(int32_t index, BracketTokenKind kind, int32_t line, + int32_t indent, bool is_at_end_of_line = false, + bool is_struct_brace = false) -> MismatchedBracketToken { + return MismatchedBracketToken{ + .token_index = TokenIndex(index), + .kind = kind, + .line = line, + .line_indent = indent, + .is_at_end_of_line = is_at_end_of_line, + .is_struct_brace = is_struct_brace, + }; + } +}; + +TEST_F(MismatchedBracketsTest, HandlesEmptyTokens) { + llvm::SmallVector tokens; + auto corrections = FixMismatchedBrackets(tokens); + EXPECT_TRUE(corrections.empty()); +} + +TEST_F(MismatchedBracketsTest, HandlesWellBalancedCode) { + // 1 fn F() { + // 2 if (x) { + // 3 y; + // 4 } + // 5 } + llvm::SmallVector tokens = { + MakeToken(0, BracketTokenKind::StatementIntroducer, 1, 1), + MakeToken(1, BracketTokenKind::Other, 1, 1), + MakeToken(2, BracketTokenKind::OpenParen, 1, 1), + MakeToken(3, BracketTokenKind::CloseParen, 1, 1), + MakeToken(4, BracketTokenKind::OpenCurlyBrace, 1, 1, + /*is_at_end_of_line=*/true), + MakeToken(5, BracketTokenKind::StatementIntroducer, 2, 3), + MakeToken(6, BracketTokenKind::OpenParen, 2, 3), + MakeToken(7, BracketTokenKind::Other, 2, 3), + MakeToken(8, BracketTokenKind::CloseParen, 2, 3), + MakeToken(9, BracketTokenKind::OpenCurlyBrace, 2, 3, + /*is_at_end_of_line=*/true), + MakeToken(10, BracketTokenKind::Other, 3, 5), + MakeToken(11, BracketTokenKind::Semi, 3, 5, /*is_at_end_of_line=*/true), + MakeToken(12, BracketTokenKind::CloseCurlyBrace, 4, 3, + /*is_at_end_of_line=*/true), + MakeToken(13, BracketTokenKind::CloseCurlyBrace, 5, 1, + /*is_at_end_of_line=*/true), + }; + + auto corrections = FixMismatchedBrackets(tokens); + EXPECT_TRUE(corrections.empty()); +} + +TEST_F(MismatchedBracketsTest, FixesMissingOpenBraceAfterIf) { + // 1 fn F() { + // 2 if (thing1) + // 3 thing2; + // 4 } + // 5 } + llvm::SmallVector tokens = { + MakeToken(0, BracketTokenKind::StatementIntroducer, 1, 1), + MakeToken(1, BracketTokenKind::Other, 1, 1), + MakeToken(2, BracketTokenKind::OpenParen, 1, 1), + MakeToken(3, BracketTokenKind::CloseParen, 1, 1), + MakeToken(4, BracketTokenKind::OpenCurlyBrace, 1, 1, + /*is_at_end_of_line=*/true), + MakeToken(5, BracketTokenKind::StatementIntroducer, 2, 3), + MakeToken(6, BracketTokenKind::OpenParen, 2, 3), + MakeToken(7, BracketTokenKind::Other, 2, 3), + MakeToken(8, BracketTokenKind::CloseParen, 2, 3, + /*is_at_end_of_line=*/true), + MakeToken(9, BracketTokenKind::Other, 3, 5), + MakeToken(10, BracketTokenKind::Semi, 3, 5, /*is_at_end_of_line=*/true), + MakeToken(11, BracketTokenKind::CloseCurlyBrace, 4, 3, + /*is_at_end_of_line=*/true), + MakeToken(12, BracketTokenKind::CloseCurlyBrace, 5, 1, + /*is_at_end_of_line=*/true), + }; + + auto corrections = FixMismatchedBrackets(tokens); + EXPECT_FALSE(corrections.empty()); + + bool inserted_open_brace = false; + for (const auto& corr : corrections) { + if (corr.fix_action == BracketFixAction::InsertBefore && + corr.fix_token_kind == TokenKind::OpenCurlyBrace) { + inserted_open_brace = true; + // Should be inserted before line 3 tokens (e.g. token 9). + EXPECT_EQ(corr.fix_token_index.index, 9); + EXPECT_EQ(corr.diagnostic_kind, BracketDiagnosticKind::UnmatchedClosing); + } + } + EXPECT_TRUE(inserted_open_brace); +} + +TEST_F(MismatchedBracketsTest, HandlesMultiLineDeclarationHeader) { + // 1 fn F[T: type] + // 2 (x: T) { + // 3 foo(); + // 4 } + llvm::SmallVector tokens = { + MakeToken(0, BracketTokenKind::StatementIntroducer, 1, 1), + MakeToken(1, BracketTokenKind::Other, 1, 1), + MakeToken(2, BracketTokenKind::OpenSquareBracket, 1, 1), + MakeToken(3, BracketTokenKind::Other, 1, 1), + MakeToken(4, BracketTokenKind::CloseSquareBracket, 1, 1, + /*is_at_end_of_line=*/true), + MakeToken(5, BracketTokenKind::OpenParen, 2, 5), + MakeToken(6, BracketTokenKind::Other, 2, 5), + MakeToken(7, BracketTokenKind::CloseParen, 2, 5), + MakeToken(8, BracketTokenKind::OpenCurlyBrace, 2, 5, + /*is_at_end_of_line=*/true), + MakeToken(9, BracketTokenKind::Other, 3, 3), + MakeToken(10, BracketTokenKind::OpenParen, 3, 3), + MakeToken(11, BracketTokenKind::CloseParen, 3, 3), + MakeToken(12, BracketTokenKind::Semi, 3, 3, /*is_at_end_of_line=*/true), + MakeToken(13, BracketTokenKind::CloseCurlyBrace, 4, 1, + /*is_at_end_of_line=*/true), + }; + + auto corrections = FixMismatchedBrackets(tokens); + EXPECT_TRUE(corrections.empty()); +} + +TEST_F(MismatchedBracketsTest, FixesOmittedOpenBraceWithMultiLineHeader) { + // 1 fn F[T: type] + // 2 (x: T) + // 3 foo(); + // 4 bar(); + // 5 } + llvm::SmallVector tokens = { + MakeToken(0, BracketTokenKind::StatementIntroducer, 1, 1), + MakeToken(1, BracketTokenKind::Other, 1, 1), + MakeToken(2, BracketTokenKind::OpenSquareBracket, 1, 1), + MakeToken(3, BracketTokenKind::Other, 1, 1), + MakeToken(4, BracketTokenKind::CloseSquareBracket, 1, 1, + /*is_at_end_of_line=*/true), + MakeToken(5, BracketTokenKind::OpenParen, 2, 5), + MakeToken(6, BracketTokenKind::Other, 2, 5), + MakeToken(7, BracketTokenKind::CloseParen, 2, 5, + /*is_at_end_of_line=*/true), + MakeToken(8, BracketTokenKind::Other, 3, 3), + MakeToken(9, BracketTokenKind::Semi, 3, 3, /*is_at_end_of_line=*/true), + MakeToken(10, BracketTokenKind::Other, 4, 3), + MakeToken(11, BracketTokenKind::Semi, 4, 3, /*is_at_end_of_line=*/true), + MakeToken(12, BracketTokenKind::CloseCurlyBrace, 5, 1, + /*is_at_end_of_line=*/true), + }; + + auto corrections = FixMismatchedBrackets(tokens); + EXPECT_FALSE(corrections.empty()); + + bool inserted_open_brace = false; + for (const auto& corr : corrections) { + if (corr.fix_action == BracketFixAction::InsertBefore && + corr.fix_token_kind == TokenKind::OpenCurlyBrace) { + inserted_open_brace = true; + // Should be inserted before line 3 statement (token 8). + EXPECT_EQ(corr.fix_token_index.index, 8); + } + } + EXPECT_TRUE(inserted_open_brace); +} + +TEST_F(MismatchedBracketsTest, HandlesUnmatchedClosingBrace) { + llvm::SmallVector tokens = { + MakeToken(0, BracketTokenKind::CloseCurlyBrace, 1, 1, + /*is_at_end_of_line=*/true), + }; + + auto corrections = FixMismatchedBrackets(tokens); + ASSERT_THAT(corrections, SizeIs(1)); + EXPECT_EQ(corrections[0].diagnostic_kind, + BracketDiagnosticKind::UnmatchedClosing); + EXPECT_EQ(corrections[0].diagnostic_token_index.index, 0); + EXPECT_EQ(corrections[0].fix_action, BracketFixAction::ReplaceWithError); + EXPECT_EQ(corrections[0].fix_token_index.index, 0); +} + +TEST_F(MismatchedBracketsTest, HandlesUnclosedOpeningBraceAtEOF) { + llvm::SmallVector tokens = { + MakeToken(0, BracketTokenKind::OpenCurlyBrace, 1, 1, + /*is_at_end_of_line=*/true), + }; + + auto corrections = FixMismatchedBrackets(tokens); + ASSERT_THAT(corrections, SizeIs(1)); + EXPECT_EQ(corrections[0].diagnostic_kind, + BracketDiagnosticKind::UnmatchedOpening); + EXPECT_EQ(corrections[0].diagnostic_token_index.index, 0); +} + +TEST_F(MismatchedBracketsTest, MissingClosingBraceBeforeSiblingFunction) { + // 1 class Grid { + // 2 fn Check4() { + // 3 return; + // // missing } + // 4 fn Check3() { + // 5 return; + // 6 } + // 7 } + llvm::SmallVector tokens = { + MakeToken(0, BracketTokenKind::StatementIntroducer, 1, 1), + MakeToken(1, BracketTokenKind::OpenCurlyBrace, 1, 1, + /*is_at_end_of_line=*/true), + MakeToken(2, BracketTokenKind::StatementIntroducer, 2, 3), + MakeToken(3, BracketTokenKind::OpenParen, 2, 3), + MakeToken(4, BracketTokenKind::CloseParen, 2, 3), + MakeToken(5, BracketTokenKind::OpenCurlyBrace, 2, 3, + /*is_at_end_of_line=*/true), + MakeToken(6, BracketTokenKind::StatementIntroducer, 3, 5), + MakeToken(7, BracketTokenKind::Semi, 3, 5, /*is_at_end_of_line=*/true), + MakeToken(8, BracketTokenKind::StatementIntroducer, 4, 3), + MakeToken(9, BracketTokenKind::OpenParen, 4, 3), + MakeToken(10, BracketTokenKind::CloseParen, 4, 3), + MakeToken(11, BracketTokenKind::OpenCurlyBrace, 4, 3, + /*is_at_end_of_line=*/true), + MakeToken(12, BracketTokenKind::StatementIntroducer, 5, 5), + MakeToken(13, BracketTokenKind::Semi, 5, 5, /*is_at_end_of_line=*/true), + MakeToken(14, BracketTokenKind::CloseCurlyBrace, 6, 3, + /*is_at_end_of_line=*/true), + MakeToken(15, BracketTokenKind::CloseCurlyBrace, 7, 1, + /*is_at_end_of_line=*/true), + }; + + auto corrections = FixMismatchedBrackets(tokens); + ASSERT_FALSE(corrections.empty()); + EXPECT_EQ(corrections[0].fix_action, BracketFixAction::InsertBefore); + EXPECT_EQ(corrections[0].fix_token_index.index, + 8); // Should insert before token 8 (fn Check3). +} + +TEST_F(MismatchedBracketsTest, PathologicalInputRecoversSafely) { + // Alternating `(` and `}`, so no bracket can ever match. The small case runs + // through the search; the large one exceeds the region size limit and takes + // the naive fallback. Either way recovery has to account for every bracket + // exactly once, and must not invent an out-of-range fix. + for (int32_t num_tokens : {200, 2000}) { + llvm::SmallVector tokens; + for (int32_t i = 0; i < num_tokens; ++i) { + tokens.push_back(MakeToken(i, + (i % 2 == 0) + ? BracketTokenKind::OpenParen + : BracketTokenKind::CloseCurlyBrace, + i + 1, (i % 4) * 2)); + } + + auto corrections = FixMismatchedBrackets(tokens); + EXPECT_THAT(corrections, SizeIs(num_tokens)); + llvm::SmallVector diagnosed(num_tokens, false); + for (const auto& correction : corrections) { + ASSERT_GE(correction.diagnostic_token_index.index, 0); + ASSERT_LT(correction.diagnostic_token_index.index, num_tokens); + ASSERT_GE(correction.fix_token_index.index, 0); + ASSERT_LT(correction.fix_token_index.index, num_tokens); + EXPECT_FALSE(diagnosed[correction.diagnostic_token_index.index]); + diagnosed[correction.diagnostic_token_index.index] = true; + } + } +} + +TEST_F(MismatchedBracketsTest, FixesMissingOpenParenAfterIf) { + // 1 if x) { + // 2 y; + // 3 } + llvm::SmallVector tokens = { + MakeToken(0, BracketTokenKind::StatementIntroducer, 1, 1), + MakeToken(1, BracketTokenKind::Leaf, 1, 1), + MakeToken(2, BracketTokenKind::CloseParen, 1, 1), + MakeToken(3, BracketTokenKind::OpenCurlyBrace, 1, 1, + /*is_at_end_of_line=*/true), + MakeToken(4, BracketTokenKind::Leaf, 2, 3), + MakeToken(5, BracketTokenKind::Semi, 2, 3, /*is_at_end_of_line=*/true), + MakeToken(6, BracketTokenKind::CloseCurlyBrace, 3, 1, + /*is_at_end_of_line=*/true), + }; + // `if` must be directly followed by `(`. + tokens[0].is_paren_keyword = true; + tokens[1].has_leading_space = true; + + auto corrections = FixMismatchedBrackets(tokens); + ASSERT_THAT(corrections, SizeIs(1)); + EXPECT_EQ(corrections[0].fix_action, BracketFixAction::InsertBefore); + EXPECT_EQ(corrections[0].fix_token_kind, TokenKind::OpenParen); + EXPECT_EQ(corrections[0].fix_token_index.index, 1); + EXPECT_FALSE(corrections[0].is_tied); +} + +TEST_F(MismatchedBracketsTest, FixesMissingCloseParenBeforeSemi) { + // 1 fn F() { + // 2 G(x; + // 3 } + llvm::SmallVector tokens = { + MakeToken(0, BracketTokenKind::StatementIntroducer, 1, 1), + MakeToken(1, BracketTokenKind::OpenParen, 1, 1), + MakeToken(2, BracketTokenKind::CloseParen, 1, 1), + MakeToken(3, BracketTokenKind::OpenCurlyBrace, 1, 1, + /*is_at_end_of_line=*/true), + MakeToken(4, BracketTokenKind::Leaf, 2, 3), + MakeToken(5, BracketTokenKind::OpenParen, 2, 3), + MakeToken(6, BracketTokenKind::Leaf, 2, 3), + MakeToken(7, BracketTokenKind::Semi, 2, 3, /*is_at_end_of_line=*/true), + MakeToken(8, BracketTokenKind::CloseCurlyBrace, 3, 1, + /*is_at_end_of_line=*/true), + }; + // A `;` can't appear inside parens, so the `)` belongs directly before it. + auto corrections = FixMismatchedBrackets(tokens); + ASSERT_THAT(corrections, SizeIs(1)); + EXPECT_EQ(corrections[0].fix_action, BracketFixAction::InsertBefore); + EXPECT_EQ(corrections[0].fix_token_kind, TokenKind::CloseParen); + EXPECT_EQ(corrections[0].fix_token_index.index, 7); + EXPECT_FALSE(corrections[0].is_tied); +} + +} // namespace +} // namespace Carbon::Lex diff --git a/toolchain/lex/testdata/fail_mismatched_brackets.carbon b/toolchain/lex/testdata/fail_mismatched_brackets.carbon index caf243fa210a..80674022ef5a 100644 --- a/toolchain/lex/testdata/fail_mismatched_brackets.carbon +++ b/toolchain/lex/testdata/fail_mismatched_brackets.carbon @@ -1,34 +1,628 @@ // Part of the Carbon Language project, under the Apache License v2.0 with LLVM // Exceptions. See /LICENSE for license information. // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception - // AUTOUPDATE // TIP: To test this file alone, run: // TIP: bazel test //toolchain/testing:file_test --test_arg=--file_tests=toolchain/lex/testdata/fail_mismatched_brackets.carbon // TIP: To dump output, run: // TIP: bazel run //toolchain/testing:file_test -- --dump_output --file_tests=toolchain/lex/testdata/fail_mismatched_brackets.carbon -// CHECK:STDOUT: - filename: fail_mismatched_brackets.carbon + +// Recovery picks the cheapest way to make a damaged region well-bracketed, so +// each case below is really about which cue makes one repair cheaper than the +// others. Cases where the repair is a `}` live in +// `fail_mismatched_brackets_close_brace.carbon` instead, because those notes +// point at the following line and so can't have `CHECK`s written there. + +// --- fail_top_level_brackets.carbon +// CHECK:STDOUT: - filename: fail_top_level_brackets.carbon // CHECK:STDOUT: tokens: -// CHECK:STDERR: fail_mismatched_brackets.carbon:[[@LINE+4]]:1: error: closing symbol without a corresponding opening symbol [UnmatchedClosing] +// The `}`s here have no plausible opener, so they're replaced with error tokens +// rather than guessed at. The trailing `[` is closed immediately, since there's +// nothing left for it to contain. + +// CHECK:STDERR: fail_top_level_brackets.carbon:[[@LINE+4]]:1: error: closing symbol without a corresponding opening symbol [UnmatchedClosing] // CHECK:STDERR: } // CHECK:STDERR: ^ // CHECK:STDERR: } -// CHECK:STDOUT: - { index: 1, kind: "Error", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", has_leading_space: true } +// CHECK:STDOUT: - { index: 1, kind: "Error", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", has_leading_space: true } -// CHECK:STDERR: fail_mismatched_brackets.carbon:[[@LINE+4]]:3: error: closing symbol without a corresponding opening symbol [UnmatchedClosing] +// CHECK:STDERR: fail_top_level_brackets.carbon:[[@LINE+4]]:3: error: closing symbol without a corresponding opening symbol [UnmatchedClosing] // CHECK:STDERR: ( } ) // CHECK:STDERR: ^ // CHECK:STDERR: ( } ) -// CHECK:STDOUT: - { index: 2, kind: "OpenParen", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "(", closing_token: 4, has_leading_space: true } -// CHECK:STDOUT: - { index: 3, kind: "Error", line: {{ *}}[[@LINE-2]], column: 3, indent: 1, spelling: "}", has_leading_space: true } -// CHECK:STDOUT: - { index: 4, kind: "CloseParen", line: {{ *}}[[@LINE-3]], column: 5, indent: 1, spelling: ")", opening_token: 2, has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "OpenParen", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "(", closing_token: 4, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "Error", line: {{ *}}[[@LINE-2]], column: 3, indent: 1, spelling: "}", has_leading_space: true } +// CHECK:STDOUT: - { index: 4, kind: "CloseParen", line: {{ *}}[[@LINE-3]], column: 5, indent: 1, spelling: ")", opening_token: 2, has_leading_space: true } -// CHECK:STDERR: fail_mismatched_brackets.carbon:[[@LINE+4]]:1: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] +// CHECK:STDERR: fail_top_level_brackets.carbon:[[@LINE+7]]:1: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] // CHECK:STDERR: [ // CHECK:STDERR: ^ +// CHECK:STDERR: fail_top_level_brackets.carbon:[[@LINE+4]]:2: note: possibly missing `]` here [PossiblyMissingBracketHere] +// CHECK:STDERR: [ +// CHECK:STDERR: ^ // CHECK:STDERR: [ -// CHECK:STDOUT: - { index: 5, kind: "Error", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "[", has_leading_space: true } +// CHECK:STDOUT: - { index: 5, kind: "OpenSquareBracket", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "[", closing_token: 6, has_leading_space: true } +// CHECK:STDOUT: - { index: 6, kind: "CloseSquareBracket", line: {{ *}}[[@LINE-2]], column: 2, indent: 1, spelling: "]", opening_token: 5, has_leading_space: true, recovery: true } + +// --- fail_semi_in_paren.carbon +// CHECK:STDOUT: - filename: fail_semi_in_paren.carbon +// CHECK:STDOUT: tokens: + +// A `;` ends a statement, so it can't appear inside parens: the group has to +// have ended before it. + +fn F() { +// CHECK:STDOUT: - { index: 1, kind: "Fn", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 4, indent: 1, spelling: "F", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenParen", line: {{ *}}[[@LINE-3]], column: 5, indent: 1, spelling: "(", closing_token: 4 } +// CHECK:STDOUT: - { index: 4, kind: "CloseParen", line: {{ *}}[[@LINE-4]], column: 6, indent: 1, spelling: ")", opening_token: 3 } +// CHECK:STDOUT: - { index: 5, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-5]], column: 8, indent: 1, spelling: "{", closing_token: 17, has_leading_space: true } + // CHECK:STDERR: fail_semi_in_paren.carbon:[[@LINE+7]]:16: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] + // CHECK:STDERR: var x: i32 = (1 + 2; + // CHECK:STDERR: ^ + // CHECK:STDERR: fail_semi_in_paren.carbon:[[@LINE+4]]:22: note: possibly missing `)` here [PossiblyMissingBracketHere] + // CHECK:STDERR: var x: i32 = (1 + 2; + // CHECK:STDERR: ^ + // CHECK:STDERR: + var x: i32 = (1 + 2; + // CHECK:STDOUT: - { index: 6, kind: "Var", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "var", has_leading_space: true } + // CHECK:STDOUT: - { index: 7, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 7, indent: 3, spelling: "x", identifier: 1, has_leading_space: true } + // CHECK:STDOUT: - { index: 8, kind: "Colon", line: {{ *}}[[@LINE-3]], column: 8, indent: 3, spelling: ":" } + // CHECK:STDOUT: - { index: 9, kind: "IntTypeLiteral", line: {{ *}}[[@LINE-4]], column: 10, indent: 3, spelling: "i32", has_leading_space: true } + // CHECK:STDOUT: - { index: 10, kind: "Equal", line: {{ *}}[[@LINE-5]], column: 14, indent: 3, spelling: "=", has_leading_space: true } + // CHECK:STDOUT: - { index: 11, kind: "OpenParen", line: {{ *}}[[@LINE-6]], column: 16, indent: 3, spelling: "(", closing_token: 15, has_leading_space: true } + // CHECK:STDOUT: - { index: 12, kind: "IntLiteral", line: {{ *}}[[@LINE-7]], column: 17, indent: 3, spelling: "1", value: "1" } + // CHECK:STDOUT: - { index: 13, kind: "Plus", line: {{ *}}[[@LINE-8]], column: 19, indent: 3, spelling: "+", has_leading_space: true } + // CHECK:STDOUT: - { index: 14, kind: "IntLiteral", line: {{ *}}[[@LINE-9]], column: 21, indent: 3, spelling: "2", value: "2", has_leading_space: true } + // CHECK:STDOUT: - { index: 15, kind: "CloseParen", line: {{ *}}[[@LINE-10]], column: 22, indent: 3, spelling: ")", opening_token: 11, recovery: true } + // CHECK:STDOUT: - { index: 16, kind: "Semi", line: {{ *}}[[@LINE-11]], column: 22, indent: 3, spelling: ";" } +} +// CHECK:STDOUT: - { index: 17, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", opening_token: 5, has_leading_space: true } + +// --- fail_scope_in_paren.carbon +// CHECK:STDOUT: - filename: fail_scope_in_paren.carbon +// CHECK:STDOUT: tokens: + +// As above, at the top level rather than inside a function. + +// CHECK:STDERR: fail_scope_in_paren.carbon:[[@LINE+7]]:14: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] +// CHECK:STDERR: var x: i32 = (1 + 2; +// CHECK:STDERR: ^ +// CHECK:STDERR: fail_scope_in_paren.carbon:[[@LINE+4]]:20: note: possibly missing `)` here [PossiblyMissingBracketHere] +// CHECK:STDERR: var x: i32 = (1 + 2; +// CHECK:STDERR: ^ +// CHECK:STDERR: +var x: i32 = (1 + 2; +// CHECK:STDOUT: - { index: 1, kind: "Var", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "var", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 5, indent: 1, spelling: "x", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "Colon", line: {{ *}}[[@LINE-3]], column: 6, indent: 1, spelling: ":" } +// CHECK:STDOUT: - { index: 4, kind: "IntTypeLiteral", line: {{ *}}[[@LINE-4]], column: 8, indent: 1, spelling: "i32", has_leading_space: true } +// CHECK:STDOUT: - { index: 5, kind: "Equal", line: {{ *}}[[@LINE-5]], column: 12, indent: 1, spelling: "=", has_leading_space: true } +// CHECK:STDOUT: - { index: 6, kind: "OpenParen", line: {{ *}}[[@LINE-6]], column: 14, indent: 1, spelling: "(", closing_token: 10, has_leading_space: true } +// CHECK:STDOUT: - { index: 7, kind: "IntLiteral", line: {{ *}}[[@LINE-7]], column: 15, indent: 1, spelling: "1", value: "1" } +// CHECK:STDOUT: - { index: 8, kind: "Plus", line: {{ *}}[[@LINE-8]], column: 17, indent: 1, spelling: "+", has_leading_space: true } +// CHECK:STDOUT: - { index: 9, kind: "IntLiteral", line: {{ *}}[[@LINE-9]], column: 19, indent: 1, spelling: "2", value: "2", has_leading_space: true } +// CHECK:STDOUT: - { index: 10, kind: "CloseParen", line: {{ *}}[[@LINE-10]], column: 20, indent: 1, spelling: ")", opening_token: 6, recovery: true } +// CHECK:STDOUT: - { index: 11, kind: "Semi", line: {{ *}}[[@LINE-11]], column: 20, indent: 1, spelling: ";" } + +fn F() { +// CHECK:STDOUT: - { index: 12, kind: "Fn", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 13, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 4, indent: 1, spelling: "F", identifier: 1, has_leading_space: true } +// CHECK:STDOUT: - { index: 14, kind: "OpenParen", line: {{ *}}[[@LINE-3]], column: 5, indent: 1, spelling: "(", closing_token: 15 } +// CHECK:STDOUT: - { index: 15, kind: "CloseParen", line: {{ *}}[[@LINE-4]], column: 6, indent: 1, spelling: ")", opening_token: 14 } +// CHECK:STDOUT: - { index: 16, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-5]], column: 8, indent: 1, spelling: "{", closing_token: 17, has_leading_space: true } +} +// CHECK:STDOUT: - { index: 17, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", opening_token: 16, has_leading_space: true } + +// --- fail_close_paren_before_brace.carbon +// CHECK:STDOUT: - filename: fail_close_paren_before_brace.carbon +// CHECK:STDOUT: tokens: + +// A `(` can't still be open when a block `{` starts, so an `if` condition's +// paren closes before the brace. + +fn F() { +// CHECK:STDOUT: - { index: 1, kind: "Fn", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 4, indent: 1, spelling: "F", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenParen", line: {{ *}}[[@LINE-3]], column: 5, indent: 1, spelling: "(", closing_token: 4 } +// CHECK:STDOUT: - { index: 4, kind: "CloseParen", line: {{ *}}[[@LINE-4]], column: 6, indent: 1, spelling: ")", opening_token: 3 } +// CHECK:STDOUT: - { index: 5, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-5]], column: 8, indent: 1, spelling: "{", closing_token: 16, has_leading_space: true } + // CHECK:STDERR: fail_close_paren_before_brace.carbon:[[@LINE+7]]:6: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] + // CHECK:STDERR: if (C { + // CHECK:STDERR: ^ + // CHECK:STDERR: fail_close_paren_before_brace.carbon:[[@LINE+4]]:8: note: possibly missing `)` here [PossiblyMissingBracketHere] + // CHECK:STDERR: if (C { + // CHECK:STDERR: ^ + // CHECK:STDERR: + if (C { + // CHECK:STDOUT: - { index: 6, kind: "If", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "if", has_leading_space: true } + // CHECK:STDOUT: - { index: 7, kind: "OpenParen", line: {{ *}}[[@LINE-2]], column: 6, indent: 3, spelling: "(", closing_token: 9, has_leading_space: true } + // CHECK:STDOUT: - { index: 8, kind: "Identifier", line: {{ *}}[[@LINE-3]], column: 7, indent: 3, spelling: "C", identifier: 1 } + // CHECK:STDOUT: - { index: 9, kind: "CloseParen", line: {{ *}}[[@LINE-4]], column: 8, indent: 3, spelling: ")", opening_token: 7, has_leading_space: true, recovery: true } + // CHECK:STDOUT: - { index: 10, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-5]], column: 9, indent: 3, spelling: "{", closing_token: 15 } + G(); + // CHECK:STDOUT: - { index: 11, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 5, indent: 5, spelling: "G", identifier: 2, has_leading_space: true } + // CHECK:STDOUT: - { index: 12, kind: "OpenParen", line: {{ *}}[[@LINE-2]], column: 6, indent: 5, spelling: "(", closing_token: 13 } + // CHECK:STDOUT: - { index: 13, kind: "CloseParen", line: {{ *}}[[@LINE-3]], column: 7, indent: 5, spelling: ")", opening_token: 12 } + // CHECK:STDOUT: - { index: 14, kind: "Semi", line: {{ *}}[[@LINE-4]], column: 8, indent: 5, spelling: ";" } + } + // CHECK:STDOUT: - { index: 15, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "}", opening_token: 10, has_leading_space: true } +} +// CHECK:STDOUT: - { index: 16, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", opening_token: 5, has_leading_space: true } + +// --- fail_close_paren_before_arrow.carbon +// CHECK:STDOUT: - filename: fail_close_paren_before_arrow.carbon +// CHECK:STDOUT: tokens: + +// `->` structures a declaration rather than an expression, so a parameter +// list's paren must have closed before it. + +// CHECK:STDERR: fail_close_paren_before_arrow.carbon:[[@LINE+7]]:5: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] +// CHECK:STDERR: fn F(x: i32 -> i32 { +// CHECK:STDERR: ^ +// CHECK:STDERR: fail_close_paren_before_arrow.carbon:[[@LINE+4]]:12: note: possibly missing `)` here [PossiblyMissingBracketHere] +// CHECK:STDERR: fn F(x: i32 -> i32 { +// CHECK:STDERR: ^ +// CHECK:STDERR: +fn F(x: i32 -> i32 { +// CHECK:STDOUT: - { index: 1, kind: "Fn", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 4, indent: 1, spelling: "F", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenParen", line: {{ *}}[[@LINE-3]], column: 5, indent: 1, spelling: "(", closing_token: 7 } +// CHECK:STDOUT: - { index: 4, kind: "Identifier", line: {{ *}}[[@LINE-4]], column: 6, indent: 1, spelling: "x", identifier: 1 } +// CHECK:STDOUT: - { index: 5, kind: "Colon", line: {{ *}}[[@LINE-5]], column: 7, indent: 1, spelling: ":" } +// CHECK:STDOUT: - { index: 6, kind: "IntTypeLiteral", line: {{ *}}[[@LINE-6]], column: 9, indent: 1, spelling: "i32", has_leading_space: true } +// CHECK:STDOUT: - { index: 7, kind: "CloseParen", line: {{ *}}[[@LINE-7]], column: 12, indent: 1, spelling: ")", opening_token: 3, has_leading_space: true, recovery: true } +// CHECK:STDOUT: - { index: 8, kind: "MinusGreater", line: {{ *}}[[@LINE-8]], column: 13, indent: 1, spelling: "->" } +// CHECK:STDOUT: - { index: 9, kind: "IntTypeLiteral", line: {{ *}}[[@LINE-9]], column: 16, indent: 1, spelling: "i32", has_leading_space: true } +// CHECK:STDOUT: - { index: 10, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-10]], column: 20, indent: 1, spelling: "{", closing_token: 14, has_leading_space: true } + return x; + // CHECK:STDOUT: - { index: 11, kind: "Return", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "return", has_leading_space: true } + // CHECK:STDOUT: - { index: 12, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 10, indent: 3, spelling: "x", identifier: 1, has_leading_space: true } + // CHECK:STDOUT: - { index: 13, kind: "Semi", line: {{ *}}[[@LINE-3]], column: 11, indent: 3, spelling: ";" } +} +// CHECK:STDOUT: - { index: 14, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", opening_token: 10, has_leading_space: true } + +// --- fail_close_paren_at_wide_gap.carbon +// CHECK:STDOUT: - filename: fail_close_paren_at_wide_gap.carbon +// CHECK:STDOUT: tokens: + +// Formatted code has no space before a `)`, so a wide mid-line gap is evidence +// that the closer used to be there. Note that the repair goes in the gap, not +// at the `;`, which would also have been well-bracketed. + +fn F() { +// CHECK:STDOUT: - { index: 1, kind: "Fn", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 4, indent: 1, spelling: "F", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenParen", line: {{ *}}[[@LINE-3]], column: 5, indent: 1, spelling: "(", closing_token: 4 } +// CHECK:STDOUT: - { index: 4, kind: "CloseParen", line: {{ *}}[[@LINE-4]], column: 6, indent: 1, spelling: ")", opening_token: 3 } +// CHECK:STDOUT: - { index: 5, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-5]], column: 8, indent: 1, spelling: "{", closing_token: 13, has_leading_space: true } + // CHECK:STDERR: fail_close_paren_at_wide_gap.carbon:[[@LINE+7]]:4: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] + // CHECK:STDERR: G(1 + 2; + // CHECK:STDERR: ^ + // CHECK:STDERR: fail_close_paren_at_wide_gap.carbon:[[@LINE+4]]:6: note: possibly missing `)` here [PossiblyMissingBracketHere] + // CHECK:STDERR: G(1 + 2; + // CHECK:STDERR: ^ + // CHECK:STDERR: + G(1 + 2; + // CHECK:STDOUT: - { index: 6, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "G", identifier: 1, has_leading_space: true } + // CHECK:STDOUT: - { index: 7, kind: "OpenParen", line: {{ *}}[[@LINE-2]], column: 4, indent: 3, spelling: "(", closing_token: 9 } + // CHECK:STDOUT: - { index: 8, kind: "IntLiteral", line: {{ *}}[[@LINE-3]], column: 5, indent: 3, spelling: "1", value: "1" } + // CHECK:STDOUT: - { index: 9, kind: "CloseParen", line: {{ *}}[[@LINE-4]], column: 6, indent: 3, spelling: ")", opening_token: 7, has_leading_space: true, recovery: true } + // CHECK:STDOUT: - { index: 10, kind: "Plus", line: {{ *}}[[@LINE-5]], column: 8, indent: 3, spelling: "+" } + // CHECK:STDOUT: - { index: 11, kind: "IntLiteral", line: {{ *}}[[@LINE-6]], column: 10, indent: 3, spelling: "2", value: "2", has_leading_space: true } + // CHECK:STDOUT: - { index: 12, kind: "Semi", line: {{ *}}[[@LINE-7]], column: 11, indent: 3, spelling: ";" } +} +// CHECK:STDOUT: - { index: 13, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", opening_token: 5, has_leading_space: true } + +// --- fail_paren_cascade.carbon +// CHECK:STDOUT: - filename: fail_paren_cascade.carbon +// CHECK:STDOUT: tokens: + +// Several groups can end at the same point, so a single `;` closes all three. + +fn F() { +// CHECK:STDOUT: - { index: 1, kind: "Fn", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 4, indent: 1, spelling: "F", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenParen", line: {{ *}}[[@LINE-3]], column: 5, indent: 1, spelling: "(", closing_token: 4 } +// CHECK:STDOUT: - { index: 4, kind: "CloseParen", line: {{ *}}[[@LINE-4]], column: 6, indent: 1, spelling: ")", opening_token: 3 } +// CHECK:STDOUT: - { index: 5, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-5]], column: 8, indent: 1, spelling: "{", closing_token: 22, has_leading_space: true } + // CHECK:STDERR: fail_paren_cascade.carbon:[[@LINE+21]]:17: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] + // CHECK:STDERR: var x: i32 = G(H(I(1; + // CHECK:STDERR: ^ + // CHECK:STDERR: fail_paren_cascade.carbon:[[@LINE+18]]:23: note: possibly missing `)` here [PossiblyMissingBracketHere] + // CHECK:STDERR: var x: i32 = G(H(I(1; + // CHECK:STDERR: ^ + // CHECK:STDERR: + // CHECK:STDERR: fail_paren_cascade.carbon:[[@LINE+14]]:19: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] + // CHECK:STDERR: var x: i32 = G(H(I(1; + // CHECK:STDERR: ^ + // CHECK:STDERR: fail_paren_cascade.carbon:[[@LINE+11]]:23: note: possibly missing `)` here [PossiblyMissingBracketHere] + // CHECK:STDERR: var x: i32 = G(H(I(1; + // CHECK:STDERR: ^ + // CHECK:STDERR: + // CHECK:STDERR: fail_paren_cascade.carbon:[[@LINE+7]]:21: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] + // CHECK:STDERR: var x: i32 = G(H(I(1; + // CHECK:STDERR: ^ + // CHECK:STDERR: fail_paren_cascade.carbon:[[@LINE+4]]:23: note: possibly missing `)` here [PossiblyMissingBracketHere] + // CHECK:STDERR: var x: i32 = G(H(I(1; + // CHECK:STDERR: ^ + // CHECK:STDERR: + var x: i32 = G(H(I(1; + // CHECK:STDOUT: - { index: 6, kind: "Var", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "var", has_leading_space: true } + // CHECK:STDOUT: - { index: 7, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 7, indent: 3, spelling: "x", identifier: 1, has_leading_space: true } + // CHECK:STDOUT: - { index: 8, kind: "Colon", line: {{ *}}[[@LINE-3]], column: 8, indent: 3, spelling: ":" } + // CHECK:STDOUT: - { index: 9, kind: "IntTypeLiteral", line: {{ *}}[[@LINE-4]], column: 10, indent: 3, spelling: "i32", has_leading_space: true } + // CHECK:STDOUT: - { index: 10, kind: "Equal", line: {{ *}}[[@LINE-5]], column: 14, indent: 3, spelling: "=", has_leading_space: true } + // CHECK:STDOUT: - { index: 11, kind: "Identifier", line: {{ *}}[[@LINE-6]], column: 16, indent: 3, spelling: "G", identifier: 2, has_leading_space: true } + // CHECK:STDOUT: - { index: 12, kind: "OpenParen", line: {{ *}}[[@LINE-7]], column: 17, indent: 3, spelling: "(", closing_token: 20 } + // CHECK:STDOUT: - { index: 13, kind: "Identifier", line: {{ *}}[[@LINE-8]], column: 18, indent: 3, spelling: "H", identifier: 3 } + // CHECK:STDOUT: - { index: 14, kind: "OpenParen", line: {{ *}}[[@LINE-9]], column: 19, indent: 3, spelling: "(", closing_token: 19 } + // CHECK:STDOUT: - { index: 15, kind: "Identifier", line: {{ *}}[[@LINE-10]], column: 20, indent: 3, spelling: "I", identifier: 4 } + // CHECK:STDOUT: - { index: 16, kind: "OpenParen", line: {{ *}}[[@LINE-11]], column: 21, indent: 3, spelling: "(", closing_token: 18 } + // CHECK:STDOUT: - { index: 17, kind: "IntLiteral", line: {{ *}}[[@LINE-12]], column: 22, indent: 3, spelling: "1", value: "1" } + // CHECK:STDOUT: - { index: 18, kind: "CloseParen", line: {{ *}}[[@LINE-13]], column: 23, indent: 3, spelling: ")", opening_token: 16, recovery: true } + // CHECK:STDOUT: - { index: 19, kind: "CloseParen", line: {{ *}}[[@LINE-14]], column: 23, indent: 3, spelling: ")", opening_token: 14, recovery: true } + // CHECK:STDOUT: - { index: 20, kind: "CloseParen", line: {{ *}}[[@LINE-15]], column: 23, indent: 3, spelling: ")", opening_token: 12, recovery: true } + // CHECK:STDOUT: - { index: 21, kind: "Semi", line: {{ *}}[[@LINE-16]], column: 23, indent: 3, spelling: ";" } +} +// CHECK:STDOUT: - { index: 22, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", opening_token: 5, has_leading_space: true } + +// --- fail_close_struct_before_semi.carbon +// CHECK:STDOUT: - filename: fail_close_struct_before_semi.carbon +// CHECK:STDOUT: tokens: + +// A `{` introducing a struct literal is tracked separately from a scope brace, +// and closes at the end of the statement. + +fn F() { +// CHECK:STDOUT: - { index: 1, kind: "Fn", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 4, indent: 1, spelling: "F", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenParen", line: {{ *}}[[@LINE-3]], column: 5, indent: 1, spelling: "(", closing_token: 4 } +// CHECK:STDOUT: - { index: 4, kind: "CloseParen", line: {{ *}}[[@LINE-4]], column: 6, indent: 1, spelling: ")", opening_token: 3 } +// CHECK:STDOUT: - { index: 5, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-5]], column: 8, indent: 1, spelling: "{", closing_token: 23, has_leading_space: true } + // CHECK:STDERR: fail_close_struct_before_semi.carbon:[[@LINE+7]]:17: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] + // CHECK:STDERR: var x: auto = {.a = 1, .b = 2; + // CHECK:STDERR: ^ + // CHECK:STDERR: fail_close_struct_before_semi.carbon:[[@LINE+4]]:32: note: possibly missing `}` here [PossiblyMissingBracketHere] + // CHECK:STDERR: var x: auto = {.a = 1, .b = 2; + // CHECK:STDERR: ^ + // CHECK:STDERR: + var x: auto = {.a = 1, .b = 2; + // CHECK:STDOUT: - { index: 6, kind: "Var", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "var", has_leading_space: true } + // CHECK:STDOUT: - { index: 7, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 7, indent: 3, spelling: "x", identifier: 1, has_leading_space: true } + // CHECK:STDOUT: - { index: 8, kind: "Colon", line: {{ *}}[[@LINE-3]], column: 8, indent: 3, spelling: ":" } + // CHECK:STDOUT: - { index: 9, kind: "Auto", line: {{ *}}[[@LINE-4]], column: 10, indent: 3, spelling: "auto", has_leading_space: true } + // CHECK:STDOUT: - { index: 10, kind: "Equal", line: {{ *}}[[@LINE-5]], column: 15, indent: 3, spelling: "=", has_leading_space: true } + // CHECK:STDOUT: - { index: 11, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-6]], column: 17, indent: 3, spelling: "{", closing_token: 21, has_leading_space: true } + // CHECK:STDOUT: - { index: 12, kind: "Period", line: {{ *}}[[@LINE-7]], column: 18, indent: 3, spelling: "." } + // CHECK:STDOUT: - { index: 13, kind: "Identifier", line: {{ *}}[[@LINE-8]], column: 19, indent: 3, spelling: "a", identifier: 2 } + // CHECK:STDOUT: - { index: 14, kind: "Equal", line: {{ *}}[[@LINE-9]], column: 21, indent: 3, spelling: "=", has_leading_space: true } + // CHECK:STDOUT: - { index: 15, kind: "IntLiteral", line: {{ *}}[[@LINE-10]], column: 23, indent: 3, spelling: "1", value: "1", has_leading_space: true } + // CHECK:STDOUT: - { index: 16, kind: "Comma", line: {{ *}}[[@LINE-11]], column: 24, indent: 3, spelling: "," } + // CHECK:STDOUT: - { index: 17, kind: "Period", line: {{ *}}[[@LINE-12]], column: 26, indent: 3, spelling: ".", has_leading_space: true } + // CHECK:STDOUT: - { index: 18, kind: "Identifier", line: {{ *}}[[@LINE-13]], column: 27, indent: 3, spelling: "b", identifier: 3 } + // CHECK:STDOUT: - { index: 19, kind: "Equal", line: {{ *}}[[@LINE-14]], column: 29, indent: 3, spelling: "=", has_leading_space: true } + // CHECK:STDOUT: - { index: 20, kind: "IntLiteral", line: {{ *}}[[@LINE-15]], column: 31, indent: 3, spelling: "2", value: "2", has_leading_space: true } + // CHECK:STDOUT: - { index: 21, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-16]], column: 32, indent: 3, spelling: "}", opening_token: 11, recovery: true } + // CHECK:STDOUT: - { index: 22, kind: "Semi", line: {{ *}}[[@LINE-17]], column: 32, indent: 3, spelling: ";" } +} +// CHECK:STDOUT: - { index: 23, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", opening_token: 5, has_leading_space: true } + +// --- fail_mismatched_kinds.carbon +// CHECK:STDOUT: - filename: fail_mismatched_kinds.carbon +// CHECK:STDOUT: tokens: + +// The `}` can't match the `(`, so the two are repaired separately: a `)` closes +// the paren, and a `{` before the `}` explains it as an empty struct literal. +// Both insertions land at the same point, and each is reported against the +// bracket it repairs. + +fn F() { +// CHECK:STDOUT: - { index: 1, kind: "Fn", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 4, indent: 1, spelling: "F", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenParen", line: {{ *}}[[@LINE-3]], column: 5, indent: 1, spelling: "(", closing_token: 4 } +// CHECK:STDOUT: - { index: 4, kind: "CloseParen", line: {{ *}}[[@LINE-4]], column: 6, indent: 1, spelling: ")", opening_token: 3 } +// CHECK:STDOUT: - { index: 5, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-5]], column: 8, indent: 1, spelling: "{", closing_token: 19, has_leading_space: true } + // CHECK:STDERR: fail_mismatched_kinds.carbon:[[@LINE+14]]:16: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] + // CHECK:STDERR: var x: i32 = (1 + 2}; + // CHECK:STDERR: ^ + // CHECK:STDERR: fail_mismatched_kinds.carbon:[[@LINE+11]]:22: note: possibly missing `)` here [PossiblyMissingBracketHere] + // CHECK:STDERR: var x: i32 = (1 + 2}; + // CHECK:STDERR: ^ + // CHECK:STDERR: + // CHECK:STDERR: fail_mismatched_kinds.carbon:[[@LINE+7]]:22: error: closing symbol without a corresponding opening symbol [UnmatchedClosing] + // CHECK:STDERR: var x: i32 = (1 + 2}; + // CHECK:STDERR: ^ + // CHECK:STDERR: fail_mismatched_kinds.carbon:[[@LINE+4]]:22: note: possibly missing `{` here [PossiblyMissingBracketHere] + // CHECK:STDERR: var x: i32 = (1 + 2}; + // CHECK:STDERR: ^ + // CHECK:STDERR: + var x: i32 = (1 + 2}; + // CHECK:STDOUT: - { index: 6, kind: "Var", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "var", has_leading_space: true } + // CHECK:STDOUT: - { index: 7, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 7, indent: 3, spelling: "x", identifier: 1, has_leading_space: true } + // CHECK:STDOUT: - { index: 8, kind: "Colon", line: {{ *}}[[@LINE-3]], column: 8, indent: 3, spelling: ":" } + // CHECK:STDOUT: - { index: 9, kind: "IntTypeLiteral", line: {{ *}}[[@LINE-4]], column: 10, indent: 3, spelling: "i32", has_leading_space: true } + // CHECK:STDOUT: - { index: 10, kind: "Equal", line: {{ *}}[[@LINE-5]], column: 14, indent: 3, spelling: "=", has_leading_space: true } + // CHECK:STDOUT: - { index: 11, kind: "OpenParen", line: {{ *}}[[@LINE-6]], column: 16, indent: 3, spelling: "(", closing_token: 15, has_leading_space: true } + // CHECK:STDOUT: - { index: 12, kind: "IntLiteral", line: {{ *}}[[@LINE-7]], column: 17, indent: 3, spelling: "1", value: "1" } + // CHECK:STDOUT: - { index: 13, kind: "Plus", line: {{ *}}[[@LINE-8]], column: 19, indent: 3, spelling: "+", has_leading_space: true } + // CHECK:STDOUT: - { index: 14, kind: "IntLiteral", line: {{ *}}[[@LINE-9]], column: 21, indent: 3, spelling: "2", value: "2", has_leading_space: true } + // CHECK:STDOUT: - { index: 15, kind: "CloseParen", line: {{ *}}[[@LINE-10]], column: 22, indent: 3, spelling: ")", opening_token: 11, recovery: true } + // CHECK:STDOUT: - { index: 16, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-11]], column: 22, indent: 3, spelling: "{", closing_token: 17, recovery: true } + // CHECK:STDOUT: - { index: 17, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-12]], column: 22, indent: 3, spelling: "}", opening_token: 16 } + // CHECK:STDOUT: - { index: 18, kind: "Semi", line: {{ *}}[[@LINE-13]], column: 23, indent: 3, spelling: ";" } +} +// CHECK:STDOUT: - { index: 19, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", opening_token: 5, has_leading_space: true } + +// --- fail_missing_open_paren_after_if.carbon +// CHECK:STDOUT: - filename: fail_missing_open_paren_after_if.carbon +// CHECK:STDOUT: tokens: + +// `if` must be followed by `(`, which makes inserting one much cheaper than +// discarding the `)`. + +fn F() { +// CHECK:STDOUT: - { index: 1, kind: "Fn", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 4, indent: 1, spelling: "F", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenParen", line: {{ *}}[[@LINE-3]], column: 5, indent: 1, spelling: "(", closing_token: 4 } +// CHECK:STDOUT: - { index: 4, kind: "CloseParen", line: {{ *}}[[@LINE-4]], column: 6, indent: 1, spelling: ")", opening_token: 3 } +// CHECK:STDOUT: - { index: 5, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-5]], column: 8, indent: 1, spelling: "{", closing_token: 18, has_leading_space: true } + // CHECK:STDERR: fail_missing_open_paren_after_if.carbon:[[@LINE+7]]:11: error: closing symbol without a corresponding opening symbol [UnmatchedClosing] + // CHECK:STDERR: if C > 0) { + // CHECK:STDERR: ^ + // CHECK:STDERR: fail_missing_open_paren_after_if.carbon:[[@LINE+4]]:6: note: possibly missing `(` here [PossiblyMissingBracketHere] + // CHECK:STDERR: if C > 0) { + // CHECK:STDERR: ^ + // CHECK:STDERR: + if C > 0) { + // CHECK:STDOUT: - { index: 6, kind: "If", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "if", has_leading_space: true } + // CHECK:STDOUT: - { index: 7, kind: "OpenParen", line: {{ *}}[[@LINE-2]], column: 5, indent: 3, spelling: "(", closing_token: 11, has_leading_space: true, recovery: true } + // CHECK:STDOUT: - { index: 8, kind: "Identifier", line: {{ *}}[[@LINE-3]], column: 6, indent: 3, spelling: "C", identifier: 1 } + // CHECK:STDOUT: - { index: 9, kind: "Greater", line: {{ *}}[[@LINE-4]], column: 8, indent: 3, spelling: ">", has_leading_space: true } + // CHECK:STDOUT: - { index: 10, kind: "IntLiteral", line: {{ *}}[[@LINE-5]], column: 10, indent: 3, spelling: "0", value: "0", has_leading_space: true } + // CHECK:STDOUT: - { index: 11, kind: "CloseParen", line: {{ *}}[[@LINE-6]], column: 11, indent: 3, spelling: ")", opening_token: 7 } + // CHECK:STDOUT: - { index: 12, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-7]], column: 13, indent: 3, spelling: "{", closing_token: 17, has_leading_space: true } + G(); + // CHECK:STDOUT: - { index: 13, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 5, indent: 5, spelling: "G", identifier: 2, has_leading_space: true } + // CHECK:STDOUT: - { index: 14, kind: "OpenParen", line: {{ *}}[[@LINE-2]], column: 6, indent: 5, spelling: "(", closing_token: 15 } + // CHECK:STDOUT: - { index: 15, kind: "CloseParen", line: {{ *}}[[@LINE-3]], column: 7, indent: 5, spelling: ")", opening_token: 14 } + // CHECK:STDOUT: - { index: 16, kind: "Semi", line: {{ *}}[[@LINE-4]], column: 8, indent: 5, spelling: ";" } + } + // CHECK:STDOUT: - { index: 17, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "}", opening_token: 12, has_leading_space: true } +} +// CHECK:STDOUT: - { index: 18, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", opening_token: 5, has_leading_space: true } + +// --- fail_missing_open_square_after_forall.carbon +// CHECK:STDOUT: - filename: fail_missing_open_square_after_forall.carbon +// CHECK:STDOUT: tokens: + +// As above for `forall`, which wants a `[` rather than a `(`. + +// CHECK:STDERR: fail_missing_open_square_after_forall.carbon:[[@LINE+7]]:21: error: closing symbol without a corresponding opening symbol [UnmatchedClosing] +// CHECK:STDERR: impl forall T:! type] X as Y { +// CHECK:STDERR: ^ +// CHECK:STDERR: fail_missing_open_square_after_forall.carbon:[[@LINE+4]]:13: note: possibly missing `[` here [PossiblyMissingBracketHere] +// CHECK:STDERR: impl forall T:! type] X as Y { +// CHECK:STDERR: ^ +// CHECK:STDERR: +impl forall T:! type] X as Y { +// CHECK:STDOUT: - { index: 1, kind: "Impl", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "impl", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Forall", line: {{ *}}[[@LINE-2]], column: 6, indent: 1, spelling: "forall", has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenSquareBracket", line: {{ *}}[[@LINE-3]], column: 12, indent: 1, spelling: "[", closing_token: 8, has_leading_space: true, recovery: true } +// CHECK:STDOUT: - { index: 4, kind: "Identifier", line: {{ *}}[[@LINE-4]], column: 13, indent: 1, spelling: "T", identifier: 0 } +// CHECK:STDOUT: - { index: 5, kind: "Colon", line: {{ *}}[[@LINE-5]], column: 14, indent: 1, spelling: ":" } +// CHECK:STDOUT: - { index: 6, kind: "Exclaim", line: {{ *}}[[@LINE-6]], column: 15, indent: 1, spelling: "!" } +// CHECK:STDOUT: - { index: 7, kind: "Type", line: {{ *}}[[@LINE-7]], column: 17, indent: 1, spelling: "type", has_leading_space: true } +// CHECK:STDOUT: - { index: 8, kind: "CloseSquareBracket", line: {{ *}}[[@LINE-8]], column: 21, indent: 1, spelling: "]", opening_token: 3 } +// CHECK:STDOUT: - { index: 9, kind: "Identifier", line: {{ *}}[[@LINE-9]], column: 23, indent: 1, spelling: "X", identifier: 1, has_leading_space: true } +// CHECK:STDOUT: - { index: 10, kind: "As", line: {{ *}}[[@LINE-10]], column: 25, indent: 1, spelling: "as", has_leading_space: true } +// CHECK:STDOUT: - { index: 11, kind: "Identifier", line: {{ *}}[[@LINE-11]], column: 28, indent: 1, spelling: "Y", identifier: 2, has_leading_space: true } +// CHECK:STDOUT: - { index: 12, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-12]], column: 30, indent: 1, spelling: "{", closing_token: 13, has_leading_space: true } +} +// CHECK:STDOUT: - { index: 13, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", opening_token: 12, has_leading_space: true } + +// --- fail_missing_brace_after_if.carbon +// CHECK:STDOUT: - filename: fail_missing_brace_after_if.carbon +// CHECK:STDOUT: tokens: + +// A statement header can be followed by a block, so the body of this `if` is +// explained by a missing `{` rather than by discarding the `}` that closes it. + +fn F() { +// CHECK:STDOUT: - { index: 1, kind: "Fn", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 4, indent: 1, spelling: "F", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenParen", line: {{ *}}[[@LINE-3]], column: 5, indent: 1, spelling: "(", closing_token: 4 } +// CHECK:STDOUT: - { index: 4, kind: "CloseParen", line: {{ *}}[[@LINE-4]], column: 6, indent: 1, spelling: ")", opening_token: 3 } +// CHECK:STDOUT: - { index: 5, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-5]], column: 8, indent: 1, spelling: "{", closing_token: 14, has_leading_space: true } + if (thing1) + // CHECK:STDOUT: - { index: 6, kind: "If", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "if", has_leading_space: true } + // CHECK:STDOUT: - { index: 7, kind: "OpenParen", line: {{ *}}[[@LINE-2]], column: 6, indent: 3, spelling: "(", closing_token: 9, has_leading_space: true } + // CHECK:STDOUT: - { index: 8, kind: "Identifier", line: {{ *}}[[@LINE-3]], column: 7, indent: 3, spelling: "thing1", identifier: 1 } + // CHECK:STDOUT: - { index: 9, kind: "CloseParen", line: {{ *}}[[@LINE-4]], column: 13, indent: 3, spelling: ")", opening_token: 7 } + // CHECK:STDOUT: - { index: 10, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-5]], column: 14, indent: 3, spelling: "{", closing_token: 13, has_leading_space: true, recovery: true } + thing2; + // CHECK:STDOUT: - { index: 11, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 5, indent: 5, spelling: "thing2", identifier: 2 } + // CHECK:STDOUT: - { index: 12, kind: "Semi", line: {{ *}}[[@LINE-2]], column: 11, indent: 5, spelling: ";" } + // CHECK:STDERR: fail_missing_brace_after_if.carbon:[[@LINE+7]]:3: error: closing symbol without a corresponding opening symbol [UnmatchedClosing] + // CHECK:STDERR: } + // CHECK:STDERR: ^ + // CHECK:STDERR: fail_missing_brace_after_if.carbon:[[@LINE-12]]:14: note: possibly missing `{` here [PossiblyMissingBracketHere] + // CHECK:STDERR: if (thing1) + // CHECK:STDERR: ^ + // CHECK:STDERR: + } + // CHECK:STDOUT: - { index: 13, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "}", opening_token: 10, has_leading_space: true } +} +// CHECK:STDOUT: - { index: 14, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", opening_token: 5, has_leading_space: true } + +// --- fail_missing_brace_multi_line_header.carbon +// CHECK:STDOUT: - filename: fail_missing_brace_multi_line_header.carbon +// CHECK:STDOUT: tokens: + +// As above, where the header spans several lines: the `{` belongs at the end of +// the header, not at the start of the body. + +fn F[T: type] +// CHECK:STDOUT: - { index: 1, kind: "Fn", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 4, indent: 1, spelling: "F", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenSquareBracket", line: {{ *}}[[@LINE-3]], column: 5, indent: 1, spelling: "[", closing_token: 7 } +// CHECK:STDOUT: - { index: 4, kind: "Identifier", line: {{ *}}[[@LINE-4]], column: 6, indent: 1, spelling: "T", identifier: 1 } +// CHECK:STDOUT: - { index: 5, kind: "Colon", line: {{ *}}[[@LINE-5]], column: 7, indent: 1, spelling: ":" } +// CHECK:STDOUT: - { index: 6, kind: "Type", line: {{ *}}[[@LINE-6]], column: 9, indent: 1, spelling: "type", has_leading_space: true } +// CHECK:STDOUT: - { index: 7, kind: "CloseSquareBracket", line: {{ *}}[[@LINE-7]], column: 13, indent: 1, spelling: "]", opening_token: 3 } + (x: T) + // CHECK:STDOUT: - { index: 8, kind: "OpenParen", line: {{ *}}[[@LINE-1]], column: 5, indent: 5, spelling: "(", closing_token: 12, has_leading_space: true } + // CHECK:STDOUT: - { index: 9, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 6, indent: 5, spelling: "x", identifier: 2 } + // CHECK:STDOUT: - { index: 10, kind: "Colon", line: {{ *}}[[@LINE-3]], column: 7, indent: 5, spelling: ":" } + // CHECK:STDOUT: - { index: 11, kind: "Identifier", line: {{ *}}[[@LINE-4]], column: 9, indent: 5, spelling: "T", identifier: 1, has_leading_space: true } + // CHECK:STDOUT: - { index: 12, kind: "CloseParen", line: {{ *}}[[@LINE-5]], column: 10, indent: 5, spelling: ")", opening_token: 8 } + // CHECK:STDOUT: - { index: 13, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-6]], column: 11, indent: 5, spelling: "{", closing_token: 22, has_leading_space: true, recovery: true } + foo(); + // CHECK:STDOUT: - { index: 14, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "foo", identifier: 3 } + // CHECK:STDOUT: - { index: 15, kind: "OpenParen", line: {{ *}}[[@LINE-2]], column: 6, indent: 3, spelling: "(", closing_token: 16 } + // CHECK:STDOUT: - { index: 16, kind: "CloseParen", line: {{ *}}[[@LINE-3]], column: 7, indent: 3, spelling: ")", opening_token: 15 } + // CHECK:STDOUT: - { index: 17, kind: "Semi", line: {{ *}}[[@LINE-4]], column: 8, indent: 3, spelling: ";" } + bar(); + // CHECK:STDOUT: - { index: 18, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "bar", identifier: 4, has_leading_space: true } + // CHECK:STDOUT: - { index: 19, kind: "OpenParen", line: {{ *}}[[@LINE-2]], column: 6, indent: 3, spelling: "(", closing_token: 20 } + // CHECK:STDOUT: - { index: 20, kind: "CloseParen", line: {{ *}}[[@LINE-3]], column: 7, indent: 3, spelling: ")", opening_token: 19 } + // CHECK:STDOUT: - { index: 21, kind: "Semi", line: {{ *}}[[@LINE-4]], column: 8, indent: 3, spelling: ";" } +// CHECK:STDERR: fail_missing_brace_multi_line_header.carbon:[[@LINE+7]]:1: error: closing symbol without a corresponding opening symbol [UnmatchedClosing] +// CHECK:STDERR: } +// CHECK:STDERR: ^ +// CHECK:STDERR: fail_missing_brace_multi_line_header.carbon:[[@LINE-20]]:11: note: possibly missing `{` here [PossiblyMissingBracketHere] +// CHECK:STDERR: (x: T) +// CHECK:STDERR: ^ +// CHECK:STDERR: +} +// CHECK:STDOUT: - { index: 22, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", opening_token: 13, has_leading_space: true } + +// --- fail_missing_open_paren_at_leaf.carbon +// CHECK:STDOUT: - filename: fail_missing_open_paren_at_leaf.carbon +// CHECK:STDOUT: tokens: + +// Two values can't be adjacent, so a `(` is missing between `G` and `1` even +// though nothing else on the line is out of place. + +fn F() { +// CHECK:STDOUT: - { index: 1, kind: "Fn", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 4, indent: 1, spelling: "F", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenParen", line: {{ *}}[[@LINE-3]], column: 5, indent: 1, spelling: "(", closing_token: 4 } +// CHECK:STDOUT: - { index: 4, kind: "CloseParen", line: {{ *}}[[@LINE-4]], column: 6, indent: 1, spelling: ")", opening_token: 3 } +// CHECK:STDOUT: - { index: 5, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-5]], column: 8, indent: 1, spelling: "{", closing_token: 16, has_leading_space: true } + // CHECK:STDERR: fail_missing_open_paren_at_leaf.carbon:[[@LINE+7]]:19: error: closing symbol without a corresponding opening symbol [UnmatchedClosing] + // CHECK:STDERR: var x: i32 = G 1); + // CHECK:STDERR: ^ + // CHECK:STDERR: fail_missing_open_paren_at_leaf.carbon:[[@LINE+4]]:18: note: possibly missing `(` here [PossiblyMissingBracketHere] + // CHECK:STDERR: var x: i32 = G 1); + // CHECK:STDERR: ^ + // CHECK:STDERR: + var x: i32 = G 1); + // CHECK:STDOUT: - { index: 6, kind: "Var", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "var", has_leading_space: true } + // CHECK:STDOUT: - { index: 7, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 7, indent: 3, spelling: "x", identifier: 1, has_leading_space: true } + // CHECK:STDOUT: - { index: 8, kind: "Colon", line: {{ *}}[[@LINE-3]], column: 8, indent: 3, spelling: ":" } + // CHECK:STDOUT: - { index: 9, kind: "IntTypeLiteral", line: {{ *}}[[@LINE-4]], column: 10, indent: 3, spelling: "i32", has_leading_space: true } + // CHECK:STDOUT: - { index: 10, kind: "Equal", line: {{ *}}[[@LINE-5]], column: 14, indent: 3, spelling: "=", has_leading_space: true } + // CHECK:STDOUT: - { index: 11, kind: "Identifier", line: {{ *}}[[@LINE-6]], column: 16, indent: 3, spelling: "G", identifier: 2, has_leading_space: true } + // CHECK:STDOUT: - { index: 12, kind: "OpenParen", line: {{ *}}[[@LINE-7]], column: 17, indent: 3, spelling: "(", closing_token: 14, has_leading_space: true, recovery: true } + // CHECK:STDOUT: - { index: 13, kind: "IntLiteral", line: {{ *}}[[@LINE-8]], column: 18, indent: 3, spelling: "1", value: "1" } + // CHECK:STDOUT: - { index: 14, kind: "CloseParen", line: {{ *}}[[@LINE-9]], column: 19, indent: 3, spelling: ")", opening_token: 12 } + // CHECK:STDOUT: - { index: 15, kind: "Semi", line: {{ *}}[[@LINE-10]], column: 20, indent: 3, spelling: ";" } +} +// CHECK:STDOUT: - { index: 16, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", opening_token: 5, has_leading_space: true } + +// --- fail_unmatched_closer.carbon +// CHECK:STDOUT: - filename: fail_unmatched_closer.carbon +// CHECK:STDOUT: tokens: + +// Nothing suggests where a matching `(` would go, so the extra `)` is replaced +// with an error token and no repair is offered. + +fn F() { +// CHECK:STDOUT: - { index: 1, kind: "Fn", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 4, indent: 1, spelling: "F", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenParen", line: {{ *}}[[@LINE-3]], column: 5, indent: 1, spelling: "(", closing_token: 4 } +// CHECK:STDOUT: - { index: 4, kind: "CloseParen", line: {{ *}}[[@LINE-4]], column: 6, indent: 1, spelling: ")", opening_token: 3 } +// CHECK:STDOUT: - { index: 5, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-5]], column: 8, indent: 1, spelling: "{", closing_token: 17, has_leading_space: true } + // CHECK:STDERR: fail_unmatched_closer.carbon:[[@LINE+4]]:20: error: closing symbol without a corresponding opening symbol [UnmatchedClosing] + // CHECK:STDERR: var x: i32 = G(x)); + // CHECK:STDERR: ^ + // CHECK:STDERR: + var x: i32 = G(x)); + // CHECK:STDOUT: - { index: 6, kind: "Var", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "var", has_leading_space: true } + // CHECK:STDOUT: - { index: 7, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 7, indent: 3, spelling: "x", identifier: 1, has_leading_space: true } + // CHECK:STDOUT: - { index: 8, kind: "Colon", line: {{ *}}[[@LINE-3]], column: 8, indent: 3, spelling: ":" } + // CHECK:STDOUT: - { index: 9, kind: "IntTypeLiteral", line: {{ *}}[[@LINE-4]], column: 10, indent: 3, spelling: "i32", has_leading_space: true } + // CHECK:STDOUT: - { index: 10, kind: "Equal", line: {{ *}}[[@LINE-5]], column: 14, indent: 3, spelling: "=", has_leading_space: true } + // CHECK:STDOUT: - { index: 11, kind: "Identifier", line: {{ *}}[[@LINE-6]], column: 16, indent: 3, spelling: "G", identifier: 2, has_leading_space: true } + // CHECK:STDOUT: - { index: 12, kind: "OpenParen", line: {{ *}}[[@LINE-7]], column: 17, indent: 3, spelling: "(", closing_token: 14 } + // CHECK:STDOUT: - { index: 13, kind: "Identifier", line: {{ *}}[[@LINE-8]], column: 18, indent: 3, spelling: "x", identifier: 1 } + // CHECK:STDOUT: - { index: 14, kind: "CloseParen", line: {{ *}}[[@LINE-9]], column: 19, indent: 3, spelling: ")", opening_token: 12 } + // CHECK:STDOUT: - { index: 15, kind: "Error", line: {{ *}}[[@LINE-10]], column: 20, indent: 3, spelling: ")" } + // CHECK:STDOUT: - { index: 16, kind: "Semi", line: {{ *}}[[@LINE-11]], column: 21, indent: 3, spelling: ";" } +} +// CHECK:STDOUT: - { index: 17, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", opening_token: 5, has_leading_space: true } + +// --- fail_two_damaged_regions.carbon +// CHECK:STDOUT: - filename: fail_two_damaged_regions.carbon +// CHECK:STDOUT: tokens: + +// A region ends at each top-level declaration, so one mistake can't smear into +// the next declaration: these two are diagnosed and repaired independently. + +fn F() { +// CHECK:STDOUT: - { index: 1, kind: "Fn", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 4, indent: 1, spelling: "F", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenParen", line: {{ *}}[[@LINE-3]], column: 5, indent: 1, spelling: "(", closing_token: 4 } +// CHECK:STDOUT: - { index: 4, kind: "CloseParen", line: {{ *}}[[@LINE-4]], column: 6, indent: 1, spelling: ")", opening_token: 3 } +// CHECK:STDOUT: - { index: 5, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-5]], column: 8, indent: 1, spelling: "{", closing_token: 13, has_leading_space: true } + // CHECK:STDERR: fail_two_damaged_regions.carbon:[[@LINE+7]]:4: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] + // CHECK:STDERR: G(1 + 2; + // CHECK:STDERR: ^ + // CHECK:STDERR: fail_two_damaged_regions.carbon:[[@LINE+4]]:6: note: possibly missing `)` here [PossiblyMissingBracketHere] + // CHECK:STDERR: G(1 + 2; + // CHECK:STDERR: ^ + // CHECK:STDERR: + G(1 + 2; + // CHECK:STDOUT: - { index: 6, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "G", identifier: 1, has_leading_space: true } + // CHECK:STDOUT: - { index: 7, kind: "OpenParen", line: {{ *}}[[@LINE-2]], column: 4, indent: 3, spelling: "(", closing_token: 9 } + // CHECK:STDOUT: - { index: 8, kind: "IntLiteral", line: {{ *}}[[@LINE-3]], column: 5, indent: 3, spelling: "1", value: "1" } + // CHECK:STDOUT: - { index: 9, kind: "CloseParen", line: {{ *}}[[@LINE-4]], column: 6, indent: 3, spelling: ")", opening_token: 7, has_leading_space: true, recovery: true } + // CHECK:STDOUT: - { index: 10, kind: "Plus", line: {{ *}}[[@LINE-5]], column: 8, indent: 3, spelling: "+" } + // CHECK:STDOUT: - { index: 11, kind: "IntLiteral", line: {{ *}}[[@LINE-6]], column: 10, indent: 3, spelling: "2", value: "2", has_leading_space: true } + // CHECK:STDOUT: - { index: 12, kind: "Semi", line: {{ *}}[[@LINE-7]], column: 11, indent: 3, spelling: ";" } +} +// CHECK:STDOUT: - { index: 13, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", opening_token: 5, has_leading_space: true } + +fn H() { +// CHECK:STDOUT: - { index: 14, kind: "Fn", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 15, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 4, indent: 1, spelling: "H", identifier: 2, has_leading_space: true } +// CHECK:STDOUT: - { index: 16, kind: "OpenParen", line: {{ *}}[[@LINE-3]], column: 5, indent: 1, spelling: "(", closing_token: 17 } +// CHECK:STDOUT: - { index: 17, kind: "CloseParen", line: {{ *}}[[@LINE-4]], column: 6, indent: 1, spelling: ")", opening_token: 16 } +// CHECK:STDOUT: - { index: 18, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-5]], column: 8, indent: 1, spelling: "{", closing_token: 31, has_leading_space: true } + // CHECK:STDERR: fail_two_damaged_regions.carbon:[[@LINE+7]]:17: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] + // CHECK:STDERR: var y: auto = {.a = 3; + // CHECK:STDERR: ^ + // CHECK:STDERR: fail_two_damaged_regions.carbon:[[@LINE+4]]:24: note: possibly missing `}` here [PossiblyMissingBracketHere] + // CHECK:STDERR: var y: auto = {.a = 3; + // CHECK:STDERR: ^ + // CHECK:STDERR: + var y: auto = {.a = 3; + // CHECK:STDOUT: - { index: 19, kind: "Var", line: {{ *}}[[@LINE-1]], column: 3, indent: 3, spelling: "var", has_leading_space: true } + // CHECK:STDOUT: - { index: 20, kind: "Identifier", line: {{ *}}[[@LINE-2]], column: 7, indent: 3, spelling: "y", identifier: 3, has_leading_space: true } + // CHECK:STDOUT: - { index: 21, kind: "Colon", line: {{ *}}[[@LINE-3]], column: 8, indent: 3, spelling: ":" } + // CHECK:STDOUT: - { index: 22, kind: "Auto", line: {{ *}}[[@LINE-4]], column: 10, indent: 3, spelling: "auto", has_leading_space: true } + // CHECK:STDOUT: - { index: 23, kind: "Equal", line: {{ *}}[[@LINE-5]], column: 15, indent: 3, spelling: "=", has_leading_space: true } + // CHECK:STDOUT: - { index: 24, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-6]], column: 17, indent: 3, spelling: "{", closing_token: 29, has_leading_space: true } + // CHECK:STDOUT: - { index: 25, kind: "Period", line: {{ *}}[[@LINE-7]], column: 18, indent: 3, spelling: "." } + // CHECK:STDOUT: - { index: 26, kind: "Identifier", line: {{ *}}[[@LINE-8]], column: 19, indent: 3, spelling: "a", identifier: 4 } + // CHECK:STDOUT: - { index: 27, kind: "Equal", line: {{ *}}[[@LINE-9]], column: 21, indent: 3, spelling: "=", has_leading_space: true } + // CHECK:STDOUT: - { index: 28, kind: "IntLiteral", line: {{ *}}[[@LINE-10]], column: 23, indent: 3, spelling: "3", value: "3", has_leading_space: true } + // CHECK:STDOUT: - { index: 29, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-11]], column: 24, indent: 3, spelling: "}", opening_token: 24, recovery: true } + // CHECK:STDOUT: - { index: 30, kind: "Semi", line: {{ *}}[[@LINE-12]], column: 24, indent: 3, spelling: ";" } +} +// CHECK:STDOUT: - { index: 31, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "}", opening_token: 18, has_leading_space: true } diff --git a/toolchain/lex/testdata/fail_mismatched_brackets_2.carbon b/toolchain/lex/testdata/fail_mismatched_brackets_2.carbon deleted file mode 100644 index 5aa52872a10a..000000000000 --- a/toolchain/lex/testdata/fail_mismatched_brackets_2.carbon +++ /dev/null @@ -1,39 +0,0 @@ -// Part of the Carbon Language project, under the Apache License v2.0 with LLVM -// Exceptions. See /LICENSE for license information. -// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception - -// TODO: For this example, we should ideally report a missing `{` on the `if` -// line, for example based on the indentation of the surrounding lines. - -fn F() { - if (thing1) - thing2; - } -} - -// The check lines are inserted at the end so that they don't disrupt the -// indentation of lines in the text. -// AUTOUPDATE -// TIP: To test this file alone, run: -// TIP: bazel test //toolchain/testing:file_test --test_arg=--file_tests=toolchain/lex/testdata/fail_mismatched_brackets_2.carbon -// TIP: To dump output, run: -// TIP: bazel run //toolchain/testing:file_test -- --dump_output --file_tests=toolchain/lex/testdata/fail_mismatched_brackets_2.carbon -// CHECK:STDERR: fail_mismatched_brackets_2.carbon:[[@LINE-9]]:1: error: closing symbol without a corresponding opening symbol [UnmatchedClosing] -// CHECK:STDERR: } -// CHECK:STDERR: ^ -// CHECK:STDERR: -// CHECK:STDOUT: - filename: fail_mismatched_brackets_2.carbon -// CHECK:STDOUT: tokens: -// CHECK:STDOUT: - { index: 1, kind: "Fn", line: {{ *}}[[@LINE-19]], column: 1, indent: 1, spelling: "fn", has_leading_space: true } -// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-20]], column: 4, indent: 1, spelling: "F", identifier: 0, has_leading_space: true } -// CHECK:STDOUT: - { index: 3, kind: "OpenParen", line: {{ *}}[[@LINE-21]], column: 5, indent: 1, spelling: "(", closing_token: 4 } -// CHECK:STDOUT: - { index: 4, kind: "CloseParen", line: {{ *}}[[@LINE-22]], column: 6, indent: 1, spelling: ")", opening_token: 3 } -// CHECK:STDOUT: - { index: 5, kind: "OpenCurlyBrace", line: {{ *}}[[@LINE-23]], column: 8, indent: 1, spelling: "{", closing_token: 12, has_leading_space: true } -// CHECK:STDOUT: - { index: 6, kind: "If", line: {{ *}}[[@LINE-23]], column: 3, indent: 3, spelling: "if", has_leading_space: true } -// CHECK:STDOUT: - { index: 7, kind: "OpenParen", line: {{ *}}[[@LINE-24]], column: 6, indent: 3, spelling: "(", closing_token: 9, has_leading_space: true } -// CHECK:STDOUT: - { index: 8, kind: "Identifier", line: {{ *}}[[@LINE-25]], column: 7, indent: 3, spelling: "thing1", identifier: 1 } -// CHECK:STDOUT: - { index: 9, kind: "CloseParen", line: {{ *}}[[@LINE-26]], column: 13, indent: 3, spelling: ")", opening_token: 7 } -// CHECK:STDOUT: - { index: 10, kind: "Identifier", line: {{ *}}[[@LINE-26]], column: 5, indent: 5, spelling: "thing2", identifier: 2, has_leading_space: true } -// CHECK:STDOUT: - { index: 11, kind: "Semi", line: {{ *}}[[@LINE-27]], column: 11, indent: 5, spelling: ";" } -// CHECK:STDOUT: - { index: 12, kind: "CloseCurlyBrace", line: {{ *}}[[@LINE-27]], column: 3, indent: 3, spelling: "}", opening_token: 5, has_leading_space: true } -// CHECK:STDOUT: - { index: 13, kind: "Error", line: {{ *}}[[@LINE-27]], column: 1, indent: 1, spelling: "}", has_leading_space: true } diff --git a/toolchain/lex/testdata/fail_mismatched_brackets_close_brace.carbon b/toolchain/lex/testdata/fail_mismatched_brackets_close_brace.carbon new file mode 100644 index 000000000000..afd580eb3ec0 --- /dev/null +++ b/toolchain/lex/testdata/fail_mismatched_brackets_close_brace.carbon @@ -0,0 +1,174 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +// AUTOUPDATE +// TIP: To test this file alone, run: +// TIP: bazel test //toolchain/testing:file_test --test_arg=--file_tests=toolchain/lex/testdata/fail_mismatched_brackets_close_brace.carbon +// TIP: To dump output, run: +// TIP: bazel run //toolchain/testing:file_test -- --dump_output --file_tests=toolchain/lex/testdata/fail_mismatched_brackets_close_brace.carbon + +// The cases from `fail_mismatched_brackets.carbon` where recovery suggests a +// `}`. A `}` goes on a line of its own, so its note points at the line after the +// scope's last token, and the `CHECK`s have to live in their own split: +// autoupdate would otherwise write them at that very position, and the note +// would quote its own `CHECK` line. + +// --- fail_dedent_inside_scope.carbon + +// A line dedented past the header of the scope it's in can't belong to that +// scope, so the `}` goes at the dedent. Indentation is the whole cue here: the +// brackets alone balance if the last `}` closes the `if` instead. + +fn F() { + if (x) { + foo(); + bar(); +} + +// --- fail_missing_brace_before_else.carbon + +// A first-on-line `else` should have had a `}` before it. Only one is inserted: +// the `else` block's own `}` still lines up, because the block's header +// indentation is taken from the `if` it continues rather than from the `else`'s +// own column, which is wherever the author left it once the `}` went missing. + +fn F() { + if (C) { + G(); + else { + H(); + } +} + +// --- fail_missing_brace_in_class.carbon + +// As above for a class body, whose `}` is missing entirely rather than +// misattributed. + +class C { + var a: i32; +fn F() { + G(); +} + +// --- fail_unclosed_at_eof.carbon + +// Whatever is still open at the end of the file is closed there. + +fn F() { + var x: i32 = 1; + + +// --- AUTOUPDATE-SPLIT + +// CHECK:STDERR: fail_dedent_inside_scope.carbon:7:10: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] +// CHECK:STDERR: if (x) { +// CHECK:STDERR: ^ +// CHECK:STDERR: fail_dedent_inside_scope.carbon:9:3: note: possibly missing `}` here [PossiblyMissingBracketHere] +// CHECK:STDERR: bar(); +// CHECK:STDERR: ^ +// CHECK:STDERR: +// CHECK:STDERR: fail_missing_brace_before_else.carbon:8:10: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] +// CHECK:STDERR: if (C) { +// CHECK:STDERR: ^ +// CHECK:STDERR: fail_missing_brace_before_else.carbon:10:3: note: possibly missing `}` here [PossiblyMissingBracketHere] +// CHECK:STDERR: else { +// CHECK:STDERR: ^ +// CHECK:STDERR: +// CHECK:STDERR: fail_missing_brace_in_class.carbon:5:9: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] +// CHECK:STDERR: class C { +// CHECK:STDERR: ^ +// CHECK:STDERR: fail_missing_brace_in_class.carbon:7:1: note: possibly missing `}` here [PossiblyMissingBracketHere] +// CHECK:STDERR: fn F() { +// CHECK:STDERR: ^ +// CHECK:STDERR: +// CHECK:STDERR: fail_unclosed_at_eof.carbon:4:8: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] +// CHECK:STDERR: fn F() { +// CHECK:STDERR: ^ +// CHECK:STDERR: fail_unclosed_at_eof.carbon:6:1: note: possibly missing `}` here [PossiblyMissingBracketHere] +// CHECK:STDERR: +// CHECK:STDERR: ^ +// CHECK:STDERR: +// CHECK:STDOUT: - filename: fail_dedent_inside_scope.carbon +// CHECK:STDOUT: tokens: +// CHECK:STDOUT: - { index: 1, kind: "Fn", line: {{ *}}6, column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}6, column: 4, indent: 1, spelling: "F", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenParen", line: {{ *}}6, column: 5, indent: 1, spelling: "(", closing_token: 4 } +// CHECK:STDOUT: - { index: 4, kind: "CloseParen", line: {{ *}}6, column: 6, indent: 1, spelling: ")", opening_token: 3 } +// CHECK:STDOUT: - { index: 5, kind: "OpenCurlyBrace", line: {{ *}}6, column: 8, indent: 1, spelling: "{", closing_token: 20, has_leading_space: true } +// CHECK:STDOUT: - { index: 6, kind: "If", line: {{ *}}7, column: 3, indent: 3, spelling: "if", has_leading_space: true } +// CHECK:STDOUT: - { index: 7, kind: "OpenParen", line: {{ *}}7, column: 6, indent: 3, spelling: "(", closing_token: 9, has_leading_space: true } +// CHECK:STDOUT: - { index: 8, kind: "Identifier", line: {{ *}}7, column: 7, indent: 3, spelling: "x", identifier: 1 } +// CHECK:STDOUT: - { index: 9, kind: "CloseParen", line: {{ *}}7, column: 8, indent: 3, spelling: ")", opening_token: 7 } +// CHECK:STDOUT: - { index: 10, kind: "OpenCurlyBrace", line: {{ *}}7, column: 10, indent: 3, spelling: "{", closing_token: 15, has_leading_space: true } +// CHECK:STDOUT: - { index: 11, kind: "Identifier", line: {{ *}}8, column: 5, indent: 5, spelling: "foo", identifier: 2, has_leading_space: true } +// CHECK:STDOUT: - { index: 12, kind: "OpenParen", line: {{ *}}8, column: 8, indent: 5, spelling: "(", closing_token: 13 } +// CHECK:STDOUT: - { index: 13, kind: "CloseParen", line: {{ *}}8, column: 9, indent: 5, spelling: ")", opening_token: 12 } +// CHECK:STDOUT: - { index: 14, kind: "Semi", line: {{ *}}8, column: 10, indent: 5, spelling: ";" } +// CHECK:STDOUT: - { index: 15, kind: "CloseCurlyBrace", line: {{ *}}8, column: 11, indent: 5, spelling: "}", opening_token: 10, has_leading_space: true, recovery: true } +// CHECK:STDOUT: - { index: 16, kind: "Identifier", line: {{ *}}9, column: 3, indent: 3, spelling: "bar", identifier: 3 } +// CHECK:STDOUT: - { index: 17, kind: "OpenParen", line: {{ *}}9, column: 6, indent: 3, spelling: "(", closing_token: 18 } +// CHECK:STDOUT: - { index: 18, kind: "CloseParen", line: {{ *}}9, column: 7, indent: 3, spelling: ")", opening_token: 17 } +// CHECK:STDOUT: - { index: 19, kind: "Semi", line: {{ *}}9, column: 8, indent: 3, spelling: ";" } +// CHECK:STDOUT: - { index: 20, kind: "CloseCurlyBrace", line: {{ *}}10, column: 1, indent: 1, spelling: "}", opening_token: 5, has_leading_space: true } +// CHECK:STDOUT: - filename: fail_missing_brace_before_else.carbon +// CHECK:STDOUT: tokens: +// CHECK:STDOUT: - { index: 1, kind: "Fn", line: {{ *}}7, column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}7, column: 4, indent: 1, spelling: "F", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenParen", line: {{ *}}7, column: 5, indent: 1, spelling: "(", closing_token: 4 } +// CHECK:STDOUT: - { index: 4, kind: "CloseParen", line: {{ *}}7, column: 6, indent: 1, spelling: ")", opening_token: 3 } +// CHECK:STDOUT: - { index: 5, kind: "OpenCurlyBrace", line: {{ *}}7, column: 8, indent: 1, spelling: "{", closing_token: 23, has_leading_space: true } +// CHECK:STDOUT: - { index: 6, kind: "If", line: {{ *}}8, column: 3, indent: 3, spelling: "if", has_leading_space: true } +// CHECK:STDOUT: - { index: 7, kind: "OpenParen", line: {{ *}}8, column: 6, indent: 3, spelling: "(", closing_token: 9, has_leading_space: true } +// CHECK:STDOUT: - { index: 8, kind: "Identifier", line: {{ *}}8, column: 7, indent: 3, spelling: "C", identifier: 1 } +// CHECK:STDOUT: - { index: 9, kind: "CloseParen", line: {{ *}}8, column: 8, indent: 3, spelling: ")", opening_token: 7 } +// CHECK:STDOUT: - { index: 10, kind: "OpenCurlyBrace", line: {{ *}}8, column: 10, indent: 3, spelling: "{", closing_token: 15, has_leading_space: true } +// CHECK:STDOUT: - { index: 11, kind: "Identifier", line: {{ *}}9, column: 5, indent: 5, spelling: "G", identifier: 2, has_leading_space: true } +// CHECK:STDOUT: - { index: 12, kind: "OpenParen", line: {{ *}}9, column: 6, indent: 5, spelling: "(", closing_token: 13 } +// CHECK:STDOUT: - { index: 13, kind: "CloseParen", line: {{ *}}9, column: 7, indent: 5, spelling: ")", opening_token: 12 } +// CHECK:STDOUT: - { index: 14, kind: "Semi", line: {{ *}}9, column: 8, indent: 5, spelling: ";" } +// CHECK:STDOUT: - { index: 15, kind: "CloseCurlyBrace", line: {{ *}}9, column: 9, indent: 5, spelling: "}", opening_token: 10, has_leading_space: true, recovery: true } +// CHECK:STDOUT: - { index: 16, kind: "Else", line: {{ *}}10, column: 3, indent: 3, spelling: "else" } +// CHECK:STDOUT: - { index: 17, kind: "OpenCurlyBrace", line: {{ *}}10, column: 8, indent: 3, spelling: "{", closing_token: 22, has_leading_space: true } +// CHECK:STDOUT: - { index: 18, kind: "Identifier", line: {{ *}}11, column: 5, indent: 5, spelling: "H", identifier: 3, has_leading_space: true } +// CHECK:STDOUT: - { index: 19, kind: "OpenParen", line: {{ *}}11, column: 6, indent: 5, spelling: "(", closing_token: 20 } +// CHECK:STDOUT: - { index: 20, kind: "CloseParen", line: {{ *}}11, column: 7, indent: 5, spelling: ")", opening_token: 19 } +// CHECK:STDOUT: - { index: 21, kind: "Semi", line: {{ *}}11, column: 8, indent: 5, spelling: ";" } +// CHECK:STDOUT: - { index: 22, kind: "CloseCurlyBrace", line: {{ *}}12, column: 3, indent: 3, spelling: "}", opening_token: 17, has_leading_space: true } +// CHECK:STDOUT: - { index: 23, kind: "CloseCurlyBrace", line: {{ *}}13, column: 1, indent: 1, spelling: "}", opening_token: 5, has_leading_space: true } +// CHECK:STDOUT: - filename: fail_missing_brace_in_class.carbon +// CHECK:STDOUT: tokens: +// CHECK:STDOUT: - { index: 1, kind: "Class", line: {{ *}}5, column: 1, indent: 1, spelling: "class", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}5, column: 7, indent: 1, spelling: "C", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenCurlyBrace", line: {{ *}}5, column: 9, indent: 1, spelling: "{", closing_token: 9, has_leading_space: true } +// CHECK:STDOUT: - { index: 4, kind: "Var", line: {{ *}}6, column: 3, indent: 3, spelling: "var", has_leading_space: true } +// CHECK:STDOUT: - { index: 5, kind: "Identifier", line: {{ *}}6, column: 7, indent: 3, spelling: "a", identifier: 1, has_leading_space: true } +// CHECK:STDOUT: - { index: 6, kind: "Colon", line: {{ *}}6, column: 8, indent: 3, spelling: ":" } +// CHECK:STDOUT: - { index: 7, kind: "IntTypeLiteral", line: {{ *}}6, column: 10, indent: 3, spelling: "i32", has_leading_space: true } +// CHECK:STDOUT: - { index: 8, kind: "Semi", line: {{ *}}6, column: 13, indent: 3, spelling: ";" } +// CHECK:STDOUT: - { index: 9, kind: "CloseCurlyBrace", line: {{ *}}6, column: 14, indent: 3, spelling: "}", opening_token: 3, has_leading_space: true, recovery: true } +// CHECK:STDOUT: - { index: 10, kind: "Fn", line: {{ *}}7, column: 1, indent: 1, spelling: "fn" } +// CHECK:STDOUT: - { index: 11, kind: "Identifier", line: {{ *}}7, column: 4, indent: 1, spelling: "F", identifier: 2, has_leading_space: true } +// CHECK:STDOUT: - { index: 12, kind: "OpenParen", line: {{ *}}7, column: 5, indent: 1, spelling: "(", closing_token: 13 } +// CHECK:STDOUT: - { index: 13, kind: "CloseParen", line: {{ *}}7, column: 6, indent: 1, spelling: ")", opening_token: 12 } +// CHECK:STDOUT: - { index: 14, kind: "OpenCurlyBrace", line: {{ *}}7, column: 8, indent: 1, spelling: "{", closing_token: 19, has_leading_space: true } +// CHECK:STDOUT: - { index: 15, kind: "Identifier", line: {{ *}}8, column: 3, indent: 3, spelling: "G", identifier: 3, has_leading_space: true } +// CHECK:STDOUT: - { index: 16, kind: "OpenParen", line: {{ *}}8, column: 4, indent: 3, spelling: "(", closing_token: 17 } +// CHECK:STDOUT: - { index: 17, kind: "CloseParen", line: {{ *}}8, column: 5, indent: 3, spelling: ")", opening_token: 16 } +// CHECK:STDOUT: - { index: 18, kind: "Semi", line: {{ *}}8, column: 6, indent: 3, spelling: ";" } +// CHECK:STDOUT: - { index: 19, kind: "CloseCurlyBrace", line: {{ *}}9, column: 1, indent: 1, spelling: "}", opening_token: 14, has_leading_space: true } +// CHECK:STDOUT: - filename: fail_unclosed_at_eof.carbon +// CHECK:STDOUT: tokens: +// CHECK:STDOUT: - { index: 1, kind: "Fn", line: {{ *}}4, column: 1, indent: 1, spelling: "fn", has_leading_space: true } +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}4, column: 4, indent: 1, spelling: "F", identifier: 0, has_leading_space: true } +// CHECK:STDOUT: - { index: 3, kind: "OpenParen", line: {{ *}}4, column: 5, indent: 1, spelling: "(", closing_token: 4 } +// CHECK:STDOUT: - { index: 4, kind: "CloseParen", line: {{ *}}4, column: 6, indent: 1, spelling: ")", opening_token: 3 } +// CHECK:STDOUT: - { index: 5, kind: "OpenCurlyBrace", line: {{ *}}4, column: 8, indent: 1, spelling: "{", closing_token: 13, has_leading_space: true } +// CHECK:STDOUT: - { index: 6, kind: "Var", line: {{ *}}5, column: 3, indent: 3, spelling: "var", has_leading_space: true } +// CHECK:STDOUT: - { index: 7, kind: "Identifier", line: {{ *}}5, column: 7, indent: 3, spelling: "x", identifier: 1, has_leading_space: true } +// CHECK:STDOUT: - { index: 8, kind: "Colon", line: {{ *}}5, column: 8, indent: 3, spelling: ":" } +// CHECK:STDOUT: - { index: 9, kind: "IntTypeLiteral", line: {{ *}}5, column: 10, indent: 3, spelling: "i32", has_leading_space: true } +// CHECK:STDOUT: - { index: 10, kind: "Equal", line: {{ *}}5, column: 14, indent: 3, spelling: "=", has_leading_space: true } +// CHECK:STDOUT: - { index: 11, kind: "IntLiteral", line: {{ *}}5, column: 16, indent: 3, spelling: "1", value: "1", has_leading_space: true } +// CHECK:STDOUT: - { index: 12, kind: "Semi", line: {{ *}}5, column: 17, indent: 3, spelling: ";" } +// CHECK:STDOUT: - { index: 13, kind: "CloseCurlyBrace", line: {{ *}}5, column: 18, indent: 3, spelling: "}", opening_token: 5, has_leading_space: true, recovery: true } diff --git a/toolchain/lex/token_info.h b/toolchain/lex/token_info.h index d97808b4e91b..d53ab15fb477 100644 --- a/toolchain/lex/token_info.h +++ b/toolchain/lex/token_info.h @@ -58,6 +58,14 @@ class TokenInfo { // state, and look at the next token to check for trailing whitespace. auto has_leading_space() const -> bool { return has_leading_space_; } + // Returns a copy of this token info with the given leading space flag. + [[nodiscard]] auto WithLeadingSpace(bool has_leading_space) const + -> TokenInfo { + TokenInfo info = *this; + info.has_leading_space_ = has_leading_space; + return info; + } + // A collection of methods to access the specific payload included with // particular kinds of tokens. Only the specific payload accessor below may // be used for an info entry of a token with a particular kind, and these diff --git a/toolchain/lex/tokenized_buffer_test.cpp b/toolchain/lex/tokenized_buffer_test.cpp index d15efead89a2..22a8ee7365c1 100644 --- a/toolchain/lex/tokenized_buffer_test.cpp +++ b/toolchain/lex/tokenized_buffer_test.cpp @@ -591,11 +591,14 @@ TEST_F(LexerTest, MatchingGroups) { TEST_F(LexerTest, MismatchedGroups) { auto& buffer1 = compile_helper_.GetTokenizedBuffer("{"); EXPECT_TRUE(buffer1.has_errors()); - EXPECT_THAT(buffer1, HasTokens(llvm::ArrayRef{ - {.kind = TokenKind::FileStart}, - {.kind = TokenKind::Error, .text = "{"}, - {.kind = TokenKind::FileEnd}, - })); + EXPECT_THAT( + buffer1, + HasTokens(llvm::ArrayRef{ + {.kind = TokenKind::FileStart}, + {.kind = TokenKind::OpenCurlyBrace, .column = 1}, + {.kind = TokenKind::CloseCurlyBrace, .column = 2, .recovery = true}, + {.kind = TokenKind::FileEnd}, + })); auto& buffer2 = compile_helper_.GetTokenizedBuffer("}"); EXPECT_TRUE(buffer2.has_errors()); @@ -618,9 +621,7 @@ TEST_F(LexerTest, MismatchedGroups) { {.kind = TokenKind::FileEnd}, })); - // Two recovery tokens inserted at the same point: each must be flagged at - // its own merged index, not the pre-insertion index of the token it was - // inserted before. + // `{((}` is recovered by closing both parens before the `}`, i.e. `{(())}`. auto& buffer3b = compile_helper_.GetTokenizedBuffer("{((}"); EXPECT_TRUE(buffer3b.has_errors()); EXPECT_THAT( @@ -672,9 +673,12 @@ TEST_F(LexerTest, MismatchedGroups) { } TEST_F(LexerTest, Whitespace) { + // The trailing `{(` is recovered by inserting `)` and `}` at end of file, + // giving `{()} {()}`. auto& buffer = compile_helper_.GetTokenizedBuffer("{( } {("); - // Whether there should be whitespace before/after each token. + // Whether there should be whitespace at each boundary, from before the + // first token to after the last. bool space[] = {false, // start-of-file true, @@ -683,13 +687,17 @@ TEST_F(LexerTest, Whitespace) { // ( true, // inserted ) - true, + false, // } true, - // error { + // { false, - // error ( + // ( true, + // inserted ) + false, + // inserted } + false, // EOF false}; int pos = 0; diff --git a/toolchain/parse/testdata/array/fail_syntax.carbon b/toolchain/parse/testdata/array/fail_syntax.carbon index 7843b3408536..3165e96a97f9 100644 --- a/toolchain/parse/testdata/array/fail_syntax.carbon +++ b/toolchain/parse/testdata/array/fail_syntax.carbon @@ -66,21 +66,16 @@ var x: array i32, 1; // --- fail_no_close_paren.carbon -// CHECK:STDERR: fail_no_close_paren.carbon:[[@LINE+16]]:13: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] +// CHECK:STDERR: fail_no_close_paren.carbon:[[@LINE+11]]:13: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] // CHECK:STDERR: var x: array(i32,; // CHECK:STDERR: ^ +// CHECK:STDERR: fail_no_close_paren.carbon:[[@LINE+8]]:18: note: possibly missing `)` here [PossiblyMissingBracketHere] +// CHECK:STDERR: var x: array(i32,; +// CHECK:STDERR: ^ // CHECK:STDERR: -// CHECK:STDERR: fail_no_close_paren.carbon:[[@LINE+12]]:13: error: expected `(` after `array` [ExpectedParenAfter] +// CHECK:STDERR: fail_no_close_paren.carbon:[[@LINE+4]]:18: error: expected expression [ExpectedExpr] // CHECK:STDERR: var x: array(i32,; -// CHECK:STDERR: ^ -// CHECK:STDERR: -// CHECK:STDERR: fail_no_close_paren.carbon:[[@LINE+8]]:13: error: expected `,` in `array(Type, Count)` [ExpectedArrayComma] -// CHECK:STDERR: var x: array(i32,; -// CHECK:STDERR: ^ -// CHECK:STDERR: -// CHECK:STDERR: fail_no_close_paren.carbon:[[@LINE+4]]:13: error: `var` declarations must end with a `;` [ExpectedDeclSemi] -// CHECK:STDERR: var x: array(i32,; -// CHECK:STDERR: ^ +// CHECK:STDERR: ^ // CHECK:STDERR: var x: array(i32,; @@ -198,14 +193,14 @@ var x: array; // CHECK:STDOUT: │ │ ╭─IdentifierNameNotBeforeSignature 'x' // CHECK:STDOUT: │ │ ├─BindingPatternTypeStart ':' // CHECK:STDOUT: │ │ │ ╭─ArrayExprKeyword 'array' -// CHECK:STDOUT: │ │ │ ├─ArrayExprOpenParen 'array' has_error -// CHECK:STDOUT: │ │ │ ├─InvalidParse '(' has_error -// CHECK:STDOUT: │ │ │ ├─ArrayExprComma '(' has_error -// CHECK:STDOUT: │ │ │ ├─InvalidParse '(' has_error -// CHECK:STDOUT: │ │ ├─ArrayExpr 'array' has_error +// CHECK:STDOUT: │ │ │ ├─ArrayExprOpenParen '(' +// CHECK:STDOUT: │ │ │ ├─IntTypeLiteral 'i32' +// CHECK:STDOUT: │ │ │ ├─ArrayExprComma ',' +// CHECK:STDOUT: │ │ │ ├─InvalidParse ')' has_error +// CHECK:STDOUT: │ │ ├─ArrayExpr ')' has_error // CHECK:STDOUT: │ │ ╭─VarBindingPattern ':' // CHECK:STDOUT: │ ├─VariablePattern 'var' -// CHECK:STDOUT: ├─VariableDecl ';' has_error +// CHECK:STDOUT: ├─VariableDecl ';' // CHECK:STDOUT: ├─FileEnd '' // CHECK:STDOUT: (root) // CHECK:STDOUT: - filename: fail_no_length.carbon diff --git a/toolchain/parse/testdata/basics/fail_bracket_recovery.carbon b/toolchain/parse/testdata/basics/fail_bracket_recovery.carbon index d469d3fca38c..da13a9f18546 100644 --- a/toolchain/parse/testdata/basics/fail_bracket_recovery.carbon +++ b/toolchain/parse/testdata/basics/fail_bracket_recovery.carbon @@ -11,22 +11,34 @@ // This is a valid parse tree even though the lex errors. fn F() { var a: array(i32, 3) = (0, 1, 2); - // CHECK:STDERR: fail_bracket_recovery.carbon:[[@LINE+4]]:5: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] + // CHECK:STDERR: fail_bracket_recovery.carbon:[[@LINE+7]]:5: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] // CHECK:STDERR: (a[1) + 1; // CHECK:STDERR: ^ + // CHECK:STDERR: fail_bracket_recovery.carbon:[[@LINE+4]]:7: note: possibly missing `]` here [PossiblyMissingBracketHere] + // CHECK:STDERR: (a[1) + 1; + // CHECK:STDERR: ^ // CHECK:STDERR: (a[1) + 1; - // CHECK:STDERR: fail_bracket_recovery.carbon:[[@LINE+12]]:15: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] + // CHECK:STDERR: fail_bracket_recovery.carbon:[[@LINE+21]]:15: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] // CHECK:STDERR: var x: {.y: (} = {.y = ((}; // CHECK:STDERR: ^ + // CHECK:STDERR: fail_bracket_recovery.carbon:[[@LINE+18]]:16: note: possibly missing `)` here [PossiblyMissingBracketHere] + // CHECK:STDERR: var x: {.y: (} = {.y = ((}; + // CHECK:STDERR: ^ // CHECK:STDERR: - // CHECK:STDERR: fail_bracket_recovery.carbon:[[@LINE+8]]:26: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] + // CHECK:STDERR: fail_bracket_recovery.carbon:[[@LINE+14]]:26: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] // CHECK:STDERR: var x: {.y: (} = {.y = ((}; // CHECK:STDERR: ^ + // CHECK:STDERR: fail_bracket_recovery.carbon:[[@LINE+11]]:28: note: possibly missing `)` here [PossiblyMissingBracketHere] + // CHECK:STDERR: var x: {.y: (} = {.y = ((}; + // CHECK:STDERR: ^ // CHECK:STDERR: - // CHECK:STDERR: fail_bracket_recovery.carbon:[[@LINE+4]]:27: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] + // CHECK:STDERR: fail_bracket_recovery.carbon:[[@LINE+7]]:27: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] // CHECK:STDERR: var x: {.y: (} = {.y = ((}; // CHECK:STDERR: ^ + // CHECK:STDERR: fail_bracket_recovery.carbon:[[@LINE+4]]:28: note: possibly missing `)` here [PossiblyMissingBracketHere] + // CHECK:STDERR: var x: {.y: (} = {.y = ((}; + // CHECK:STDERR: ^ // CHECK:STDERR: var x: {.y: (} = {.y = ((}; } diff --git a/toolchain/parse/testdata/function/declaration.carbon b/toolchain/parse/testdata/function/declaration.carbon index 40026576d214..b5ca491d4c9f 100644 --- a/toolchain/parse/testdata/function/declaration.carbon +++ b/toolchain/parse/testdata/function/declaration.carbon @@ -53,13 +53,12 @@ fn foo bar; // --- fail_missing_implicit_close.carbon -// CHECK:STDERR: fail_missing_implicit_close.carbon:[[@LINE+8]]:7: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] +// CHECK:STDERR: fail_missing_implicit_close.carbon:[[@LINE+7]]:7: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] // CHECK:STDERR: fn Div[(); // CHECK:STDERR: ^ -// CHECK:STDERR: -// CHECK:STDERR: fail_missing_implicit_close.carbon:[[@LINE+4]]:7: error: `fn` declarations must either end with a `;` or have a `{ ... }` block for a definition [ExpectedDeclSemiOrDefinition] +// CHECK:STDERR: fail_missing_implicit_close.carbon:[[@LINE+4]]:10: note: possibly missing `]` here [PossiblyMissingBracketHere] // CHECK:STDERR: fn Div[(); -// CHECK:STDERR: ^ +// CHECK:STDERR: ^ // CHECK:STDERR: fn Div[(); @@ -317,8 +316,12 @@ fn ComplexReturnForm() ->? X.Y(Z); // CHECK:STDOUT: - filename: fail_missing_implicit_close.carbon // CHECK:STDOUT: ╭─FileStart '' // CHECK:STDOUT: │ ╭─FunctionIntroducer 'fn' -// CHECK:STDOUT: │ ├─IdentifierNameNotBeforeSignature 'Div' -// CHECK:STDOUT: ├─FunctionDecl ';' has_error +// CHECK:STDOUT: │ ├─IdentifierNameMaybeBeforeSignature 'Div' +// CHECK:STDOUT: │ │ ╭─ImplicitParamListStart '[' +// CHECK:STDOUT: │ │ │ ╭─TuplePatternStart '(' +// CHECK:STDOUT: │ │ ├─TuplePattern ')' +// CHECK:STDOUT: │ ├─ImplicitParamList ']' +// CHECK:STDOUT: ├─FunctionDecl ';' // CHECK:STDOUT: ├─FileEnd '' // CHECK:STDOUT: (root) // CHECK:STDOUT: - filename: fail_missing_name.carbon diff --git a/toolchain/parse/testdata/match/fail_missing_guard_close_paren.carbon b/toolchain/parse/testdata/match/fail_missing_guard_close_paren.carbon index 83f77a839924..83b12b735fb4 100644 --- a/toolchain/parse/testdata/match/fail_missing_guard_close_paren.carbon +++ b/toolchain/parse/testdata/match/fail_missing_guard_close_paren.carbon @@ -10,9 +10,12 @@ fn f() -> i32 { match (true) { - // CHECK:STDERR: fail_missing_guard_close_paren.carbon:[[@LINE+8]]:21: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] + // CHECK:STDERR: fail_missing_guard_close_paren.carbon:[[@LINE+11]]:21: error: opening symbol without a corresponding closing symbol [UnmatchedOpening] // CHECK:STDERR: case x: bool if (false => { return 1; } // CHECK:STDERR: ^ + // CHECK:STDERR: fail_missing_guard_close_paren.carbon:[[@LINE+8]]:30: note: possibly missing `)` here [PossiblyMissingBracketHere] + // CHECK:STDERR: case x: bool if (false => { return 1; } + // CHECK:STDERR: ^ // CHECK:STDERR: // CHECK:STDERR: fail_missing_guard_close_paren.carbon:[[@LINE+4]]:28: error: expected `)` [ExpectedMatchCaseGuardCloseParen] // CHECK:STDERR: case x: bool if (false => { return 1; } diff --git a/toolchain/parse/testdata/match/fail_missing_guard_open_paren.carbon b/toolchain/parse/testdata/match/fail_missing_guard_open_paren.carbon index 8636f8d7e67d..9627d483bd5f 100644 --- a/toolchain/parse/testdata/match/fail_missing_guard_open_paren.carbon +++ b/toolchain/parse/testdata/match/fail_missing_guard_open_paren.carbon @@ -10,13 +10,12 @@ fn f() -> i32 { match (true) { - // CHECK:STDERR: fail_missing_guard_open_paren.carbon:[[@LINE+8]]:21: error: expected `(` after `if` [ExpectedMatchCaseGuardOpenParen] - // CHECK:STDERR: case x: bool if false) => { return 1; } - // CHECK:STDERR: ^~~~~ - // CHECK:STDERR: - // CHECK:STDERR: fail_missing_guard_open_paren.carbon:[[@LINE+4]]:26: error: closing symbol without a corresponding opening symbol [UnmatchedClosing] + // CHECK:STDERR: fail_missing_guard_open_paren.carbon:[[@LINE+7]]:26: error: closing symbol without a corresponding opening symbol [UnmatchedClosing] // CHECK:STDERR: case x: bool if false) => { return 1; } // CHECK:STDERR: ^ + // CHECK:STDERR: fail_missing_guard_open_paren.carbon:[[@LINE+4]]:21: note: possibly missing `(` here [PossiblyMissingBracketHere] + // CHECK:STDERR: case x: bool if false) => { return 1; } + // CHECK:STDERR: ^ // CHECK:STDERR: case x: bool if false) => { return 1; } } @@ -43,12 +42,15 @@ fn f() -> i32 { // CHECK:STDOUT: │ │ │ │ ├─BoolTypeLiteral 'bool' // CHECK:STDOUT: │ │ │ ├─LetBindingPattern ':' // CHECK:STDOUT: │ │ │ │ ╭─MatchCaseGuardIntroducer 'if' -// CHECK:STDOUT: │ │ │ │ ├─MatchCaseGuardStart 'false' has_error -// CHECK:STDOUT: │ │ │ │ ├─InvalidParse 'false' has_error -// CHECK:STDOUT: │ │ │ ├─MatchCaseGuard 'false' has_error -// CHECK:STDOUT: │ │ │ ╭─MatchCase 'false' has_error -// CHECK:STDOUT: │ │ │ ╭─MatchHandlerStart 'false' has_error -// CHECK:STDOUT: │ │ ├─MatchHandler 'false' has_error +// CHECK:STDOUT: │ │ │ │ ├─MatchCaseGuardStart '(' +// CHECK:STDOUT: │ │ │ │ ├─BoolLiteralFalse 'false' +// CHECK:STDOUT: │ │ │ ├─MatchCaseGuard ')' +// CHECK:STDOUT: │ │ │ ╭─MatchCase '=>' +// CHECK:STDOUT: │ │ │ ╭─MatchHandlerStart '{' +// CHECK:STDOUT: │ │ │ │ ╭─ReturnStatementStart 'return' +// CHECK:STDOUT: │ │ │ │ ├─IntLiteral '1' +// CHECK:STDOUT: │ │ │ ├─ReturnStatement ';' +// CHECK:STDOUT: │ │ ├─MatchHandler '}' // CHECK:STDOUT: │ ├─MatchStatement '}' // CHECK:STDOUT: │ │ ╭─ReturnStatementStart 'return' // CHECK:STDOUT: │ │ ├─IntLiteral '0'