From 22721da92bbf1f0ccb4e06504b54cfdb11dfe00a Mon Sep 17 00:00:00 2001 From: Jon Meow <46229924+jonmeow@users.noreply.github.com> Date: Tue, 25 Jan 2022 10:20:42 -0800 Subject: [PATCH] Avoid unnecessary relexing of the last line (#999) Co-authored-by: Richard Smith --- toolchain/lexer/tokenized_buffer.cpp | 23 ++++++++++-------- toolchain/lexer/tokenized_buffer.h | 17 ++++++++++--- .../7fb2ef54332fca52e531ccaf4d3d4a8a9d664634 | Bin 0 -> 396075 bytes toolchain/parser/parse_tree.cpp | 3 ++- 4 files changed, 28 insertions(+), 15 deletions(-) create mode 100644 toolchain/parser/fuzzer_corpus/7fb2ef54332fca52e531ccaf4d3d4a8a9d664634 diff --git a/toolchain/lexer/tokenized_buffer.cpp b/toolchain/lexer/tokenized_buffer.cpp index a4e27ad4e2d9..12d6f6f1731e 100644 --- a/toolchain/lexer/tokenized_buffer.cpp +++ b/toolchain/lexer/tokenized_buffer.cpp @@ -115,9 +115,9 @@ class TokenizedBuffer::Lexer { Lexer(TokenizedBuffer& buffer, DiagnosticConsumer& consumer) : buffer_(buffer), - translator_(buffer), + translator_(buffer, ¤t_column_), emitter_(translator_, consumer), - token_translator_(buffer), + token_translator_(buffer, ¤t_column_), token_emitter_(token_translator_, consumer), current_line_(buffer.AddLine({0, 0, 0})), current_line_info_(&buffer.GetLineInfo(current_line_)) {} @@ -883,11 +883,12 @@ auto TokenizedBuffer::SourceBufferLocationTranslator::GetLocation( // Find the first line starting after the given location. Note that we can't // inspect `line.length` here because it is not necessarily correct for the - // final line. + // final line during lexing (but will be correct later for the parse tree). auto line_it = std::partition_point( buffer_->line_infos_.begin(), buffer_->line_infos_.end(), [offset](const LineInfo& line) { return line.start <= offset; }); - bool incomplete_line_info = line_it == buffer_->line_infos_.end(); + bool incomplete_line_info = last_line_lexed_to_column_ != nullptr && + line_it == buffer_->line_infos_.end(); // Step back one line to find the line containing the given position. CHECK(line_it != buffer_->line_infos_.begin()) @@ -897,11 +898,12 @@ auto TokenizedBuffer::SourceBufferLocationTranslator::GetLocation( int column_number = offset - line_it->start; // We might still be lexing the last line. If so, check to see if there are - // any newline characters between the start of this line and the given - // location. - if (incomplete_line_info) { - column_number = 0; - for (int64_t i = line_it->start; i != offset; ++i) { + // any newline characters between the position we've finished lexing up to + // and the given location. + if (incomplete_line_info && column_number > *last_line_lexed_to_column_) { + column_number = *last_line_lexed_to_column_; + for (int64_t i = line_it->start + *last_line_lexed_to_column_; i != offset; + ++i) { if (buffer_->source_->Text()[i] == '\n') { ++line_number; column_number = 0; @@ -927,7 +929,8 @@ auto TokenizedBuffer::TokenLocationTranslator::GetLocation(Token token) // Find the corresponding file location. // TODO: Should we somehow indicate in the diagnostic location if this token // is a recovery token that doesn't correspond to the original source? - return SourceBufferLocationTranslator(*buffer_).GetLocation(token_start); + return SourceBufferLocationTranslator(*buffer_, last_line_lexed_to_column_) + .GetLocation(token_start); } } // namespace Carbon diff --git a/toolchain/lexer/tokenized_buffer.h b/toolchain/lexer/tokenized_buffer.h index 8b546bee586c..b0d6638acb2c 100644 --- a/toolchain/lexer/tokenized_buffer.h +++ b/toolchain/lexer/tokenized_buffer.h @@ -246,14 +246,18 @@ class TokenizedBuffer { class TokenLocationTranslator : public DiagnosticLocationTranslator { public: - explicit TokenLocationTranslator(TokenizedBuffer& buffer) - : buffer_(&buffer) {} + explicit TokenLocationTranslator(TokenizedBuffer& buffer, + int* last_line_lexed_to_column) + : buffer_(&buffer), + last_line_lexed_to_column_(last_line_lexed_to_column) {} // Map the given token into a diagnostic location. auto GetLocation(Token token) -> Diagnostic::Location override; private: TokenizedBuffer* buffer_; + // Passed to SourceBufferLocationTranslator. + int* last_line_lexed_to_column_; }; // Lexes a buffer of source code into a tokenized buffer. @@ -366,8 +370,10 @@ class TokenizedBuffer { class SourceBufferLocationTranslator : public DiagnosticLocationTranslator { public: - explicit SourceBufferLocationTranslator(TokenizedBuffer& buffer) - : buffer_(&buffer) {} + explicit SourceBufferLocationTranslator(TokenizedBuffer& buffer, + int* last_line_lexed_to_column) + : buffer_(&buffer), + last_line_lexed_to_column_(last_line_lexed_to_column) {} // Map the given position within the source buffer into a diagnostic // location. @@ -375,6 +381,9 @@ class TokenizedBuffer { private: TokenizedBuffer* buffer_; + // The last lexed column, for determining whether the last line should be + // checked for unlexed newlines. May be null after lexing is complete. + int* last_line_lexed_to_column_; }; // Specifies minimum widths to use when printing a token's fields via diff --git a/toolchain/parser/fuzzer_corpus/7fb2ef54332fca52e531ccaf4d3d4a8a9d664634 b/toolchain/parser/fuzzer_corpus/7fb2ef54332fca52e531ccaf4d3d4a8a9d664634 new file mode 100644 index 0000000000000000000000000000000000000000..df9560bb43669329f1d37d64b3b2b36fe934355e GIT binary patch literal 396075 zcmeI(&59dW8V2CrnaCFsVZ}uCqrZWM!eBaJw>Lv*2)T$r?q+Ty`&ng^ z^%#;V{j}>lr8-7B-EBMMQ6N6wDXAnqs;}NEsrusL;$Ij4d;aR#n>V}vKXW5MfB*pk z1PBoL$O0Gp+vFoZt*0YEfWXHWsM~e3UbLG<{l$LQH$CTe*j;|`pWOOhUR}Ryj<_^u`GE%(`oLlOeSqx)$Ov`)SIT+ zG<8*N+D%i{%~#1QIx22o&(VU@CUUPQ&Y!lX6Ytd-PVV@cPkg5LcUw;;v&rP*o7s1V z+wAZezQ4!&LHBvTxqkij`St8_Hkr+y%x-?TxtUyE{{H9ww%s85+kVq{|KJY|qGxXW zp~vT!9bVbspZFZP$BuSLx}K?qh?0kK4zy@qLyQRTr;0qM=@xuj#V5(F_k@P%R+rs} z=U4mo#mC5y-CG>Kh*S4<{od}!eg~J1GD$eU-y|&lpoZ^!wBB{wW^~a*O?L5BwplxjOFxddk;bvZxYX%B zD~EC8no?_fj4rjtw~o?TsTKF@xp7W+v?qP}FyP9m#`(cR?p8Y2inPhN4EFZyD*ZRE zIJe%>4oO$zTU=|({#GWna@*5>UTVeO%Z+n}qdh6L23$GSI6rvE-AbtyX_Ijo>`}}} ztx?F_4njhq?Xi@T1iKg6;dlns#uhP zO0DQ7wW5a9l3G$L>4>sIYQ>;R#$}*VE4oRos3Enamefi*qO6cwNm9k43{+}GH>njh zq?Xi@T1iKg6;dk(RWdFEm0HnFYDEpHCAI!Gsr8@#-oxHB>7KrjT8V8edZcqkQY&gn zt*9Zjq}Jaiwc@U{8z+5m*FtIybo;W5R%%5zsTDP(mei73Nk^0wQmc=t!Ke&WYDG7x z6*Z)m)RJ0BN0b#(YmilaSq3V#qMOu;8d6JYNv)(K$_lB~$JAg{1}e3po79RLQcG${ zt)wH$3aK^7s=h1(m0HnFYDEpHCAFkh(h+5a)aqkuFe(F;TG35vMGdJXwWLdP`vsTJL%R@9JMQcG$j z9Z^&2>Bi~}*KeO+&n{<^?!VcS+073(H&UtYa^@zc-0 z{Q8H~l3KY<>x}tbA02jMXFTjKckTLUmUdG1`MX94J<@l`?}uzi8|BfvZD*YH(nfZ; zM^h)Q=;%(Lb!t6i%k2Dy98xP0ky^jq?6$$Db=9ugb+c>_8UD1Y+j&(ti>9vY)qJ_A zSDU6;w~MB2)^&H!^`hM@>XHMkqsAVYNIs=dyu7-8@%rl7t1va4ewkFq#blDqZFf@p zUDNx}exL)#$)s&N$7QprJG@O(SJkH7G#%?#$tyZ4ZeP#Qg3~5OK#!(QThod6l3GW; zw-YA{4(=wkB8b$A8onWPuJw=yj#4Y>wtFp{YmIY`9@ia(;Pf_BhyRYRbUHoycINiL z$~S~V!aLsWsY9d9H#bQft(BJ6WLFJ>)26q}C{8 zcLW+PwCOKKbFMYYRXph$!fbd#%Z1SHADIi+jEx%GDT2x7};uYK(Et z(QhM;PTjJ<$c(SlHH70XAI#pv!tqDC8pB^1=eCUfysI%ry6eMaXD)TFH6py9I#cp+ z_oJBcyVfYA-?c_r?tRklTG<8<3v%qEN0NTm8gK>3>~QdqyOmPwo}Hyu+#r2J7&ZK^ z<##Q=Yb6~~R!A+Wb>5bf_1Z2Nu1abpB7WEMSH|5o;IE96t~kGtTG_qu`Iy^BIM<49 zQY&gW*K)4qTr25_vO;P}t@E~=tk-tIa8*((5pk~NT+6vu(h^05)XMIKi#(lcC2UeF z;gDKVOKK%8QB+7Rsde6lll9sz7_LfcB_dKwYDul6C5j5EmE8*$c{%0vo>$P1lT$R*HM5LC~l3Gbi6cth{yB99{;L`~F J^J72l{{f})W|{y1 literal 0 HcmV?d00001 diff --git a/toolchain/parser/parse_tree.cpp b/toolchain/parser/parse_tree.cpp index a4d14b2fbddf..da1f8cffde5e 100644 --- a/toolchain/parser/parse_tree.cpp +++ b/toolchain/parser/parse_tree.cpp @@ -22,7 +22,8 @@ namespace Carbon { auto ParseTree::Parse(TokenizedBuffer& tokens, DiagnosticConsumer& consumer) -> ParseTree { - TokenizedBuffer::TokenLocationTranslator translator(tokens); + TokenizedBuffer::TokenLocationTranslator translator( + tokens, /*last_line_lexed_to_column=*/nullptr); TokenDiagnosticEmitter emitter(translator, consumer); // Delegate to the parser.