mirror of
https://github.com/carbon-language/carbon-lang.git
synced 2026-10-04 15:51:04 +01:00
This moves over to the vanilla upstream GoogleTest pulled in the more expected manner with Bazel. It also adds Abseil and Google Benchmark libraries in the same fashion (there are cross dependencies here). As part of this, also introduce a dependency check test that can enforce basic layering of dependencies. For example, this lets us ensure that non-test Carbon code only depends on LLVM and Clang despite having other libraries available. There remains some cleanup to improve the way these dependency tests work, but this at least ensures we don't regress. I've also provided workarounds to allow both Carbon code and LLVM code to freely be used with GoogleTest (and other `std::ostream` based output code). This is done by extending the code in `//common/ostream.h`. One downside is that it requires opening the `llvm` namespace and adding an ADL_found overload there. I think on balance this is still a win and doesn't make me too nervous. The new version of GoogleTest requires printing more often from matchers and so I've also added several printing routines to types that previously didn't require them. Otherwise, most of the updates are just using the more conventional upstream style of including the headers and adding `ostream.h` where it is needed. I did consider moving code over to use `std::ostream` instead of LLVM's `raw_ostream`, but the advantages of not doing virtual dispatch still seem significant, and it also seems good to retain access to LLVM's formatting utilities built around `raw_ostream` given that we can't pull arbitrary dependencies into Carbon code outside of test code. All of this was slightly motivated by requests for newer features in GoogleTest, but much more-so by my desire to have access to Google Benchmark and Abseil when writing benchmarks. For example, using Abseil's random number generator seems extremely helpful when generating inputs for benchmarks. The growing dependencies between these packages further motivated me to just pull them all in and ensure they worked well.
297 lines
7.0 KiB
C++
297 lines
7.0 KiB
C++
// Part of the Carbon Language project, under the Apache License v2.0 with LLVM
|
|
// Exceptions. See /LICENSE for license information.
|
|
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
|
|
|
#include "toolchain/lexer/string_literal.h"
|
|
|
|
#include <gmock/gmock.h>
|
|
#include <gtest/gtest.h>
|
|
|
|
#include "common/ostream.h"
|
|
#include "toolchain/diagnostics/diagnostic_emitter.h"
|
|
#include "toolchain/lexer/test_helpers.h"
|
|
|
|
namespace Carbon {
|
|
namespace {
|
|
|
|
struct StringLiteralTest : ::testing::Test {
|
|
StringLiteralTest() : error_tracker(ConsoleDiagnosticConsumer()) {}
|
|
|
|
ErrorTrackingDiagnosticConsumer error_tracker;
|
|
|
|
auto Lex(llvm::StringRef text) -> LexedStringLiteral {
|
|
llvm::Optional<LexedStringLiteral> result = LexedStringLiteral::Lex(text);
|
|
assert(result);
|
|
EXPECT_EQ(result->Text(), text);
|
|
return *result;
|
|
}
|
|
|
|
auto Parse(llvm::StringRef text) -> std::string {
|
|
LexedStringLiteral token = Lex(text);
|
|
Testing::SingleTokenDiagnosticTranslator translator(text);
|
|
DiagnosticEmitter<const char*> emitter(translator, error_tracker);
|
|
return token.ComputeValue(emitter);
|
|
}
|
|
};
|
|
|
|
TEST_F(StringLiteralTest, StringLiteralBounds) {
|
|
llvm::StringLiteral valid[] = {
|
|
R"("")",
|
|
R"("""
|
|
""")",
|
|
R"("""
|
|
"foo"
|
|
""")",
|
|
|
|
// Escaped terminators don't end the string.
|
|
R"("\"")",
|
|
R"("\\")",
|
|
R"("\\\"")",
|
|
R"("""
|
|
\"""
|
|
""")",
|
|
R"("""
|
|
"\""
|
|
""")",
|
|
R"("""
|
|
""\"
|
|
""")",
|
|
R"("""
|
|
""\
|
|
""")",
|
|
R"(#"""
|
|
"""\#n
|
|
"""#)",
|
|
|
|
// Only a matching number of '#'s terminates the string.
|
|
R"(#""#)",
|
|
R"(#"xyz"foo"#)",
|
|
R"(##"xyz"#foo"##)",
|
|
R"(#"\""#)",
|
|
|
|
// Escape sequences likewise require a matching number of '#'s.
|
|
R"(#"\#"#"#)",
|
|
R"(#"\"#)",
|
|
R"(#"""
|
|
\#"""#
|
|
"""#)",
|
|
|
|
// #"""# does not start a multiline string literal.
|
|
R"(#"""#)",
|
|
R"(##"""##)",
|
|
};
|
|
|
|
for (llvm::StringLiteral test : valid) {
|
|
llvm::Optional<LexedStringLiteral> result = LexedStringLiteral::Lex(test);
|
|
EXPECT_TRUE(result.hasValue()) << test;
|
|
if (result) {
|
|
EXPECT_EQ(result->Text(), test);
|
|
}
|
|
}
|
|
|
|
llvm::StringLiteral invalid[] = {
|
|
R"(")",
|
|
R"("""
|
|
"")",
|
|
R"("\)", //
|
|
R"("\")",
|
|
R"("\\)", //
|
|
R"("\\\")",
|
|
R"("""
|
|
)",
|
|
R"(#"""
|
|
""")",
|
|
R"(" \
|
|
")",
|
|
};
|
|
|
|
for (llvm::StringLiteral test : invalid) {
|
|
EXPECT_FALSE(LexedStringLiteral::Lex(test).hasValue())
|
|
<< "`" << test << "`";
|
|
}
|
|
}
|
|
|
|
TEST_F(StringLiteralTest, StringLiteralContents) {
|
|
// We use ""s strings to handle embedded nul characters below.
|
|
using std::operator""s;
|
|
|
|
std::pair<llvm::StringLiteral, llvm::StringLiteral> testcases[] = {
|
|
// Empty strings.
|
|
{R"("")", ""},
|
|
|
|
{R"(
|
|
"""
|
|
"""
|
|
)",
|
|
""},
|
|
|
|
// Nearly-empty strings.
|
|
{R"(
|
|
"""
|
|
|
|
"""
|
|
)",
|
|
"\n"},
|
|
|
|
// Lines containing only whitespace are treated as empty even if they
|
|
// contain tabs.
|
|
{"\"\"\"\n\t \t\n\"\"\"", "\n"},
|
|
|
|
// Indent removal.
|
|
{R"(
|
|
"""file type indicator
|
|
indented contents \
|
|
"""
|
|
)",
|
|
" indented contents "},
|
|
|
|
// Removal of tabs in indent and suffix.
|
|
{"\"\"\"\n \t hello \t \n \t \"\"\"", " hello\n"},
|
|
|
|
{R"(
|
|
"""
|
|
hello
|
|
world
|
|
|
|
end of test
|
|
"""
|
|
)",
|
|
" hello\nworld\n\n end of test\n"},
|
|
|
|
// Escape sequences.
|
|
{R"(
|
|
"\x14,\u{1234},\u{00000010},\n,\r,\t,\0,\",\',\\"
|
|
)",
|
|
llvm::StringLiteral::withInnerNUL(
|
|
"\x14,\xE1\x88\xB4,\x10,\x0A,\x0D,\x09,\x00,\x22,\x27,\x5C")},
|
|
|
|
{R"(
|
|
"\0A\x1234"
|
|
)",
|
|
llvm::StringLiteral::withInnerNUL("\0A\x12"
|
|
"34")},
|
|
|
|
{R"(
|
|
"\u{D7FF},\u{E000},\u{10FFFF}"
|
|
)",
|
|
"\xED\x9F\xBF,\xEE\x80\x80,\xF4\x8F\xBF\xBF"},
|
|
|
|
// Escape sequences in 'raw' strings.
|
|
{R"(
|
|
#"\#x00,\#xFF,\#u{56789},\#u{ABCD},\#u{00000000000000000EF}"#
|
|
)",
|
|
llvm::StringLiteral::withInnerNUL(
|
|
"\x00,\xFF,\xF1\x96\x9E\x89,\xEA\xAF\x8D,\xC3\xAF")},
|
|
|
|
{R"(
|
|
##"\n,\#n,\##n,\##\##n,\##\###n"##
|
|
)",
|
|
"\\n,\\#n,\n,\\##n,\\###n"},
|
|
|
|
// Trailing whitespace handling.
|
|
{"\"\"\"\n Hello \\\n World \t \n Bye! \\\n \"\"\"",
|
|
"Hello World\nBye! "},
|
|
};
|
|
|
|
for (auto [test, contents] : testcases) {
|
|
error_tracker.Reset();
|
|
auto value = Parse(test.trim());
|
|
EXPECT_FALSE(error_tracker.SeenError()) << "`" << test << "`";
|
|
EXPECT_EQ(value, contents);
|
|
}
|
|
}
|
|
|
|
TEST_F(StringLiteralTest, StringLiteralBadIndent) {
|
|
std::pair<llvm::StringLiteral, llvm::StringLiteral> testcases[] = {
|
|
// Indent doesn't match the last line.
|
|
{"\"\"\"\n \tx\n \"\"\"", "x\n"},
|
|
{"\"\"\"\n x\n \"\"\"", "x\n"},
|
|
{"\"\"\"\n x\n\t\"\"\"", "x\n"},
|
|
{"\"\"\"\n ok\n bad\n \"\"\"", "ok\nbad\n"},
|
|
{"\"\"\"\n bad\n ok\n \"\"\"", "bad\nok\n"},
|
|
{"\"\"\"\n escaped,\\\n bad\n \"\"\"", "escaped,bad\n"},
|
|
|
|
// Indent on last line is followed by text.
|
|
{"\"\"\"\n x\n x\"\"\"", "x\nx"},
|
|
{"\"\"\"\n x\n x\"\"\"", " x\nx"},
|
|
{"\"\"\"\n x\n x\"\"\"", "x\nx"},
|
|
};
|
|
|
|
for (auto [test, contents] : testcases) {
|
|
error_tracker.Reset();
|
|
auto value = Parse(test);
|
|
EXPECT_TRUE(error_tracker.SeenError()) << "`" << test << "`";
|
|
EXPECT_EQ(value, contents);
|
|
}
|
|
}
|
|
|
|
TEST_F(StringLiteralTest, StringLiteralBadEscapeSequence) {
|
|
llvm::StringLiteral testcases[] = {
|
|
R"("\a")",
|
|
R"("\b")",
|
|
R"("\e")",
|
|
R"("\f")",
|
|
R"("\v")",
|
|
R"("\?")",
|
|
R"("\1")",
|
|
R"("\9")",
|
|
|
|
// \0 can't be followed by a decimal digit.
|
|
R"("\01")",
|
|
R"("\09")",
|
|
|
|
// \x requires two (uppercase) hexadecimal digits.
|
|
R"("\x")",
|
|
R"("\x0")",
|
|
R"("\x0G")",
|
|
R"("\xab")",
|
|
R"("\x\n")",
|
|
R"("\x\"")",
|
|
|
|
// \u requires a braced list of one or more hexadecimal digits.
|
|
R"("\u")",
|
|
R"("\u?")",
|
|
R"("\u\"")",
|
|
R"("\u{")",
|
|
R"("\u{}")",
|
|
R"("\u{A")",
|
|
R"("\u{G}")",
|
|
R"("\u{0000012323127z}")",
|
|
R"("\u{-3}")",
|
|
|
|
// \u must specify a non-surrogate code point.
|
|
R"("\u{110000}")",
|
|
R"("\u{000000000000000000000000000000000110000}")",
|
|
R"("\u{D800}")",
|
|
R"("\u{DFFF}")",
|
|
};
|
|
|
|
for (llvm::StringLiteral test : testcases) {
|
|
error_tracker.Reset();
|
|
auto value = Parse(test);
|
|
EXPECT_TRUE(error_tracker.SeenError()) << "`" << test << "`";
|
|
// TODO: Test value produced by error recovery.
|
|
}
|
|
}
|
|
|
|
TEST_F(StringLiteralTest, TabInString) {
|
|
auto value = Parse("\"x\ty\"");
|
|
EXPECT_TRUE(error_tracker.SeenError());
|
|
EXPECT_EQ(value, "x\ty");
|
|
}
|
|
|
|
TEST_F(StringLiteralTest, TabAtEndOfString) {
|
|
auto value = Parse("\"\t\t\t\"");
|
|
EXPECT_TRUE(error_tracker.SeenError());
|
|
EXPECT_EQ(value, "\t\t\t");
|
|
}
|
|
|
|
TEST_F(StringLiteralTest, TabInBlockString) {
|
|
auto value = Parse("\"\"\"\nx\ty\n\"\"\"");
|
|
EXPECT_TRUE(error_tracker.SeenError());
|
|
EXPECT_EQ(value, "x\ty\n");
|
|
}
|
|
|
|
} // namespace
|
|
} // namespace Carbon
|