diff --git a/.codespell_ignore b/.codespell_ignore index c612f438f3ff..4660b644c9ff 100644 --- a/.codespell_ignore +++ b/.codespell_ignore @@ -18,6 +18,7 @@ groupt indext inout isELF +iterm parameteras pullrequest rightt diff --git a/common/filesystem.cpp b/common/filesystem.cpp index 904beb2c5db0..7548abcc0856 100644 --- a/common/filesystem.cpp +++ b/common/filesystem.cpp @@ -103,15 +103,8 @@ auto Internal::FileRefBase::ReadFileToString() auto Internal::FileRefBase::WriteFileFromString(llvm::StringRef str) -> ErrorOr { CARBON_RETURN_IF_ERROR(SeekFromBeginning(0)); - auto bytes = llvm::ArrayRef( - reinterpret_cast(str.data()), str.size()); - while (!bytes.empty()) { - auto write_result = WriteFromBuffer(bytes); - if (!write_result.ok()) { - return std::move(write_result).error(); - } - bytes = *write_result; - } + CARBON_RETURN_IF_ERROR(WriteCompleteBuffer(llvm::ArrayRef( + reinterpret_cast(str.data()), str.size()))); CARBON_RETURN_IF_ERROR(Truncate(str.size())); return Success(); } diff --git a/common/filesystem.h b/common/filesystem.h index 33713954d425..436616dccf4f 100644 --- a/common/filesystem.h +++ b/common/filesystem.h @@ -219,6 +219,24 @@ namespace Internal { class FileRefBase; } // namespace Internal +// Convenience type defs for the three access combinations. +using ReadFileRef = FileRef; +using WriteFileRef = FileRef; +using ReadWriteFileRef = FileRef; + +// Returns constant references to the standard streams the process is started +// with. +// +// The returned references are non-owning: the process shares these descriptors +// with whatever started it, closing them is never correct, and unrelated code +// throughout the process may be reading or writing the same descriptor. +// +// Their descriptor numbers are fixed by the platform rather than discovered at +// runtime, so these are constant expressions. +consteval auto Stdin() -> ReadFileRef; +consteval auto Stdout() -> WriteFileRef; +consteval auto Stderr() -> WriteFileRef; + // Returns a constant `Dir` object that models the open current working // directory. // @@ -348,7 +366,13 @@ class Internal::FileRefBase { FileRefBase() = default; // Returns true if this refers to a valid open file, and false otherwise. - auto is_valid() const -> bool { return fd_ != -1; } + constexpr auto is_valid() const -> bool { return fd_ != -1; } + + // Non-portable API only available on Unix-like systems. Returns the + // underlying file descriptor, for the platform calls this type doesn't wrap, + // such as `isatty` and `ioctl`. The descriptor remains owned by whatever owns + // this file. + constexpr auto unix_fd() const -> int { return fd_; } // Reads the file status. // @@ -405,6 +429,24 @@ class Internal::FileRefBase { auto WriteFromBuffer(llvm::ArrayRef buffer) -> ErrorOr, FdError>; + // Writes the complete contents of the provided buffer. + // + // Unlike `WriteFromBuffer`, this doesn't return until every byte has been + // written or an error occurs. It repeats `WriteFromBuffer` over whatever is + // left, so each write is issued for as much of the buffer as remains and the + // whole is written in as few writes as the file allows. Anything else writing + // to the same file can only interleave between those writes, which leaves no + // room to interleave at all when the file accepts the buffer in one write. + // + // On an error, an unspecified prefix of the buffer has already been written + // and can't be un-written. How much isn't reported; a caller that needs to + // know should drive `WriteFromBuffer` itself. + // + // This method retries `EINTR` on Unix-like systems and returns other errors + // to the caller. + auto WriteCompleteBuffer(llvm::ArrayRef buffer) + -> ErrorOr; + // Returns an LLVM `raw_fd_ostream` that writes to this file. // // Note that this doesn't expose any write errors here, those will surface @@ -458,7 +500,7 @@ class Internal::FileRefBase { Duration poll_interval = {}) -> ErrorOr; protected: - explicit FileRefBase(int fd) : fd_(fd) {} + explicit constexpr FileRefBase(int fd) : fd_(fd) {} // Note: this should only be used or made part of the public API by subclasses // that provide *ownership* of the open file. It is implemented here to @@ -536,6 +578,9 @@ class FileRef : public Internal::FileRefBase { auto WriteFromBuffer(llvm::ArrayRef buffer) -> ErrorOr, FdError> requires Writeable; + auto WriteCompleteBuffer(llvm::ArrayRef buffer) + -> ErrorOr + requires Writeable; auto WriteStream() -> llvm::raw_fd_ostream requires Writeable; auto ReadFileToString() -> ErrorOr @@ -546,16 +591,14 @@ class FileRef : public Internal::FileRefBase { protected: friend File; friend DirRef; + friend consteval auto Stdin() -> ReadFileRef; + friend consteval auto Stdout() -> WriteFileRef; + friend consteval auto Stderr() -> WriteFileRef; // Other constructors from the base are also available, but remain protected. using FileRefBase::FileRefBase; }; -// Convenience type defs for the three access combinations. -using ReadFileRef = FileRef; -using WriteFileRef = FileRef; -using ReadWriteFileRef = FileRef; - // An owning handle to an open file. // // This extends the `FileRef` API to provide ownership of the file handle. Most @@ -1320,6 +1363,10 @@ inline auto DurationToTimespec(Duration d) -> timespec { } // namespace Internal +consteval auto Stdin() -> ReadFileRef { return ReadFileRef(STDIN_FILENO); } +consteval auto Stdout() -> WriteFileRef { return WriteFileRef(STDOUT_FILENO); } +consteval auto Stderr() -> WriteFileRef { return WriteFileRef(STDERR_FILENO); } + consteval auto Cwd() -> Dir { return Dir(AT_FDCWD); } inline auto FileLock::Destroy() -> void { @@ -1431,6 +1478,14 @@ inline auto Internal::FileRefBase::WriteFromBuffer( } } +inline auto Internal::FileRefBase::WriteCompleteBuffer( + llvm::ArrayRef buffer) -> ErrorOr { + while (!buffer.empty()) { + CARBON_ASSIGN_OR_RETURN(buffer, WriteFromBuffer(buffer)); + } + return Success(); +} + inline auto Internal::FileRefBase::WriteStream() -> llvm::raw_fd_ostream { return llvm::raw_fd_ostream(fd_, /*shouldClose=*/false); } @@ -1495,6 +1550,14 @@ auto FileRef::WriteFromBuffer(llvm::ArrayRef buffer) return FileRefBase::WriteFromBuffer(buffer); } +template +auto FileRef::WriteCompleteBuffer(llvm::ArrayRef buffer) + -> ErrorOr + requires Writeable +{ + return FileRefBase::WriteCompleteBuffer(buffer); +} + template auto FileRef::WriteStream() -> llvm::raw_fd_ostream requires Writeable diff --git a/common/filesystem_test.cpp b/common/filesystem_test.cpp index 1554a808cb88..09fa20f304d5 100644 --- a/common/filesystem_test.cpp +++ b/common/filesystem_test.cpp @@ -411,6 +411,58 @@ TEST_F(FilesystemTest, WriteStream) { EXPECT_THAT(dir_.ReadFileToString("test"), IsSuccess(Eq(content_str))); } +TEST_F(FilesystemTest, WriteCompleteBuffer) { + std::string content_str = "0123456789"; + auto bytes = llvm::ArrayRef( + reinterpret_cast(content_str.data()), + content_str.size()); + + auto write = dir_.OpenWriteOnly("test", CreationOptions::CreateNew); + ASSERT_THAT(write, IsSuccess(_)); + EXPECT_THAT(write->WriteCompleteBuffer(bytes), IsSuccess(_)); + // Writing appends rather than replacing, unlike `WriteFileFromString`. + EXPECT_THAT(write->WriteCompleteBuffer(bytes), IsSuccess(_)); + // An empty buffer is a no-op rather than an error. + EXPECT_THAT(write->WriteCompleteBuffer(llvm::ArrayRef()), + IsSuccess(_)); + (*std::move(write)).Close().Check(); + + EXPECT_THAT(dir_.ReadFileToString("test"), + IsSuccess(Eq(content_str + content_str))); +} + +TEST_F(FilesystemTest, StandardStreams) { + // The standard streams name descriptors the process already has, so these + // are constants and never open or close anything. + static_assert(Stdin().unix_fd() == STDIN_FILENO); + static_assert(Stdout().unix_fd() == STDOUT_FILENO); + static_assert(Stderr().unix_fd() == STDERR_FILENO); + EXPECT_TRUE(Stderr().is_valid()); + + // Writing through one reaches the descriptor. Tests run with stdout captured, + // so this uses a pipe put in its place for the duration. + int fds[2]; + ASSERT_EQ(pipe(fds), 0); + int saved = dup(STDOUT_FILENO); + ASSERT_GE(saved, 0); + ASSERT_GE(dup2(fds[1], STDOUT_FILENO), 0); + + llvm::StringRef message = "through stdout"; + auto result = Stdout().WriteCompleteBuffer(llvm::ArrayRef( + reinterpret_cast(message.data()), message.size())); + + ASSERT_GE(dup2(saved, STDOUT_FILENO), 0); + ASSERT_EQ(close(saved), 0); + ASSERT_EQ(close(fds[1]), 0); + EXPECT_THAT(result, IsSuccess(_)); + + char buffer[64]; + ssize_t n = read(fds[0], buffer, sizeof(buffer)); + ASSERT_EQ(close(fds[0]), 0); + ASSERT_GT(n, 0); + EXPECT_EQ(llvm::StringRef(buffer, n), message); +} + TEST_F(FilesystemTest, Rename) { // Rename a file within a directory. ASSERT_THAT(dir_.WriteFileFromString("file1", "content1"), IsSuccess(_)); diff --git a/common/terminal/BUILD b/common/terminal/BUILD new file mode 100644 index 000000000000..034d50a6b970 --- /dev/null +++ b/common/terminal/BUILD @@ -0,0 +1,195 @@ +# Part of the Carbon Language project, under the Apache License v2.0 with LLVM +# Exceptions. See /LICENSE for license information. +# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +# Terminal rendering: what the attached terminal can do, and how to draw styled +# text for it. Used for diagnostic rendering and for command line output. + +load("@rules_shell//shell:sh_test.bzl", "sh_test") +load("//bazel/cc_rules:defs.bzl", "cc_binary", "cc_library", "cc_test") + +package(default_visibility = ["//visibility:public"]) + +cc_library( + name = "output_buffer_ref", + hdrs = ["output_buffer_ref.h"], + deps = ["@llvm-project//llvm:Support"], +) + +cc_test( + name = "output_buffer_ref_test", + size = "small", + srcs = ["output_buffer_ref_test.cpp"], + deps = [ + ":output_buffer_ref", + "//testing/base:gtest_main", + "@googletest//:gtest", + "@llvm-project//llvm:Support", + ], +) + +cc_library( + name = "color", + srcs = ["color.cpp"], + hdrs = ["color.h"], + deps = [ + ":output_buffer_ref", + "//common:check", + "//common:ostream", + "@llvm-project//llvm:Support", + ], +) + +cc_test( + name = "color_test", + size = "small", + srcs = ["color_test.cpp"], + deps = [ + ":color", + "//common:ostream", + "//testing/base:gtest_main", + "@googletest//:gtest", + "@llvm-project//llvm:Support", + ], +) + +cc_library( + name = "style", + srcs = ["style.cpp"], + hdrs = ["style.h"], + deps = [ + ":color", + ":output_buffer_ref", + "//common:check", + "//common:ostream", + "@llvm-project//llvm:Support", + ], +) + +cc_test( + name = "style_test", + size = "small", + srcs = ["style_test.cpp"], + deps = [ + ":style", + "//common:ostream", + "//common:raw_string_ostream", + "//testing/base:gtest_main", + "@googletest//:gtest", + "@llvm-project//llvm:Support", + ], +) + +cc_library( + name = "capabilities", + srcs = ["capabilities.cpp"], + hdrs = ["capabilities.h"], + deps = [ + ":color", + "//common:filesystem", + "@llvm-project//llvm:Support", + ], +) + +cc_test( + name = "capabilities_test", + size = "small", + srcs = ["capabilities_test.cpp"], + deps = [ + ":capabilities", + "//common:filesystem", + "//testing/base:gtest_main", + "@googletest//:gtest", + ], +) + +cc_library( + name = "metrics", + srcs = ["metrics.cpp"], + hdrs = ["metrics.h"], + deps = [ + ":capabilities", + "//common:check", + "@llvm-project//llvm:Support", + ], +) + +cc_test( + name = "metrics_test", + size = "small", + srcs = ["metrics_test.cpp"], + deps = [ + ":metrics", + "//testing/base:gtest_main", + "@googletest//:gtest", + "@llvm-project//llvm:Support", + ], +) + +cc_library( + name = "buffer", + srcs = ["buffer.cpp"], + hdrs = ["buffer.h"], + deps = [ + ":capabilities", + ":color", + ":metrics", + ":output_buffer_ref", + ":style", + "//common:check", + "//common:filesystem", + "@llvm-project//llvm:Support", + ], +) + +cc_test( + name = "buffer_test", + size = "small", + srcs = ["buffer_test.cpp"], + deps = [ + ":buffer", + ":metrics", + "//common:filesystem", + "//testing/base:gtest_main", + "@googletest//:gtest", + "@llvm-project//llvm:Support", + ], +) + +cc_test( + name = "pressure_test", + size = "small", + srcs = ["pressure_test.cpp"], + deps = [ + ":buffer", + ":capabilities", + ":metrics", + ":style", + "//testing/base:gtest_main", + "@googletest//:gtest", + "@llvm-project//llvm:Support", + ], +) + +cc_binary( + name = "terminal_benchmark", + testonly = 1, + srcs = ["terminal_benchmark.cpp"], + deps = [ + ":buffer", + ":capabilities", + ":color", + ":style", + "//testing/base:benchmark_main", + "@abseil-cpp//absl/random", + "@google_benchmark//:benchmark", + "@llvm-project//llvm:Support", + ], +) + +sh_test( + name = "terminal_benchmark_test", + size = "small", + srcs = [":terminal_benchmark"], + args = ["--benchmark_dry_run"], +) diff --git a/common/terminal/buffer.cpp b/common/terminal/buffer.cpp new file mode 100644 index 000000000000..2cb0c5998f80 --- /dev/null +++ b/common/terminal/buffer.cpp @@ -0,0 +1,554 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#include "common/terminal/buffer.h" + +#include +#include +#include +#include + +#include "common/check.h" +#include "llvm/ADT/STLExtras.h" +#include "llvm/ADT/Sequence.h" +#include "llvm/ADT/SmallString.h" +#include "llvm/Support/ConvertUTF.h" +#include "llvm/Support/Unicode.h" + +namespace Carbon::Terminal { + +// The most bytes of combining marks kept on one cell. Text stacking more than +// this is either adversarial or already illegible, and keeping all of it would +// let a single column of output carry unbounded bytes. +static constexpr size_t MaxCombiningBytes = 32; + +// Glyphs for every combination of line directions, indexed by the direction +// bits. +static constexpr std::array Utf8LineGlyphs = { + U'·', // (none): a line between one center and itself, which is a point + U'╴', // left + U'╶', // right + U'─', // left, right + U'╵', // up + U'╯', // left, up + U'╰', // right, up + U'┴', // left, right, up + U'╷', // down + U'╮', // left, down + U'╭', // right, down + U'┬', // left, right, down + U'│', // up, down + U'┤', // left, up, down + U'├', // right, up, down + U'┼', // left, right, up, down +}; + +// The ASCII stand-ins, which can only distinguish horizontal, vertical, and +// everything else. +static constexpr std::array AsciiLineGlyphs = { + U'+', U'-', U'-', U'-', U'|', U'+', U'+', U'+', + U'|', U'+', U'+', U'+', U'|', U'+', U'+', U'+', +}; + +// Returns the next tab stop after `x` on a line whose stops are `tab_width` +// columns apart counting from `origin`, which `x` must not be left of. +static auto NextTabStop(int x, int origin, int tab_width) -> int { + CARBON_DCHECK(x >= origin, "Column {0} is left of the origin {1}.", x, + origin); + return origin + ((x - origin) / tab_width + 1) * tab_width; +} + +Buffer::Buffer(int columns, Charset charset, int tab_width) + : columns_(columns), + width_(columns), + tab_width_(tab_width), + metrics_(charset) { + CARBON_CHECK(columns > 0 && columns <= MaxColumns, + "Buffer width must be in [1, {0}], but was {1}.", MaxColumns, + columns); + CARBON_CHECK(tab_width > 0 && tab_width <= MaxTabWidth, + "Tab width must be in [1, {0}], but was {1}.", MaxTabWidth, + tab_width); +} + +auto Buffer::height() const -> int { + return static_cast(cells_.size()) / width_; +} + +auto Buffer::EnsureRow(int y) -> void { + CARBON_CHECK(y >= 0 && y < MaxRows, "Row {0} is outside [0, {1}).", y, + MaxRows); + if (y < height()) { + return; + } + // Rows are added at the end and nothing already in the grid moves, so this + // asks for exactly the rows wanted and lets the vector amortize the growing. + cells_.resize(static_cast(y + 1) * width_); +} + +auto Buffer::EnsureColumn(int x) -> void { + CARBON_CHECK(x >= 0 && x < MaxColumns, "Column {0} is outside [0, {1}).", x, + MaxColumns); + if (x < width_) { + return; + } + // Widening moves every row, so it grows by halves rather than to exactly what + // was asked: a row drawn one code point at a time would otherwise copy the + // whole grid on every one of them. Growth stops at the bound, which is what + // holds the product of the two dimensions inside what a cell index can + // represent. + int width = std::min(std::max(x + 1, width_ + width_ / 2), MaxColumns); + + int rows = height(); + llvm::SmallVector new_cells(static_cast(rows) * width); + for (int y : llvm::seq(rows)) { + llvm::copy( + llvm::ArrayRef(cells_).slice(static_cast(y) * width_, width_), + new_cells.begin() + static_cast(y) * width); + } + cells_ = std::move(new_cells); + + // A mark's key is a cell index, which depends on the width, so each is + // recomputed for the new one. + llvm::DenseMap new_combining_marks; + new_combining_marks.reserve(combining_marks_.size()); + for (auto& [index, marks] : combining_marks_) { + new_combining_marks.insert( + {index / width_ * width + index % width_, std::move(marks)}); + } + combining_marks_ = std::move(new_combining_marks); + + width_ = width; +} + +auto Buffer::ClearCells(int x, int y, int width) -> void { + CARBON_CHECK( + x >= 0 && width >= 0 && x + width <= width_ && y >= 0 && y < height(), + "Clearing [{0}, {1}) of row {2} reaches outside the {3}x{4} cells the " + "buffer holds.", + x, x + width, y, width_, height()); + + // A cleared range must not leave half of a double-width character behind, so + // it extends over either half that crosses its edges. + int begin = x; + if (begin > 0 && CellAt(begin, y).is_continuation) { + --begin; + } + int end = x + width; + if (end < width_ && CellAt(end, y).is_continuation) { + ++end; + } + + for (int i = begin; i < end; ++i) { + CellAt(i, y) = Cell(); + combining_marks_.erase(CellIndex(i, y)); + } +} + +auto Buffer::AttachCombiningMark(int x, int y, char32_t code_point) -> void { + // A mark has nowhere to go when no cell precedes it, so it is dropped. + if (x <= 0 || x > width_ || y < 0 || y >= height()) { + return; + } + // The left half of a double-width character is never itself a continuation, + // so stepping back from one always lands on a real character. + int base = x - 1; + if (CellAt(base, y).is_continuation) { + --base; + } + CARBON_CHECK(base >= 0, "A continuation cell at column zero has no base."); + + Utf8Storage storage; + llvm::StringRef encoded = EncodeUtf8(code_point, storage); + std::string& marks = combining_marks_[CellIndex(base, y)]; + if (marks.size() + encoded.size() > MaxCombiningBytes) { + return; + } + marks.append(encoded.data(), encoded.size()); +} + +auto Buffer::DrawCodePoint(int x, int y, char32_t code_point, + const Style& style) -> DrawEnd { + CheckOrigin(x, y); + return {.x = PlaceCodePoint(x, y, code_point, style), .y = y}; +} + +auto Buffer::PlaceCodePoint(int x, int y, char32_t code_point, + const Style& style) -> int { + CARBON_DCHECK(x >= 0 && y >= 0, + "Placing at ({0}, {1}), which no walk should reach.", x, y); + + int width = metrics_.CodePointWidth(code_point); + if (width == 0) { + AttachCombiningMark(x, y, code_point); + return x; + } + code_point = metrics_.RenderedCodePoint(code_point); + + // Both bounds are reached by what the text holds rather than by where the + // caller aimed -- a word overhanging the target width, or newlines running + // past the rows a grid can index -- so past either one nothing is drawn and + // the column still advances, which is what keeps measuring and drawing + // answering the same thing. A double-width character needs both its columns, + // so one that would only half fit is past the edge like any other: splitting + // it would leave the terminal rendering half a character. + if (y >= MaxRows || x > MaxColumns - width) { + return x + width; + } + + EnsureColumn(x + width - 1); + EnsureRow(y); + ClearCells(x, y, width); + + Cell& cell = CellAt(x, y); + cell.code_point = code_point; + cell.style = style; + // Nothing is wider than two columns, so the second is the only continuation + // there can be. + if (width > 1) { + Cell& continuation = CellAt(x + 1, y); + continuation.style = style; + continuation.is_continuation = true; + } + return x + width; +} + +// Returns the glyphs a cell's directions are read from. +static auto LineGlyphs(Charset charset) -> const std::array& { + return charset == Charset::Utf8 ? Utf8LineGlyphs : AsciiLineGlyphs; +} + +auto Buffer::DrawLine(int x, int y, uint8_t directions, const Style& style) + -> void { + CARBON_DCHECK(directions <= LineDirections, + "Direction bits {0} name no glyph.", directions); + EnsureColumn(x); + EnsureRow(y); + + uint8_t existing = CellAt(x, y).lines; + if (existing == 0) { + // Whatever is here isn't a line. Clearing also removes either half of a + // double-width character the cell was part of. + ClearCells(x, y, 1); + } + + Cell& cell = CellAt(x, y); + cell.lines = existing | directions | LineCell; + cell.code_point = LineGlyphs(metrics_.charset())[cell.lines & LineDirections]; + cell.style = style; +} + +// Checks that a line of `length` starting at `position` stays within `limit`, +// which is the width for a horizontal line and `MaxRows` for a vertical one. +// +// Unlike text, a line has no reason to reach outside what it is being drawn +// into: nothing about it is unbreakable, and a layout that put one there +// computed the wrong extent. +static auto CheckLineFits(int position, int length, int limit) -> void { + CARBON_CHECK(length >= 0 && position <= limit - length, + "A line of {0} at {1} runs outside the {2} available to it.", + length, position, limit); +} + +auto Buffer::DrawHorizontalLine(int x, int y, int length, const Style& style, + LineEnd start, LineEnd end) -> DrawEnd { + CheckOrigin(x, y); + CheckLineFits(x, length, columns_); + for (int i : llvm::seq(length)) { + // A cell in the middle of the line is entered from one side and left by the + // other. An end cell is only left towards the rest of the line, unless that + // end runs out through the cell's own side. + uint8_t directions = + (i > 0 || start == LineEnd::Edge ? LineLeft : 0) | + (i + 1 < length || end == LineEnd::Edge ? LineRight : 0); + DrawLine(x + i, y, directions, style); + } + return {.x = x + length, .y = y}; +} + +auto Buffer::DrawVerticalLine(int x, int y, int length, const Style& style, + LineEnd start, LineEnd end) -> DrawEnd { + CheckOrigin(x, y); + CheckLineFits(y, length, MaxRows); + for (int i : llvm::seq(length)) { + uint8_t directions = + (i > 0 || start == LineEnd::Edge ? LineUp : 0) | + (i + 1 < length || end == LineEnd::Edge ? LineDown : 0); + DrawLine(x, y + i, directions, style); + } + return {.x = x, .y = y + length}; +} + +auto Buffer::DrawBox(int x, int y, int box_width, int box_height, + const Style& style) -> DrawEnd { + CheckOrigin(x, y); + CheckLineFits(x, box_width, columns_); + CheckLineFits(y, box_height, MaxRows); + if (box_width == 0 || box_height == 0) { + return {.x = x, .y = y}; + } + DrawHorizontalLine(x, y, box_width, style); + DrawHorizontalLine(x, y + box_height - 1, box_width, style); + DrawVerticalLine(x, y, box_height, style); + DrawVerticalLine(x + box_width - 1, y, box_height, style); + return {.x = x + box_width, .y = y + box_height}; +} + +template +auto Buffer::WalkText(int x, int y, int margin, llvm::StringRef text, + PlaceFn place) const -> DrawEnd { + CheckTextSize(text); + CARBON_CHECK( + margin >= 0 && margin <= x && x < columns_ && y >= 0 && y < MaxRows, + "Text at ({0}, {1}) with a margin of {2} is outside the {3} " + "columns and {4} rows a buffer covers, or left of its margin.", + x, y, margin, columns_, MaxRows); + + int cur_x = x; + int cur_y = y; + + while (!text.empty()) { + char32_t code_point = metrics_.TakeCodePoint(text); + if (code_point == '\n') { + cur_x = margin; + ++cur_y; + continue; + } + if (code_point == '\r') { + cur_x = margin; + continue; + } + if (code_point == '\t') { + int stop = NextTabStop(cur_x, margin, tab_width_); + for (; cur_x < stop; ++cur_x) { + place(cur_x, cur_y, U' '); + } + continue; + } + + cur_x = place(cur_x, cur_y, code_point); + } + + return {.x = cur_x, .y = cur_y}; +} + +auto Buffer::DrawText(int x, int y, int margin, llvm::StringRef text, + const Style& style) -> DrawEnd { + return WalkText(x, y, margin, text, + [&](int cur_x, int cur_y, char32_t code_point) { + return PlaceCodePoint(cur_x, cur_y, code_point, style); + }); +} + +auto Buffer::MeasureText(int x, int y, int margin, llvm::StringRef text) const + -> DrawEnd { + return WalkText(x, y, margin, text, + [&](int cur_x, int /*cur_y*/, char32_t code_point) { + return cur_x + metrics_.CodePointWidth(code_point); + }); +} + +// Returns whether wrapped text can be broken at `c`. +// +// This is the one definition of where wrapping may introduce a break, so that +// measuring what text wraps into and drawing it wrapped agree about it. +// Carriage returns count so that a CRLF ending is whitespace rather than part +// of the word before it; what becomes of the `\r` is then up to the drawing. +static constexpr auto IsWrapBreak(char c) -> bool { + return c == ' ' || c == '\t' || c == '\r'; +} + +template +auto Buffer::WalkWrappedText(int x, int y, int margin, int max_width, + llvm::StringRef text, PlaceFn place) const + -> DrawEnd { + CheckTextSize(text); + // The block runs from the margin to `margin + max_width`, lies within the + // buffer, and holds the column the text starts in, which is every bound on + // the three of them read in one order. + CARBON_CHECK(llvm::is_sorted(std::array{0, margin, x, x + 1, + margin + max_width, columns_}) && + y >= 0 && y < MaxRows, + "A block of {0} columns at {1} holding text from ({2}, {3}) " + "does not fit the {4} columns and {5} rows a buffer covers.", + max_width, margin, x, y, columns_, MaxRows); + + // The column a row runs out of room at. The block lies within the buffer's + // width, so this is a column like any other rather than a sum that has to be + // kept from overflowing. + int limit = margin + max_width; + + int cur_x = x; + int cur_y = y; + + // Splitting on bytes is safe because every character text can break at is + // ASCII, and UTF-8 never encodes anything else using an ASCII byte. Only + // words are decoded; whitespace is handled a byte at a time. + while (!text.empty()) { + if (text.front() == '\n') { + text = text.drop_front(); + cur_x = margin; + ++cur_y; + continue; + } + + if (IsWrapBreak(text.front())) { + llvm::StringRef breaks = text.take_while(IsWrapBreak); + text = text.drop_front(breaks.size()); + for (char c : breaks) { + if (c == '\r') { + continue; + } + // Whitespace stops at the block's edge, leaving the word after it to + // wrap. + int next = std::min( + c == '\t' ? NextTabStop(cur_x, margin, tab_width_) : cur_x + 1, + limit); + while (cur_x < next) { + cur_x = place(cur_x, cur_y, U' '); + } + } + + // A combining mark renders into the column before it, so one following + // whitespace belongs to that whitespace and goes with it. Left to begin + // the next word, it would move to another row whenever that word wrapped + // and attach to whatever preceded it there. + while (!text.empty()) { + llvm::StringRef rest = text; + char32_t code_point = metrics_.TakeCodePoint(rest); + if (metrics_.CodePointWidth(code_point) != 0) { + break; + } + text = rest; + cur_x = place(cur_x, cur_y, code_point); + } + continue; + } + + llvm::StringRef word = + text.take_until([](char c) { return c == '\n' || IsWrapBreak(c); }); + text = text.drop_front(word.size()); + + // Move a word that doesn't fit down to the next row, which minimizes the + // overhang when it doesn't fit there either. The word is drawn into the row + // this starts before anything else can reach it, so a wrapped row begins at + // the margin rather than with the whitespace the wrap came after. + if (cur_x > margin && cur_x + metrics_.Width(word) > limit) { + cur_x = margin; + ++cur_y; + } + + while (!word.empty()) { + cur_x = place(cur_x, cur_y, metrics_.TakeCodePoint(word)); + } + } + + return {.x = cur_x, .y = cur_y}; +} + +auto Buffer::DrawWrappedText(int x, int y, int margin, int max_width, + llvm::StringRef text, const Style& style) + -> DrawEnd { + return WalkWrappedText(x, y, margin, max_width, text, + [&](int cur_x, int cur_y, char32_t code_point) { + return PlaceCodePoint(cur_x, cur_y, code_point, + style); + }); +} + +auto Buffer::MeasureWrappedText(int x, int y, int margin, int max_width, + llvm::StringRef text) const -> DrawEnd { + return WalkWrappedText(x, y, margin, max_width, text, + [&](int cur_x, int /*cur_y*/, char32_t code_point) { + return cur_x + metrics_.CodePointWidth(code_point); + }); +} + +auto Buffer::MeasureWrapWidth(llvm::StringRef text) const -> int { + int width = 0; + while (!text.empty()) { + llvm::StringRef word = + text.take_until([](char c) { return c == '\n' || IsWrapBreak(c); }); + width = std::max(width, metrics_.Width(word)); + text = text.drop_front(std::max(word.size(), 1)); + } + return width; +} + +auto Buffer::LastVisibleColumn(int y, ColorMode mode) const -> int { + // A style only paints a blank cell if it is rendered at all, so with color + // off a blank cell is padding whatever style it carries. + bool styles_render = mode != ColorMode::NoColor; + for (int x = width_ - 1; x >= 0; --x) { + const Cell& cell = CellAt(x, y); + if (cell.is_continuation || cell.code_point != ' ' || + (styles_render && cell.style.IsVisibleOnBlank()) || + (!combining_marks_.empty() && + combining_marks_.contains(CellIndex(x, y)))) { + return x; + } + } + return -1; +} + +auto Buffer::Render(OutputBufferRef out, ColorMode mode) const -> void { + Utf8Storage storage; + + // The style a terminal starts in, and the one it is left in. + const Style default_style; + + // Cells outlive this loop, so the active style is tracked by pointing at one + // rather than copying a whole style per cell. It carries across rows: a style + // is usually still in use on the row below, and turning it off and back on + // costs a reset and a fresh start for nothing. + const Style* active = &default_style; + + int rows = height(); + for (int y = 0; y < rows; ++y) { + int last = LastVisibleColumn(y, mode); + for (int x = 0; x <= last; ++x) { + const Cell& cell = CellAt(x, y); + if (cell.is_continuation) { + continue; + } + + active->AppendTransitionTo(out, cell.style, mode); + active = &cell.style; + out.Append(EncodeUtf8(cell.code_point, storage)); + + // Almost nothing has combining marks, so the lookup is worth skipping + // outright rather than doing it for every cell on the screen. + if (!combining_marks_.empty()) { + auto marks = combining_marks_.find(CellIndex(x, y)); + if (marks != combining_marks_.end()) { + out.Append(marks->second); + } + } + } + + // A style is turned off before the newline in two cases. On the last row, + // so that nothing is left set for whatever is printed after this and the + // escape that turns it off still falls inside the rendering. And whenever + // it paints where there is no glyph, because a terminal fills the rest of + // the row with the background it is in when the row ends, so leaving one + // set would paint a stripe out to the right edge that nothing asked for. + if (y + 1 == rows || active->IsVisibleOnBlank()) { + active->AppendTransitionTo(out, default_style, mode); + active = &default_style; + } + out.Append("\n"); + } +} + +auto Buffer::WriteTo(Filesystem::WriteFileRef file, ColorMode mode) const + -> ErrorOr { + // Sized for the few short lines a diagnostic renders to. A full screen with + // color runs well past it and allocates once. + llvm::SmallString<1024> bytes; + Render(bytes, mode); + return file.WriteCompleteBuffer(llvm::ArrayRef( + reinterpret_cast(bytes.data()), bytes.size())); +} + +} // namespace Carbon::Terminal diff --git a/common/terminal/buffer.h b/common/terminal/buffer.h new file mode 100644 index 000000000000..31495ec17fa7 --- /dev/null +++ b/common/terminal/buffer.h @@ -0,0 +1,537 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#ifndef CARBON_COMMON_TERMINAL_BUFFER_H_ +#define CARBON_COMMON_TERMINAL_BUFFER_H_ + +#include +#include +#include + +#include "common/check.h" +#include "common/filesystem.h" +#include "common/terminal/capabilities.h" +#include "common/terminal/color.h" +#include "common/terminal/metrics.h" +#include "common/terminal/output_buffer_ref.h" +#include "common/terminal/style.h" +#include "llvm/ADT/DenseMap.h" +#include "llvm/ADT/SmallVector.h" +#include "llvm/ADT/StringRef.h" + +namespace Carbon::Terminal { + +// Where a line stops within the cell at one of its ends. +// +// A line runs between points, and in a grid of cells the two points it can +// name are a cell's center and a cell's outer edge. Which one an end is decides +// what a line meeting it there becomes: a line ending at a center and another +// leaving that center form a corner, while a line running out through an edge +// carries on past whatever meets it, which is a tee. +// +// This is the distinction a vector graphics stroke draws between a butt cap and +// a square cap, where the square cap extends the stroke by half its width past +// the endpoint. Half a stroke here is half a cell. +// +// Unicode has a glyph for a line reaching only the middle of its cell (U+2574 +// through U+2577), so a `Center` end is drawn as one and the reader sees where +// the line really stops rather than having to infer it from the junctions. With +// `Charset::Ascii` there is nothing to draw half a line with, so both ends fill +// their cell and only the junctions around them say which was which. +enum class LineEnd : int8_t { + // The line stops at the center of its end cell. Lines meeting there corner. + Center, + // The line runs out through the outer edge of its end cell, joining whatever + // is beyond it. Lines meeting there tee. + Edge, +}; + +// A grid of styled cells staged for rendering to a terminal. +// +// Coordinates are 0-based with (0, 0) at the top left, `x` counting terminal +// columns and `y` counting rows. +// +// A buffer renders once, top to bottom, the way a compiler writes diagnostics. +// There is no cursor addressing and nothing is ever redrawn, so a rendered +// buffer is just as valid in a file or a pipe as on a terminal. +// +// Every row is a line, ended by a newline of its own, so nothing is left for +// the terminal to break. A break introduced to fit a width is an ordinary +// newline like any other, which is what lets wrapped text carry an indent or +// sit in a column beside a gutter: a terminal wrapping a row of its own accord +// continues at column zero, under the gutter rather than beside it. It also +// means text copied out of the output holds the lines that were displayed. +// +// The cost is that such a break is in whatever a reader copies, so wrapping +// never puts one inside a word. A path or a URL stays whole and overhangs the +// width when it doesn't fit, which is what keeps it selectable in one piece and +// clickable where a terminal recognizes one. Wrapping only adds breaks as well: +// the newlines already in a caller's text are kept as they are. A row is a row +// once something is drawn into it, so a break the text ends with closes its +// last line rather than opening an empty one after it. +// +// Staging into a grid lets layout position content directly, rather than +// interleaving text, padding, and escape sequences as it goes. That separation +// is what makes the two hard parts tractable: escape sequences are minimized +// once, in `Render`, and the drawing APIs reason about columns on screen rather +// than bytes in a stream. +// +// Which bytes make up a column depends on the charset, and the buffer handles +// that rather than leaving it to callers, because getting it wrong misaligns +// everything downstream of it: +// +// - Under `Charset::Ascii` no UTF-8 processing happens at all. Every byte is +// one column, exactly as a terminal decoding some single-byte encoding will +// treat it, and bytes outside printable ASCII are replaced with `?` because +// there is no telling what such a terminal would draw for them. +// - Under `Charset::Utf8` bytes are decoded as UTF-8. Double-width characters +// occupy both of the columns they will really take, and drawing over either +// column erases the whole character instead of leaving half of one behind. +// Combining marks render into the column before them, so a base character +// and its marks stay in one cell. Carbon source is in Unicode normalization +// form C, which still spells out marks for characters that have no +// precomposed form, so this comes up in ordinary input. Anything with no +// printable rendering, including invalid UTF-8, becomes U+FFFD. +// +// A buffer is `columns()` wide, and that width is the whole point of it: it is +// what wrapping fits text into, and it comes from the terminal where one was +// measured and from `DefaultColumns` where none was. Rows are the direction +// there is no bound in -- a buffer grows downward to whatever is drawn into it, +// up to `MaxRows` -- so laying out is a question of how many rows something +// takes, never of how wide the grid will turn out to be. +// +// Coordinates are the caller's to get right. Drawing a line outside the width, +// or starting text outside it, is a programming error and is checked: a caller +// deciding where to put something already knows the width, since it is what +// decided the layout, and a drawing that lands outside it is a bug in that +// layout rather than something to silently clip. Origins are checked against +// `MaxRows` the same way, though text that runs off the bottom on its own +// newlines is clipped rather than checked, as an overhang is. +// +// A row can still end up wider than `columns()`. Text that starts inside the +// width may run off the right of it: a quoted source line longer than the room +// left, a double-width character in the last column, and above all a word +// wrapping cannot break, which is moved to a row of its own and then overhangs +// it. Breaking that word is the alternative, and it costs a reader the ability +// to copy or click it. So `width()` can exceed `columns()`, while nothing is +// ever drawn left of the origin or beyond `MaxColumns`. +// +// A combining mark renders into the cell before it, so one with no cell before +// it -- at column zero, or on a row nothing has been drawn on -- has nowhere to +// go and is dropped. That is data rather than a coordinate, which is why it is +// dropped rather than checked: source files contain such text. +// +// TODO: None of this handles bidirectional text. A right-to-left run reorders +// on screen, so the column a character occupies stops following from the +// characters before it, which is the assumption every position here rests on: +// that drawing advances left to right by the width of what was drawn. Getting +// this right needs the reordering to happen before anything is placed, which +// makes it a question about where the boundary between a client's layout and +// this buffer should sit -- whether the buffer takes runs that are already in +// visual order, or takes logical order and reorders as it draws, and what it +// then means for a caller to name a column at all. Marking a span and drawing a +// line under it are the hard cases, since a logically contiguous span need not +// be contiguous on screen. +class Buffer { + public: + // The bounds a buffer exists within. + // + // These are far past anything a terminal displays, and exist so that a cell + // index stays representable rather than to ration anything. `columns()` and + // every row drawn into are checked against them, so a caller cannot reach + // outside them by asking. What can reach `MaxColumns` without being asked for + // is a word overhanging the target width, and that alone is clipped rather + // than checked, since how far it overhangs is a fact about the text. + static constexpr int MaxColumns = 1 << 14; + static constexpr int MaxRows = 1 << 16; + + // The most bytes of text one operation draws or measures. + // + // The column advances by the width of what was drawn whether or not a cell + // was written, so without this a long enough run would carry it past what an + // `int` holds and come back negative. Far more text than any terminal shows, + // and a caller with this much has built it rather than read it off a line. + static constexpr int MaxTextBytes = 1 << 24; + + // The widest tab stops a buffer draws to. + // + // Far past any terminal, and small enough that even text made entirely of + // tabs measures into a column an `int` holds: a tab is the one character + // that occupies more columns than it does bytes, so this is what bounds + // `MaxTextBytes` of them. + static constexpr int MaxTabWidth = 64; + + // Where a drawing ended: for text, the row it ended on and the column after + // its last code point there; for a line or a box, the cell past the end of + // what it drew. + // + // Everything that draws returns one, so that a caller placing something + // after a drawing advances from this rather than measuring the same text a + // second time. The `Measure` operations return one too, and answer for text + // that hasn't been drawn yet what drawing it would answer. + struct DrawEnd { + int x; + int y; + + friend auto operator==(DrawEnd lhs, DrawEnd rhs) -> bool = default; + }; + + // Constructs an empty buffer holding `charset`, laying out for + // `DefaultColumns`. + explicit Buffer(Charset charset) : Buffer(DefaultColumns, charset) {} + + // Constructs an empty buffer `columns` wide, which must be in + // [1, `MaxColumns`], and whose tabs advance to stops `tab_width` columns + // apart. + // + // The width is what everything drawn into the buffer is laid out for and + // checked against, not a starting size. The grid holds it from the start, so + // a row is only ever reallocated for something that overhangs it. + Buffer(int columns, Charset charset, int tab_width = DefaultTabWidth); + + // Constructs an empty buffer holding `capabilities`'s charset and tab stops, + // laying out for its width, or for `DefaultColumns` where it has none. + // + // Both numbers are clamped rather than checked. They describe a terminal + // rather than coming from a caller -- `columns` by way of `COLUMNS`, which + // anyone can export as anything -- so a value a grid cannot hold is bad input + // rather than a mistake, and the nearest usable one lays out no worse than + // the fallback would. + explicit Buffer(const Capabilities& capabilities) + : Buffer(std::clamp(capabilities.columns.value_or(DefaultColumns), 1, + MaxColumns), + capabilities.charset, + std::clamp(capabilities.tab_width, 1, MaxTabWidth)) {} + + // Returns the width everything drawn into the buffer is laid out for. + auto columns() const -> int { return columns_; } + + // Returns the columns the grid currently holds: `columns()` until something + // overhangs it, and at least enough to hold the overhang after that. + auto width() const -> int { return width_; } + + // Returns the number of rows the grid holds, which is one past the last row + // drawn into. + auto height() const -> int; + + auto charset() const -> Charset { return metrics_.charset(); } + + // Returns how text is measured for this buffer's charset. + // + // The buffer lays its cells out with this, so a caller deciding where to put + // something asks the same thing the drawing will. + auto metrics() const -> Metrics { return metrics_; } + + // Returns where `DrawText` would end for these arguments, without drawing. + // + // Measuring and drawing walk the text with the same code, differing only in + // whether they write a cell, so a layout decision made from this can't + // disagree with what drawing then does. + // + // This is for text that a tab, a newline, or a carriage return makes + // positional. Text with none of them is as wide wherever it is drawn, and + // `Metrics::Width` answers for it without a buffer to draw into. + auto MeasureText(int x, int y, int margin, llvm::StringRef text) const + -> DrawEnd; + + // Returns where the `DrawText` taking no margin would end, which draws `text` + // as text of its own beginning at (x, y). + auto MeasureText(int x, int y, llvm::StringRef text) const -> DrawEnd { + return MeasureText(x, y, x, text); + } + + // Returns where `DrawWrappedText` would end for these arguments, without + // drawing. + // + // The block and the origin are checked as drawing checks them, so measuring + // answers only for arguments drawing would accept. + auto MeasureWrappedText(int x, int y, int margin, int max_width, + llvm::StringRef text) const -> DrawEnd; + + // Returns the fewest columns `text` wraps into without overhanging them, + // which is the width of its widest word since wrapping never breaks one. + // + // Wrapping into fewer columns still draws everything; the excess overhangs. + // So this is a layout preference rather than a minimum. + auto MeasureWrapWidth(llvm::StringRef text) const -> int; + + // Draws `code_point` at (x, y), which must be inside `columns()` and + // `MaxRows`, adding rows as needed to reach it. + // + // Returns the column after it, which is `x` again for a combining mark since + // one renders into the column before it. A double-width character starting in + // the last column is drawn rather than refused, and takes the column after + // it: half a character is not something a terminal can render, so the choice + // is between the whole of it and none, and this is the same overhang wrapping + // allows a word that fits no row. + auto DrawCodePoint(int x, int y, char32_t code_point, const Style& style) + -> DrawEnd; + + // Draws a horizontal line across `length` columns starting at (x, y). + // + // By default the line runs between the centers of its first and last cells, + // which is what a line connecting two things is: `DrawBox` draws its four + // sides this way, and each pair meets at a corner. `LineEnd::Edge` instead + // runs that end out through the side of its cell, which is what a line + // bounding `length` whole columns of something is, and what makes a line + // meeting it there a tee. A line of one column between two centers is a + // point, and is drawn as one. + // + // Lines join wherever they overlap: a cell records which directions lines + // leave it in, and its glyph follows from those bits alone, so crossings, + // corners, and tees all appear without being asked for and whatever order + // the lines were drawn in. This is the only way to produce a junction, and + // it suffices because a junction in real line art always has the lines that + // imply it running through it. Only line drawing records directions, so text + // containing `-` or `+` is never redrawn as line art. + // + // A cell's style is whatever was drawn there last, so crossing lines of + // different styles do depend on order. + auto DrawHorizontalLine(int x, int y, int length, const Style& style, + LineEnd start = LineEnd::Center, + LineEnd end = LineEnd::Center) -> DrawEnd; + + // Draws a vertical line down `length` rows starting at (x, y), with the same + // meaning for its ends. Returns the row after it, in the column it ran down. + auto DrawVerticalLine(int x, int y, int length, const Style& style, + LineEnd start = LineEnd::Center, + LineEnd end = LineEnd::Center) -> DrawEnd; + + // Draws the outline of a box with its top-left corner at (x, y). + // + // Each side runs between the centers of the cells it ends in, so the four + // corners come out of the sides meeting there. A box with no interior is + // then the single line that bounds it, and one with no extent in either + // direction is a point, without either being a case of its own. + auto DrawBox(int x, int y, int box_width, int box_height, const Style& style) + -> DrawEnd; + + // Draws `text` starting at (x, y), which must be inside `columns()`, as part + // of text whose left edge is `margin`. + // + // Nothing here wraps, so text with no newline in it runs off the right of the + // width when it is longer than the room left, exactly as an overhanging word + // does. That is what this is for: a source line is quoted as it was written, + // and deciding how much of one to show is the caller's, made against + // `columns()` before the quoting starts. + // + // Newlines return to column `margin` on the next row, carriage returns to + // column `margin` on the same row, and tabs advance to the next tab stop, + // with stops measured from `margin` so that a quoted source line keeps the + // tab alignment it had in the file wherever the quote is placed. Returns + // where it ended, which for text with a newline in it is on a later row than + // it started. + // + // The margin is what lets text with newlines in it be drawn as differently + // styled spans, each starting where the last ended and all naming the same + // margin, the way `DrawWrappedText` does for a block: a newline in the middle + // of such a run returns to the text's own left edge rather than to wherever + // the span it fell in happened to start. + auto DrawText(int x, int y, int margin, llvm::StringRef text, + const Style& style) -> DrawEnd; + + // Draws `text` as text of its own beginning at (x, y), which is then both + // where it starts and the margin its later rows return to. + auto DrawText(int x, int y, llvm::StringRef text, const Style& style) + -> DrawEnd { + return DrawText(x, y, x, text, style); + } + + // Draws `text` starting at (x, y), into the block of `max_width` columns + // beginning at `margin`. + // + // The block must lie within `columns()` and `x` within the block, so + // `0 <= margin <= x < margin + max_width <= columns()`. A block is a division + // of the width rather than something that can exceed it: what a caller wants + // when it has nothing to divide is `max_width` of `columns() - margin`, the + // whole of what is left. + // + // The block is what the text wraps within, and (x, y) is only where this run + // of it starts: rows after the first begin at `margin`, and how much room a + // row has is measured from there. A block whose spans are styled differently + // is drawn as one call per span, each starting where the last ended and all + // naming the same margin and width. Passing `x` as the margin draws a block + // in one call. + // + // Wrapping breaks at ASCII spaces, tabs, and carriage returns, and only + // there. A word here is whatever lies between two of them, so a URL is one + // word, and one too long for a row of its own is moved down to one and then + // overhangs it rather than being broken. + // + // Whitespace stops at the block's edge rather than running past it, so the + // spaces between two words stay on the row the first of them ended and the + // row the second wraps onto begins at the margin. Spaces the text opens with, + // or that follow a newline in it, are kept as they are, since those are + // indentation the caller wrote. + // + // Newlines are breaks the caller already made, and are kept as they are: + // wrapping only adds breaks to the text it is given. They break the line as a + // wrap does, continuing at `margin` on the next row, and carriage returns are + // dropped so that CRLF endings break exactly once. + // + // A tab is both a break opportunity and a jump to the next tab stop, with + // stops measured from `margin` rather than from `x`. The margin is the one + // column every row of the block begins at, so the stops are the same on each + // of them and a tabbed column stays a column however the text wraps; stops + // from `x` would move with the span that happened to be drawn first. A tab + // that would reach past the block stops at its edge, like the spaces do, + // leaving the word after it to wrap. + // + // `DrawText` is the way to draw text that should not wrap at all, and differs + // in more than that: it keeps every space, and returns to the margin on a + // carriage return rather than dropping it. + // + // Returns where it ended. + // + // TODO: There is no mode that reflows, treating the newlines in `text` as + // breaks to be chosen again rather than kept. Text that arrives wrapped to + // some other width keeps that wrapping, which is wrong for it wherever that + // width isn't the one it is being drawn into. Add one when there is a caller + // with such text, since which breaks a reflow may discard -- every newline, + // or only those a previous wrapping introduced -- is a question about where + // that text came from. + auto DrawWrappedText(int x, int y, int margin, int max_width, + llvm::StringRef text, const Style& style) -> DrawEnd; + + // Renders the grid, appending the bytes that draw it to `out`. + // + // Each row ends in a newline, with trailing blank cells dropped so output + // carries no invisible padding. The rendering ends with the style turned off + // so nothing bleeds into what is printed next, and a style that paints blank + // cells is turned off at each row's end so a background does not run to the + // right edge. Color is chosen here rather than at construction because it + // affects only how cells are serialized, while the charset decides how + // content is laid out into them. + auto Render(OutputBufferRef out, ColorMode mode) const -> void; + + // Renders the grid and writes it to `file`. + // + // The whole grid goes out in one `write` where the destination accepts it, + // which is what gives the output whatever atomicity the descriptor offers + // against other writers: a terminal or a pipe interleaves at write + // boundaries, so one call per rendered buffer is the most that can be had + // without a lock. + auto WriteTo(Filesystem::WriteFileRef file, ColorMode mode) const + -> ErrorOr; + + private: + // The directions in which drawn lines leave a cell, and whether the cell + // holds line art at all. A cell's glyph is a function of the directions + // alone. + enum LineDirection : uint8_t { + LineLeft = 1 << 0, + LineRight = 1 << 1, + LineUp = 1 << 2, + LineDown = 1 << 3, + LineDirections = 0b1111, + // Set on every cell line drawing writes. A cell can hold line art and no + // directions -- a line between one center and itself is a point -- and + // without this such a cell would be indistinguishable from one holding + // text, so nothing drawn later would join it. + LineCell = 1 << 4, + }; + + struct Cell { + // The code point rendered here. For a cell with `lines` set, this is + // derived from those bits and the charset. + char32_t code_point = ' '; + + Style style; + + // Which directions drawn lines leave this cell in, with `LineCell` set, + // or zero for a cell holding text. + uint8_t lines = 0; + + // Whether this cell is the right half of a double-width character, and so + // renders nothing of its own. + bool is_continuation = false; + }; + + // Checks that `text` is short enough to measure without overflowing a column. + static auto CheckTextSize(llvm::StringRef text) -> void { + CARBON_CHECK(text.size() <= MaxTextBytes, + "Laying out {0} bytes of text is past the {1} one operation " + "handles.", + text.size(), MaxTextBytes); + } + + auto CellIndex(int x, int y) const -> int { return y * width_ + x; } + auto CellAt(int x, int y) -> Cell& { return cells_[CellIndex(x, y)]; } + auto CellAt(int x, int y) const -> const Cell& { + return cells_[CellIndex(x, y)]; + } + + // Checks that (x, y) is somewhere a drawing may start. + // + // The text walks check this themselves, together with the bounds particular + // to each: they are inlined into every text operation, and one check there + // costs measurably less than two. + auto CheckOrigin(int x, int y) const -> void { + CARBON_CHECK( + x >= 0 && x < columns_ && y >= 0 && y < MaxRows, + "Drawing at ({0}, {1}) is outside the {2} columns and {3} rows " + "a buffer covers.", + x, y, columns_, MaxRows); + } + + // Places `code_point` at (x, y) without checking it against the target width + // or `MaxRows`, which text reaches on its own by overhanging or by carrying + // newlines. Past either, nothing is drawn and the column still advances. The + // coordinates must be non-negative, which follows from the origin the walk + // was checked at. + auto PlaceCodePoint(int x, int y, char32_t code_point, const Style& style) + -> int; + + // The walks behind the text operations, over which drawing and measuring are + // the same code. `place` is called with each code point and where it goes, + // and returns the column after it: `PlaceCodePoint` when drawing, and the + // width alone when measuring. + template + auto WalkText(int x, int y, int margin, llvm::StringRef text, + PlaceFn place) const -> DrawEnd; + template + auto WalkWrappedText(int x, int y, int margin, int max_width, + llvm::StringRef text, PlaceFn place) const -> DrawEnd; + + // Adds rows until row `y` exists. + auto EnsureRow(int y) -> void; + + // Widens the grid until column `x` exists, reflowing the rows it already + // holds, which are stored back to back. Only something overhanging the target + // width reaches past it, so this runs for nothing else. + auto EnsureColumn(int x) -> void; + + // Resets the cells in row `y` spanning columns [x, x + width), along with + // either half of a double-width character that straddles the range's edges. + auto ClearCells(int x, int y, int width) -> void; + + // Appends `code_point` to the marks rendered with the cell before column `x`. + auto AttachCombiningMark(int x, int y, char32_t code_point) -> void; + + // Adds `directions` to the lines through (x, y) and updates its glyph. + auto DrawLine(int x, int y, uint8_t directions, const Style& style) -> void; + + // Returns the last column in row `y` that renders anything under `mode`, or + // -1 when the row renders nothing. + auto LastVisibleColumn(int y, ColorMode mode) const -> int; + + // The width laid out for, and the width the grid holds. They differ only + // where something overhung the first. + int columns_; + int width_; + + int tab_width_; + Metrics metrics_; + + llvm::SmallVector cells_; + + // Combining marks, as UTF-8, for the few cells that have any, keyed by cell + // index. Kept out of `Cell` so that the common case of no marks costs + // nothing per cell. Always empty under `Charset::Ascii`. + llvm::DenseMap combining_marks_; +}; + +} // namespace Carbon::Terminal + +#endif // CARBON_COMMON_TERMINAL_BUFFER_H_ diff --git a/common/terminal/buffer_test.cpp b/common/terminal/buffer_test.cpp new file mode 100644 index 000000000000..822ce8f5d798 --- /dev/null +++ b/common/terminal/buffer_test.cpp @@ -0,0 +1,1104 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#include "common/terminal/buffer.h" + +#include + +#include +#include +#include + +#include "common/filesystem.h" +#include "common/terminal/metrics.h" +#include "llvm/ADT/SmallString.h" +#include "llvm/ADT/SmallVector.h" +#include "llvm/ADT/StringRef.h" + +namespace Carbon::Terminal { +namespace { + +// "e" followed by U+0301 COMBINING ACUTE ACCENT. Spelled out because the +// precomposed U+00E9 is a single code point and wouldn't exercise marks at all. +constexpr char32_t CombiningAcute = 0x0301; +constexpr llvm::StringLiteral AcuteE = "e\xcc\x81"; + +using DrawEnd = Buffer::DrawEnd; + +auto Render(const Buffer& buffer, ColorMode mode = ColorMode::NoColor) + -> std::string { + llvm::SmallString<256> out; + buffer.Render(out, mode); + return std::string(out); +} + +TEST(BufferTest, Empty) { + Buffer buffer(10, Charset::Ascii); + EXPECT_EQ(buffer.width(), 10); + EXPECT_EQ(buffer.height(), 0); + EXPECT_EQ(Render(buffer), ""); +} + +TEST(BufferTest, GrowsRowsToFitWhatIsDrawn) { + Buffer buffer(4, Charset::Ascii); + EXPECT_EQ(buffer.DrawCodePoint(0, 2, 'x', Style()).x, 1); + EXPECT_EQ(buffer.height(), 3); + EXPECT_EQ(Render(buffer), "\n\nx\n"); +} + +TEST(BufferTest, WidthIsTheTargetUntilSomethingOverhangs) { + // `width()` tracks `columns()` while everything stays inside it. + Buffer buffer(6, Charset::Ascii); + EXPECT_EQ(buffer.columns(), 6); + EXPECT_EQ(buffer.width(), 6); + buffer.DrawHorizontalLine(1, 0, 5, Style()); + buffer.DrawVerticalLine(0, 1, 5, Style()); + EXPECT_EQ(buffer.width(), 6); + EXPECT_EQ(Render(buffer), + " -----\n" + "|\n" + "|\n" + "|\n" + "|\n" + "|\n"); + + // Text that starts inside the width can run past it, and a word wrapping + // cannot break is the case that produces. + Buffer overhang(4, Charset::Ascii); + overhang.DrawWrappedText(0, 0, 0, 4, "ab wordthatislong", Style()); + EXPECT_EQ(overhang.columns(), 4); + EXPECT_GE(overhang.width(), 14); + EXPECT_EQ(Render(overhang), + "ab\n" + "wordthatislong\n"); +} + +TEST(BufferTest, WidthGrowsAcrossRowsAlreadyDrawn) { + // Rows are stored back to back, so widening has to reflow the ones already + // there rather than leaving them where their old width put them. + Buffer buffer(2, Charset::Ascii); + buffer.DrawText(0, 0, "ab", Style()); + buffer.DrawText(0, 1, "cd", Style()); + buffer.DrawText(0, 2, "efghij", Style()); + EXPECT_EQ(Render(buffer), + "ab\n" + "cd\n" + "efghij\n"); +} + +TEST(BufferTest, GrowthFromEveryStartingWidth) { + // Growth runs in steps, so every starting width reaches the same place by a + // different route. + for (int start : {1, 2, 3, 5, 17}) { + Buffer buffer(start, Charset::Utf8); + for (int row = 0; row < 4; ++row) { + buffer.DrawText(0, row, "abcdefghijklmnopqrstuvwxyz", Style()); + } + EXPECT_EQ(Render(buffer), + "abcdefghijklmnopqrstuvwxyz\n" + "abcdefghijklmnopqrstuvwxyz\n" + "abcdefghijklmnopqrstuvwxyz\n" + "abcdefghijklmnopqrstuvwxyz\n") + << start; + } +} + +TEST(BufferTest, LinesJoinWhereTheyMeet) { + Buffer buffer(3, Charset::Utf8); + buffer.DrawHorizontalLine(0, 1, 3, Style()); + buffer.DrawVerticalLine(1, 0, 3, Style()); + + EXPECT_EQ(Render(buffer), + " ╷\n" + "╶┼╴\n" + " ╵\n"); +} + +TEST(BufferTest, LineEndsFormCorners) { + Buffer buffer(3, Charset::Utf8); + buffer.DrawHorizontalLine(1, 0, 2, Style()); + buffer.DrawVerticalLine(1, 0, 3, Style()); + + EXPECT_EQ(Render(buffer), + " ╭╴\n" + " │\n" + " ╵\n"); +} + +TEST(BufferTest, LinesOnlyJoinWhereTheyOverlap) { + // A cell's glyph follows from the directions lines leave it in, so joining + // is a matter of drawing into the same cell rather than of being adjacent. + Buffer separate(3, Charset::Utf8); + separate.DrawHorizontalLine(0, 0, 3, Style()); + separate.DrawVerticalLine(0, 1, 2, Style()); + EXPECT_EQ(Render(separate), + "╶─╴\n" + "╷\n" + "╵\n"); + + Buffer overlapping(3, Charset::Utf8); + overlapping.DrawHorizontalLine(0, 0, 3, Style()); + overlapping.DrawVerticalLine(0, 0, 3, Style()); + EXPECT_EQ(Render(overlapping), + "╭─╴\n" + "│\n" + "╵\n"); +} + +TEST(BufferTest, DrawOrderDoesNotMatter) { + Buffer vertical_first(3, Charset::Utf8); + vertical_first.DrawVerticalLine(1, 0, 3, Style()); + vertical_first.DrawHorizontalLine(0, 1, 3, Style()); + + EXPECT_EQ(Render(vertical_first), + " ╷\n" + "╶┼╴\n" + " ╵\n"); +} + +TEST(BufferTest, Tees) { + // Every junction shape comes out of lines overlapping, so a table of them + // exercises all four tees and the cross together. + Buffer buffer(5, Charset::Utf8); + buffer.DrawBox(0, 0, 5, 5, Style()); + buffer.DrawHorizontalLine(0, 2, 5, Style()); + buffer.DrawVerticalLine(2, 0, 5, Style()); + EXPECT_EQ(Render(buffer), + "╭─┬─╮\n" + "│ │ │\n" + "├─┼─┤\n" + "│ │ │\n" + "╰─┴─╯\n"); +} + +TEST(BufferTest, ALineBetweenOneCenterAndItselfIsAPoint) { + // A line of one cell between two centers has no length and no direction, so + // it is a point. It is still line art, so anything drawn through it later + // joins it rather than replacing it. + Buffer horizontal(3, Charset::Utf8); + horizontal.DrawHorizontalLine(1, 0, 1, Style()); + EXPECT_EQ(Render(horizontal), " ·\n"); + + Buffer vertical(3, Charset::Utf8); + vertical.DrawVerticalLine(1, 0, 1, Style()); + EXPECT_EQ(Render(vertical), " ·\n"); + + Buffer joined(3, Charset::Utf8); + joined.DrawHorizontalLine(1, 0, 1, Style()); + joined.DrawHorizontalLine(0, 0, 3, Style()); + EXPECT_EQ(Render(joined), "╶─╴\n"); + + // ASCII has one glyph for everything that isn't a plain segment. + Buffer ascii(3, Charset::Ascii); + ascii.DrawVerticalLine(1, 0, 1, Style()); + EXPECT_EQ(Render(ascii), " +\n"); + + // A line with no length draws nothing at all. + Buffer empty(3, Charset::Utf8); + empty.DrawHorizontalLine(0, 0, 0, Style()); + empty.DrawVerticalLine(0, 0, 0, Style()); + EXPECT_EQ(Render(empty), ""); +} + +TEST(BufferTest, LineEndsDecideWhatMeetsThemAtTheEnd) { + // A line ending at a center is met with a corner, because the two lines stop + // at the same point. One running out through the edge of its last cell is met + // with a tee, because it carries on past whatever arrives there. + Buffer corner(4, Charset::Utf8); + corner.DrawHorizontalLine(0, 0, 3, Style()); + corner.DrawVerticalLine(2, 0, 2, Style()); + EXPECT_EQ(Render(corner), + "╶─╮\n" + " ╵\n"); + + Buffer tee(4, Charset::Utf8); + tee.DrawHorizontalLine(0, 0, 3, Style(), LineEnd::Center, LineEnd::Edge); + tee.DrawVerticalLine(2, 0, 2, Style()); + EXPECT_EQ(Render(tee), + "╶─┬\n" + " ╵\n"); + + // `start` decides the first cell the way `end` decides the last. + Buffer edge_start(4, Charset::Utf8); + edge_start.DrawHorizontalLine(1, 0, 2, Style(), LineEnd::Edge); + EXPECT_EQ(Render(edge_start), " ─╴\n"); +} + +TEST(BufferTest, AnEndAtACenterDrawsHalfALine) { + // Two half-lines laid end to end show the gap that says they do not connect, + // which ASCII cannot draw. + Buffer buffer(6, Charset::Utf8); + buffer.DrawHorizontalLine(0, 0, 4, Style()); + buffer.DrawVerticalLine(0, 1, 3, Style()); + EXPECT_EQ(Render(buffer), + "╶──╴\n" + "╷\n" + "│\n" + "╵\n"); + + // Two lines that each stop at their own center, laid end to end, are drawn + // with the gap between them that says they do not connect. + Buffer apart(6, Charset::Utf8); + apart.DrawHorizontalLine(0, 0, 2, Style()); + apart.DrawHorizontalLine(2, 0, 2, Style()); + EXPECT_EQ(Render(apart), "╶╴╶╴\n"); + + // ASCII has nothing to draw half a line with, so there the two read as one. + Buffer ascii(6, Charset::Ascii); + ascii.DrawHorizontalLine(0, 0, 2, Style()); + ascii.DrawHorizontalLine(2, 0, 2, Style()); + EXPECT_EQ(Render(ascii), "----\n"); +} + +TEST(BufferTest, AnEdgeToEdgeLineSpansWholeCells) { + // Bounding a run of columns is edge to edge: the line covers all of them + // rather than stopping halfway into the first and last. + Buffer buffer(4, Charset::Utf8); + buffer.DrawHorizontalLine(0, 0, 1, Style(), LineEnd::Edge, LineEnd::Edge); + buffer.DrawVerticalLine(0, 1, 1, Style(), LineEnd::Edge, LineEnd::Edge); + EXPECT_EQ(Render(buffer), + "─\n" + "│\n"); + + // Two edge-to-edge runs that meet end to end read as one line, without + // either having to reach into the other's cells. + Buffer split(6, Charset::Utf8); + split.DrawVerticalLine(0, 0, 2, Style(), LineEnd::Edge, LineEnd::Edge); + split.DrawVerticalLine(0, 2, 2, Style(), LineEnd::Edge, LineEnd::Edge); + EXPECT_EQ(Render(split), + "│\n" + "│\n" + "│\n" + "│\n"); +} + +TEST(BufferTest, LinesDrawOverContent) { + // A line replaces whatever text was in the cell, including both halves of a + // double-width character it lands on. + Buffer over_text(5, Charset::Utf8); + over_text.DrawText(0, 0, "abcde", Style()); + over_text.DrawHorizontalLine(1, 0, 3, Style()); + EXPECT_EQ(Render(over_text), "a╶─╴e\n"); + + Buffer over_wide(5, Charset::Utf8); + over_wide.DrawText(0, 0, "中中", Style()); + over_wide.DrawVerticalLine(1, 0, 1, Style(), LineEnd::Edge, LineEnd::Edge); + EXPECT_EQ(Render(over_wide), " │中\n"); +} + +TEST(BufferTest, Box) { + Buffer buffer(4, Charset::Utf8); + buffer.DrawBox(0, 0, 4, 4, Style()); + EXPECT_EQ(Render(buffer), + "╭──╮\n" + "│ │\n" + "│ │\n" + "╰──╯\n"); + + // ASCII can only tell horizontal and vertical apart from everything else. + Buffer ascii(4, Charset::Ascii); + ascii.DrawBox(0, 0, 4, 3, Style()); + EXPECT_EQ(Render(ascii), + "+--+\n" + "| |\n" + "+--+\n"); + + // A box with no interior is the single line that bounds it. + Buffer flat(4, Charset::Utf8); + flat.DrawBox(0, 0, 4, 1, Style()); + flat.DrawBox(0, 2, 1, 2, Style()); + EXPECT_EQ(Render(flat), + "╶──╴\n" + "\n" + "╷\n" + "╵\n"); + + Buffer degenerate(4, Charset::Utf8); + degenerate.DrawBox(0, 0, 0, 4, Style()); + degenerate.DrawBox(0, 0, 4, 0, Style()); + EXPECT_EQ(Render(degenerate), ""); +} + +TEST(BufferTest, TextIsNeverRedrawnAsLines) { + Buffer buffer(10, Charset::Utf8); + buffer.DrawText(0, 0, "a -+- b", Style()); + EXPECT_EQ(Render(buffer), "a -+- b\n"); +} + +TEST(BufferTest, Text) { + Buffer buffer(15, Charset::Utf8); + EXPECT_EQ(buffer.DrawText(0, 0, "Hello, World!", Style()).y, 0); + EXPECT_EQ(buffer.DrawText(0, 1, "A\tB\nC", Style()).y, 2); + + EXPECT_EQ(Render(buffer), + "Hello, World!\n" + "A B\n" + "C\n"); + + // Drawing nothing ends where it began. + EXPECT_EQ(buffer.DrawText(0, 0, "", Style()).y, 0); +} + +TEST(BufferTest, TabStopsAreMeasuredFromWhereTextBegins) { + // Tab stops follow the text's own origin, not the left edge, so a source + // line quoted beside a gutter keeps the tab alignment it had in the file. + Buffer buffer(20, Charset::Utf8); + buffer.DrawText(3, 0, "A\tB", Style()); + EXPECT_EQ(Render(buffer), " A B\n"); +} + +TEST(BufferTest, TextRowsReturnToTheMargin) { + // Text drawn as differently styled spans names one margin across all of them, + // so a newline in the middle of it returns to the text's own left edge rather + // than to wherever the span it fell in started. + Buffer buffer(20, Charset::Ascii); + buffer.DrawText(0, 0, "| ", Style()); + DrawEnd end = buffer.DrawText(2, 0, 2, "plain\nmore ", Style()); + buffer.DrawText(end.x, end.y, 2, "bold\ntext", Style().Bold()); + EXPECT_EQ(Render(buffer), + "| plain\n" + " more bold\n" + " text\n"); + + // Carriage returns return there too, and tab stops are counted from it. + Buffer positional(20, Charset::Ascii); + positional.DrawText(0, 0, "| ", Style()); + positional.DrawText(2, 0, 2, "a\tb\rc", Style()); + EXPECT_EQ(Render(positional), "| c b\n"); +} + +TEST(BufferTest, TextControlCharacters) { + Buffer buffer(10, Charset::Utf8); + // A carriage return returns to the column the text started in. + buffer.DrawText(2, 0, "abc\rx", Style()); + EXPECT_EQ(Render(buffer), " xbc\n"); + + // Anything else with no printable rendering is replaced, so a stray byte + // can't shift the columns after it. + Buffer utf8(10, Charset::Utf8); + utf8.DrawText(0, 0, + "a\x01" + "b", + Style()); + EXPECT_EQ(Render(utf8), "a�b\n"); + + Buffer ascii(10, Charset::Ascii); + ascii.DrawText(0, 0, + "a\x01" + "b", + Style()); + EXPECT_EQ(Render(ascii), "a?b\n"); +} + +TEST(BufferTest, AsciiDoesNoUtf8Processing) { + // A terminal that isn't decoding UTF-8 renders each byte as some character + // of its own, so every byte has to be counted as one column. Replacing the + // bytes keeps the column count honest without guessing at an encoding. + Buffer buffer(10, Charset::Ascii); + buffer.DrawText(0, 0, "a中b", Style()); + EXPECT_EQ(Render(buffer), "a???b\n"); + EXPECT_EQ(buffer.MeasureText(0, 0, "a中b").x, 5); + + // The same goes for text with combining marks, which occupy no columns only + // because a UTF-8 terminal folds them into the one before. + Buffer marks(10, Charset::Ascii); + marks.DrawText(0, 0, AcuteE, Style()); + EXPECT_EQ(Render(marks), "e??\n"); + + // Invalid UTF-8 is not even a category here; bytes are bytes. + Buffer invalid(10, Charset::Ascii); + invalid.DrawText(0, 0, llvm::StringRef("\xc0\x80z", 3), Style()); + EXPECT_EQ(Render(invalid), "??z\n"); + + // Every code point is one column wide, whatever it is. + EXPECT_EQ(buffer.DrawCodePoint(0, 1, U'中', Style()).x, 1); + EXPECT_EQ(buffer.DrawCodePoint(1, 1, CombiningAcute, Style()).x, 2); + EXPECT_EQ(Render(buffer), + "a???b\n" + "??\n"); +} + +TEST(BufferTest, TextInvalidUtf8) { + Buffer buffer(10, Charset::Utf8); + // An overlong encoding of NUL. It is rejected, and decoding resynchronizes + // one byte at a time rather than giving up on the rest of the text. + buffer.DrawText(0, 0, llvm::StringRef("\xc0\x80z", 3), Style()); + EXPECT_EQ(Render(buffer), "��z\n"); + + Buffer surrogate(10, Charset::Utf8); + surrogate.DrawText(0, 0, llvm::StringRef("\xed\xa0\x80z", 4), Style()); + EXPECT_EQ(Render(surrogate), "���z\n"); +} + +TEST(BufferTest, CodePointsWithNoEncoding) { + // Decoding text never yields these, but `DrawCodePoint` takes any code point, + // including the surrogates and the values past the last one that UTF-8 has no + // encoding for. + Buffer buffer(10, Charset::Utf8); + EXPECT_EQ( + buffer.DrawCodePoint(0, 0, static_cast(0xd800), Style()).x, 1); + EXPECT_EQ( + buffer.DrawCodePoint(1, 0, static_cast(0xdfff), Style()).x, 2); + EXPECT_EQ( + buffer.DrawCodePoint(2, 0, static_cast(0x110000), Style()).x, + 3); + EXPECT_EQ(Render(buffer), "���\n"); +} + +TEST(BufferTest, DoubleWidthCharacters) { + Buffer buffer(6, Charset::Utf8); + buffer.DrawText(0, 0, "中A🔥", Style()); + EXPECT_EQ(Render(buffer), "中A🔥\n"); + EXPECT_EQ(buffer.DrawCodePoint(0, 1, U'中', Style()).x, 2); +} + +TEST(BufferTest, DrawingOverADoubleWidthCharacterErasesAllOfIt) { + // Overwriting either half has to erase the whole character. Either half left + // behind misaligns what follows, because the columns the grid counts for the + // cell and the ones the terminal paints stop agreeing. + Buffer over_head(4, Charset::Utf8); + over_head.DrawCodePoint(0, 0, U'中', Style()); + over_head.DrawCodePoint(0, 0, 'A', Style()); + EXPECT_EQ(Render(over_head), "A\n"); + + Buffer over_tail(4, Charset::Utf8); + over_tail.DrawCodePoint(0, 0, U'中', Style()); + over_tail.DrawCodePoint(1, 0, 'B', Style()); + EXPECT_EQ(Render(over_tail), " B\n"); + + // The same holds when a double-width character lands on another one. + Buffer over_both(6, Charset::Utf8); + over_both.DrawCodePoint(0, 0, U'中', Style()); + over_both.DrawCodePoint(2, 0, U'中', Style()); + over_both.DrawCodePoint(1, 0, U'国', Style()); + EXPECT_EQ(Render(over_both), " 国\n"); +} + +TEST(BufferTest, DoubleWidthCharacterInTheLastColumnTakesBothColumns) { + // It starts inside the width, and splitting one would leave the terminal + // rendering half a character, so it overhangs by a column rather than being + // refused. + Buffer buffer(3, Charset::Utf8); + EXPECT_EQ(buffer.DrawCodePoint(2, 0, U'中', Style()).x, 4); + buffer.DrawCodePoint(0, 0, 'a', Style()); + EXPECT_EQ(Render(buffer), "a 中\n"); + EXPECT_GE(buffer.width(), 4); +} + +TEST(BufferTest, CombiningMarks) { + // Marks render into the column before them, so a base character and its + // marks stay in one cell and don't shift what follows. + Buffer buffer(10, Charset::Utf8); + buffer.DrawText(0, 0, AcuteE, Style()); + EXPECT_EQ(Render(buffer), AcuteE.str() + "\n"); + EXPECT_EQ(buffer.MeasureText(0, 0, AcuteE).x, 1); + EXPECT_EQ(buffer.DrawCodePoint(5, 0, CombiningAcute, Style()).x, 5); + + // Drawing over the base takes its marks with it. + Buffer overwritten(10, Charset::Utf8); + overwritten.DrawText(0, 0, AcuteE, Style()); + overwritten.DrawText(1, 0, "x", Style()); + overwritten.DrawCodePoint(0, 0, 'o', Style()); + EXPECT_EQ(Render(overwritten), "ox\n"); +} + +TEST(BufferTest, CombiningMarksOnADoubleWidthBase) { + // A mark following a double-width character arrives at the column past its + // continuation, and has to reach back to the character itself. + Buffer buffer(10, Charset::Utf8); + buffer.DrawText(0, 0, ("中" + AcuteE.drop_front(1) + "x").str(), Style()); + EXPECT_EQ(Render(buffer), ("中" + AcuteE.drop_front(1) + "x\n").str()); + EXPECT_EQ(buffer.MeasureText(0, 0, ("中" + AcuteE.drop_front(1)).str()).x, 2); +} + +TEST(BufferTest, CombiningMarksAreCapped) { + // Marks stack without bound in adversarial text, and every one of them would + // otherwise land in a single cell's output. + std::string zalgo = "e"; + for (int i = 0; i < 100; ++i) { + zalgo += AcuteE.drop_front(1); + } + + Buffer buffer(10, Charset::Utf8); + buffer.DrawText(0, 0, zalgo, Style()); + // The base, some bounded run of marks, and the newline. + EXPECT_LT(Render(buffer).size(), 64U); + EXPECT_GT(Render(buffer).size(), 1U); +} + +TEST(BufferTest, WrappedText) { + Buffer buffer(20, Charset::Ascii); + EXPECT_EQ(buffer + .DrawWrappedText(0, 0, 0, 10, + "This is a long sentence that should be " + "wrapped.", + Style()) + .y, + 5); + + EXPECT_EQ(Render(buffer), + "This is a\n" + "long\n" + "sentence\n" + "that\n" + "should be\n" + "wrapped.\n"); +} + +TEST(BufferTest, WrappedTextKeepsWordsTooLongToFitWhole) { + // A break is a newline in the rendered text, and one inside a word stops it + // being copied out in one piece, so the word overhangs the width instead and + // the buffer grows to hold it. + Buffer buffer(10, Charset::Ascii); + EXPECT_EQ(buffer.DrawWrappedText(0, 0, 0, 5, "abcdefghij", Style()).y, 0); + EXPECT_EQ(Render(buffer), "abcdefghij\n"); +} + +TEST(BufferTest, WrappedTextStartsAnOverlongWordOnItsOwnRow) { + // A word too long for any row moves to one of its own anyway, so it overhangs + // from the margin rather than from wherever the previous word ended. + Buffer buffer(10, Charset::Ascii); + EXPECT_EQ(buffer.DrawWrappedText(0, 0, 0, 5, "ab abcdefghij", Style()).y, 1); + EXPECT_EQ(Render(buffer), + "ab\n" + "abcdefghij\n"); +} + +TEST(BufferTest, WrappedSpansShareOneBlock) { + // A block whose spans are styled differently is drawn one span at a time, + // each continuing where the last ended and all naming the same margin, so + // the rows after the first line up with the block rather than with wherever + // the span happened to start. + Buffer buffer(20, Charset::Ascii); + Buffer::DrawEnd end = + buffer.DrawWrappedText(2, 0, 2, 12, "plain words", Style()); + end = buffer.DrawWrappedText(end.x, end.y, 2, 12, " emphasized more", + Style().Bold()); + buffer.DrawWrappedText(end.x, end.y, 2, 12, " plain again", Style()); + + EXPECT_EQ(Render(buffer), + " plain words\n" + " emphasized\n" + " more plain\n" + " again\n"); +} + +TEST(BufferTest, WrappedTextIndents) { + // Wrapping is relative to the margin rather than to where the text starts, + // which is what lets a wrapped block sit beside a gutter. + Buffer buffer(12, Charset::Ascii); + buffer.DrawText(0, 0, "| ", Style()); + buffer.DrawWrappedText(2, 0, 2, 6, "alpha beta gamma", Style()); + EXPECT_EQ(Render(buffer), + "| alpha\n" + " beta\n" + " gamma\n"); +} + +TEST(BufferTest, WrappedTextExpandsTabsToStops) { + // A tab advances to the next stop, and one following a newline the text wrote + // is indentation and is kept. + Buffer buffer(40, Charset::Ascii); + buffer.DrawWrappedText(0, 0, 0, 30, "a\tb\nlonger\tc", Style()); + EXPECT_EQ(Render(buffer), + "a b\n" + "longer c\n"); + + // A tab the text wrote after a newline is indentation, and is kept. + Buffer indented(40, Charset::Ascii); + indented.DrawWrappedText(0, 0, 0, 30, "a\n\tb", Style()); + EXPECT_EQ(Render(indented), + "a\n" + " b\n"); +} + +TEST(BufferTest, WrappedTextTabStopsFollowTheMargin) { + // Stops are counted from the block's margin rather than from the buffer's + // left edge. + Buffer buffer(40, Charset::Ascii); + buffer.DrawText(0, 0, "| ", Style()); + buffer.DrawWrappedText(2, 0, 2, 20, "a\tb", Style()); + EXPECT_EQ(Render(buffer), "| a b\n"); +} + +TEST(BufferTest, WrappedTextBreaksAtTabs) { + // A tab is a break opportunity as well as a jump to a stop. + Buffer buffer(40, Charset::Ascii); + buffer.DrawWrappedText(0, 0, 0, 8, "aaaa\tbbbb", Style()); + EXPECT_EQ(Render(buffer), + "aaaa\n" + "bbbb\n"); + + // One reaching past the block stops at its edge, so the span after it + // continues from there rather than from a stop outside the block. + Buffer clipped(40, Charset::Ascii); + EXPECT_EQ(clipped.DrawWrappedText(0, 0, 0, 4, "ab\t", Style()), + DrawEnd(4, 0)); +} + +TEST(BufferTest, TabWidthComesFromCapabilities) { + Capabilities capabilities; + capabilities.charset = Charset::Ascii; + capabilities.tab_width = 4; + + Buffer buffer(capabilities); + buffer.DrawText(0, 0, "a\tb", Style()); + buffer.DrawWrappedText(0, 1, 0, 20, "a\tb", Style()); + EXPECT_EQ(Render(buffer), + "a b\n" + "a b\n"); + + // A terminal claiming stops a buffer can't draw to is clamped rather than + // trusted, the same as one claiming an impossible width. + capabilities.tab_width = 0; + Buffer clamped(capabilities); + clamped.DrawText(0, 0, "a\tb", Style()); + EXPECT_EQ(Render(clamped), "a b\n"); +} + +TEST(BufferTest, WrappedTextKeepsTheBreaksItIsGiven) { + // Wrapping only adds breaks. Text that arrives wrapped to some other width + // keeps that wrapping rather than being reflowed into this one, even where + // its lines would fit together. + Buffer buffer(40, Charset::Ascii); + EXPECT_EQ( + buffer.DrawWrappedText(0, 0, 0, 30, "already\nwrapped\nnarrow", Style()) + .y, + 2); + EXPECT_EQ(Render(buffer), + "already\n" + "wrapped\n" + "narrow\n"); + + // A break the text ends with closes its last line, and a row nothing was + // drawn into is not a row, so it doesn't also open an empty one. + Buffer trailing(40, Charset::Ascii); + trailing.DrawWrappedText(0, 0, 0, 30, "one\ntwo\n", Style()); + EXPECT_EQ(Render(trailing), + "one\n" + "two\n"); +} + +TEST(BufferTest, WrappedTextLineBreaks) { + // Carriage returns are dropped, so CRLF endings break exactly once. + Buffer buffer(10, Charset::Ascii); + EXPECT_EQ(buffer.DrawWrappedText(0, 0, 0, 10, "a\r\nb", Style()).y, 1); + EXPECT_EQ(Render(buffer), + "a\n" + "b\n"); + + // Whitespace stops at the block's edge, so a wrapped row starts at the + // margin. + Buffer spaces(10, Charset::Ascii); + spaces.DrawWrappedText(0, 0, 0, 5, "aaaaa bbbbb", Style()); + EXPECT_EQ(Render(spaces), + "aaaaa\n" + "bbbbb\n"); +} + +TEST(BufferTest, WrappedTextFillsTheWidthLeftOfTheMargin) { + // A caller with nothing to divide the width between gives the block all of + // what is left of it, which is what wrapping to the terminal is. + Buffer buffer(20, Charset::Ascii); + buffer.DrawText(0, 0, "-> ", Style()); + buffer.DrawWrappedText(3, 0, 3, buffer.columns() - 3, + "several words that would otherwise fit", Style()); + EXPECT_EQ(Render(buffer), + "-> several words\n" + " that would\n" + " otherwise fit\n"); +} + +TEST(BufferTest, WrappedTextKeepsAMarkWithTheWhitespaceBeforeIt) { + // A combining mark renders into the column before it, which for a mark + // following whitespace is that whitespace. Left to begin the word after it, + // the mark would move to another row whenever that word wrapped and attach to + // whatever preceded it there. + Buffer buffer(10, Charset::Utf8); + std::string mark = AcuteE.drop_front(1).str(); + buffer.DrawWrappedText(0, 0, 0, 5, "aaa " + mark + "bbbb", Style()); + EXPECT_EQ(Render(buffer), "aaa " + mark + "\nbbbb\n"); + + // Whitespace stops at the block's edge, so a mark can arrive where none of it + // was drawn. It attaches to the last cell written rather than being carried + // to the row the next word wraps onto. + Buffer filled(10, Charset::Utf8); + filled.DrawWrappedText(0, 0, 0, 5, "aaaaa " + mark + "bb", Style()); + EXPECT_EQ(Render(filled), "aaaaa" + mark + "\nbb\n"); +} + +TEST(BufferTest, WrappedTextKeepsCharactersWiderThanTheRegion) { + // No row in a one-column region could hold a double-width character. Drawing + // it anyway overruns the region, which is what keeping the text costs here. + Buffer buffer(10, Charset::Utf8); + buffer.DrawWrappedText(0, 0, 0, 1, "中中", Style()); + EXPECT_EQ(Render(buffer), "中中\n"); +} + +TEST(BufferTest, WrappedTextWithDoubleWidthCharacters) { + // Wrapping counts columns, not characters, so half as many double-width ones + // fit a row. + Buffer buffer(10, Charset::Utf8); + EXPECT_EQ(buffer.DrawWrappedText(0, 0, 0, 4, "中中 中中", Style()).y, 1); + EXPECT_EQ(Render(buffer), + "中中\n" + "中中\n"); +} + +TEST(BufferTest, TrailingBlanksAreDropped) { + // Padding out to the buffer's width would put invisible whitespace into + // every line of output, which shows up in diffs and in copied text. + Buffer buffer(40, Charset::Ascii); + buffer.DrawText(0, 0, "hi", Style()); + EXPECT_EQ(Render(buffer), "hi\n"); +} + +TEST(BufferTest, StyledBlanksAreKept) { + // A blank cell with a background still paints, so it isn't padding. + Buffer buffer(10, Charset::Ascii); + buffer.DrawCodePoint(0, 0, 'x', Style()); + buffer.DrawCodePoint(3, 0, ' ', Style().Background(AnsiColor::Red)); + EXPECT_EQ(Render(buffer, ColorMode::Ansi16), "x \x1b[41m \x1b[0m\n"); + + // With color off the background paints nothing, so those cells are padding + // again and must not reach the output as trailing spaces. + EXPECT_EQ(Render(buffer, ColorMode::NoColor), "x\n"); +} + +TEST(BufferTest, RenderMinimizesEscapes) { + Buffer buffer(4, Charset::Ascii); + Style red_bold = Style().Bold().Foreground(Color(255, 0, 0)); + buffer.DrawCodePoint(0, 0, 'A', red_bold); + // Sharing the attributes and changing only the color costs one escape. + buffer.DrawCodePoint(1, 0, 'B', red_bold.Foreground(Color(0, 0, 255))); + // Dropping bold costs a reset and a fresh start. + buffer.DrawCodePoint(2, 0, 'C', Style().Foreground(Color(0, 0, 255))); + buffer.DrawCodePoint(3, 0, 'D', Style()); + + EXPECT_EQ(Render(buffer, ColorMode::Truecolor), + "\x1b[1m\x1b[38;2;255;0;0m" + "A" + "\x1b[38;2;0;0;255m" + "B" + "\x1b[0m\x1b[38;2;0;0;255m" + "C" + "\x1b[0m" + "D\n"); +} + +TEST(BufferTest, MeasureText) { + Buffer utf8(10, Charset::Utf8); + EXPECT_EQ(utf8.MeasureText(0, 0, ""), DrawEnd(0, 0)); + EXPECT_EQ(utf8.MeasureText(0, 0, "hello"), DrawEnd(5, 0)); + EXPECT_EQ(utf8.MeasureText(0, 0, "中A🔥"), DrawEnd(5, 0)); + EXPECT_EQ(utf8.MeasureText(0, 0, AcuteE), DrawEnd(1, 0)); + EXPECT_EQ(utf8.MeasureText(0, 0, llvm::StringRef("\xc0\x80", 2)), + DrawEnd(2, 0)); + + // Newlines, carriage returns, and tabs are interpreted as drawing does. + EXPECT_EQ(utf8.MeasureText(0, 0, "a\nb"), DrawEnd(1, 1)); + EXPECT_EQ(utf8.MeasureText(0, 0, "a\r\nb"), DrawEnd(1, 1)); + EXPECT_EQ(utf8.MeasureText(0, 0, "a\tb"), DrawEnd(9, 0)); + + // Measuring starts from where it is told to, which is the column a newline + // returns to and the origin the tab stops count from. + EXPECT_EQ(utf8.MeasureText(3, 2, "a\tb"), DrawEnd(12, 2)); + EXPECT_EQ(utf8.MeasureText(3, 2, "a\nb"), DrawEnd(4, 3)); + + // Every byte is a column when the terminal isn't decoding UTF-8. + Buffer ascii(10, Charset::Ascii); + EXPECT_EQ(ascii.MeasureText(0, 0, "hello"), DrawEnd(5, 0)); + EXPECT_EQ(ascii.MeasureText(0, 0, "中A🔥"), DrawEnd(8, 0)); + EXPECT_EQ(ascii.MeasureText(0, 0, AcuteE), DrawEnd(3, 0)); +} + +TEST(BufferTest, MeasuringMatchesDrawing) { + // The point of measuring is to answer what drawing would, so check the two + // against each other on the text most likely to make them disagree. + for (llvm::StringRef text : + {"hello", "a\tb\tc", "a\nb\r\nc", "中A🔥", "a b", ""}) { + Buffer buffer(10, Charset::Utf8); + EXPECT_EQ(buffer.MeasureText(2, 1, text), + buffer.DrawText(2, 1, text, Style())) + << text; + + // A margin left of where the text starts moves what the positional + // characters answer to, and moves it for both of them alike. + Buffer block(10, Charset::Utf8); + EXPECT_EQ(block.MeasureText(2, 1, 1, text), + block.DrawText(2, 1, 1, text, Style())) + << text; + } + for (llvm::StringRef text : + {"one two three", "a\nlonger line here", "verylongunbreakableword", + "a\tb\tc", "col\tone\nrow\ttwo", "e\xcc\x81 \xcc\x81word", ""}) { + Buffer buffer(10, Charset::Utf8); + EXPECT_EQ(buffer.MeasureWrappedText(2, 1, 2, 8, text), + buffer.DrawWrappedText(2, 1, 2, 8, text, Style())) + << text; + } +} + +TEST(BufferTest, MeasureWrapWidth) { + Buffer buffer(10, Charset::Utf8); + EXPECT_EQ(buffer.MeasureWrapWidth(""), 0); + EXPECT_EQ(buffer.MeasureWrapWidth("a bb ccc"), 3); + EXPECT_EQ(buffer.MeasureWrapWidth(" spaced out "), 6); + + // Newlines and tabs bound a word without taking columns of their own. + EXPECT_EQ(buffer.MeasureWrapWidth("a\nbb\tccc"), 3); + + // A word is measured in the columns it takes, not the bytes it holds. + EXPECT_EQ(buffer.MeasureWrapWidth("中中 a"), 4); +} + +TEST(BufferTest, WrapWidthIsWhatWrappingDoesNotOverhang) { + // The width answered for is a fact about wrapping, so it is checked against + // the wrapping it describes. + Buffer buffer(10, Charset::Utf8); + // Tabs don't widen the answer: one stops at the block's edge rather than + // running to a stop outside it, so it can't overhang either. + for (llvm::StringRef text : + {"some quite long words here", "some\tquite\tlong words here"}) { + int width = buffer.MeasureWrapWidth(text); + Buffer drawn(width, Charset::Utf8); + drawn.DrawWrappedText(0, 0, 0, width, text, Style()); + llvm::SmallVector rows; + llvm::StringRef(Render(drawn)).split(rows, '\n'); + for (llvm::StringRef row : rows) { + EXPECT_LE(static_cast(row.size()), width) << row; + } + } +} + +TEST(BufferTest, BuiltFromCapabilities) { + Capabilities capabilities; + capabilities.columns = 5; + capabilities.charset = Charset::Utf8; + + Buffer buffer(capabilities); + EXPECT_EQ(buffer.columns(), 5); + EXPECT_EQ(buffer.charset(), Charset::Utf8); + buffer.DrawHorizontalLine(0, 0, 5, Style()); + EXPECT_EQ(Render(buffer, capabilities.color_mode), "╶───╴\n"); + + // A terminal that said nothing about its width gets one chosen to be safe + // wherever the output ends up, rather than no width at all. + capabilities.columns = std::nullopt; + Buffer fallback(capabilities); + EXPECT_EQ(fallback.columns(), DefaultColumns); + + // One claiming a width no grid can hold gets the nearest that can be held. + capabilities.columns = Buffer::MaxColumns + 1; + Buffer clamped(capabilities); + EXPECT_EQ(clamped.columns(), Buffer::MaxColumns); +} + +TEST(BufferTest, CombiningMarkPastTheWidthItStartedWith) { + // The buffer grows to hold the text, so a mark arriving past the width it + // was constructed with attaches to its base like any other. + Buffer buffer(4, Charset::Utf8); + buffer.DrawText(0, 0, ("abcde" + AcuteE.drop_front(1)).str(), Style()); + EXPECT_EQ(Render(buffer), ("abcde" + AcuteE.drop_front(1) + "\n").str()); +} + +TEST(BufferTest, CombiningMarksSurviveGrowthAndOverdraw) { + // Marks live in a side table keyed by cell index, so widening has to move + // them with the rows and overdrawing has to take them with the cell. + for (int start : {1, 2, 3}) { + Buffer buffer(start, Charset::Utf8); + buffer.DrawText(0, 0, std::string(AcuteE) + "e" + std::string(AcuteE), + Style()); + buffer.DrawText(0, 1, "xxxxxxxxxxxx", Style()); + buffer.DrawCodePoint(0, 0, U'z', Style()); + std::string rendered = Render(buffer); + // The mark on the overdrawn cell went with it; the later one stayed. + EXPECT_EQ(rendered.substr(0, rendered.find('\n')), + "ze" + std::string(AcuteE)) + << start; + } +} + +TEST(BufferTest, CombiningMarkWithNoBase) { + // Marks render into the column before them, so one at the start of a row has + // nowhere to go rather than attaching to the end of the row above. This is + // ordinary input rather than a caller mistake -- a source file can open a + // line with a mark -- so it is dropped and drawing goes on. + Buffer buffer(4, Charset::Utf8); + buffer.DrawText(0, 0, "ab", Style()); + buffer.DrawText(0, 1, AcuteE.drop_front(1).str(), Style()); + EXPECT_EQ(Render(buffer), "ab\n"); + + // Nor does it disturb what is already drawn on the row it lands on. + Buffer after(4, Charset::Utf8); + after.DrawText(0, 0, "ab", Style()); + after.DrawCodePoint(0, 0, CombiningAcute, Style()); + EXPECT_EQ(Render(after), "ab\n"); +} + +TEST(BufferTest, OverhangStopsAtTheGridBound) { + // An overhanging word is the only thing that reaches the grid's far edge, and + // how far it reaches is a fact about the text rather than something a caller + // could have checked, so it is clipped there rather than being a mistake. + // The column still advances past it, which is what keeps measuring and + // drawing answering the same thing. + Buffer buffer(4, Charset::Ascii); + std::string word(Buffer::MaxColumns + 10, 'x'); + DrawEnd past(Buffer::MaxColumns + 10, 0); + EXPECT_EQ(buffer.MeasureWrappedText(0, 0, 0, 4, word), past); + EXPECT_EQ(buffer.DrawWrappedText(0, 0, 0, 4, word, Style()), past); + EXPECT_EQ(buffer.width(), Buffer::MaxColumns); + EXPECT_EQ(buffer.height(), 1); +} + +TEST(BufferTest, ACharacterStraddlingTheGridBoundIsNotDrawn) { + // Whether a character fits the far edge depends on how wide it turns out to + // be, and half of one is not something a terminal can render, so one + // straddling the bound is past it entirely. + Buffer buffer(4, Charset::Utf8); + std::string word(Buffer::MaxColumns - 1, 'x'); + word += "中"; + EXPECT_EQ(buffer.DrawWrappedText(0, 0, 0, 4, word, Style()), + DrawEnd(Buffer::MaxColumns + 1, 0)); + EXPECT_EQ(buffer.width(), Buffer::MaxColumns); + // Only the run before it reached the output; the character itself did not. + EXPECT_EQ(Render(buffer).size(), static_cast(Buffer::MaxColumns)); +} + +TEST(BufferTest, APointBoxLeavesALineAlone) { + // A point is line art with no directions, so drawing one where a line + // already runs adds nothing and leaves the line's own directions in place. + Buffer buffer(3, Charset::Utf8); + buffer.DrawHorizontalLine(0, 0, 3, Style()); + buffer.DrawBox(1, 0, 1, 1, Style()); + EXPECT_EQ(Render(buffer), "╶─╴\n"); + + // With nothing there, the point is drawn. + Buffer empty(3, Charset::Utf8); + empty.DrawBox(1, 0, 1, 1, Style()); + EXPECT_EQ(Render(empty), " ·\n"); +} + +TEST(BufferDeathTest, WidthMustFitTheGrid) { + // A width comes from `COLUMNS` by way of `Capabilities`, which clamps it. A + // caller reaching this constructor has computed the width itself, so a value + // the grid can't hold is a mistake rather than bad input. + EXPECT_DEATH(Buffer(0, Charset::Ascii), "Buffer width must be in"); + EXPECT_DEATH(Buffer(-1, Charset::Ascii), "Buffer width must be in"); + EXPECT_DEATH(Buffer(Buffer::MaxColumns + 1, Charset::Ascii), + "Buffer width must be in"); +} + +TEST(BufferDeathTest, DrawingMustStartInsideTheGrid) { + // A caller placing something already knows the width, since it is what + // decided the layout, so landing outside it is a bug in that layout rather + // than something to quietly drop. Rows are checked the same way. + Buffer buffer(10, Charset::Ascii); + EXPECT_DEATH(buffer.DrawCodePoint(10, 0, 'a', Style()), "is outside the"); + EXPECT_DEATH(buffer.DrawCodePoint(-1, 0, 'a', Style()), "is outside the"); + EXPECT_DEATH(buffer.DrawCodePoint(0, -1, 'a', Style()), "is outside the"); + EXPECT_DEATH(buffer.DrawCodePoint(0, Buffer::MaxRows, 'a', Style()), + "is outside the"); + EXPECT_DEATH(buffer.DrawText(10, 0, "a", Style()), "is outside the"); + EXPECT_DEATH(buffer.MeasureText(10, 0, "a"), "is outside the"); + + // Text begins at or right of the margin its rows return to. + EXPECT_DEATH(buffer.DrawText(2, 0, 3, "a", Style()), "left of its margin"); + EXPECT_DEATH(buffer.MeasureText(2, 0, -1, "a"), "left of its margin"); +} + +TEST(BufferDeathTest, LinesMustFitWhatTheyAreDrawnInto) { + // Nothing about a line is unbreakable, so unlike text it has no reason to + // reach outside the width, and one that does came from a wrong extent. + Buffer buffer(10, Charset::Ascii); + EXPECT_DEATH(buffer.DrawHorizontalLine(6, 0, 5, Style()), "runs outside the"); + EXPECT_DEATH(buffer.DrawHorizontalLine(0, 0, -1, Style()), + "runs outside the"); + EXPECT_DEATH(buffer.DrawVerticalLine(0, 0, Buffer::MaxRows + 1, Style()), + "runs outside the"); + EXPECT_DEATH(buffer.DrawBox(0, 0, 11, 2, Style()), "runs outside the"); +} + +TEST(BufferDeathTest, WrappedBlocksMustFitTheWidth) { + // A block is a division of the width rather than something that can exceed + // it, and the text has to start inside the block it wraps in. + Buffer buffer(10, Charset::Ascii); + EXPECT_DEATH(buffer.DrawWrappedText(0, 0, 0, 11, "a", Style()), + "does not fit the"); + EXPECT_DEATH(buffer.DrawWrappedText(4, 0, 4, 8, "a", Style()), + "does not fit the"); + EXPECT_DEATH(buffer.DrawWrappedText(0, 0, 0, 0, "a", Style()), + "does not fit the"); + EXPECT_DEATH(buffer.DrawWrappedText(2, 0, 4, 6, "a", Style()), + "does not fit the"); + EXPECT_DEATH(buffer.MeasureWrappedText(0, 0, 0, 11, "a"), "does not fit the"); +} + +TEST(BufferDeathTest, TextMustBeShortEnoughToMeasure) { + // Measuring walks the text adding widths, so a long enough run would carry + // the column past what an `int` holds. Text this long was built rather than + // read off a line. + Buffer buffer(10, Charset::Ascii); + std::string huge(Buffer::MaxTextBytes + 1, 'x'); + EXPECT_DEATH(buffer.DrawText(0, 0, huge, Style()), "is past the"); + EXPECT_DEATH(buffer.MeasureText(0, 0, huge), "is past the"); +} + +TEST(BufferTest, WriteTo) { + Buffer buffer(20, Charset::Ascii); + buffer.DrawText(0, 0, "hello", Style().Bold()); + buffer.DrawText(0, 1, "world", Style()); + + auto dir = Filesystem::MakeTmpDir(); + ASSERT_TRUE(dir.ok()) << dir.error(); + auto file = dir->OpenWriteOnly("out", Filesystem::CreationOptions::CreateNew); + ASSERT_TRUE(file.ok()) << file.error(); + auto written = buffer.WriteTo(*file, ColorMode::Ansi16); + EXPECT_TRUE(written.ok()) << written.error(); + (*std::move(file)).Close().Check(); + + // Byte for byte what `Render` produces. + auto read_back = dir->ReadFileToString("out"); + ASSERT_TRUE(read_back.ok()) << read_back.error(); + EXPECT_EQ(*read_back, Render(buffer, ColorMode::Ansi16)); + EXPECT_EQ(*read_back, "\x1b[1mhello\n\x1b[0mworld\n"); +} + +TEST(BufferTest, StyleCarriesAcrossRows) { + // A style is usually still in use on the row below, so turning it off at the + // end of a row and back on at the start of the next costs a reset and a fresh + // start for nothing. + Buffer buffer(4, Charset::Ascii); + buffer.DrawText(0, 0, "ab", Style().Bold()); + buffer.DrawText(0, 1, "cd", Style().Bold()); + EXPECT_EQ(Render(buffer, ColorMode::Ansi16), "\x1b[1mab\ncd\x1b[0m\n"); +} + +TEST(BufferTest, StyleThatPaintsBlanksStopsAtTheRowEnd) { + // A terminal fills the rest of a row with the background it is in when the + // row ends, so a style that paints where there is no glyph can't be left on + // across the newline the way an attribute that only affects a glyph can. + Buffer buffer(4, Charset::Ascii); + buffer.DrawText(0, 0, "ab", Style().Background(AnsiColor::Red)); + buffer.DrawText(0, 1, "cd", Style().Background(AnsiColor::Red)); + EXPECT_EQ(Render(buffer, ColorMode::Ansi16), + "\x1b[41mab\x1b[0m\n" + "\x1b[41mcd\x1b[0m\n"); +} + +TEST(BufferTest, RenderAppends) { + // Rendering appends, so several buffers can be gathered into one write. + Buffer first(10, Charset::Ascii); + first.DrawText(0, 0, "one", Style()); + Buffer second(10, Charset::Ascii); + second.DrawText(0, 0, "two", Style()); + + llvm::SmallString<64> out; + first.Render(out, ColorMode::NoColor); + second.Render(out, ColorMode::NoColor); + EXPECT_EQ(std::string(out), "one\ntwo\n"); +} + +} // namespace +} // namespace Carbon::Terminal diff --git a/common/terminal/capabilities.cpp b/common/terminal/capabilities.cpp new file mode 100644 index 000000000000..20e4fe719264 --- /dev/null +++ b/common/terminal/capabilities.cpp @@ -0,0 +1,216 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#include "common/terminal/capabilities.h" + +#include +#include + +#include + +#include "llvm/ADT/StringExtras.h" +#include "llvm/ADT/StringSwitch.h" + +namespace Carbon::Terminal { + +// Returns the value of `name` in the process environment, empty when unset. +static auto GetEnv(const char* name) -> llvm::StringRef { + const char* value = std::getenv(name); + return value ? llvm::StringRef(value) : llvm::StringRef(); +} + +auto ColorEnvironment::FromProcess() -> ColorEnvironment { + return {.no_color = GetEnv("NO_COLOR"), + .clicolor_force = GetEnv("CLICOLOR_FORCE"), + .force_color = GetEnv("FORCE_COLOR"), + .clicolor = GetEnv("CLICOLOR"), + .colorterm = GetEnv("COLORTERM"), + .term_program = GetEnv("TERM_PROGRAM"), + .term = GetEnv("TERM")}; +} + +// Returns whether the environment and the stream call for color, ignoring any +// explicit preference. See `ChooseColorMode` for the precedence this +// implements and where it comes from. +static auto EnvironmentEnablesColor(const ColorEnvironment& env, + bool is_terminal) -> bool { + if (!env.no_color.empty()) { + return false; + } + // The forcing variables use `0` to decline to force, and `FORCE_COLOR` takes + // it further as a request to disable. + if (env.force_color == "0") { + return false; + } + if (!env.force_color.empty() || + (!env.clicolor_force.empty() && env.clicolor_force != "0")) { + return true; + } + if (env.clicolor == "0") { + return false; + } + if (!is_terminal) { + return false; + } + + // `dumb` says outright that escape sequences won't render, which outranks + // anything below claiming they will. + if (env.term == "dumb") { + return false; + } + + // Any of these identifies the terminal as something that renders escapes. A + // terminal that none of them describe can't be assumed to. + // + // `COLORTERM` and `TERM_PROGRAM` stand on their own rather than refining + // `TERM`: `TERM` names a terminfo entry, while these name the emulator and + // the color it handles. + return !env.term.empty() || !env.colorterm.empty() || + !env.term_program.empty(); +} + +// Returns the richest color escapes the terminal is believed to accept. +// +// Every signal here is a heuristic: there is no way to ask a terminal what it +// supports without writing to it and parsing a reply, which would be far too +// invasive for a compiler. Guessing too high garbles color on a terminal that +// can't keep up, and guessing too low only makes output plainer, so unknown +// terminals get the conservative answer. +static auto DetectColorDepth(const ColorEnvironment& env) -> ColorMode { + // `FORCE_COLOR`'s levels name a depth outright. + if (auto mode = llvm::StringSwitch>(env.force_color) + .Case("1", ColorMode::Ansi16) + .Case("2", ColorMode::Ansi256) + .Case("3", ColorMode::Truecolor) + .Default(std::nullopt)) { + return *mode; + } + + // The convention documented at + // https://github.com/termstandard/colors#checking-for-colorterm. + if (env.colorterm == "truecolor" || env.colorterm == "24bit") { + return ColorMode::Truecolor; + } + + // `TERM_PROGRAM` identifies the emulator regardless of how `TERM` is set, + // which matters because several of these ship a conservative `TERM` while + // rendering far more than it claims. + { + if (auto mode = + llvm::StringSwitch>(env.term_program) + .Case("vscode", ColorMode::Truecolor) + .Case("iTerm.app", ColorMode::Truecolor) + .Case("WarpTerminal", ColorMode::Truecolor) + .Case("Hyper", ColorMode::Truecolor) + .Case("Tabby", ColorMode::Truecolor) + .Case("Terminus", ColorMode::Truecolor) + // Apple's Terminal renders only the 256-color palette. + .Case("Apple_Terminal", ColorMode::Ansi256) + .Default(std::nullopt)) { + return *mode; + } + } + + // The enumerated terminals that stand in for a terminfo lookup. + { + if (auto mode = llvm::StringSwitch>(env.term) + .Case("xterm-kitty", ColorMode::Truecolor) + .Case("alacritty", ColorMode::Truecolor) + .Case("wezterm", ColorMode::Truecolor) + .Case("ghostty", ColorMode::Truecolor) + .StartsWith("foot", ColorMode::Truecolor) + .StartsWith("contour", ColorMode::Truecolor) + .StartsWith("vte", ColorMode::Truecolor) + .EndsWith("-direct", ColorMode::Truecolor) + .EndsWith("-truecolor", ColorMode::Truecolor) + .EndsWith("-256color", ColorMode::Ansi256) + .EndsWith("-256", ColorMode::Ansi256) + .Default(std::nullopt)) { + return *mode; + } + } + + // Color is called for, but nothing said how much of it works. + return ColorMode::Ansi16; +} + +auto ChooseColorMode(Preference preference, const ColorEnvironment& env, + bool is_terminal) -> ColorMode { + switch (preference) { + case Preference::Never: + return ColorMode::NoColor; + case Preference::Always: + break; + case Preference::Auto: + if (!EnvironmentEnablesColor(env, is_terminal)) { + return ColorMode::NoColor; + } + break; + } + return DetectColorDepth(env); +} + +auto ChooseCharset(Preference preference, llvm::StringRef locale) -> Charset { + switch (preference) { + case Preference::Never: + return Charset::Ascii; + case Preference::Always: + return Charset::Utf8; + case Preference::Auto: + break; + } + + // Locale names spell the encoding several ways: `en_US.UTF-8`, `C.utf8`, and + // bare `UTF-8` all appear in the wild. + return locale.contains_insensitive("utf-8") || + locale.contains_insensitive("utf8") + ? Charset::Utf8 + : Charset::Ascii; +} + +// Returns the locale that determines the terminal's character encoding, +// following the precedence POSIX defines for `LC_CTYPE`. +static auto GetLocale() -> llvm::StringRef { + for (const char* name : {"LC_ALL", "LC_CTYPE", "LANG"}) { + if (llvm::StringRef value = GetEnv(name); !value.empty()) { + return value; + } + } + return ""; +} + +// Returns the terminal's width in columns, or nullopt when there is nothing to +// ask. +// +// `COLUMNS` comes first: when it is exported, the user has deliberately +// overridden the real width. +static auto GetColumns(int fd) -> std::optional { + int columns = 0; + if (llvm::to_integer(GetEnv("COLUMNS"), columns) && columns > 0) { + return columns; + } + + struct winsize size = {}; + if (ioctl(fd, TIOCGWINSZ, &size) == 0 && size.ws_col > 0) { + return size.ws_col; + } + return std::nullopt; +} + +auto Capabilities::Detect(Filesystem::WriteFileRef file, + Preferences preferences) -> Capabilities { + int fd = file.unix_fd(); + + Capabilities capabilities; + capabilities.is_terminal = isatty(fd) != 0; + capabilities.color_mode = + ChooseColorMode(preferences.color, ColorEnvironment::FromProcess(), + capabilities.is_terminal); + capabilities.charset = ChooseCharset(preferences.utf8, GetLocale()); + capabilities.columns = GetColumns(fd); + + return capabilities; +} + +} // namespace Carbon::Terminal diff --git a/common/terminal/capabilities.h b/common/terminal/capabilities.h new file mode 100644 index 000000000000..51bac57f7367 --- /dev/null +++ b/common/terminal/capabilities.h @@ -0,0 +1,203 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#ifndef CARBON_COMMON_TERMINAL_CAPABILITIES_H_ +#define CARBON_COMMON_TERMINAL_CAPABILITIES_H_ + +#include +#include + +#include "common/filesystem.h" +#include "common/terminal/color.h" +#include "llvm/ADT/StringRef.h" + +namespace Carbon::Terminal { + +// The encoding the terminal decodes output with. +// +// This decides far more than which characters can be drawn: rendering has to +// count the columns a run of bytes will occupy, and that count only follows +// from the code points those bytes encode if the terminal agrees about the +// encoding. Disagreeing misaligns the entire line rather than drawing a single +// character wrong. +// +// No other encoding is modeled. A terminal decoding something else, an ISO 8859 +// part for example, is treated as `Ascii`. That is correct output for any of +// them, as they all encode printable ASCII as itself, and rendering in one +// natively would mean carrying its conversion and column-width tables to gain +// nothing but nicer line drawing, which `Ascii` already has a fallback for. So +// `Utf8` is used only where the environment says outright that the terminal +// decodes UTF-8, and `Ascii` does no UTF-8 processing at all. +enum class Charset : int8_t { + // Every byte is one column, and lines are drawn from `-`, `|`, and `+`. + // + // Bytes outside printable ASCII are replaced rather than passed through, + // because a terminal decoding some single-byte encoding will render them as + // something, and there is no way to know what. + Ascii, + // Bytes are decoded as UTF-8, giving double-width characters two columns and + // combining marks none, and lines are drawn with box-drawing characters. + Utf8, +}; + +// Whether to use one of the terminal features detection decides about, where +// an explicit request overrides what detection would conclude. +// +// This is the tri-state that a `--color=never` style flag parses into. It says +// nothing about which feature is being requested; `Preferences` holds one of +// these per feature. +enum class Preference : int8_t { + // Decide from the environment and the stream. + Auto, + // Never use the feature, whatever the environment says. + Never, + // Use the feature even when the stream isn't a terminal. This is what a + // caller wants when piping into a pager, capturing output for later replay, + // or writing a test. + Always, +}; + +// An explicit preference for each feature detection decides about, normally +// parsed from command line flags. +struct Preferences { + Preference color = Preference::Auto; + Preference utf8 = Preference::Auto; +}; + +// The environment variables that control whether and how color is used. +// +// An unset variable and one set to the empty string mean the same thing +// throughout: no opinion. Values point into the process environment and are +// invalidated by anything that modifies it. +struct ColorEnvironment { + // Reads the variables from the process environment. + static auto FromProcess() -> ColorEnvironment; + + llvm::StringRef no_color; + llvm::StringRef clicolor_force; + llvm::StringRef force_color; + llvm::StringRef clicolor; + llvm::StringRef colorterm; + llvm::StringRef term_program; + llvm::StringRef term; +}; + +// Returns the color mode to render with. +// +// Whether to use color at all is decided first, from highest priority to +// lowest: +// +// - An explicit `Never` or `Always` preference. +// - `NO_COLOR` set to any non-empty value disables color: see +// https://no-color.org. +// - `FORCE_COLOR=0` disables color, following the Node convention. +// - Any other non-empty `FORCE_COLOR`, or a non-empty `CLICOLOR_FORCE` other +// than `0`, enables color even when the stream isn't a terminal. +// - `CLICOLOR=0` disables color. +// - `TERM=dumb` disables color, being an explicit statement that escape +// sequences won't render. +// - Otherwise color is used only when the stream is a terminal and something +// identifies that terminal: `TERM` set to anything else, or `COLORTERM` or +// `TERM_PROGRAM` set at all. The latter two stand on their own rather than +// refining `TERM`, because the emulator sets them itself and they are +// specifically about color, while `TERM` is left unset by anything not +// launched from a shell. +// +// How much color to use is then guessed from `FORCE_COLOR`'s level, +// `COLORTERM`, `TERM_PROGRAM`, and `TERM`, falling back to `Ansi16` when color +// is called for but nothing says how much of it works. Apart from +// `FORCE_COLOR`, which is a request rather than a description, none of these +// enable color on their own, so a rich `COLORTERM` inherited by a redirected +// stream can't put escape sequences into it. +// +// Depth comes from enumerating known terminals rather than from terminfo, +// which trades a list to maintain here for not depending on databases that are +// routinely absent from the containers and CI images this runs in. An +// unrecognized terminal gets the conservative answer. +// +// This is separated from `Capabilities::Detect` so that the policy can be +// tested without touching the process environment. +auto ChooseColorMode(Preference preference, const ColorEnvironment& env, + bool is_terminal) -> ColorMode; + +// Returns the encoding to render with, where `locale` is the value of the +// first set variable among `LC_ALL`, `LC_CTYPE`, and `LANG`. +// +// Only a locale that names UTF-8 gets `Utf8`. Guessing wrong in that direction +// costs alignment on every line that isn't pure ASCII, while guessing wrong +// the other way only makes output plainer. +auto ChooseCharset(Preference preference, llvm::StringRef locale) -> Charset; + +// The width to lay out for when nothing says how wide the output is. +// +// Layout always has a width to fit, because the alternative is output laid out +// as if nothing bounded it, which a terminal then wraps at column zero -- +// breaking every indent and gutter it was given, and in the middle of whatever +// word it lands on. The cost of guessing is asymmetric: a viewer wider than +// this sees slack on the right, while one narrower sees the wrapping done +// twice, ours and then its own. +// +// Eighty is the traditional terminal width, and narrower ones are rare enough +// that fitting them would cost more in wasted width everywhere else. +inline constexpr int DefaultColumns = 80; + +// The columns between tab stops, absent anything saying otherwise. +// +// Eight is the interval terminfo records as `it#8` for all but a handful of +// legacy entries. Nothing measures a terminal's stops, so unlike its width this +// stands in for no measurement: `Capabilities` carries it as a plain value +// rather than as one a caller can tell apart from an absence. +inline constexpr int DefaultTabWidth = 8; + +// What the terminal behind a stream can render, and how wide it is. +// +// Detect this once per stream at startup and pass it down; the fields come from +// environment queries and system calls that shouldn't be repeated per +// diagnostic. +struct Capabilities { + // Detects the capabilities of the terminal behind `file`, honoring + // `preferences`. + // + // Detection reads the descriptor directly, because `isatty` and `TIOCGWINSZ` + // are what answer the question and no stream abstraction exposes them. + // LLVM's `raw_ostream::has_colors()` is not a substitute for the enablement + // rule above, which recognizes terminals its `TERM` list doesn't. + static auto Detect(Filesystem::WriteFileRef file, + Preferences preferences = {}) -> Capabilities; + + // The richest color escapes the terminal is believed to understand. + ColorMode color_mode = ColorMode::NoColor; + + // The encoding the terminal decodes output with. + Charset charset = Charset::Ascii; + + // Whether the stream is attached to a terminal at all. Note that color can + // still be in use when this is false, if the environment forces it. + bool is_terminal = false; + + // The terminal's width, or none when nothing says how wide the output is. + // Positive whenever it is set, so layout can divide by it freely. + // + // This says what was measured, and nothing is invented to fill it in: an + // absence is a real answer about a pipe nobody described. Laying out still + // needs a width, and `DefaultColumns` is what a layout falls back to, so + // whether this is set decides whether output is fitted to the terminal in + // front of it or to a width chosen to be safe wherever it ends up. + std::optional columns; + + // The columns between the terminal's tab stops, which is what a tab in text + // advances to the next of. + // + // TODO: Nothing sets this away from `DefaultTabWidth`. A terminal's stops are + // mutable at runtime -- `hts` sets one and `tbc` clears them -- so the only + // report of the live ones is `DECRQPSR`, which few emulators outside `xterm` + // answer, or a `DSR-CPR` round trip after writing a tab, which nearly all do. + // Either means putting the descriptor in raw mode and reading a reply with a + // timeout. Add it when a terminal that disagrees with eight is worth that. + int tab_width = DefaultTabWidth; +}; + +} // namespace Carbon::Terminal + +#endif // CARBON_COMMON_TERMINAL_CAPABILITIES_H_ diff --git a/common/terminal/capabilities_test.cpp b/common/terminal/capabilities_test.cpp new file mode 100644 index 000000000000..96aa272f0bd7 --- /dev/null +++ b/common/terminal/capabilities_test.cpp @@ -0,0 +1,287 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#include "common/terminal/capabilities.h" + +#include + +#include + +#include "common/filesystem.h" + +namespace Carbon::Terminal { +namespace { + +// Detection policy is a pure function of the environment and whether the +// stream is a terminal, so these tests build the environment directly instead +// of mutating the process environment, which would leak between tests and race +// with anything else running. +auto OnTerminal(const ColorEnvironment& env) -> ColorMode { + return ChooseColorMode(Preference::Auto, env, /*is_terminal=*/true); +} + +auto OffTerminal(const ColorEnvironment& env) -> ColorMode { + return ChooseColorMode(Preference::Auto, env, /*is_terminal=*/false); +} + +// A terminal that supports color, for tests varying one other variable. +auto ColorTerminal() -> ColorEnvironment { return {.term = "xterm-256color"}; } + +TEST(CapabilitiesTest, ColorNeedsATerminal) { + EXPECT_EQ(OnTerminal(ColorTerminal()), ColorMode::Ansi256); + + // Writing to a file or a pipe must stay plain, or every redirected build log + // fills with escape sequences. + EXPECT_EQ(OffTerminal(ColorTerminal()), ColorMode::NoColor); + + // A terminal that nothing says anything about can't be assumed to render + // escapes. + EXPECT_EQ(OnTerminal({}), ColorMode::NoColor); + EXPECT_EQ(OnTerminal({.term = ""}), ColorMode::NoColor); + EXPECT_EQ(OnTerminal({.term = "dumb"}), ColorMode::NoColor); +} + +TEST(CapabilitiesTest, ColorFromTheEmulatorWithoutTerm) { + // `TERM` is unset for anything not launched from a shell, but the emulator + // sets `COLORTERM` and `TERM_PROGRAM` itself, and both are specifically + // about color. Either one identifies the terminal on its own, at whatever + // depth it names. + EXPECT_EQ(OnTerminal({.colorterm = "truecolor"}), ColorMode::Truecolor); + EXPECT_EQ(OnTerminal({.colorterm = "yes"}), ColorMode::Ansi16); + EXPECT_EQ(OnTerminal({.term_program = "vscode"}), ColorMode::Truecolor); + EXPECT_EQ(OnTerminal({.term_program = "unknown"}), ColorMode::Ansi16); + + // An empty value says nothing at all. + EXPECT_EQ(OnTerminal({.colorterm = "", .term_program = ""}), + ColorMode::NoColor); + + // `dumb` outranks them: it states outright that escapes won't render. + EXPECT_EQ(OnTerminal({.colorterm = "truecolor", .term = "dumb"}), + ColorMode::NoColor); + + // And none of them enable color off a terminal. + EXPECT_EQ(OffTerminal({.colorterm = "truecolor"}), ColorMode::NoColor); + EXPECT_EQ(OffTerminal({.term_program = "vscode"}), ColorMode::NoColor); +} + +TEST(CapabilitiesTest, ExplicitPreferenceWins) { + ColorEnvironment forcing = {.force_color = "3", .term = "xterm-256color"}; + EXPECT_EQ(ChooseColorMode(Preference::Never, forcing, /*is_terminal=*/true), + ColorMode::NoColor); + + ColorEnvironment disabling = {.no_color = "1", .term = "dumb"}; + EXPECT_EQ(ChooseColorMode(Preference::Always, disabling, + /*is_terminal=*/false), + ColorMode::Ansi16); + + // Forcing color on without any hint of what the terminal handles gets the + // depth every color terminal supports. + EXPECT_EQ(ChooseColorMode(Preference::Always, {}, /*is_terminal=*/false), + ColorMode::Ansi16); + EXPECT_EQ(ChooseColorMode(Preference::Always, ColorTerminal(), + /*is_terminal=*/false), + ColorMode::Ansi256); +} + +TEST(CapabilitiesTest, NoColor) { + // https://no-color.org: any non-empty value disables color, whatever it is. + EXPECT_EQ(OnTerminal({.no_color = "1", .term = "xterm-256color"}), + ColorMode::NoColor); + EXPECT_EQ(OnTerminal({.no_color = "0", .term = "xterm-256color"}), + ColorMode::NoColor); + + // Being set to the empty string carries no meaning, so it must not disable + // color: an empty variable inherited from a wrapper script would otherwise + // silently turn color off everywhere. + EXPECT_EQ(OnTerminal({.no_color = "", .term = "xterm-256color"}), + ColorMode::Ansi256); + + // It outranks the forcing variables. + EXPECT_EQ(OnTerminal({.no_color = "1", .force_color = "3"}), + ColorMode::NoColor); + EXPECT_EQ(OnTerminal({.no_color = "1", .clicolor_force = "1"}), + ColorMode::NoColor); +} + +TEST(CapabilitiesTest, ForceColor) { + // Color even without a terminal, at the depth the level names. + EXPECT_EQ(OffTerminal({.force_color = "1"}), ColorMode::Ansi16); + EXPECT_EQ(OffTerminal({.force_color = "2"}), ColorMode::Ansi256); + EXPECT_EQ(OffTerminal({.force_color = "3"}), ColorMode::Truecolor); + + // The level overrides what the terminal claims. + EXPECT_EQ(OnTerminal({.force_color = "1", .colorterm = "truecolor"}), + ColorMode::Ansi16); + + // Any other non-empty value enables color without naming a depth. + EXPECT_EQ(OffTerminal({.force_color = "true"}), ColorMode::Ansi16); + EXPECT_EQ(OffTerminal({.force_color = "true", .term = "xterm-256color"}), + ColorMode::Ansi256); + + // Zero disables color outright, even on a capable terminal. + EXPECT_EQ(OnTerminal({.force_color = "0", .term = "xterm-256color"}), + ColorMode::NoColor); +} + +TEST(CapabilitiesTest, EmptyValuesCarryNoOpinion) { + // An empty variable means the same as an unset one throughout, so a wrapper + // script that exports one without a value changes nothing. + EXPECT_EQ(OffTerminal({.force_color = ""}), ColorMode::NoColor); + EXPECT_EQ(OffTerminal({.clicolor_force = ""}), ColorMode::NoColor); + EXPECT_EQ(OnTerminal({.no_color = "", .term = "xterm-256color"}), + ColorMode::Ansi256); + EXPECT_EQ(OnTerminal({.clicolor = "", .term = "xterm-256color"}), + ColorMode::Ansi256); + EXPECT_EQ(OnTerminal({.force_color = "", .term = "xterm-256color"}), + ColorMode::Ansi256); +} + +TEST(CapabilitiesTest, CliColor) { + // The BSD convention: `CLICOLOR_FORCE` enables color off a terminal, and + // `CLICOLOR=0` disables it on one. + EXPECT_EQ(OffTerminal({.clicolor_force = "1"}), ColorMode::Ansi16); + EXPECT_EQ(OffTerminal({.clicolor_force = "1", .term = "xterm-256color"}), + ColorMode::Ansi256); + // `0` means "don't force", not "disable", so a terminal still gets color. + EXPECT_EQ(OnTerminal({.clicolor_force = "0", .term = "xterm-256color"}), + ColorMode::Ansi256); + EXPECT_EQ(OffTerminal({.clicolor_force = "0"}), ColorMode::NoColor); + + EXPECT_EQ(OnTerminal({.clicolor = "0", .term = "xterm-256color"}), + ColorMode::NoColor); + EXPECT_EQ(OnTerminal({.clicolor = "1", .term = "xterm-256color"}), + ColorMode::Ansi256); + + // Forcing beats disabling. + EXPECT_EQ(OnTerminal({.clicolor_force = "1", .clicolor = "0"}), + ColorMode::Ansi16); +} + +TEST(CapabilitiesTest, ColorDepthFromColorterm) { + EXPECT_EQ(OnTerminal({.colorterm = "truecolor", .term = "xterm"}), + ColorMode::Truecolor); + EXPECT_EQ(OnTerminal({.colorterm = "24bit", .term = "xterm"}), + ColorMode::Truecolor); + EXPECT_EQ(OnTerminal({.colorterm = "", .term = "xterm"}), ColorMode::Ansi16); + + // `COLORTERM` can't enable color off a terminal, so a rich value inherited + // by a redirected stream can't smuggle escapes into it. + EXPECT_EQ(OffTerminal({.colorterm = "truecolor", .term = "xterm"}), + ColorMode::NoColor); +} + +TEST(CapabilitiesTest, ColorDepthFromTermProgram) { + EXPECT_EQ(OnTerminal({.term_program = "vscode", .term = "xterm"}), + ColorMode::Truecolor); + EXPECT_EQ(OnTerminal({.term_program = "iTerm.app", .term = "xterm"}), + ColorMode::Truecolor); + EXPECT_EQ(OnTerminal({.term_program = "WarpTerminal", .term = "xterm"}), + ColorMode::Truecolor); + EXPECT_EQ(OnTerminal({.term_program = "Hyper", .term = "xterm"}), + ColorMode::Truecolor); + EXPECT_EQ(OnTerminal({.term_program = "Tabby", .term = "xterm"}), + ColorMode::Truecolor); + EXPECT_EQ(OnTerminal({.term_program = "Terminus", .term = "xterm"}), + ColorMode::Truecolor); + // Apple's Terminal renders only the 256-color palette. + EXPECT_EQ(OnTerminal({.term_program = "Apple_Terminal", .term = "xterm"}), + ColorMode::Ansi256); + EXPECT_EQ(OnTerminal({.term_program = "unknown", .term = "xterm-256color"}), + ColorMode::Ansi256); +} + +TEST(CapabilitiesTest, ColorDepthFromTerm) { + EXPECT_EQ(OnTerminal({.term = "xterm-kitty"}), ColorMode::Truecolor); + EXPECT_EQ(OnTerminal({.term = "alacritty"}), ColorMode::Truecolor); + EXPECT_EQ(OnTerminal({.term = "wezterm"}), ColorMode::Truecolor); + EXPECT_EQ(OnTerminal({.term = "foot-extra"}), ColorMode::Truecolor); + EXPECT_EQ(OnTerminal({.term = "xterm-direct"}), ColorMode::Truecolor); + EXPECT_EQ(OnTerminal({.term = "ghostty"}), ColorMode::Truecolor); + EXPECT_EQ(OnTerminal({.term = "contour-latest"}), ColorMode::Truecolor); + EXPECT_EQ(OnTerminal({.term = "xterm-truecolor"}), ColorMode::Truecolor); + + EXPECT_EQ(OnTerminal({.term = "xterm-256color"}), ColorMode::Ansi256); + EXPECT_EQ(OnTerminal({.term = "screen-256color"}), ColorMode::Ansi256); + EXPECT_EQ(OnTerminal({.term = "putty-256"}), ColorMode::Ansi256); + + // A terminal matching both a truecolor and a 256-color pattern takes the + // richer one, so the order these are tried in is load-bearing. + EXPECT_EQ(OnTerminal({.term = "vte-256color"}), ColorMode::Truecolor); + + // Known to render color, but with nothing saying how much. + EXPECT_EQ(OnTerminal({.term = "xterm"}), ColorMode::Ansi16); + EXPECT_EQ(OnTerminal({.term = "linux"}), ColorMode::Ansi16); +} + +TEST(CapabilitiesTest, Charset) { + EXPECT_EQ(ChooseCharset(Preference::Auto, "en_US.UTF-8"), Charset::Utf8); + EXPECT_EQ(ChooseCharset(Preference::Auto, "C.utf8"), Charset::Utf8); + EXPECT_EQ(ChooseCharset(Preference::Auto, "en_US.utf-8"), Charset::Utf8); + + // Drawing box characters into a terminal decoding something else turns them + // into several bytes of mojibake and destroys the alignment they were for. + EXPECT_EQ(ChooseCharset(Preference::Auto, "C"), Charset::Ascii); + EXPECT_EQ(ChooseCharset(Preference::Auto, "POSIX"), Charset::Ascii); + EXPECT_EQ(ChooseCharset(Preference::Auto, "en_US.ISO-8859-1"), + Charset::Ascii); + EXPECT_EQ(ChooseCharset(Preference::Auto, ""), Charset::Ascii); + + EXPECT_EQ(ChooseCharset(Preference::Never, "en_US.UTF-8"), Charset::Ascii); + EXPECT_EQ(ChooseCharset(Preference::Always, "C"), Charset::Utf8); +} + +TEST(CapabilitiesTest, Defaults) { + // The defaults describe a plain-text sink, which is what a file or a pipe + // gets and what tests should use unless exercising something richer. + Capabilities capabilities; + EXPECT_EQ(capabilities.color_mode, ColorMode::NoColor); + EXPECT_EQ(capabilities.charset, Charset::Ascii); + EXPECT_FALSE(capabilities.is_terminal); + EXPECT_FALSE(capabilities.columns.has_value()); +} + +TEST(CapabilitiesTest, Detect) { + // Detection reads the process environment and the descriptor it is handed, so + // only what neither can change is pinned here. What the policy decides from + // given inputs is tested above, against `ChooseColorMode` and `ChooseCharset` + // directly. + // + // It detects against a file rather than the process's own streams: those are + // a pipe under the test runner but a terminal under a debugger, and an + // exported `FORCE_COLOR` turns color on for either. + auto dir = Filesystem::MakeTmpDir(); + ASSERT_TRUE(dir.ok()) << dir.error(); + auto file = dir->OpenWriteOnly("out", Filesystem::CreationOptions::CreateNew); + ASSERT_TRUE(file.ok()) << file.error(); + + Capabilities capabilities = Capabilities::Detect(*file); + // A file is never a terminal. + EXPECT_FALSE(capabilities.is_terminal); + // `COLUMNS` reaches detection from the environment, so whether a width is + // found depends on it, but one that is found is usable. + if (capabilities.columns) { + EXPECT_GT(*capabilities.columns, 0); + } + EXPECT_GT(capabilities.tab_width, 0); + + // A preference decides on its own, whatever the environment holds. Color + // forced on picks a depth from the environment, so only that it is on can be + // pinned here. + EXPECT_EQ(Capabilities::Detect( + *file, {.color = Preference::Never, .utf8 = Preference::Never}) + .color_mode, + ColorMode::NoColor); + EXPECT_NE( + Capabilities::Detect(*file, {.color = Preference::Always}).color_mode, + ColorMode::NoColor); + EXPECT_EQ(Capabilities::Detect(*file, {.utf8 = Preference::Always}).charset, + Charset::Utf8); + EXPECT_EQ(Capabilities::Detect(*file, {.utf8 = Preference::Never}).charset, + Charset::Ascii); + + (*std::move(file)).Close().Check(); +} + +} // namespace +} // namespace Carbon::Terminal diff --git a/common/terminal/color.cpp b/common/terminal/color.cpp new file mode 100644 index 000000000000..14ae42b60d9e --- /dev/null +++ b/common/terminal/color.cpp @@ -0,0 +1,239 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#include "common/terminal/color.h" + +#include +#include + +#include "common/check.h" +#include "llvm/ADT/StringRef.h" +#include "llvm/Support/Format.h" + +namespace Carbon::Terminal { + +static constexpr int AnsiColorCount = 16; + +// Reference values for the 16 ANSI colors. +// +// Nothing standardizes these: a terminal draws them from the user's palette, +// which is exactly what makes them worth using. But downsampling an RGB color +// still needs some notion of where each named color sits, so these use the +// xterm defaults, which terminals vary from but stay recognizably near. +static constexpr std::array AnsiColorRgbs = {{ + {.r = 0, .g = 0, .b = 0}, // Black + {.r = 205, .g = 0, .b = 0}, // Red + {.r = 0, .g = 205, .b = 0}, // Green + {.r = 205, .g = 205, .b = 0}, // Yellow + {.r = 0, .g = 0, .b = 238}, // Blue + {.r = 205, .g = 0, .b = 205}, // Magenta + {.r = 0, .g = 205, .b = 205}, // Cyan + {.r = 229, .g = 229, .b = 229}, // White + {.r = 127, .g = 127, .b = 127}, // BrightBlack + {.r = 255, .g = 0, .b = 0}, // BrightRed + {.r = 0, .g = 255, .b = 0}, // BrightGreen + {.r = 255, .g = 255, .b = 0}, // BrightYellow + {.r = 92, .g = 92, .b = 255}, // BrightBlue + {.r = 255, .g = 0, .b = 255}, // BrightMagenta + {.r = 0, .g = 255, .b = 255}, // BrightCyan + {.r = 255, .g = 255, .b = 255}, // BrightWhite +}}; + +static constexpr std::array AnsiColorNames = { + "Black", "Red", "Green", "Yellow", + "Blue", "Magenta", "Cyan", "White", + "BrightBlack", "BrightRed", "BrightGreen", "BrightYellow", + "BrightBlue", "BrightMagenta", "BrightCyan", "BrightWhite"}; + +// Returns the "redmean" distance between two colors, squared and scaled by 256 +// to keep it in integer arithmetic. +// +// Treating the channels as orthogonal axes is cheaper but sits a long way from +// perceived difference, and downsampling is exactly where that shows: a color +// picked for a diagnostic lands on whichever of a small fixed set the +// arithmetic says is closest, and a plain Euclidean fit underweights green, +// where the eye is most sensitive. Redmean weights the +// channels by where the pair sits on the red axis, which tracks perception far +// better for a couple of extra multiplies: +// https://en.wikipedia.org/wiki/Color_difference#sRGB +// +// The formula ends in a square root, which is dropped because only the ordering +// is used. Scaling by 256 turns the two fractional weights into integers; the +// result peaks just under 150 million, well inside the range. +static auto DistanceSquared(Color::RgbValue lhs, Color::RgbValue rhs) -> int { + int red_mean = (static_cast(lhs.r) + static_cast(rhs.r)) / 2; + int dr = static_cast(lhs.r) - static_cast(rhs.r); + int dg = static_cast(lhs.g) - static_cast(rhs.g); + int db = static_cast(lhs.b) - static_cast(rhs.b); + return (512 + red_mean) * dr * dr + 1024 * dg * dg + + (767 - red_mean) * db * db; +} + +// Returns the ANSI color whose reference value is nearest to `rgb`. +static auto NearestAnsiColor(Color::RgbValue rgb) -> AnsiColor { + int best_index = 0; + int best_distance = DistanceSquared(rgb, AnsiColorRgbs[0]); + for (int i = 1; i < AnsiColorCount; ++i) { + int distance = DistanceSquared(rgb, AnsiColorRgbs[i]); + if (distance < best_distance) { + best_distance = distance; + best_index = i; + } + } + return static_cast(best_index); +} + +// The channel values of the 6x6x6 color cube at palette indices 16 through +// 231. The first step is much larger than the rest, so a channel can't be +// rounded to the nearest level by dividing. +static constexpr std::array CubeLevels = {0, 95, 135, + 175, 215, 255}; + +// The midpoints between adjacent entries of `CubeLevels`, which are where the +// nearest level changes. +static constexpr std::array CubeLevelMidpoints = {48, 115, 155, 195, + 235}; + +static_assert( + [] { + for (size_t i = 0; i < CubeLevelMidpoints.size(); ++i) { + // Rounded up, so that a value exactly between two levels takes the + // higher one. + if (CubeLevelMidpoints[i] != + (CubeLevels[i] + CubeLevels[i + 1] + 1) / 2) { + return false; + } + } + return true; + }(), + "Midpoints must stay in step with the levels they separate."); + +// Returns the index into `CubeLevels` of the level nearest `value`. +static auto NearestCubeLevel(uint8_t value) -> int { + int level = 0; + while (level < static_cast(CubeLevelMidpoints.size()) && + value >= CubeLevelMidpoints[level]) { + ++level; + } + return level; +} + +// Returns the 256-color palette index whose color is nearest to `rgb`. +static auto NearestPaletteIndex(Color::RgbValue rgb) -> uint8_t { + // Only the color cube and the gray ramp are considered. Indices 0 through 15 + // alias the ANSI colors, whose appearance comes from the user's palette, so + // an exact RGB request must never be answered with one. + int r_level = NearestCubeLevel(rgb.r); + int g_level = NearestCubeLevel(rgb.g); + int b_level = NearestCubeLevel(rgb.b); + Color::RgbValue cube = {.r = CubeLevels[r_level], + .g = CubeLevels[g_level], + .b = CubeLevels[b_level]}; + + // The gray ramp at indices 232 through 255 runs from 8 to 238 in steps of + // 10, and is finer than the cube's gray diagonal for near-neutral colors. + int average = (static_cast(rgb.r) + static_cast(rgb.g) + + static_cast(rgb.b)) / + 3; + int gray_step = std::clamp((average - 8 + 5) / 10, 0, 23); + auto gray_value = static_cast(8 + 10 * gray_step); + Color::RgbValue gray = {.r = gray_value, .g = gray_value, .b = gray_value}; + + if (DistanceSquared(rgb, gray) < DistanceSquared(rgb, cube)) { + return 232 + gray_step; + } + return 16 + 36 * r_level + 6 * g_level + b_level; +} + +// Returns the SGR parameter selecting `color` for `target`. +// +// The original ANSI codes cover the first eight colors, and the later "bright" +// codes cover the rest at a fixed offset. +static auto AnsiSgrCode(AnsiColor color, ColorTarget target) -> uint8_t { + CARBON_CHECK(target != ColorTarget::Underline, + "Underline color has no direct ANSI form."); + int index = static_cast(color); + int base = target == ColorTarget::Background ? 40 : 30; + if (index >= 8) { + // Bright foregrounds are 90-97 and bright backgrounds 100-107. + base += 60; + } + return base + (index % 8); +} + +// Returns the SGR parameter introducing an extended color for `target`, which +// is followed by either `;5;` or `;2;;;`. +static auto ExtendedSgrCode(ColorTarget target) -> uint8_t { + switch (target) { + case ColorTarget::Foreground: + return 38; + case ColorTarget::Background: + return 48; + case ColorTarget::Underline: + return 58; + } +} + +auto Color::AppendEscape(OutputBufferRef out, ColorMode mode, + ColorTarget target) const -> void { + if (mode == ColorMode::NoColor) { + return; + } + + // Underline colors are only expressible through the extended-color escape, + // which `Ansi16` doesn't use. + if (target == ColorTarget::Underline && mode == ColorMode::Ansi16) { + return; + } + + CARBON_CHECK(is_set(), "Only a color that is set can be selected."); + + if (kind_ == Kind::Ansi) { + if (target == ColorTarget::Underline) { + // Named underline colors go through the palette form of the extended + // escape, as there is no direct code for them. + out.Append("\x1b[58;5;", static_cast(ansi()), "m"); + } else { + out.Append("\x1b[", AnsiSgrCode(ansi(), target), "m"); + } + return; + } + + switch (mode) { + case ColorMode::Truecolor: + out.Append("\x1b[", ExtendedSgrCode(target), ";2;", channels_.r, ";", + channels_.g, ";", channels_.b, "m"); + break; + + case ColorMode::Ansi256: + out.Append("\x1b[", ExtendedSgrCode(target), ";5;", + NearestPaletteIndex(channels_), "m"); + break; + + case ColorMode::Ansi16: + out.Append("\x1b[", AnsiSgrCode(NearestAnsiColor(channels_), target), + "m"); + break; + + case ColorMode::NoColor: + CARBON_FATAL("Returned above without emitting anything."); + } +} + +auto Color::Print(llvm::raw_ostream& out) const -> void { + switch (kind_) { + case Kind::None: + out << "None"; + return; + case Kind::Ansi: + out << AnsiColorNames[static_cast(ansi())]; + return; + case Kind::Rgb: + out << llvm::format("#%02x%02x%02x", channels_.r, channels_.g, + channels_.b); + return; + } +} + +} // namespace Carbon::Terminal diff --git a/common/terminal/color.h b/common/terminal/color.h new file mode 100644 index 000000000000..358f457d0192 --- /dev/null +++ b/common/terminal/color.h @@ -0,0 +1,167 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#ifndef CARBON_COMMON_TERMINAL_COLOR_H_ +#define CARBON_COMMON_TERMINAL_COLOR_H_ + +#include + +#include "common/check.h" +#include "common/ostream.h" +#include "common/terminal/output_buffer_ref.h" + +namespace Carbon::Terminal { + +// The color escape sequences a terminal understands. +// +// Colors, like the other text attributes, are selected with Select Graphic +// Rendition (SGR) escape sequences. The sequence and the codes for the first +// eight colors come from ECMA-48, published in parallel as ANSI X3.64; +// terminal emulators added the bright variants, the 256-color palette, and the +// 24-bit form: +// https://ecma-international.org/publications-and-standards/standards/ecma-48/ +// +// These form a ladder: each mode can express everything the modes before it +// can. Colors that the active mode can't express exactly are downsampled to +// the nearest color it can, so callers author in the richest form and let +// rendering degrade on its own. +enum class ColorMode : int8_t { + // Emit no escape sequences at all, producing plain text. + NoColor, + // The 16 colors with SGR codes of their own. + Ansi16, + // The 256-color palette: the 16 ANSI colors, a 6x6x6 RGB cube, and a 24-step + // gray ramp. + Ansi256, + // Direct 24-bit RGB, commonly called "truecolor". + Truecolor, +}; + +// The 16 colors with SGR codes of their own. +// +// Terminals render these through the user's configured palette, which makes +// them the right choice for output that should blend with the user's theme. +// The tradeoff is that their rendered appearance is outside our control: a +// user's "red" may be any color at all. +enum class AnsiColor : uint8_t { + Black, + Red, + Green, + Yellow, + Blue, + Magenta, + Cyan, + White, + BrightBlack, + BrightRed, + BrightGreen, + BrightYellow, + BrightBlue, + BrightMagenta, + BrightCyan, + BrightWhite, +}; + +// Which part of a cell's rendering a color applies to. +enum class ColorTarget : int8_t { + Foreground, + Background, + // The color of the underline itself, independent of the foreground. Only + // `Ansi256` and richer modes can express this. + Underline, +}; + +// A color to render with: one of the 16 named ANSI colors, a 24-bit RGB value, +// or no color at all. +// +// RGB colors render exactly where the terminal supports them, and are +// downsampled where it doesn't. Downsampling to `Ansi16` measures distance +// against fixed reference values, but the terminal renders the result from the +// user's palette, so a downsampled color can land far from the original. +// Prefer `AnsiColor` wherever output should track the user's theme, and RGB +// only where an exact color matters. +// +// A default-constructed color selects nothing. That is how a `Style` spells +// leaving one of its colors to the terminal, so this is a value with an empty +// state rather than something wrapped in an `optional` to get one. +class Color : public Printable { + public: + // Whether a color names a palette entry, gives channel values directly, or + // selects nothing. + enum class Kind : uint8_t { + None, + Ansi, + Rgb, + }; + + // The channel values of a 24-bit color. + struct RgbValue { + uint8_t r; + uint8_t g; + uint8_t b; + + friend auto operator==(RgbValue lhs, RgbValue rhs) -> bool = default; + }; + + constexpr Color() = default; + + // Colors convert implicitly from `AnsiColor` so that call sites can read as + // `style.Foreground(AnsiColor::Red)`. + // + // NOLINTNEXTLINE(google-explicit-constructor) + constexpr Color(AnsiColor ansi) + : kind_(Kind::Ansi), channels_{.r = static_cast(ansi)} {} + + constexpr Color(uint8_t r, uint8_t g, uint8_t b) + : kind_(Kind::Rgb), channels_{.r = r, .g = g, .b = b} {} + + auto kind() const -> Kind { return kind_; } + + // Returns whether this selects a color at all. + auto is_set() const -> bool { return kind_ != Kind::None; } + + // Returns the named color. Valid only when `kind()` is `Ansi`. + auto ansi() const -> AnsiColor { + CARBON_CHECK(kind_ == Kind::Ansi, + "Only a named color has a palette index."); + return static_cast(channels_.r); + } + + // Returns the channel values. Valid only when `kind()` is `Rgb`. + auto rgb() const -> RgbValue { + CARBON_CHECK(kind_ == Kind::Rgb, "Only an RGB color has channel values."); + return channels_; + } + + // Appends the escape sequence selecting this color for `target`, which + // requires that one is set. + // + // Appends nothing when `mode` is `NoColor`, or when `target` is `Underline` + // and `mode` is `Ansi16`, which has no way to express an underline color. + auto AppendEscape(OutputBufferRef out, ColorMode mode, + ColorTarget target) const -> void; + + auto Print(llvm::raw_ostream& out) const -> void; + + // Written out rather than defaulted because the `Printable` base has no + // comparison of its own, which would leave a defaulted one deleted. + friend auto operator==(Color lhs, Color rhs) -> bool { + return lhs.kind_ == rhs.kind_ && lhs.channels_ == rhs.channels_; + } + + private: + Kind kind_ = Kind::None; + + // The palette index in `r` with the rest zero for `Ansi`, the channel values + // for `Rgb`, and all zero for `None`. + // + // Overlapping the two in a union would leave the bytes past a palette index + // unwritten. Every byte carrying part of the value is what lets a whole + // `Style` be compared as bytes, and an index fits in a channel anyway. + RgbValue channels_ = {.r = 0, .g = 0, .b = 0}; +}; + +} // namespace Carbon::Terminal + +#endif // CARBON_COMMON_TERMINAL_COLOR_H_ diff --git a/common/terminal/color_test.cpp b/common/terminal/color_test.cpp new file mode 100644 index 000000000000..c97cd990ef28 --- /dev/null +++ b/common/terminal/color_test.cpp @@ -0,0 +1,171 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#include "common/terminal/color.h" + +#include + +#include "llvm/ADT/SmallString.h" +#include "llvm/ADT/StringExtras.h" + +namespace Carbon::Terminal { +namespace { + +auto Escape(Color color, ColorMode mode, + ColorTarget target = ColorTarget::Foreground) -> std::string { + llvm::SmallString<32> escape; + color.AppendEscape(escape, mode, target); + return std::string(escape); +} + +TEST(ColorTest, AnsiEscapes) { + // Named colors use their own SGR codes rather than the extended forms, in + // every mode that has color at all, so the terminal renders them from the + // user's palette. + for (ColorMode mode : + {ColorMode::Ansi16, ColorMode::Ansi256, ColorMode::Truecolor}) { + EXPECT_EQ(Escape(AnsiColor::Red, mode), "\x1b[31m"); + EXPECT_EQ(Escape(AnsiColor::Black, mode), "\x1b[30m"); + EXPECT_EQ(Escape(AnsiColor::BrightCyan, mode), "\x1b[96m"); + EXPECT_EQ(Escape(AnsiColor::Red, mode, ColorTarget::Background), + "\x1b[41m"); + EXPECT_EQ(Escape(AnsiColor::BrightWhite, mode, ColorTarget::Background), + "\x1b[107m"); + } + + EXPECT_EQ(Escape(AnsiColor::Red, ColorMode::NoColor), ""); +} + +TEST(ColorTest, RgbEscapes) { + Color red(255, 0, 0); + EXPECT_EQ(Escape(red, ColorMode::Truecolor), "\x1b[38;2;255;0;0m"); + EXPECT_EQ(Escape(red, ColorMode::Truecolor, ColorTarget::Background), + "\x1b[48;2;255;0;0m"); + EXPECT_EQ(Escape(red, ColorMode::Ansi256), "\x1b[38;5;196m"); + EXPECT_EQ(Escape(red, ColorMode::Ansi16), "\x1b[91m"); + EXPECT_EQ(Escape(red, ColorMode::NoColor), ""); +} + +TEST(ColorTest, UnderlineEscapes) { + // Underline colors only exist in the extended-color escapes, so in `Ansi16` + // the terminal draws the underline in the foreground color. + EXPECT_EQ( + Escape(AnsiColor::Red, ColorMode::Truecolor, ColorTarget::Underline), + "\x1b[58;5;1m"); + EXPECT_EQ(Escape(AnsiColor::Red, ColorMode::Ansi256, ColorTarget::Underline), + "\x1b[58;5;1m"); + EXPECT_EQ(Escape(AnsiColor::Red, ColorMode::Ansi16, ColorTarget::Underline), + ""); + + Color green(0, 255, 0); + EXPECT_EQ(Escape(green, ColorMode::Truecolor, ColorTarget::Underline), + "\x1b[58;2;0;255;0m"); + EXPECT_EQ(Escape(green, ColorMode::Ansi256, ColorTarget::Underline), + "\x1b[58;5;46m"); + EXPECT_EQ(Escape(green, ColorMode::Ansi16, ColorTarget::Underline), ""); +} + +TEST(ColorTest, DownsampleToAnsi16) { + // The reference value of each ANSI color must come back as that color, or + // downsampling would shift colors that were already expressible. Spelling + // the values out here rather than reading them back from the same table the + // implementation uses is what makes this catch a wrong table. + struct Expected { + Color color; + llvm::StringRef escape; + }; + Expected cases[] = { + {Color(0, 0, 0), "\x1b[30m"}, {Color(205, 0, 0), "\x1b[31m"}, + {Color(0, 205, 0), "\x1b[32m"}, {Color(205, 205, 0), "\x1b[33m"}, + {Color(0, 0, 238), "\x1b[34m"}, {Color(205, 0, 205), "\x1b[35m"}, + {Color(0, 205, 205), "\x1b[36m"}, {Color(229, 229, 229), "\x1b[37m"}, + {Color(127, 127, 127), "\x1b[90m"}, {Color(255, 0, 0), "\x1b[91m"}, + {Color(0, 255, 0), "\x1b[92m"}, {Color(255, 255, 0), "\x1b[93m"}, + {Color(92, 92, 255), "\x1b[94m"}, {Color(255, 0, 255), "\x1b[95m"}, + {Color(0, 255, 255), "\x1b[96m"}, {Color(255, 255, 255), "\x1b[97m"}, + }; + for (const Expected& expected : cases) { + EXPECT_EQ(Escape(expected.color, ColorMode::Ansi16), expected.escape) + << expected.color; + } + + // Colors between the reference values land on the nearest one. + EXPECT_EQ(Escape(Color(250, 10, 10), ColorMode::Ansi16), "\x1b[91m"); + EXPECT_EQ(Escape(Color(10, 10, 10), ColorMode::Ansi16), "\x1b[30m"); + EXPECT_EQ(Escape(Color(120, 120, 120), ColorMode::Ansi16), "\x1b[90m"); +} + +TEST(ColorTest, DownsampleToPalette) { + // The corners of the 6x6x6 cube are exactly representable. + EXPECT_EQ(Escape(Color(0, 0, 0), ColorMode::Ansi256), "\x1b[38;5;16m"); + EXPECT_EQ(Escape(Color(255, 255, 255), ColorMode::Ansi256), "\x1b[38;5;231m"); + EXPECT_EQ(Escape(Color(255, 0, 0), ColorMode::Ansi256), "\x1b[38;5;196m"); + EXPECT_EQ(Escape(Color(0, 0, 255), ColorMode::Ansi256), "\x1b[38;5;21m"); + + // The cube's levels are unevenly spaced, so rounding has to account for that + // rather than divide: 95 and 135 are adjacent levels only 40 apart. + EXPECT_EQ(Escape(Color(95, 0, 0), ColorMode::Ansi256), "\x1b[38;5;52m"); + EXPECT_EQ(Escape(Color(130, 0, 0), ColorMode::Ansi256), "\x1b[38;5;88m"); + + // Near-neutral colors land on the gray ramp, which is far finer than the + // cube's diagonal, except at the ends where the cube wins. + EXPECT_EQ(Escape(Color(8, 8, 8), ColorMode::Ansi256), "\x1b[38;5;232m"); + EXPECT_EQ(Escape(Color(128, 128, 128), ColorMode::Ansi256), "\x1b[38;5;244m"); + EXPECT_EQ(Escape(Color(238, 238, 238), ColorMode::Ansi256), "\x1b[38;5;255m"); +} + +TEST(ColorTest, DownsampleAvoidsPaletteEntries) { + // Indices 0 through 15 render from the user's palette, so an exact RGB + // request must never be answered with one. + for (int r = 0; r < 256; r += 17) { + for (int g = 0; g < 256; g += 17) { + for (int b = 0; b < 256; b += 17) { + Color color(r, g, b); + std::string escape = Escape(color, ColorMode::Ansi256); + int index = 0; + ASSERT_TRUE(llvm::to_integer( + llvm::StringRef(escape).drop_front(7).drop_back(1), index)) + << color; + EXPECT_GE(index, 16) << color; + } + } + } +} + +TEST(ColorTest, Equality) { + EXPECT_EQ(Color(AnsiColor::Red), Color(AnsiColor::Red)); + EXPECT_NE(Color(AnsiColor::Red), Color(AnsiColor::Blue)); + EXPECT_EQ(Color(1, 2, 3), Color(1, 2, 3)); + EXPECT_NE(Color(1, 2, 3), Color(1, 2, 4)); + + // A named color and its reference value are different colors: the terminal + // renders one from the palette and the other exactly. + EXPECT_NE(Color(AnsiColor::BrightRed), Color(255, 0, 0)); + + // A palette index occupies the same byte as the red channel, so these pairs + // hold identical channel bytes and are told apart only by their kind. + EXPECT_NE(Color(AnsiColor::Red), Color(1, 0, 0)); + EXPECT_NE(Color(AnsiColor::Black), Color(0, 0, 0)); + EXPECT_NE(Color(AnsiColor::Black), Color()); + EXPECT_NE(Color(0, 0, 0), Color()); + EXPECT_EQ(Color(), Color()); +} + +TEST(ColorTest, Unset) { + EXPECT_FALSE(Color().is_set()); + EXPECT_EQ(Color().kind(), Color::Kind::None); + + // Black is a color like any other, however little of it there is. + EXPECT_TRUE(Color(AnsiColor::Black).is_set()); + EXPECT_TRUE(Color(0, 0, 0).is_set()); +} + +TEST(ColorTest, Print) { + EXPECT_EQ(PrintToString(Color(AnsiColor::BrightMagenta)), "BrightMagenta"); + EXPECT_EQ(PrintToString(Color(0x12, 0xab, 0xff)), "#12abff"); + EXPECT_EQ(PrintToString(Color()), "None"); +} + +} // namespace +} // namespace Carbon::Terminal diff --git a/common/terminal/metrics.cpp b/common/terminal/metrics.cpp new file mode 100644 index 000000000000..221a2849d1e0 --- /dev/null +++ b/common/terminal/metrics.cpp @@ -0,0 +1,190 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#include "common/terminal/metrics.h" + +#include "common/check.h" +#include "llvm/Support/ConvertUTF.h" +#include "llvm/Support/Unicode.h" + +namespace Carbon::Terminal { + +// Stands in for anything a UTF-8 terminal has no rendering for: invalid UTF-8, +// control characters, and unassigned code points. +static constexpr char32_t Utf8Replacement = U'�'; + +// Returns whether an ASCII terminal renders `code_point` as itself, in one +// column. +static auto IsPrintableAscii(char32_t code_point) -> bool { + return code_point >= 0x20 && code_point < 0x7f; +} + +// Spelled out rather than handed to a general converter, which walks a range +// and checks bounds this already knows. Box-drawing characters go through here +// for every cell of every line drawn. +// +// TODO: Offer this to LLVM, whose `ConvertCodePointToUTF8` is the general +// converter this replaces. Encoding one code point at a time is what anything +// writing UTF-8 out of a grid does, so this belongs beside it rather than +// here; drop this once it is there. +auto EncodeUtf8(char32_t code_point, Utf8Storage& storage) -> llvm::StringRef { + // Most of what gets rendered is ASCII, and encoding it is a single byte. + if (code_point < 0x80) { + storage[0] = static_cast(code_point); + return llvm::StringRef(storage.data(), 1); + } + + // Surrogates have no encoding of their own, and nothing past the last code + // point has one at all. + if (code_point > 0x10ffff || (code_point >= 0xd800 && code_point < 0xe000)) { + code_point = Utf8Replacement; + } + + auto trailing = [code_point](int shift) { + return static_cast(0b1000'0000 | ((code_point >> shift) & 0b11'1111)); + }; + if (code_point < 0x800) { + storage[0] = static_cast(0b1100'0000 | (code_point >> 6)); + storage[1] = trailing(0); + return llvm::StringRef(storage.data(), 2); + } + if (code_point < 0x10000) { + storage[0] = static_cast(0b1110'0000 | (code_point >> 12)); + storage[1] = trailing(6); + storage[2] = trailing(0); + return llvm::StringRef(storage.data(), 3); + } + storage[0] = static_cast(0b1111'0000 | (code_point >> 18)); + storage[1] = trailing(12); + storage[2] = trailing(6); + storage[3] = trailing(0); + return llvm::StringRef(storage.data(), 4); +} + +// Returns the columns `code_point` occupies on a UTF-8 terminal: zero for a +// combining mark, one or two for one with a glyph of its own, and a +// negative value when there is no printable rendering for it. +// +// TODO: This encodes a code point only for LLVM to decode it again. +// `llvm::sys::unicode::charWidth` computes exactly this and is what +// `columnWidthUTF8` calls once per code point, but it is file-local to LLVM's +// `Unicode.cpp`. Exposing it there would let this call it directly. LLVM's own +// contract already says a string's width is the sum of its code points', so +// there is nothing in the way of it. +static auto Utf8CodePointWidth(char32_t code_point) -> int { + // Printable ASCII is one column, and is most of what gets measured. The + // general path parses a UTF-8 sequence and searches several code point + // range tables, which is far more than this needs. + if (IsPrintableAscii(code_point)) { + return 1; + } + + Utf8Storage storage; + return llvm::sys::unicode::columnWidthUTF8(EncodeUtf8(code_point, storage)); +} + +// Removes the first UTF-8 sequence from `text` and returns the code point it +// encodes. +static auto TakeUtf8CodePoint(llvm::StringRef& text) -> char32_t { + const auto* begin = reinterpret_cast(text.data()); + const auto* pos = begin; + llvm::UTF32 code_point = 0; + if (llvm::convertUTF8Sequence(&pos, begin + text.size(), &code_point, + llvm::strictConversion) != llvm::conversionOK) { + text = text.drop_front(1); + return Utf8Replacement; + } + text = text.drop_front(pos - begin); + return code_point; +} + +auto Metrics::TakeCodePoint(llvm::StringRef& text) const -> char32_t { + CARBON_CHECK(!text.empty(), "No code point to take."); + if (charset_ == Charset::Ascii) { + auto byte = static_cast(text.front()); + text = text.drop_front(); + return byte; + } + return TakeUtf8CodePoint(text); +} + +auto Metrics::CodePointWidth(char32_t code_point) const -> int { + if (charset_ == Charset::Ascii) { + return 1; + } + int width = Utf8CodePointWidth(code_point); + // A code point with no rendering is drawn as the replacement character, which + // takes one column. + return width < 0 ? 1 : width; +} + +auto Metrics::RenderedCodePoint(char32_t code_point) const -> char32_t { + // Printable ASCII is most of what gets drawn, and settling it here keeps it + // out of the range tables the general answer searches. + if (IsPrintableAscii(code_point)) { + return code_point; + } + + // Fallback if we can't use unicode. + if (charset_ == Charset::Ascii) { + return U'?'; + } + + // Which code points have no rendering is what a negative width names as well, + // asked directly rather than through a width that has to encode one to + // answer. + return llvm::sys::unicode::isPrintable(static_cast(code_point)) + ? code_point + : Utf8Replacement; +} + +auto Metrics::Width(llvm::StringRef text) const -> int { + // Checked rather than debug-checked: text with one of these in it measures as + // though each took one column, which is not what drawing does, and measuring + // wrong is invisible in the output. The scan is one more linear pass over + // text that is walked linearly anyway. + CARBON_CHECK( + text.find_first_of("\t\n\r") == llvm::StringRef::npos, + "Width is only for text whose width is its code points', but got `{0}`.", + text); + if (charset_ == Charset::Ascii) { + return static_cast(text.size()); + } + + // Text that is valid UTF-8 throughout and printable throughout is the common + // case, and LLVM measures a whole run of it in one pass. It answers with a + // negative value rather than a width when the text holds anything it can't + // measure, which is what the walk below is for: each such code point still + // takes the one column the replacement character drawn for it will. + int width = llvm::sys::unicode::columnWidthUTF8(text); + if (width >= 0) { + return width; + } + + width = 0; + while (!text.empty()) { + width += CodePointWidth(TakeUtf8CodePoint(text)); + } + return width; +} + +auto Metrics::TakeColumns(llvm::StringRef& text, int columns) const + -> llvm::StringRef { + llvm::StringRef rest = text; + int taken = 0; + while (!rest.empty()) { + llvm::StringRef next = rest; + int width = CodePointWidth(TakeCodePoint(next)); + if (taken + width > columns) { + break; + } + taken += width; + rest = next; + } + llvm::StringRef prefix = text.drop_back(rest.size()); + text = rest; + return prefix; +} + +} // namespace Carbon::Terminal diff --git a/common/terminal/metrics.h b/common/terminal/metrics.h new file mode 100644 index 000000000000..ae9250d9a55f --- /dev/null +++ b/common/terminal/metrics.h @@ -0,0 +1,114 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#ifndef CARBON_COMMON_TERMINAL_METRICS_H_ +#define CARBON_COMMON_TERMINAL_METRICS_H_ + +#include +#include + +#include "common/terminal/capabilities.h" +#include "llvm/ADT/StringRef.h" + +namespace Carbon::Terminal { + +// The most bytes one code point encodes to in UTF-8, and storage for one. +inline constexpr size_t MaxUtf8Bytes = 4; +using Utf8Storage = std::array; + +// Encodes `code_point` as UTF-8 into `storage`, returning the bytes written. +// +// Code points with no valid encoding, including surrogates and anything past +// U+10FFFF, become the replacement character. +auto EncodeUtf8(char32_t code_point, Utf8Storage& storage) -> llvm::StringRef; + +// How many columns a terminal spends on text, given the charset it decodes +// with. +// +// Which bytes make up a column depends on the charset, so every question about +// the size of text is a question about the charset as well, and this is what +// answers both at once. `Buffer` holds one and lays its cells out with it; +// anything deciding where to put something asks one directly rather than +// keeping its own idea of how wide a string is. +// +// Nothing here converts between a byte offset and a column: which byte a column +// lands on depends on the encoding, and which column a byte lands in depends on +// the width of everything before it. `TakeColumns` hands back the text it cut +// rather than an offset into it, so a caller never holds one count where the +// other belongs. +// +// TODO: Every width here is a sum over code points taken in logical order, +// which is only the width on screen for left-to-right text. Bidirectional text +// reorders, so a run's width still adds up but `TakeColumns` has no meaning: +// the prefix occupying the first N columns need not be a prefix of the string. +// Settle this together with the question `Buffer`'s own TODO describes, since +// both turn on what a client hands over. +class Metrics { + public: + explicit constexpr Metrics(Charset charset) : charset_(charset) {} + + constexpr auto charset() const -> Charset { return charset_; } + + // Removes the next code point from `text`, which must not be empty, and + // returns it: one byte under `Charset::Ascii`, and one decoded code point + // under `Charset::Utf8`. + // + // A byte that doesn't start a valid sequence yields the replacement + // character and is consumed on its own, so decoding resynchronizes at the + // next byte rather than discarding the rest of the text. + auto TakeCodePoint(llvm::StringRef& text) const -> char32_t; + + // Returns the columns `code_point` occupies once drawn, which is what drawing + // it advances by. + // + // Under `Charset::Ascii` every code point is one column. Under + // `Charset::Utf8` a combining mark is zero, since it renders into the column + // before it, and anything with no printable rendering is one, since it is + // drawn as a replacement character. + // + // A combining mark is the only thing zero is ever the answer for, which is + // what lets `Buffer` read a zero as one: a code point to fold into the cell + // before it rather than give a cell of its own. A code point that takes no + // column without combining with anything, such as U+200C ZERO WIDTH + // NON-JOINER, has no printable rendering here and takes the column its + // replacement character does. Terminals disagree about those -- some give + // them a column and some don't -- so drawing one as itself would leave the + // columns counted here and the columns painted disagreeing from there on. + auto CodePointWidth(char32_t code_point) const -> int; + + // Returns the code point to render for `code_point`, which is a replacement + // character where it has no dependable rendering of its own. + // + // Under `Charset::Ascii` that is anything outside printable ASCII, because a + // terminal decoding some single-byte encoding will draw such a byte as + // something and there is no way to know what. Under `Charset::Utf8` it is + // anything with no printable rendering at all, which includes the surrogates + // and so covers everything UTF-8 has no encoding for as well. + auto RenderedCodePoint(char32_t code_point) const -> char32_t; + + // Returns the columns `text` occupies once drawn. + // + // `text` must hold no character that drawing gives a width other than its + // code points', so no tab, newline, or carriage return. Those are positional + // -- what a tab advances by depends on where the text began -- which makes + // them questions about a drawing rather than about the text, and `Buffer` + // answers those. + auto Width(llvm::StringRef text) const -> int; + + // Removes and returns the longest prefix of `text` that occupies at most + // `columns` columns. + // + // A code point that would straddle the end stops the walk before it, so a cut + // never lands inside one and the prefix is never wider than asked for -- it + // can be one column narrower, where a double-width character sits on the + // boundary. + auto TakeColumns(llvm::StringRef& text, int columns) const -> llvm::StringRef; + + private: + Charset charset_; +}; + +} // namespace Carbon::Terminal + +#endif // CARBON_COMMON_TERMINAL_METRICS_H_ diff --git a/common/terminal/metrics_test.cpp b/common/terminal/metrics_test.cpp new file mode 100644 index 000000000000..e8d3016d6958 --- /dev/null +++ b/common/terminal/metrics_test.cpp @@ -0,0 +1,161 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#include "common/terminal/metrics.h" + +#include +#include + +#include + +#include "llvm/ADT/StringRef.h" + +namespace Carbon::Terminal { +namespace { + +// "e" followed by U+0301 COMBINING ACUTE ACCENT, which is one column because +// the mark renders into the column the "e" is in. +static constexpr llvm::StringLiteral AcuteE = "é"; + +TEST(MetricsTest, Width) { + Metrics utf8(Charset::Utf8); + EXPECT_EQ(utf8.Width(""), 0); + EXPECT_EQ(utf8.Width("hello"), 5); + EXPECT_EQ(utf8.Width("中中"), 4); + EXPECT_EQ(utf8.Width("a中b"), 4); + EXPECT_EQ(utf8.Width(AcuteE), 1); + + // Every byte is a column when the terminal isn't decoding UTF-8. + Metrics ascii(Charset::Ascii); + EXPECT_EQ(ascii.Width("hello"), 5); + EXPECT_EQ(ascii.Width("中中"), 6); + EXPECT_EQ(ascii.Width(AcuteE), 3); +} + +TEST(MetricsTest, CodePointWidth) { + Metrics utf8(Charset::Utf8); + EXPECT_EQ(utf8.CodePointWidth(U'a'), 1); + EXPECT_EQ(utf8.CodePointWidth(U'中'), 2); + // A combining mark renders into the column before it. + EXPECT_EQ(utf8.CodePointWidth(U'́'), 0); + // Something with no rendering is drawn as a replacement, which is a column. + EXPECT_EQ(utf8.CodePointWidth(U''), 1); + + Metrics ascii(Charset::Ascii); + EXPECT_EQ(ascii.CodePointWidth(U'a'), 1); + EXPECT_EQ(ascii.CodePointWidth(U'中'), 1); + EXPECT_EQ(ascii.CodePointWidth(U'́'), 1); +} + +TEST(MetricsTest, OnlyCombiningMarksAreZeroColumns) { + // Drawing reads a width of zero as "renders into the cell before this one", + // so a code point that takes no column without combining with anything has to + // measure as something else. Terminals disagree about these -- Terminal.app + // gives U+200C a column and VS Code's terminal gives it none -- so each is + // drawn as a replacement character, which takes exactly one. + Metrics utf8(Charset::Utf8); + for (char32_t code_point : {U'\u200b', U'\u200c', U'\u200d', U'\ufeff'}) { + EXPECT_EQ(utf8.CodePointWidth(code_point), 1) + << static_cast(code_point); + EXPECT_EQ(utf8.RenderedCodePoint(code_point), U'�') + << static_cast(code_point); + } +} + +TEST(MetricsTest, RenderedCodePoint) { + Metrics utf8(Charset::Utf8); + EXPECT_EQ(utf8.RenderedCodePoint(U'a'), U'a'); + EXPECT_EQ(utf8.RenderedCodePoint(U'中'), U'中'); + EXPECT_EQ(utf8.RenderedCodePoint(U''), U'�'); + + // Code points that UTF-8 has no encoding for have no rendering either. + EXPECT_EQ(utf8.RenderedCodePoint(static_cast(0xd800)), U'�'); + EXPECT_EQ(utf8.RenderedCodePoint(static_cast(0x110000)), U'�'); + + // An ASCII terminal is only given what it draws as itself, because there is + // no telling what it would draw for anything else. + Metrics ascii(Charset::Ascii); + EXPECT_EQ(ascii.RenderedCodePoint(U'a'), U'a'); + EXPECT_EQ(ascii.RenderedCodePoint(U'中'), U'?'); + EXPECT_EQ(ascii.RenderedCodePoint(U''), U'?'); +} + +TEST(MetricsTest, TakeColumns) { + Metrics utf8(Charset::Utf8); + llvm::StringRef text = "abcde"; + EXPECT_EQ(utf8.TakeColumns(text, 3), "abc"); + EXPECT_EQ(text, "de"); + + // Taking more than there is takes all of it. + EXPECT_EQ(utf8.TakeColumns(text, 10), "de"); + EXPECT_EQ(text, ""); + + // Taking nothing takes nothing, and a negative width is no different. + text = "abcde"; + EXPECT_EQ(utf8.TakeColumns(text, 0), ""); + EXPECT_EQ(utf8.TakeColumns(text, -1), ""); + EXPECT_EQ(text, "abcde"); +} + +TEST(MetricsTest, TakeColumnsKeepsWideCharactersWhole) { + Metrics utf8(Charset::Utf8); + // A character that would straddle the end stops the walk before it, so the + // prefix comes back a column short rather than half a character wide. + llvm::StringRef text = "中中中"; + llvm::StringRef prefix = utf8.TakeColumns(text, 3); + EXPECT_EQ(prefix, "中"); + EXPECT_EQ(utf8.Width(prefix), 2); + EXPECT_EQ(text, "中中"); + + // A request landing on a character boundary takes the whole prefix. + text = "中中中"; + EXPECT_EQ(utf8.TakeColumns(text, 4), "中中"); + EXPECT_EQ(text, "中"); +} + +TEST(MetricsTest, TakeColumnsUnderAscii) { + // Every byte is a column, so a multi-byte character is cut like any other + // run of bytes. + Metrics ascii(Charset::Ascii); + llvm::StringRef text = "中"; + EXPECT_EQ(ascii.TakeColumns(text, 2).size(), 2U); + EXPECT_EQ(text.size(), 1U); +} + +TEST(MetricsTest, TakeCodePointResynchronizesOnInvalidUtf8) { + Metrics utf8(Charset::Utf8); + // A byte that starts no valid sequence is consumed on its own, so the text + // after it is still decoded rather than being discarded. + llvm::StringRef text = + "\xff" + "a"; + EXPECT_EQ(utf8.TakeCodePoint(text), U'�'); + EXPECT_EQ(utf8.TakeCodePoint(text), U'a'); + EXPECT_TRUE(text.empty()); +} + +TEST(MetricsTest, EncodeUtf8) { + Utf8Storage storage; + EXPECT_EQ(EncodeUtf8(U'a', storage), "a"); + EXPECT_EQ(EncodeUtf8(U'é', storage), "é"); + EXPECT_EQ(EncodeUtf8(U'中', storage), "中"); + EXPECT_EQ(EncodeUtf8(U'\U0001f525', storage), "\U0001f525"); + + // A code point with no encoding of its own becomes the replacement. + EXPECT_EQ(EncodeUtf8(static_cast(0xd800), storage), "�"); + EXPECT_EQ(EncodeUtf8(static_cast(0x110000), storage), "�"); +} + +TEST(MetricsDeathTest, WidthRejectsPositionalCharacters) { + // A tab's width is a fact about a drawing rather than about the text, so + // answering for one here would be answering a question this can't see the + // inputs to. + Metrics metrics(Charset::Utf8); + EXPECT_DEATH((void)metrics.Width("a\tb"), "Width is only for text whose"); + EXPECT_DEATH((void)metrics.Width("a\nb"), "Width is only for text whose"); + EXPECT_DEATH((void)metrics.Width("a\rb"), "Width is only for text whose"); +} + +} // namespace +} // namespace Carbon::Terminal diff --git a/common/terminal/output_buffer_ref.h b/common/terminal/output_buffer_ref.h new file mode 100644 index 000000000000..f12c9ff0ed7d --- /dev/null +++ b/common/terminal/output_buffer_ref.h @@ -0,0 +1,163 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#ifndef CARBON_COMMON_TERMINAL_OUTPUT_BUFFER_REF_H_ +#define CARBON_COMMON_TERMINAL_OUTPUT_BUFFER_REF_H_ + +#include +#include +#include +#include + +#include "llvm/ADT/SmallVector.h" +#include "llvm/ADT/StringRef.h" + +namespace Carbon::Terminal { + +// A reference to the buffer a terminal rendering is assembled into. +// +// This owns nothing. It refers to a buffer the caller holds, which must outlive +// it, and converts implicitly from one so that rendering goes into storage the +// caller already has. +// +// Rendering assembles bytes here rather than streaming them: a stream call per +// literal and per number costs measurably more than handing over a finished +// sequence, and code that wants a stream prints the buffer once it is complete. +// +// Appending is shaped around what terminal output is made of, which is a great +// many short escape sequences, each a handful of literal bytes around a number +// that never exceeds 255. Taking a whole sequence at a time grows the buffer +// once per sequence rather than once per byte, and that difference is much of +// what rendering costs. +class OutputBufferRef { + public: + // Implicit, so that call sites pass the buffer they already hold rather than + // naming this type. + // + // NOLINTNEXTLINE(google-explicit-constructor) + OutputBufferRef(llvm::SmallVectorImpl& bytes) : bytes_(&bytes) {} + + // Appends `pieces`, each of which is either text, appended as it is, or a + // `uint8_t`, appended in decimal. + // + // No other type is accepted, so the two can never be taken for each other, + // and nothing needs one: the literal bytes of an escape sequence are always + // text, and every number one carries is a channel value, a palette index, or + // an SGR code, none of which exceed 255. + // + // No piece may point into the buffer, which appending can reallocate. + template + auto Append(const PieceT&... pieces) -> void { + if constexpr (sizeof...(pieces) == 1) { + // A lone piece has nothing to assemble, and the buffer's own append is + // already the single growth and single copy this is after. + (AppendPiece(pieces), ...); + } else { + // Growing to the bound before writing keeps how far the buffer grows + // independent of the piece values, so computing one can't hold that up. + // Only the trim afterwards depends on how many digits a number took. + size_t begin = bytes_->size(); + bytes_->resize_for_overwrite(begin + (AppendedSize(pieces) + ... + 0)); + char* data = bytes_->data(); + char* cursor = data + begin; + ((cursor = WritePiece(cursor, pieces)), ...); + bytes_->truncate(cursor - data); + } + } + + private: + // The room a number needs: its three digits, plus one more because it is + // written as a single four-byte store whose last byte is discarded. + static constexpr size_t NumberBytes = 4; + + // The decimal text of a number, and how many digits it took. The digits are + // at the front and the length in the byte after them, so a whole entry is one + // store and the length says how far of it to keep. + struct NumberText { + std::array digits; + uint8_t length; + }; + static_assert(sizeof(NumberText) == NumberBytes, + "A number is written by storing a whole entry at once."); + + // The text of every value a number piece can hold. A kilobyte of table, in + // exchange for a lookup where computing the digits would branch on the value + // three times. + static constexpr std::array NumberTexts = [] { + std::array texts = {}; + for (int value = 0; value < 256; ++value) { + NumberText& text = texts[value]; + text.length = 1 + (value >= 10) + (value >= 100); + int rest = value; + for (int digit = text.length; digit > 0; --digit) { + text.digits[digit - 1] = static_cast('0' + rest % 10); + rest /= 10; + } + } + return texts; + }(); + + // Returns the most bytes a piece can append. A number contributes the bound + // above rather than the digits it will take, so the bound for a sequence + // doesn't depend on any of the values in it. + template + static constexpr auto AppendedSize(const char (& /*piece*/)[N]) -> size_t { + return N - 1; + } + static constexpr auto AppendedSize(llvm::StringRef piece) -> size_t { + return piece.size(); + } + template T> + static constexpr auto AppendedSize(T /*piece*/) -> size_t { + return NumberBytes; + } + + // Writes a piece at `out` and returns the position past it. There must be + // `AppendedSize(piece)` bytes of room, as nothing here checks. + template + static auto WritePiece(char* out, const char (&piece)[N]) -> char* { + std::memcpy(out, piece, N - 1); + return out + N - 1; + } + static auto WritePiece(char* out, llvm::StringRef piece) -> char* { + // An empty `StringRef` may hold a null pointer, which `memcpy` doesn't + // accept even for an empty copy. + if (!piece.empty()) { + std::memcpy(out, piece.data(), piece.size()); + } + return out + piece.size(); + } + template T> + static auto WritePiece(char* out, T piece) -> char* { + // One load and one store, with no branch on the value. Escape sequences + // carry color channels and palette indices, which are spread across the + // whole range, so a branch per digit is one the processor can't predict, + // and there are four numbers in a truecolor escape. The store always covers + // four bytes, which is why a number reserves that many, and the cursor + // advances only over the digits that count. + const NumberText& text = NumberTexts[piece]; + std::memcpy(out, &text, sizeof(text)); + return out + text.length; + } + + // Appends a piece on its own, growing the buffer to fit it. + template + auto AppendPiece(const char (&piece)[N]) -> void { + bytes_->append(piece, piece + N - 1); + } + auto AppendPiece(llvm::StringRef piece) -> void { + bytes_->append(piece.begin(), piece.end()); + } + template T> + auto AppendPiece(T piece) -> void { + std::array digits; + bytes_->append(digits.data(), WritePiece(digits.data(), piece)); + } + + llvm::SmallVectorImpl* bytes_; +}; + +} // namespace Carbon::Terminal + +#endif // CARBON_COMMON_TERMINAL_OUTPUT_BUFFER_REF_H_ diff --git a/common/terminal/output_buffer_ref_test.cpp b/common/terminal/output_buffer_ref_test.cpp new file mode 100644 index 000000000000..17c916e49cf9 --- /dev/null +++ b/common/terminal/output_buffer_ref_test.cpp @@ -0,0 +1,99 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#include "common/terminal/output_buffer_ref.h" + +#include +#include + +#include "llvm/ADT/SmallString.h" + +namespace Carbon::Terminal { +namespace { + +using ::testing::Eq; + +// A single piece and several pieces are appended by different code, so both +// appear throughout these tests rather than in one case of their own. + +TEST(OutputBufferRefTest, Text) { + llvm::SmallString<16> bytes; + OutputBufferRef out = bytes; + out.Append("one"); + out.Append(llvm::StringRef(" two")); + out.Append(std::string(" three")); + out.Append(" four", llvm::StringRef(" five")); + EXPECT_THAT(bytes, Eq("one two three four five")); +} + +TEST(OutputBufferRefTest, EmptyPieces) { + llvm::SmallString<16> bytes; + OutputBufferRef out = bytes; + out.Append(); + out.Append(""); + out.Append("", llvm::StringRef(), "kept", llvm::StringRef("")); + EXPECT_THAT(bytes, Eq("kept")); +} + +TEST(OutputBufferRefTest, NumbersUseEveryDigitCount) { + llvm::SmallString<16> bytes; + OutputBufferRef out = bytes; + for (uint8_t value : {0, 9, 10, 99, 100, 255}) { + out.Append(value); + out.Append(" ", value, " "); + } + EXPECT_THAT(bytes, Eq("0 0 9 9 10 10 99 99 100 100 255 255 ")); +} + +// A number always writes fewer bytes than it reserves, so pieces after one in +// the same call are what catch a misplaced write. +TEST(OutputBufferRefTest, NumbersFollowedByMorePieces) { + llvm::SmallString<32> bytes; + OutputBufferRef out = bytes; + out.Append("\x1b[", static_cast(38), ";2;", static_cast(1), + ";", static_cast(22), ";", static_cast(255), + "m"); + EXPECT_THAT(bytes, Eq("\x1b[38;2;1;22;255m")); +} + +TEST(OutputBufferRefTest, AppendsAfterExistingContents) { + llvm::SmallString<16> bytes = llvm::StringRef("before:"); + OutputBufferRef out = bytes; + out.Append(static_cast(7)); + EXPECT_THAT(bytes, Eq("before:7")); +} + +// Appending has to work the same however the buffer is laid out, and a number +// leaves the buffer grown further than it wrote, so reallocation is where a +// size mistake would show up. +TEST(OutputBufferRefTest, AppendsPastInlineCapacity) { + llvm::SmallString<8> bytes; + OutputBufferRef out = bytes; + std::string expected; + for (int i = 0; i < 100; ++i) { + out.Append("x", static_cast(i)); + expected += "x" + std::to_string(i); + } + EXPECT_THAT(bytes, Eq(expected)); + EXPECT_THAT(bytes.size(), Eq(expected.size())); +} + +// References to one buffer all append to it, and none of them own it, so the +// buffer keeps everything written through any of them. +TEST(OutputBufferRefTest, ReferencesShareTheirBuffer) { + llvm::SmallString<16> bytes; + OutputBufferRef first = bytes; + first.Append("a"); + { + OutputBufferRef second = bytes; + second.Append("b"); + } + OutputBufferRef copy = first; + copy.Append("c"); + first.Append("d"); + EXPECT_THAT(bytes, Eq("abcd")); +} + +} // namespace +} // namespace Carbon::Terminal diff --git a/common/terminal/pressure_test.cpp b/common/terminal/pressure_test.cpp new file mode 100644 index 000000000000..10360844e950 --- /dev/null +++ b/common/terminal/pressure_test.cpp @@ -0,0 +1,197 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +// Adversarial inputs across the terminal library's surface. +// +// The library draws whatever a source file contains, for a terminal whose width +// came from an environment variable. Both are things a user controls, and +// neither may crash, read out of bounds, or quietly produce a position that is +// wrong. The tests here feed each entry point the inputs most likely to do one +// of those, and assert that what comes back is coherent rather than asserting +// any particular rendering. +// +// What is deliberately not here: anything a caller is checked for getting +// wrong, which is text past `MaxTextBytes` and coordinates outside the width +// laid out for. Those are death tests in `buffer_test`. What remains is the +// text and the width, neither of which a caller can validate ahead of drawing. + +#include + +#include +#include +#include +#include + +#include "common/terminal/buffer.h" +#include "common/terminal/capabilities.h" +#include "common/terminal/metrics.h" +#include "common/terminal/style.h" +#include "llvm/ADT/SmallString.h" +#include "llvm/ADT/StringRef.h" + +namespace Carbon::Terminal { +namespace { + +// Byte sequences that stress column accounting. +auto HostileText() -> std::vector { + return { + "", + "\x01\x02\x7f", // C0 controls and delete + "\xff\xfe\xfd", // never valid UTF-8 + "\xe4\xb8", // truncated multi-byte sequence + "\xe4\xb8\x96\xe4\xb8", // valid then truncated + "\xcc\x81", // combining mark with no base + "e\xcc\x81\xcc\x82\xcc\x83", // a base with several marks + "\xf0\x9f\x94\xa5", // outside the basic plane + "\xed\xa0\x80", // a surrogate, which UTF-8 forbids + "\xc0\x80", // overlong encoding of NUL + "中中中", // double-width throughout + "a\tb\nc\r\nd", // every positional character + "\t\t\t\t\t\t\t\t", // nothing but tabs + "\n\n\n", // nothing but newlines + std::string(4096, ' '), // a long run of blanks + std::string(1024, '\t'), // a long run of tabs + }; +} + +// Renders `buffer` and checks the bytes are coherent. Everything here draws +// with the default style, so what a style leaves behind is `buffer_test`'s to +// cover; this is about the text surviving at all. +auto RenderAndCheck(const Buffer& buffer, ColorMode mode) -> std::string { + llvm::SmallString<256> out; + buffer.Render(out, mode); + std::string rendered(out); + if (rendered.empty()) { + return rendered; + } + EXPECT_EQ(rendered.back(), '\n'); + return rendered; +} + +TEST(PressureTest, DrawTextSurvivesHostileBytes) { + for (Charset charset : {Charset::Ascii, Charset::Utf8}) { + for (const std::string& text : HostileText()) { + for (int x : {0, 1, 7}) { + Buffer buffer(8, charset); + Buffer::DrawEnd end = buffer.DrawText(x, 0, text, Style()); + + // The end is where drawing would carry on, which is not always a cell + // that exists: text ending in a newline leaves it on a row nothing was + // drawn on. Nor does the width follow from it, since the grid grows by + // halves and overshoots. What must hold is that the end names a + // non-negative cell, and that no row was created past where drawing + // ended. + EXPECT_GE(end.y, 0) << text; + EXPECT_LE(buffer.height(), end.y + 1) << text; + EXPECT_GE(end.x, 0) << text; + + // Measuring answers what drawing did. + EXPECT_EQ(buffer.MeasureText(x, 0, text), end) << text; + + RenderAndCheck(buffer, ColorMode::Ansi16); + } + } + } +} + +TEST(PressureTest, WrappedTextSurvivesHostileBytes) { + for (Charset charset : {Charset::Ascii, Charset::Utf8}) { + for (const std::string& text : HostileText()) { + // A width of one is the tightest anything can be asked to wrap into. + for (int width : {1, 2, 3, 80}) { + Buffer buffer(width, charset); + Buffer::DrawEnd end = + buffer.DrawWrappedText(0, 0, 0, width, text, Style()); + + EXPECT_GE(end.y, 0) << text; + EXPECT_LE(buffer.height(), end.y + 1) << text; + EXPECT_GE(end.x, 0) << text; + EXPECT_EQ(buffer.MeasureWrappedText(0, 0, 0, width, text), end) << text; + // The width wrapping this wouldn't overhang is a fact about the text + // rather than about the block it was drawn into. + EXPECT_GE(buffer.MeasureWrapWidth(text), 0) << text; + + RenderAndCheck(buffer, ColorMode::Truecolor); + } + } + } +} + +TEST(PressureTest, MetricsSurviveHostileBytes) { + for (Charset charset : {Charset::Ascii, Charset::Utf8}) { + Metrics metrics(charset); + for (const std::string& text : HostileText()) { + // `Width` requires text with no positional characters in it, which is + // checked, so only the rest is measured here. + if (llvm::StringRef(text).find_first_of("\t\n\r") == + llvm::StringRef::npos) { + int width = metrics.Width(text); + EXPECT_GE(width, 0) << text; + + // Cutting at any column gives back a prefix that is no wider than + // asked for and that leaves the rest of the string behind it. + for (int columns : {-1, 0, 1, 2, width, width + 1}) { + llvm::StringRef rest = text; + llvm::StringRef prefix = metrics.TakeColumns(rest, columns); + EXPECT_EQ(prefix.size() + rest.size(), text.size()) << text; + EXPECT_LE(metrics.Width(prefix), std::max(columns, 0)) << text; + } + } + + // Taking code points consumes the whole string however invalid it is, + // rather than stalling on a byte it can't decode. + llvm::StringRef rest = text; + size_t steps = 0; + while (!rest.empty()) { + metrics.TakeCodePoint(rest); + ++steps; + ASSERT_LE(steps, text.size()) << "TakeCodePoint failed to consume"; + } + } + } +} + +TEST(PressureTest, OverlappingDrawsLeaveNoHalfCharacters) { + // Double-width characters, lines, and text all writing over each other is + // where a stale continuation cell would show up as a rendering with half a + // character in it. + Buffer buffer(8, Charset::Utf8); + for (int pass = 0; pass < 3; ++pass) { + buffer.DrawText(0, 0, "中中中中", Style()); + buffer.DrawHorizontalLine(1, 0, 3, Style()); + buffer.DrawText(2, 0, "中", Style()); + buffer.DrawVerticalLine(3, 0, 2, Style()); + buffer.DrawCodePoint(4, 0, U'中', Style()); + buffer.DrawCodePoint(5, 0, U'x', Style()); + buffer.DrawText(0, 0, "ab", Style()); + } + + // `Render` encodes every cell it emits, so this checks the overdraws leave a + // grid that renders at all rather than that no half character survived. + std::string rendered = RenderAndCheck(buffer, ColorMode::NoColor); + Metrics metrics(Charset::Utf8); + llvm::StringRef rest = rendered; + while (!rest.empty()) { + EXPECT_NE(metrics.TakeCodePoint(rest), 0xfffd) << rendered; + } +} + +TEST(PressureTest, CapabilitiesWidthNeverBreaksTheBuffer) { + // A width claimed by the environment can be anything at all. + for (int columns : + {-1, 0, 1, 2, 80, Buffer::MaxColumns - 1, Buffer::MaxColumns, + Buffer::MaxColumns + 1, 1 << 20, std::numeric_limits::max()}) { + Capabilities capabilities = {.charset = Charset::Utf8, .columns = columns}; + Buffer buffer(capabilities); + // Whatever was claimed, what comes out is a width that can be laid out for + // and drawn into. + EXPECT_GE(buffer.columns(), 1) << columns; + EXPECT_LE(buffer.columns(), Buffer::MaxColumns) << columns; + buffer.DrawText(0, 0, "中x", Style()); + RenderAndCheck(buffer, ColorMode::Ansi256); + } +} + +} // namespace +} // namespace Carbon::Terminal diff --git a/common/terminal/style.cpp b/common/terminal/style.cpp new file mode 100644 index 000000000000..04521acacf33 --- /dev/null +++ b/common/terminal/style.cpp @@ -0,0 +1,173 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#include "common/terminal/style.h" + +#include + +#include "common/check.h" +#include "llvm/ADT/SmallString.h" +#include "llvm/ADT/StringExtras.h" + +namespace Carbon::Terminal { + +// Returns the SGR parameter selecting `shape`, or an empty string for `None`. +// +// The shaped underlines are colon-separated subparameters of the plain +// underline code. Terminals that predate them are also the ones limited to the +// 16 ANSI colors, and they mishandle the subparameters rather than ignoring +// them, so in that mode every shape degrades to a plain underline. +static auto UnderlineSgrParam(UnderlineShape shape, ColorMode mode) + -> llvm::StringRef { + if (mode == ColorMode::Ansi16 && shape != UnderlineShape::None) { + return "4"; + } + switch (shape) { + case UnderlineShape::None: + return ""; + case UnderlineShape::Single: + return "4"; + case UnderlineShape::Double: + return "4:2"; + case UnderlineShape::Curly: + return "4:3"; + case UnderlineShape::Dotted: + return "4:4"; + case UnderlineShape::Dashed: + return "4:5"; + } +} + +auto Style::NeedsResetFrom(const Style& from) const -> bool { + auto drops = [](bool from_set, bool to_set) { return from_set && !to_set; }; + return drops(from.bold_, bold_) || drops(from.dim_, dim_) || + drops(from.italic_, italic_) || drops(from.reverse_, reverse_) || + drops(from.strikethrough_, strikethrough_) || + drops(from.underline(), underline()) || + drops(from.foreground_.is_set(), foreground_.is_set()) || + drops(from.background_.is_set(), background_.is_set()) || + drops(from.underline_color_.is_set(), underline_color_.is_set()); +} + +auto Style::AppendDiff(OutputBufferRef out, ColorMode mode, + const Style& from) const -> void { + CARBON_CHECK(!NeedsResetFrom(from), + "Cannot reach this style from `from` without a reset."); + + // Attributes combine into a single SGR sequence, in ascending code order. + // Every write goes through `add` so the bound is checked in one place. + std::array params; + int param_count = 0; + auto add = [&](llvm::StringRef param) { + CARBON_CHECK(param_count < static_cast(params.size()), + "More SGR parameters than the {0} there is room for.", + params.size()); + params[param_count++] = param; + }; + auto add_if = [&](bool from_set, bool to_set, llvm::StringRef param) { + if (to_set && !from_set) { + add(param); + } + }; + add_if(from.bold_, bold_, "1"); + add_if(from.dim_, dim_, "2"); + add_if(from.italic_, italic_, "3"); + llvm::StringRef underline_param = UnderlineSgrParam(underline_shape_, mode); + if (underline_param != UnderlineSgrParam(from.underline_shape_, mode)) { + // Turning an underline off needs a reset, which is the caller's to do, so + // reaching here with nothing to select would emit an empty parameter. + CARBON_CHECK(!underline_param.empty(), + "Removing an underline cannot be done with a diff."); + add(underline_param); + } + add_if(from.reverse_, reverse_, "7"); + add_if(from.strikethrough_, strikethrough_, "9"); + + if (param_count > 0) { + out.Append("\x1b[", params[0]); + for (int i = 1; i < param_count; ++i) { + out.Append(";", params[i]); + } + out.Append("m"); + } + + if (foreground_.is_set() && foreground_ != from.foreground_) { + foreground_.AppendEscape(out, mode, ColorTarget::Foreground); + } + if (background_.is_set() && background_ != from.background_) { + background_.AppendEscape(out, mode, ColorTarget::Background); + } + if (underline_color_.is_set() && underline_color_ != from.underline_color_) { + underline_color_.AppendEscape(out, mode, ColorTarget::Underline); + } +} + +auto Style::AppendColorTransitionTo(OutputBufferRef out, const Style& target, + ColorMode mode) const -> void { + CARBON_CHECK(mode != ColorMode::NoColor, + "Color transitions are only reached when color is in use."); + if (*this == target) { + return; + } + + if (target.NeedsResetFrom(*this)) { + out.Append(ResetEscape); + target.AppendDiff(out, mode, Style()); + return; + } + target.AppendDiff(out, mode, *this); +} + +// Returns the name of `shape`, for printing a style. +static auto UnderlineShapeName(UnderlineShape shape) -> llvm::StringRef { + switch (shape) { + case UnderlineShape::None: + return "None"; + case UnderlineShape::Single: + return "Single"; + case UnderlineShape::Double: + return "Double"; + case UnderlineShape::Curly: + return "Curly"; + case UnderlineShape::Dotted: + return "Dotted"; + case UnderlineShape::Dashed: + return "Dashed"; + } +} + +auto Style::Print(llvm::raw_ostream& out) const -> void { + out << "Style("; + llvm::ListSeparator sep; + if (bold_) { + out << sep << "bold"; + } + if (dim_) { + out << sep << "dim"; + } + if (italic_) { + out << sep << "italic"; + } + if (reverse_) { + out << sep << "reverse"; + } + if (strikethrough_) { + out << sep << "strikethrough"; + } + if (underline()) { + out << sep << "underline=" << UnderlineShapeName(underline_shape_); + } + if (foreground_.is_set()) { + out << sep << "foreground=" << foreground_; + } + if (background_.is_set()) { + out << sep << "background=" << background_; + } + if (underline_color_.is_set()) { + out << sep << "underline_color=" << underline_color_; + } + out << ")"; +} + +} // namespace Carbon::Terminal diff --git a/common/terminal/style.h b/common/terminal/style.h new file mode 100644 index 000000000000..023580e5efaf --- /dev/null +++ b/common/terminal/style.h @@ -0,0 +1,249 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#ifndef CARBON_COMMON_TERMINAL_STYLE_H_ +#define CARBON_COMMON_TERMINAL_STYLE_H_ + +#include +#include +#include + +#include "common/ostream.h" +#include "common/terminal/color.h" +#include "common/terminal/output_buffer_ref.h" +#include "llvm/ADT/SmallString.h" +#include "llvm/ADT/StringRef.h" + +namespace Carbon::Terminal { + +// The shapes an underline can take. +// +// Only `Single` is universally understood. The rest are selected with +// colon-separated subparameters of the Select Graphic Rendition (SGR) +// underline code, which terminals limited to 16 colors mishandle, so in that +// mode they degrade to `Single` rather than disappearing. +enum class UnderlineShape : int8_t { + None, + Single, + Double, + Curly, + Dotted, + Dashed, +}; + +// A set of colors and text attributes to render with. +// +// Styles are values, and are composed by chaining, which keeps a named style +// readable at its definition: +// +// ```cpp +// const Style Error = Style().Bold().Foreground(AnsiColor::BrightRed); +// const Style ErrorSquiggle = Error.Underline(UnderlineShape::Curly); +// ``` +// +// A style authored for a rich terminal stays meaningful on a poor one. Colors +// the active `ColorMode` can't express are downsampled, an underline shape it +// can't express becomes a plain underline, and an underline color it can't +// express is left to the terminal. `NoColor` drops everything. +class Style : public Printable