diff --git a/utils/textmate/README.md b/utils/textmate/README.md index af979d7ce855..a06528693526 100644 --- a/utils/textmate/README.md +++ b/utils/textmate/README.md @@ -33,3 +33,17 @@ If you are using Atom, you can convert the bundle to an Atom-compatible one. See For other editors that support TextMate bundles you can consult your editors documentation to see how to use the bundle. + +## Samples + +`Samples/` holds Carbon sources that exercise the grammar, each with an SVG +rendering of how this bundle highlights it. Some deliberately contain invalid +code, to show that highlighting stays sensible while something is being typed. + +The renderings are generated, not screenshotted, so they always reflect the +grammar in this repository. Regenerate them in the same commit as any change to +the grammar, so a reviewer can see what the change does to real code: + +```shell +utils/textmate/render_sample.py utils/textmate/Samples/*.carbon +``` diff --git a/utils/textmate/Samples/choices.jpg b/utils/textmate/Samples/choices.jpg deleted file mode 100644 index 5244a6734e45..000000000000 Binary files a/utils/textmate/Samples/choices.jpg and /dev/null differ diff --git a/utils/textmate/Samples/choices.svg b/utils/textmate/Samples/choices.svg new file mode 100644 index 000000000000..3ef8c06a9fa3 --- /dev/null +++ b/utils/textmate/Samples/choices.svg @@ -0,0 +1,13 @@ + + + +1 // Part of the Carbon Language project, under the Apache License v2.0 with LLVM +2 // Exceptions. See /LICENSE for license information. +3 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +4 +5 choice SomeChoice(T: type) { +6 Val1, +7 Val2(a: i32, b: array(T, 5), t: T), +8 Val3 +9 } + diff --git a/utils/textmate/Samples/comments.carbon b/utils/textmate/Samples/comments.carbon index 2598264bee85..e089e9cc197b 100644 --- a/utils/textmate/Samples/comments.carbon +++ b/utils/textmate/Samples/comments.carbon @@ -7,8 +7,10 @@ // class C; fn F[T: A](a: T) -> i32 { return 86; } //@dump-sem-ir-begin //@dump-sem-ir-end +//@include-in-dumps //@dump-sem-ir- //no whitespace after double slash /* invalid comment pattern. */ /// invalid comment pattern. class C; // Comment after a statement/expression/declaration. +var x: i32 = 1;// Trailing, with no space before the introducer. diff --git a/utils/textmate/Samples/comments.jpg b/utils/textmate/Samples/comments.jpg deleted file mode 100644 index c55e11b61c4e..000000000000 Binary files a/utils/textmate/Samples/comments.jpg and /dev/null differ diff --git a/utils/textmate/Samples/comments.svg b/utils/textmate/Samples/comments.svg new file mode 100644 index 000000000000..d389aba94246 --- /dev/null +++ b/utils/textmate/Samples/comments.svg @@ -0,0 +1,20 @@ + + + + 1 // Part of the Carbon Language project, under the Apache License v2.0 with LLVM + 2 // Exceptions. See /LICENSE for license information. + 3 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + 4 + 5 // This is a comment. + 6 // Comment with whitespace. + 7 // class C; fn F[T: A](a: T) -> i32 { return 86; } + 8 //@dump-sem-ir-begin + 9 //@dump-sem-ir-end +10 //@include-in-dumps +11 //@dump-sem-ir- +12 //no whitespace after double slash +13 /* invalid comment pattern. */ +14 /// invalid comment pattern. +15 class C; // Comment after a statement/expression/declaration. +16 var x: i32 = 1;// Trailing, with no space before the introducer. + diff --git a/utils/textmate/Samples/customs.carbon b/utils/textmate/Samples/customs.carbon index 546c79c7d868..0b62fb3abe52 100644 --- a/utils/textmate/Samples/customs.carbon +++ b/utils/textmate/Samples/customs.carbon @@ -2,7 +2,7 @@ // Exceptions. See /LICENSE for license information. // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception -package SomePackage api; +package SomePackage; Core SomeCore; import SomePackage; diff --git a/utils/textmate/Samples/customs.jpg b/utils/textmate/Samples/customs.jpg deleted file mode 100644 index 0cebb9a79f25..000000000000 Binary files a/utils/textmate/Samples/customs.jpg and /dev/null differ diff --git a/utils/textmate/Samples/customs.svg b/utils/textmate/Samples/customs.svg new file mode 100644 index 000000000000..24c7cc87dfd3 --- /dev/null +++ b/utils/textmate/Samples/customs.svg @@ -0,0 +1,15 @@ + + + + 1 // Part of the Carbon Language project, under the Apache License v2.0 with LLVM + 2 // Exceptions. See /LICENSE for license information. + 3 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + 4 + 5 package SomePackage; + 6 Core SomeCore; + 7 + 8 import SomePackage; + 9 +10 CallFunction(a, b, c.d); +11 unidentified; + diff --git a/utils/textmate/Samples/functions_variables.carbon b/utils/textmate/Samples/functions_variables.carbon index 907bc2031988..1054a90b015a 100644 --- a/utils/textmate/Samples/functions_variables.carbon +++ b/utils/textmate/Samples/functions_variables.carbon @@ -37,3 +37,16 @@ fn F(ref a: A, const ref b: B); fn F() -> T; fn F() -> T.U { return val; } fn F() -> i32 => return val; + +// --- lambdas and positional parameters + +var explicit: auto = fn (x: i32) -> i32 { return x + 1; }; +var expression: auto = fn (x: i32) => x + 1; +var positional: auto = fn { Print($0, $1, $42); }; + +// --- raw identifiers + +r#class +r#if +r#i32 +var r#var: i32 = r#class; diff --git a/utils/textmate/Samples/functions_variables.jpg b/utils/textmate/Samples/functions_variables.jpg deleted file mode 100644 index a242857b4817..000000000000 Binary files a/utils/textmate/Samples/functions_variables.jpg and /dev/null differ diff --git a/utils/textmate/Samples/functions_variables.svg b/utils/textmate/Samples/functions_variables.svg new file mode 100644 index 000000000000..4e3aa845622d --- /dev/null +++ b/utils/textmate/Samples/functions_variables.svg @@ -0,0 +1,56 @@ + + + + 1 // Part of the Carbon Language project, under the Apache License v2.0 with LLVM + 2 // Exceptions. See /LICENSE for license information. + 3 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + 4 + 5 fn F; + 6 fn F1(); + 7 fn Func(a: i32, b: A); + 8 fn Generic[T: type, U: I](a: T, b: U) -> U; + 9 fn Lambda(a: i32) -> i32 => return a + 5; +10 fn +11 +12 MultiLine[ +13 A: type, +14 B: C1, +15 C: C2 +16 ](a: A(B, C), +17 b: B, +18 c: i32); +19 fn Invalid#+?Name(); +20 x.(I.F)(); +21 +22 a: A; +23 a: A = (b as A); +24 a: i32; +25 a: array(A, 5); +26 generic a: type; +27 generic a: T = b; +28 a: A in some_list; +29 var a: T; +30 let b: T; +31 binding: T; +32 generic binding: T; +33 var +34 multiline: T; +35 +36 fn F(ref a: A, const ref b: B); +37 fn F() -> T; +38 fn F() -> T.U { return val; } +39 fn F() -> i32 => return val; +40 +41 // --- lambdas and positional parameters +42 +43 var explicit: auto = fn (x: i32) -> i32 { return x + 1; }; +44 var expression: auto = fn (x: i32) => x + 1; +45 var positional: auto = fn { Print($0, $1, $42); }; +46 +47 // --- raw identifiers +48 +49 r#class +50 r#if +51 r#i32 +52 var r#var: i32 = r#class; + diff --git a/utils/textmate/Samples/interop.carbon b/utils/textmate/Samples/interop.carbon new file mode 100644 index 000000000000..58fa046f050f --- /dev/null +++ b/utils/textmate/Samples/interop.carbon @@ -0,0 +1,29 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +// +// Inline C++ is scoped `meta.embedded.block.cpp`, so an editor that has a C++ +// grammar highlights them as C++; these renderings come from the Carbon grammar +// alone, so that content shows plain. + +package Sample library "interop"; + +import Cpp library ""; +import Cpp library ""; + +import Cpp inline '''c++ +int FromImport() { return 1; } +'''; + +import Cpp inline "int OneLiner() { return 2; }"; + +inline Cpp ''' +int FromBlock() { return 3; } +'''; + +fn Uses() { + var value: Cpp.int = 1 as Cpp.int; + var widened: Cpp.size_t = value as Cpp.size_t; + var back: i32 = value unsafe as i32; + Cpp.printf("%d\n", value); +} diff --git a/utils/textmate/Samples/interop.svg b/utils/textmate/Samples/interop.svg new file mode 100644 index 000000000000..84c6cec24c33 --- /dev/null +++ b/utils/textmate/Samples/interop.svg @@ -0,0 +1,33 @@ + + + + 1 // Part of the Carbon Language project, under the Apache License v2.0 with LLVM + 2 // Exceptions. See /LICENSE for license information. + 3 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + 4 // + 5 // Inline C++ is scoped `meta.embedded.block.cpp`, so an editor that has a C++ + 6 // grammar highlights them as C++; these renderings come from the Carbon grammar + 7 // alone, so that content shows plain. + 8 + 9 package Sample library "interop"; +10 +11 import Cpp library "<cstdio>"; +12 import Cpp library "<string_view>"; +13 +14 import Cpp inline '''c++ +15 int FromImport() { return 1; } +16 '''; +17 +18 import Cpp inline "int OneLiner() { return 2; }"; +19 +20 inline Cpp ''' +21 int FromBlock() { return 3; } +22 '''; +23 +24 fn Uses() { +25 var value: Cpp.int = 1 as Cpp.int; +26 var widened: Cpp.size_t = value as Cpp.size_t; +27 var back: i32 = value unsafe as i32; +28 Cpp.printf("%d\n", value); +29 } + diff --git a/utils/textmate/Samples/keywords.carbon b/utils/textmate/Samples/keywords.carbon index b4cf3e071f98..ec8d30452388 100644 --- a/utils/textmate/Samples/keywords.carbon +++ b/utils/textmate/Samples/keywords.carbon @@ -4,8 +4,9 @@ // --- operators ->>= <=> <<= &= == != >= >> <= << -= -> -- %= \ = += ++ /= *= & ^ = > < - % . | + / * -and as impls in like not or partial ref template where +>>= <=> <<= ->? &= ^= := :? == => != >= >> <= <> << <- -= -> -- %= |= += ++ /= +*= ~= & @ ^ : = ! > < - % . | + ? / * ~ \ +and as impls in like not or where // --- special-keywords @@ -14,7 +15,8 @@ default => // --- introducer-keywords -adapt alias choice class constraint fn import interface let library namespace var +adapt alias choice class constraint fn import inline interface let library +match_first namespace observe require var base export impl @@ -22,7 +24,8 @@ package // --- modifier-keywords -abstract const extend extern final private protected virtual +abstract const eval extend extern final generic musteval override partial +private protected ref runtime static template unsafe unused val virtual base class default export import @@ -32,13 +35,17 @@ impl package // --- misc-keywords -auto destructor forall friend observe override require +forall form friend .base // --- self-keywords self Self +// --- package-roots + +Core Cpp + // --- underscore ; diff --git a/utils/textmate/Samples/keywords.jpg b/utils/textmate/Samples/keywords.jpg deleted file mode 100644 index 2678cefd595f..000000000000 Binary files a/utils/textmate/Samples/keywords.jpg and /dev/null differ diff --git a/utils/textmate/Samples/keywords.svg b/utils/textmate/Samples/keywords.svg new file mode 100644 index 000000000000..0e3a591c126a --- /dev/null +++ b/utils/textmate/Samples/keywords.svg @@ -0,0 +1,57 @@ + + + + 1 // Part of the Carbon Language project, under the Apache License v2.0 with LLVM + 2 // Exceptions. See /LICENSE for license information. + 3 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + 4 + 5 // --- operators + 6 + 7 >>= <=> <<= ->? &= ^= := :? == => != >= >> <= <> << <- -= -> -- %= |= += ++ /= + 8 *= ~= & @ ^ : = ! > < - % . | + ? / * ~ \ + 9 and as impls in like not or where +10 +11 // --- special-keywords +12 +13 break case continue else if for match return returned then while +14 default => +15 +16 // --- introducer-keywords +17 +18 adapt alias choice class constraint fn import inline interface let library +19 match_first namespace observe require var +20 base +21 export +22 impl +23 package +24 +25 // --- modifier-keywords +26 +27 abstract const eval extend extern final generic musteval override partial +28 private protected ref runtime static template unsafe unused val virtual +29 base class +30 default +31 export import +32 impl fn +33 impl library +34 impl package +35 +36 // --- misc-keywords +37 +38 forall form friend +39 .base +40 +41 // --- self-keywords +42 +43 self Self +44 +45 // --- package-roots +46 +47 Core Cpp +48 +49 // --- underscore +50 +51 ; +52 _ +53 _some_var + diff --git a/utils/textmate/Samples/literals.carbon b/utils/textmate/Samples/literals.carbon index 117269c0ea6e..382f52b7b63d 100644 --- a/utils/textmate/Samples/literals.carbon +++ b/utils/textmate/Samples/literals.carbon @@ -11,6 +11,24 @@ ''' Multi line string ''' +'''py + Block literal with a file type indicator. +''' + +#"a raw \n literal, where \#n is the escape"# +##"nested "# stays inside, and \##n is the escape"## +#''' + A raw block literal. +'''# + +"unterminated + +// --- characters + +'a' '\n' '\0' '\u{1F600}' +'\x70' +'' +'abcde' // --- true-false @@ -18,7 +36,7 @@ true false // --- type-literals -array bool char str type ; +array auto bool char str type i8 i16 i32 i64 i128 i10 u8 u16 u32 u64 u128 u11 f8 f16 f32 f64 f128 f12 @@ -34,6 +52,10 @@ f8 f16 f32 f64 f128 f12 .123 .1e3 +0o755 +0o_17 +0o8 + 0b1011_0010 0b_010 0b2345 diff --git a/utils/textmate/Samples/literals.jpg b/utils/textmate/Samples/literals.jpg deleted file mode 100644 index d78f94cacab6..000000000000 Binary files a/utils/textmate/Samples/literals.jpg and /dev/null differ diff --git a/utils/textmate/Samples/literals.svg b/utils/textmate/Samples/literals.svg new file mode 100644 index 000000000000..39291b821308 --- /dev/null +++ b/utils/textmate/Samples/literals.svg @@ -0,0 +1,72 @@ + + + + 1 // Part of the Carbon Language project, under the Apache License v2.0 with LLVM + 2 // Exceptions. See /LICENSE for license information. + 3 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + 4 + 5 // --- string + 6 + 7 'C' + 8 "Hello\nWorld\n" + 9 "\"\'\t\r\n" +10 '''Single line string''' +11 ''' Multi line +12 string +13 ''' +14 '''py +15 Block literal with a file type indicator. +16 ''' +17 +18 #"a raw \n literal, where \#n is the escape"# +19 ##"nested "# stays inside, and \##n is the escape"## +20 #''' +21 A raw block literal. +22 '''# +23 +24 "unterminated +25 +26 // --- characters +27 +28 'a' '\n' '\0' '\u{1F600}' +29 '\x70' +30 '' +31 'abcde' +32 +33 // --- true-false +34 +35 true false +36 +37 // --- type-literals +38 +39 array auto bool char str type +40 i8 i16 i32 i64 i128 i10 +41 u8 u16 u32 u64 u128 u11 +42 f8 f16 f32 f64 f128 f12 +43 +44 // --- numbers +45 +46 12345 6789 00012345 +47 +48 1234.56789 +49 0.875 +50 0.3e5 +51 1e5 +52 .123 +53 .1e3 +54 +55 0o755 +56 0o_17 +57 0o8 +58 +59 0b1011_0010 +60 0b_010 +61 0b2345 +62 0B011 +63 +64 0x2AE5 +65 0x_F109 +66 0xD8_E2.5F_7Bp8 +67 0xabcdef.abcdef +68 0XA85CE + diff --git a/utils/textmate/Samples/main.carbon b/utils/textmate/Samples/main.carbon index 73fa0feca659..808d9aa0e839 100644 --- a/utils/textmate/Samples/main.carbon +++ b/utils/textmate/Samples/main.carbon @@ -4,7 +4,7 @@ // More single line comments -package Carbon api; +package Carbon; interface HasValueParam(T: type, V: T) { fn Go(self) -> T; @@ -25,7 +25,7 @@ class Point { fn Procedure() -> i32 { returned var zoop: i32 = 0; - while (DoSomeJob() { + while (DoSomeJob()) { zoop += 1; } @@ -38,9 +38,9 @@ fn Main() -> i32 { let bin = 0b0000111001010010; let dec = 123456789012345678; - let big: Carbon.Int(1024) = 1234; + let big: Core.Int(1024) = 1234; let aaa: auto = "Carbon"; - let view: StringView = "Carbon"; + let view: str = "Carbon"; return Procedure(); } diff --git a/utils/textmate/Samples/main.jpg b/utils/textmate/Samples/main.jpg deleted file mode 100644 index 196d76bd0044..000000000000 Binary files a/utils/textmate/Samples/main.jpg and /dev/null differ diff --git a/utils/textmate/Samples/main.svg b/utils/textmate/Samples/main.svg new file mode 100644 index 000000000000..4ad28ec5653d --- /dev/null +++ b/utils/textmate/Samples/main.svg @@ -0,0 +1,50 @@ + + + + 1 // Part of the Carbon Language project, under the Apache License v2.0 with LLVM + 2 // Exceptions. See /LICENSE for license information. + 3 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + 4 + 5 // More single line comments + 6 + 7 package Carbon; + 8 + 9 interface HasValueParam(T: type, V: T) { +10 fn Go(self) -> T; +11 } +12 +13 impl () as HasValueParam(i32, 5) { +14 fn Go(self) -> i32 { return 42; } +15 } +16 +17 class Point { +18 fn Origin() -> Self { +19 return {.x = 0, .y = 0}; +20 } +21 +22 var x: i32; +23 var y: i32; +24 } +25 +26 fn Procedure() -> i32 { +27 returned var zoop: i32 = 0; +28 while (DoSomeJob()) { +29 zoop += 1; +30 } +31 +32 return var; +33 } +34 +35 fn Main() -> i32 { +36 let str = "Hello world"; +37 let hex = 0xABCDEF1234567890; +38 let bin = 0b0000111001010010; +39 let dec = 123456789012345678; +40 +41 let big: Core.Int(1024) = 1234; +42 let aaa: auto = "Carbon"; +43 let view: str = "Carbon"; +44 +45 return Procedure(); +46 } + diff --git a/utils/textmate/Samples/types.jpg b/utils/textmate/Samples/types.jpg deleted file mode 100644 index 5c8c8ec62e5f..000000000000 Binary files a/utils/textmate/Samples/types.jpg and /dev/null differ diff --git a/utils/textmate/Samples/types.svg b/utils/textmate/Samples/types.svg new file mode 100644 index 000000000000..970b20b49638 --- /dev/null +++ b/utils/textmate/Samples/types.svg @@ -0,0 +1,45 @@ + + + + 1 // Part of the Carbon Language project, under the Apache License v2.0 with LLVM + 2 // Exceptions. See /LICENSE for license information. + 3 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + 4 + 5 class C1; + 6 class C2(T1: type, T2: C1) {} + 7 class C3 { + 8 extend base: C1; + 9 } +10 class +11 (C4( +12 (T1: type), +13 T2: +14 (A.(B(E, F(G)).C) +15 .D))) ; +16 +17 class C5 { +18 impl as I; +19 extend impl as I; +20 } +21 +22 alias a1 = C1; +23 alias a2 = i32; +24 +25 adapt SomeClass; +26 adapt {.a: SomeClass, .b: i32}; +27 adapt i32; +28 +29 interface I; +30 impl C as (I(A, B, C.(D.E(F, G)).H)) {} +31 impl C as I where .A = (B where .Self = {}) {} +32 impl C as I where .A = {} and .B = C {} +33 impl Self as I; +34 impl as I; +35 +36 constraint Constraint { +37 require T impls I; +38 require Self impls I; +39 require impls I; +40 extern require impls I; +41 } + diff --git a/utils/textmate/render_sample.py b/utils/textmate/render_sample.py new file mode 100755 index 000000000000..dee7fbc3e5e3 --- /dev/null +++ b/utils/textmate/render_sample.py @@ -0,0 +1,187 @@ +#!/usr/bin/env -S uv run --script + +# /// script +# requires-python = ">=3.12" +# /// + +"""Renders Carbon sources as highlighted SVG. + +This regenerates the renderings next to the TextMate samples, so they show what +the grammar in this repository actually produces rather than whatever an editor +looked like when someone last took a screenshot by hand. + +The SVG holds the source as text rather than as outlines, so whatever displays +it lays the text out and draws the glyphs; nothing here rasterizes. Colors are +VS Code's Dark+. A scope the theme does not style resolves outward through the +scope stack, which is what makes a string's quotes take the string color and a +comment's `//` take the comment color. +""" + +__copyright__ = """ +Part of the Carbon Language project, under the Apache License v2.0 with LLVM +Exceptions. See /LICENSE for license information. +SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +""" + +import argparse +import sys +from pathlib import Path +from typing import Optional + +import tmlanguage + +# The grammar is part of the VS Code extension, which must all be contained +# under the extension's root directory. +_GRAMMAR_PATH = ( + Path(__file__).resolve().parents[1] / "vscode" / "carbon.tmLanguage.json" +) + +# VS Code's Dark+, keyed by the selectors the theme itself uses. Matching is by +# longest dotted prefix, as a real theme does, so this stays correct for any +# grammar rather than only the scope names in use today. +_THEME = { + "comment": "#6a9955", + "constant.character.escape": "#d7ba7d", + "constant.language": "#569cd6", + "constant.numeric": "#b5cea8", + "entity.name.function": "#dcdcaa", + "entity.name.namespace": "#4ec9b0", + "entity.name.tag": "#569cd6", + "entity.name.type": "#4ec9b0", + "keyword.control": "#c586c0", + "keyword.operator": "#d4d4d4", + "keyword.other": "#569cd6", + "meta.embedded": "#d4d4d4", + "storage.modifier": "#569cd6", + "storage.type": "#569cd6", + "string": "#ce9178", + "support.class": "#4ec9b0", + "support.function": "#dcdcaa", + "support.type": "#4ec9b0", + "support.type.property-name": "#9cdcfe", + "support.variable": "#9cdcfe", + "variable.language": "#569cd6", + "variable.other": "#9cdcfe", + "variable.other.enummember": "#4fc1ff", + "variable.parameter": "#9cdcfe", +} + +_BACKGROUND = "#1f1f1f" +_FOREGROUND = "#d4d4d4" +_GUTTER = "#6e7681" +# Whatever monospace font the viewer has: nothing can be fetched, because +# GitHub serves SVG under `default-src 'none'` and a web font would be blocked. +# The text simply flows, so the font's own metrics lay each line out. +_FONT = "ui-monospace, SFMono-Regular, Menlo, Consolas, monospace" +_FONT_SIZE = 14 +# Only used to size the canvas. Monospace advances cluster near 0.6em, so a +# font a little wider than this just runs closer to the right edge. +_CHAR_WIDTH = 8.5 +_LINE_HEIGHT = 19 +# How far above the bottom of its line the baseline sits, leaving room for the +# descenders of `g` and `y` at this line height. +_DESCENT = 5 +_PAD = 12 + + +def _color_for(scopes: list[str]) -> str: + """Resolves a scope stack to a color, innermost scope first. + + Within a scope the longest matching prefix wins, so that a selector such as + `variable.other.enummember` beats `variable.other`. A scope the theme does + not style resolves outward to its enclosing scope, which is what gives a + string's quotes the string color. + """ + for scope in reversed(scopes): + parts = scope.split(".") + for end in range(len(parts), 0, -1): + color = _THEME.get(".".join(parts[:end])) + if color: + return color + return _FOREGROUND + + +def _escape(text: str) -> str: + """Escapes source text for an SVG text node, as one would for HTML.""" + return text.replace("&", "&").replace("<", "<").replace(">", ">") + + +def _runs(colors: list[str]) -> list[tuple[int, int, str]]: + """Merges a per-character color list into `(start, end, color)` runs.""" + runs: list[tuple[int, int, str]] = [] + for index, color in enumerate(colors): + if runs and runs[-1][2] == color: + runs[-1] = (runs[-1][0], index + 1, color) + else: + runs.append((index, index + 1, color)) + return runs + + +def render(grammar: tmlanguage.Grammar, source: str) -> str: + """Renders a Carbon source as a standalone SVG document.""" + lines = tmlanguage.split_lines(source) + colored = [[_FOREGROUND] * len(line) for line in lines] + for token in tmlanguage.tokenize(grammar, source): + color = _color_for(token.scopes) + row = colored[token.line] + # A token runs one past the line, over the newline it was tokenized + # with, and there is no column there to color. + for column in range(token.start, min(token.end, len(row))): + row[column] = color + + # Trailing blank lines would only pad the bottom of the image. + while lines and not lines[-1].strip(): + lines.pop() + colored.pop() + + digits = len(str(len(lines))) if lines else 1 + longest = max((len(line) for line in lines), default=0) + width = round(_PAD * 2 + (digits + 1 + longest) * _CHAR_WIDTH) + height = _PAD * 2 + len(lines) * _LINE_HEIGHT + + out = [ + f'', + f'', + # Indentation has to survive two eras of the spec. SVG 1.1 renderers + # read `xml:space`, and honor it here on the group. Browsers follow + # SVG 2, where whitespace is a CSS property and neither form is + # inherited into text from an ancestor, so each `text` repeats it. + f'', + ] + for linenum, linetext in enumerate(lines): + baseline = _PAD + (linenum + 1) * _LINE_HEIGHT - _DESCENT + spans = [ + f'{str(linenum + 1).rjust(digits)} ' + ] + [ + f'{_escape(linetext[start:end])}' + for start, end, color in _runs(colored[linenum]) + ] + out.append( + f'' + f"{''.join(spans)}" + ) + out.append("") + return "\n".join(out) + "\n" + + +def main(argv: Optional[list[str]] = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("sources", nargs="+", type=Path) + parser.add_argument("--grammar", type=Path, default=_GRAMMAR_PATH) + args = parser.parse_args(argv) + + grammar = tmlanguage.Grammar.load(args.grammar) + for source_path in args.sources: + out_path = source_path.with_suffix(".svg") + out_path.write_text( + render(grammar, source_path.read_text(encoding="utf-8")), + encoding="utf-8", + ) + print(f"wrote {out_path}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/utils/textmate/tmlanguage.py b/utils/textmate/tmlanguage.py new file mode 100644 index 000000000000..3cdcfe5657cc --- /dev/null +++ b/utils/textmate/tmlanguage.py @@ -0,0 +1,405 @@ +#!/usr/bin/env python3 + +"""A minimal TextMate grammar tokenizer. + +TextMate grammars are normally run by Oniguruma, which we cannot depend on +hermetically. Carbon's grammar happens to use no Oniguruma-only regex syntax, +so `re` compiles its regexes unchanged. That is necessary but not sufficient, +since the two engines can still read shared syntax differently, so what +establishes that this reproduces VS Code's output is `conformance_test.py`, +which tokenizes the repository with both and compares. `check_regex_dialect` +catches the half of that a hermetic test can reach: if it fails, the grammar +has left the shared subset and this tokenizer can no longer be trusted. + +Only the grammar features Carbon uses are implemented: `match`, `begin`/`end` +with captures and `contentName`, `include` into the repository, and `\\N` +backreferences from `begin` into `end`. +""" + +__copyright__ = """ +Part of the Carbon Language project, under the Apache License v2.0 with LLVM +Exceptions. See /LICENSE for license information. +SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +""" + +import json +import re +from pathlib import Path +from typing import Any, Iterator, NamedTuple, Optional + +# Regex constructs Oniguruma supports and `re` either rejects or interprets +# differently. A grammar using any of these would invalidate this tokenizer. +_FOREIGN_SYNTAX = { + "possessive quantifier": r"(?:[*+?}])\+", + "atomic group": r"\(\?>", + r"\G anchor": r"\\G", + r"\h horizontal space": r"\\[hH]", + "conditional": r"\(\?\(", + "named backreference": r"\\k<", + "variable-length lookbehind": r"\(\?<[=!][^)]*[*+]", +} + + +# Every key a TextMate grammar can put a regex under. `match`, `begin`, and +# `end` are the three this tokenizer runs; an editor also runs the folding +# markers, so they are checked for dialect even though nothing here reads them. +_REGEX_KEYS = ( + "match", + "begin", + "end", + "foldingStartMarker", + "foldingStopMarker", +) + + +class Token(NamedTuple): + """A scope assignment covering `[start, end)` of line `line`. + + `scopes` runs outermost first, starting with the grammar's own scope name, + so the last entry is the most specific. A theme matches from there + outward, taking the first entry it has a color for. + """ + + line: int + start: int + end: int + scopes: list[str] + + +class Rule(NamedTuple): + """An entry of the region stack: an open `begin`/`end`, or the grammar. + + The grammar's own entry sits at the bottom with an `end_re` of `None`, + since nothing closes it. + """ + + scopes: list[str] + content_scopes: list[str] + end_re: Optional[re.Pattern[str]] + end_captures: dict[str, Any] + patterns: list[dict[str, Any]] + + +class _RuleMatch(NamedTuple): + """A rule that matched, and where it matched. + + `kind` names the grammar key the matching regex came from: `"end"` for + the open region's own `end`, in which case `pattern` is empty, or + `"match"` or `"begin"` for one of the rules in the region's `patterns`. + `match` is always a real match; it is what `kind` describes. + """ + + kind: str + pattern: dict[str, Any] + match: re.Match[str] + + +class Grammar: + """A parsed TextMate grammar, ready to tokenize with.""" + + def __init__(self, raw: dict[str, Any]) -> None: + self.raw = raw + self.scope_name: str = raw.get("scopeName", "") + self._repository: dict[str, Any] = raw.get("repository", {}) + self._compiled: dict[str, re.Pattern[str]] = {} + + @staticmethod + def load(path: Path) -> "Grammar": + with path.open(encoding="utf-8") as f: + return Grammar(json.load(f)) + + def compile(self, regex: str) -> re.Pattern[str]: + compiled = self._compiled.get(regex) + if compiled is None: + compiled = self._compiled[regex] = re.compile(regex) + return compiled + + def resolve( + self, + patterns: Optional[list[dict[str, Any]]], + active: frozenset[str] = frozenset(), + ) -> list[dict[str, Any]]: + """Flattens `include` directives into a list of concrete rules. + + A `name` on the repository entry itself is deliberately dropped: VS + Code inlines an include-only rule's patterns without pushing its scope, + so honoring it here would disagree with the real tokenizer. `active` + breaks a cycle between mutually including entries. + + An include of a different grammar, such as `source.cpp`, contributes + nothing: we tokenize Carbon, and embedded regions are left to whatever + scope their `contentName` assigns. + """ + resolved: list[dict[str, Any]] = [] + for pattern in patterns or []: + target = pattern.get("include") + if target is None: + resolved.append(pattern) + elif target.startswith("#"): + name = target[1:] + if name in active: + continue + entry = self._repository.get(name, {}) + resolved.extend( + self.resolve(entry.get("patterns"), active | {name}) + ) + elif target == "$self": + resolved.extend(self.resolve(self.raw.get("patterns"), active)) + return resolved + + def initial_stack(self) -> list[Rule]: + base = [self.scope_name] + return [ + Rule(base, base, None, {}, self.resolve(self.raw.get("patterns"))) + ] + + def all_regexes(self) -> Iterator[tuple[str, str]]: + """Yields every `(key, regex)` the grammar holds, for validation. + + A grammar keeps regexes only under the keys in `_REGEX_KEYS`, so + finding them all means walking every dictionary in it and picking + those keys out. `key` says which one it was, which is all a message + needs to point at the right place. + """ + + def walk(node: Any) -> Iterator[tuple[str, str]]: + if isinstance(node, dict): + for key in _REGEX_KEYS: + value = node.get(key) + if isinstance(value, str): + yield key, value + for value in node.values(): + yield from walk(value) + elif isinstance(node, list): + for value in node: + yield from walk(value) + + yield from walk(self.raw) + + +def check_regex_dialect(grammar: Grammar) -> list[str]: + """Returns a message per regex `re` cannot stand in for Oniguruma on.""" + problems: list[str] = [] + for key, regex in grammar.all_regexes(): + # A `\N` backreference is substituted before compiling, so do the + # same here with an arbitrary value before validating. + probe = re.sub(r"\\\d", "x", regex) + try: + re.compile(probe) + except re.error as e: + problems.append(f"`{key}` regex does not compile: {regex}: {e}") + continue + for name, foreign in _FOREIGN_SYNTAX.items(): + if re.search(foreign, regex): + problems.append( + f"`{key}` regex uses {name}, which `re` does not share " + f"with Oniguruma: {regex}" + ) + return problems + + +def _substitute_backrefs(regex: str, begin_match: re.Match[str]) -> str: + """Replaces `\\N` in an `end` with the text `begin` captured, as VS Code + does before compiling the `end` regex.""" + + def replace(ref: re.Match[str]) -> str: + return re.escape(begin_match.group(int(ref.group(1))) or "") + + return re.sub(r"\\(\d)", replace, regex) + + +def _append_token( + tokens: list[Token], linenum: int, start: int, end: int, scopes: list[str] +) -> None: + """Appends a token, unless it would be empty. + + Callers append the text between two things without first checking that + there is any, and about a quarter of the time there is none, so the check + lives here rather than at every call. + """ + if end > start: + tokens.append(Token(linenum, start, end, scopes)) + + +def _append_capture_tokens( + tokens: list[Token], + linenum: int, + match: re.Match[str], + captures: dict[str, Any], + scopes: list[str], +) -> None: + """Appends the tokens for one match, splitting it at its capture groups. + + `captures` maps a group number, written as a string, to the scope the + group's text takes on top of `scopes`; group `"0"` is the whole match. The + tokens tile the match with no gaps: text not covered by a listed group is + still appended, under `scopes` alone. + """ + if not captures: + _append_token(tokens, linenum, match.start(), match.end(), scopes) + return + pos = match.start() + for group in range((match.re.groups or 0) + 1): + spec = captures.get(str(group)) + # Skip a group the grammar does not name, one that did not + # participate in the match (`start` is then -1), and one that lies + # behind text already appended, since tokens come out in order. + if spec is None or match.start(group) < pos: + continue + if match.start(group) == match.end(group): + continue + _append_token(tokens, linenum, pos, match.start(group), scopes) + _append_token( + tokens, + linenum, + match.start(group), + match.end(group), + scopes + [spec["name"]], + ) + pos = match.end(group) + _append_token(tokens, linenum, pos, match.end(), scopes) + + +def _push_scope( + scopes: list[str], pattern: dict[str, Any], key: str +) -> list[str]: + """Adds the scope `pattern[key]` names, if it names one.""" + name = pattern.get(key) + return scopes + [name] if name is not None else scopes + + +def _find_earliest( + grammar: Grammar, rule: Rule, text: str, pos: int +) -> Optional[_RuleMatch]: + """Finds which of `rule`'s regexes matches soonest at or after `pos`. + + The open region's own `end` is tried first and ties are broken towards + whatever was tried earlier, which gives TextMate's two ordering rules: a + region that can close here closes here, and otherwise the first rule + listed in the grammar wins. + """ + best: Optional[_RuleMatch] = None + if rule.end_re is not None: + match = rule.end_re.search(text, pos) + if match is not None: + best = _RuleMatch("end", {}, match) + for pattern in rule.patterns: + # The key a regex came from is what the rule does with it, so take + # both from the same lookup. + if (regex := pattern.get("match")) is not None: + kind = "match" + elif (regex := pattern.get("begin")) is not None: + kind = "begin" + else: + continue + match = grammar.compile(regex).search(text, pos) + if match is None: + continue + if best is not None and match.start() >= best.match.start(): + continue + best = _RuleMatch(kind, pattern, match) + return best + + +def tokenize_line( + grammar: Grammar, linenum: int, text: str, stack: list[Rule] +) -> tuple[list[Token], list[Rule]]: + """Tokenizes one line, given the regions open when it starts. + + `text` must carry its trailing newline: VS Code tokenizes a line together + with its terminator, so that newline gets a token of its own, and carrying + it is what keeps the two token streams identical. `stack` is the chain of + `begin`/`end` regions open at the start of the line, outermost first and + never empty: its first entry is the grammar itself. The stack returned is + the one open at the start of the next line, which is how a region spans + lines. + """ + tokens: list[Token] = [] + pos = 0 + # Counts region transitions that consumed nothing, which make no progress. + stalls = 0 + while pos <= len(text): + rule = stack[-1] + found = _find_earliest(grammar, rule, text, pos) + if found is None: + break + # Text before the match belongs to the region containing it. + _append_token( + tokens, linenum, pos, found.match.start(), rule.content_scopes + ) + if found.kind == "end": + _append_capture_tokens( + tokens, linenum, found.match, rule.end_captures, rule.scopes + ) + stack = stack[:-1] + else: + scopes = _push_scope(rule.content_scopes, found.pattern, "name") + captures = "captures" if found.kind == "match" else "beginCaptures" + _append_capture_tokens( + tokens, + linenum, + found.match, + found.pattern.get(captures, {}), + scopes, + ) + if found.kind == "begin": + stack = stack + [ + Rule( + scopes, + _push_scope(scopes, found.pattern, "contentName"), + grammar.compile( + _substitute_backrefs( + found.pattern["end"], found.match + ) + ), + found.pattern.get("endCaptures", {}), + grammar.resolve(found.pattern.get("patterns")), + ) + ] + + if found.match.end() > found.match.start(): + pos = found.match.end() + elif found.kind == "match": + # A `match` that consumed nothing would match there forever, so + # step over a character. + pos = found.match.start() + 1 + else: + # Entering or leaving a region is progress in itself, and the + # character stays available to the region now on top: a lookahead + # `end` such as `(?=[\[\(])` depends on that. + pos = found.match.start() + stalls += 1 + # The bound is arbitrary; it only has to exceed the transitions a + # real line could ask for. Reaching it means a rule opens and + # closes forever, which is a grammar bug: say so rather than + # silently truncating the line. + if stalls > len(text) + 64: + raise RuntimeError( + f"line {linenum} stopped making progress at offset {pos}: " + f"rule {found.pattern!r} neither consumes nor terminates" + ) + # Whatever is left over, including the newline, belongs to the region the + # line ends inside of. + _append_token(tokens, linenum, pos, len(text), stack[-1].content_scopes) + return tokens, stack + + +def split_lines(text: str) -> list[str]: + """Splits a file into lines the way an editor numbers them. + + Terminators are not kept on the lines, and a file ending in a newline + yields an empty final line, which is the blank line an editor shows there. + + Not `str.splitlines`: that also breaks on `\\f`, `\\v`, and a handful of + Unicode separators, which a TextMate grammar sees as ordinary characters + within a line, and it drops that final empty line. + """ + return text.split("\n") + + +def tokenize(grammar: Grammar, text: str) -> Iterator[Token]: + """Tokenizes a whole file.""" + stack = grammar.initial_stack() + for linenum, line in enumerate(split_lines(text)): + tokens, stack = tokenize_line(grammar, linenum, line + "\n", stack) + yield from tokens