diff --git a/utils/textmate/README.md b/utils/textmate/README.md
index af979d7ce855..a06528693526 100644
--- a/utils/textmate/README.md
+++ b/utils/textmate/README.md
@@ -33,3 +33,17 @@ If you are using Atom, you can convert the bundle to an Atom-compatible one. See
For other editors that support TextMate bundles you can consult your editors
documentation to see how to use the bundle.
+
+## Samples
+
+`Samples/` holds Carbon sources that exercise the grammar, each with an SVG
+rendering of how this bundle highlights it. Some deliberately contain invalid
+code, to show that highlighting stays sensible while something is being typed.
+
+The renderings are generated, not screenshotted, so they always reflect the
+grammar in this repository. Regenerate them in the same commit as any change to
+the grammar, so a reviewer can see what the change does to real code:
+
+```shell
+utils/textmate/render_sample.py utils/textmate/Samples/*.carbon
+```
diff --git a/utils/textmate/Samples/choices.jpg b/utils/textmate/Samples/choices.jpg
deleted file mode 100644
index 5244a6734e45..000000000000
Binary files a/utils/textmate/Samples/choices.jpg and /dev/null differ
diff --git a/utils/textmate/Samples/choices.svg b/utils/textmate/Samples/choices.svg
new file mode 100644
index 000000000000..3ef8c06a9fa3
--- /dev/null
+++ b/utils/textmate/Samples/choices.svg
@@ -0,0 +1,13 @@
+
diff --git a/utils/textmate/Samples/comments.carbon b/utils/textmate/Samples/comments.carbon
index 2598264bee85..e089e9cc197b 100644
--- a/utils/textmate/Samples/comments.carbon
+++ b/utils/textmate/Samples/comments.carbon
@@ -7,8 +7,10 @@
// class C; fn F[T: A](a: T) -> i32 { return 86; }
//@dump-sem-ir-begin
//@dump-sem-ir-end
+//@include-in-dumps
//@dump-sem-ir-
//no whitespace after double slash
/* invalid comment pattern. */
/// invalid comment pattern.
class C; // Comment after a statement/expression/declaration.
+var x: i32 = 1;// Trailing, with no space before the introducer.
diff --git a/utils/textmate/Samples/comments.jpg b/utils/textmate/Samples/comments.jpg
deleted file mode 100644
index c55e11b61c4e..000000000000
Binary files a/utils/textmate/Samples/comments.jpg and /dev/null differ
diff --git a/utils/textmate/Samples/comments.svg b/utils/textmate/Samples/comments.svg
new file mode 100644
index 000000000000..d389aba94246
--- /dev/null
+++ b/utils/textmate/Samples/comments.svg
@@ -0,0 +1,20 @@
+
diff --git a/utils/textmate/Samples/customs.carbon b/utils/textmate/Samples/customs.carbon
index 546c79c7d868..0b62fb3abe52 100644
--- a/utils/textmate/Samples/customs.carbon
+++ b/utils/textmate/Samples/customs.carbon
@@ -2,7 +2,7 @@
// Exceptions. See /LICENSE for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
-package SomePackage api;
+package SomePackage;
Core SomeCore;
import SomePackage;
diff --git a/utils/textmate/Samples/customs.jpg b/utils/textmate/Samples/customs.jpg
deleted file mode 100644
index 0cebb9a79f25..000000000000
Binary files a/utils/textmate/Samples/customs.jpg and /dev/null differ
diff --git a/utils/textmate/Samples/customs.svg b/utils/textmate/Samples/customs.svg
new file mode 100644
index 000000000000..24c7cc87dfd3
--- /dev/null
+++ b/utils/textmate/Samples/customs.svg
@@ -0,0 +1,15 @@
+
diff --git a/utils/textmate/Samples/functions_variables.carbon b/utils/textmate/Samples/functions_variables.carbon
index 907bc2031988..1054a90b015a 100644
--- a/utils/textmate/Samples/functions_variables.carbon
+++ b/utils/textmate/Samples/functions_variables.carbon
@@ -37,3 +37,16 @@ fn F(ref a: A, const ref b: B);
fn F() -> T;
fn F() -> T.U { return val; }
fn F() -> i32 => return val;
+
+// --- lambdas and positional parameters
+
+var explicit: auto = fn (x: i32) -> i32 { return x + 1; };
+var expression: auto = fn (x: i32) => x + 1;
+var positional: auto = fn { Print($0, $1, $42); };
+
+// --- raw identifiers
+
+r#class
+r#if
+r#i32
+var r#var: i32 = r#class;
diff --git a/utils/textmate/Samples/functions_variables.jpg b/utils/textmate/Samples/functions_variables.jpg
deleted file mode 100644
index a242857b4817..000000000000
Binary files a/utils/textmate/Samples/functions_variables.jpg and /dev/null differ
diff --git a/utils/textmate/Samples/functions_variables.svg b/utils/textmate/Samples/functions_variables.svg
new file mode 100644
index 000000000000..4e3aa845622d
--- /dev/null
+++ b/utils/textmate/Samples/functions_variables.svg
@@ -0,0 +1,56 @@
+
diff --git a/utils/textmate/Samples/interop.carbon b/utils/textmate/Samples/interop.carbon
new file mode 100644
index 000000000000..58fa046f050f
--- /dev/null
+++ b/utils/textmate/Samples/interop.carbon
@@ -0,0 +1,29 @@
+// Part of the Carbon Language project, under the Apache License v2.0 with LLVM
+// Exceptions. See /LICENSE for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+// Inline C++ is scoped `meta.embedded.block.cpp`, so an editor that has a C++
+// grammar highlights them as C++; these renderings come from the Carbon grammar
+// alone, so that content shows plain.
+
+package Sample library "interop";
+
+import Cpp library "";
+import Cpp library "";
+
+import Cpp inline '''c++
+int FromImport() { return 1; }
+''';
+
+import Cpp inline "int OneLiner() { return 2; }";
+
+inline Cpp '''
+int FromBlock() { return 3; }
+''';
+
+fn Uses() {
+ var value: Cpp.int = 1 as Cpp.int;
+ var widened: Cpp.size_t = value as Cpp.size_t;
+ var back: i32 = value unsafe as i32;
+ Cpp.printf("%d\n", value);
+}
diff --git a/utils/textmate/Samples/interop.svg b/utils/textmate/Samples/interop.svg
new file mode 100644
index 000000000000..84c6cec24c33
--- /dev/null
+++ b/utils/textmate/Samples/interop.svg
@@ -0,0 +1,33 @@
+
diff --git a/utils/textmate/Samples/keywords.carbon b/utils/textmate/Samples/keywords.carbon
index b4cf3e071f98..ec8d30452388 100644
--- a/utils/textmate/Samples/keywords.carbon
+++ b/utils/textmate/Samples/keywords.carbon
@@ -4,8 +4,9 @@
// --- operators
->>= <=> <<= &= == != >= >> <= << -= -> -- %= \ = += ++ /= *= & ^ = > < - % . | + / *
-and as impls in like not or partial ref template where
+>>= <=> <<= ->? &= ^= := :? == => != >= >> <= <> << <- -= -> -- %= |= += ++ /=
+*= ~= & @ ^ : = ! > < - % . | + ? / * ~ \
+and as impls in like not or where
// --- special-keywords
@@ -14,7 +15,8 @@ default =>
// --- introducer-keywords
-adapt alias choice class constraint fn import interface let library namespace var
+adapt alias choice class constraint fn import inline interface let library
+match_first namespace observe require var
base
export
impl
@@ -22,7 +24,8 @@ package
// --- modifier-keywords
-abstract const extend extern final private protected virtual
+abstract const eval extend extern final generic musteval override partial
+private protected ref runtime static template unsafe unused val virtual
base class
default
export import
@@ -32,13 +35,17 @@ impl package
// --- misc-keywords
-auto destructor forall friend observe override require
+forall form friend
.base
// --- self-keywords
self Self
+// --- package-roots
+
+Core Cpp
+
// --- underscore
;
diff --git a/utils/textmate/Samples/keywords.jpg b/utils/textmate/Samples/keywords.jpg
deleted file mode 100644
index 2678cefd595f..000000000000
Binary files a/utils/textmate/Samples/keywords.jpg and /dev/null differ
diff --git a/utils/textmate/Samples/keywords.svg b/utils/textmate/Samples/keywords.svg
new file mode 100644
index 000000000000..0e3a591c126a
--- /dev/null
+++ b/utils/textmate/Samples/keywords.svg
@@ -0,0 +1,57 @@
+
diff --git a/utils/textmate/Samples/literals.carbon b/utils/textmate/Samples/literals.carbon
index 117269c0ea6e..382f52b7b63d 100644
--- a/utils/textmate/Samples/literals.carbon
+++ b/utils/textmate/Samples/literals.carbon
@@ -11,6 +11,24 @@
''' Multi line
string
'''
+'''py
+ Block literal with a file type indicator.
+'''
+
+#"a raw \n literal, where \#n is the escape"#
+##"nested "# stays inside, and \##n is the escape"##
+#'''
+ A raw block literal.
+'''#
+
+"unterminated
+
+// --- characters
+
+'a' '\n' '\0' '\u{1F600}'
+'\x70'
+''
+'abcde'
// --- true-false
@@ -18,7 +36,7 @@ true false
// --- type-literals
-array bool char str type ;
+array auto bool char str type
i8 i16 i32 i64 i128 i10
u8 u16 u32 u64 u128 u11
f8 f16 f32 f64 f128 f12
@@ -34,6 +52,10 @@ f8 f16 f32 f64 f128 f12
.123
.1e3
+0o755
+0o_17
+0o8
+
0b1011_0010
0b_010
0b2345
diff --git a/utils/textmate/Samples/literals.jpg b/utils/textmate/Samples/literals.jpg
deleted file mode 100644
index d78f94cacab6..000000000000
Binary files a/utils/textmate/Samples/literals.jpg and /dev/null differ
diff --git a/utils/textmate/Samples/literals.svg b/utils/textmate/Samples/literals.svg
new file mode 100644
index 000000000000..39291b821308
--- /dev/null
+++ b/utils/textmate/Samples/literals.svg
@@ -0,0 +1,72 @@
+
diff --git a/utils/textmate/Samples/main.carbon b/utils/textmate/Samples/main.carbon
index 73fa0feca659..808d9aa0e839 100644
--- a/utils/textmate/Samples/main.carbon
+++ b/utils/textmate/Samples/main.carbon
@@ -4,7 +4,7 @@
// More single line comments
-package Carbon api;
+package Carbon;
interface HasValueParam(T: type, V: T) {
fn Go(self) -> T;
@@ -25,7 +25,7 @@ class Point {
fn Procedure() -> i32 {
returned var zoop: i32 = 0;
- while (DoSomeJob() {
+ while (DoSomeJob()) {
zoop += 1;
}
@@ -38,9 +38,9 @@ fn Main() -> i32 {
let bin = 0b0000111001010010;
let dec = 123456789012345678;
- let big: Carbon.Int(1024) = 1234;
+ let big: Core.Int(1024) = 1234;
let aaa: auto = "Carbon";
- let view: StringView = "Carbon";
+ let view: str = "Carbon";
return Procedure();
}
diff --git a/utils/textmate/Samples/main.jpg b/utils/textmate/Samples/main.jpg
deleted file mode 100644
index 196d76bd0044..000000000000
Binary files a/utils/textmate/Samples/main.jpg and /dev/null differ
diff --git a/utils/textmate/Samples/main.svg b/utils/textmate/Samples/main.svg
new file mode 100644
index 000000000000..4ad28ec5653d
--- /dev/null
+++ b/utils/textmate/Samples/main.svg
@@ -0,0 +1,50 @@
+
diff --git a/utils/textmate/Samples/types.jpg b/utils/textmate/Samples/types.jpg
deleted file mode 100644
index 5c8c8ec62e5f..000000000000
Binary files a/utils/textmate/Samples/types.jpg and /dev/null differ
diff --git a/utils/textmate/Samples/types.svg b/utils/textmate/Samples/types.svg
new file mode 100644
index 000000000000..970b20b49638
--- /dev/null
+++ b/utils/textmate/Samples/types.svg
@@ -0,0 +1,45 @@
+
diff --git a/utils/textmate/render_sample.py b/utils/textmate/render_sample.py
new file mode 100755
index 000000000000..dee7fbc3e5e3
--- /dev/null
+++ b/utils/textmate/render_sample.py
@@ -0,0 +1,187 @@
+#!/usr/bin/env -S uv run --script
+
+# /// script
+# requires-python = ">=3.12"
+# ///
+
+"""Renders Carbon sources as highlighted SVG.
+
+This regenerates the renderings next to the TextMate samples, so they show what
+the grammar in this repository actually produces rather than whatever an editor
+looked like when someone last took a screenshot by hand.
+
+The SVG holds the source as text rather than as outlines, so whatever displays
+it lays the text out and draws the glyphs; nothing here rasterizes. Colors are
+VS Code's Dark+. A scope the theme does not style resolves outward through the
+scope stack, which is what makes a string's quotes take the string color and a
+comment's `//` take the comment color.
+"""
+
+__copyright__ = """
+Part of the Carbon Language project, under the Apache License v2.0 with LLVM
+Exceptions. See /LICENSE for license information.
+SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+"""
+
+import argparse
+import sys
+from pathlib import Path
+from typing import Optional
+
+import tmlanguage
+
+# The grammar is part of the VS Code extension, which must all be contained
+# under the extension's root directory.
+_GRAMMAR_PATH = (
+ Path(__file__).resolve().parents[1] / "vscode" / "carbon.tmLanguage.json"
+)
+
+# VS Code's Dark+, keyed by the selectors the theme itself uses. Matching is by
+# longest dotted prefix, as a real theme does, so this stays correct for any
+# grammar rather than only the scope names in use today.
+_THEME = {
+ "comment": "#6a9955",
+ "constant.character.escape": "#d7ba7d",
+ "constant.language": "#569cd6",
+ "constant.numeric": "#b5cea8",
+ "entity.name.function": "#dcdcaa",
+ "entity.name.namespace": "#4ec9b0",
+ "entity.name.tag": "#569cd6",
+ "entity.name.type": "#4ec9b0",
+ "keyword.control": "#c586c0",
+ "keyword.operator": "#d4d4d4",
+ "keyword.other": "#569cd6",
+ "meta.embedded": "#d4d4d4",
+ "storage.modifier": "#569cd6",
+ "storage.type": "#569cd6",
+ "string": "#ce9178",
+ "support.class": "#4ec9b0",
+ "support.function": "#dcdcaa",
+ "support.type": "#4ec9b0",
+ "support.type.property-name": "#9cdcfe",
+ "support.variable": "#9cdcfe",
+ "variable.language": "#569cd6",
+ "variable.other": "#9cdcfe",
+ "variable.other.enummember": "#4fc1ff",
+ "variable.parameter": "#9cdcfe",
+}
+
+_BACKGROUND = "#1f1f1f"
+_FOREGROUND = "#d4d4d4"
+_GUTTER = "#6e7681"
+# Whatever monospace font the viewer has: nothing can be fetched, because
+# GitHub serves SVG under `default-src 'none'` and a web font would be blocked.
+# The text simply flows, so the font's own metrics lay each line out.
+_FONT = "ui-monospace, SFMono-Regular, Menlo, Consolas, monospace"
+_FONT_SIZE = 14
+# Only used to size the canvas. Monospace advances cluster near 0.6em, so a
+# font a little wider than this just runs closer to the right edge.
+_CHAR_WIDTH = 8.5
+_LINE_HEIGHT = 19
+# How far above the bottom of its line the baseline sits, leaving room for the
+# descenders of `g` and `y` at this line height.
+_DESCENT = 5
+_PAD = 12
+
+
+def _color_for(scopes: list[str]) -> str:
+ """Resolves a scope stack to a color, innermost scope first.
+
+ Within a scope the longest matching prefix wins, so that a selector such as
+ `variable.other.enummember` beats `variable.other`. A scope the theme does
+ not style resolves outward to its enclosing scope, which is what gives a
+ string's quotes the string color.
+ """
+ for scope in reversed(scopes):
+ parts = scope.split(".")
+ for end in range(len(parts), 0, -1):
+ color = _THEME.get(".".join(parts[:end]))
+ if color:
+ return color
+ return _FOREGROUND
+
+
+def _escape(text: str) -> str:
+ """Escapes source text for an SVG text node, as one would for HTML."""
+ return text.replace("&", "&").replace("<", "<").replace(">", ">")
+
+
+def _runs(colors: list[str]) -> list[tuple[int, int, str]]:
+ """Merges a per-character color list into `(start, end, color)` runs."""
+ runs: list[tuple[int, int, str]] = []
+ for index, color in enumerate(colors):
+ if runs and runs[-1][2] == color:
+ runs[-1] = (runs[-1][0], index + 1, color)
+ else:
+ runs.append((index, index + 1, color))
+ return runs
+
+
+def render(grammar: tmlanguage.Grammar, source: str) -> str:
+ """Renders a Carbon source as a standalone SVG document."""
+ lines = tmlanguage.split_lines(source)
+ colored = [[_FOREGROUND] * len(line) for line in lines]
+ for token in tmlanguage.tokenize(grammar, source):
+ color = _color_for(token.scopes)
+ row = colored[token.line]
+ # A token runs one past the line, over the newline it was tokenized
+ # with, and there is no column there to color.
+ for column in range(token.start, min(token.end, len(row))):
+ row[column] = color
+
+ # Trailing blank lines would only pad the bottom of the image.
+ while lines and not lines[-1].strip():
+ lines.pop()
+ colored.pop()
+
+ digits = len(str(len(lines))) if lines else 1
+ longest = max((len(line) for line in lines), default=0)
+ width = round(_PAD * 2 + (digits + 1 + longest) * _CHAR_WIDTH)
+ height = _PAD * 2 + len(lines) * _LINE_HEIGHT
+
+ out = [
+ f'")
+ return "\n".join(out) + "\n"
+
+
+def main(argv: Optional[list[str]] = None) -> int:
+ parser = argparse.ArgumentParser(description=__doc__)
+ parser.add_argument("sources", nargs="+", type=Path)
+ parser.add_argument("--grammar", type=Path, default=_GRAMMAR_PATH)
+ args = parser.parse_args(argv)
+
+ grammar = tmlanguage.Grammar.load(args.grammar)
+ for source_path in args.sources:
+ out_path = source_path.with_suffix(".svg")
+ out_path.write_text(
+ render(grammar, source_path.read_text(encoding="utf-8")),
+ encoding="utf-8",
+ )
+ print(f"wrote {out_path}")
+ return 0
+
+
+if __name__ == "__main__":
+ sys.exit(main())
diff --git a/utils/textmate/tmlanguage.py b/utils/textmate/tmlanguage.py
new file mode 100644
index 000000000000..3cdcfe5657cc
--- /dev/null
+++ b/utils/textmate/tmlanguage.py
@@ -0,0 +1,405 @@
+#!/usr/bin/env python3
+
+"""A minimal TextMate grammar tokenizer.
+
+TextMate grammars are normally run by Oniguruma, which we cannot depend on
+hermetically. Carbon's grammar happens to use no Oniguruma-only regex syntax,
+so `re` compiles its regexes unchanged. That is necessary but not sufficient,
+since the two engines can still read shared syntax differently, so what
+establishes that this reproduces VS Code's output is `conformance_test.py`,
+which tokenizes the repository with both and compares. `check_regex_dialect`
+catches the half of that a hermetic test can reach: if it fails, the grammar
+has left the shared subset and this tokenizer can no longer be trusted.
+
+Only the grammar features Carbon uses are implemented: `match`, `begin`/`end`
+with captures and `contentName`, `include` into the repository, and `\\N`
+backreferences from `begin` into `end`.
+"""
+
+__copyright__ = """
+Part of the Carbon Language project, under the Apache License v2.0 with LLVM
+Exceptions. See /LICENSE for license information.
+SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+"""
+
+import json
+import re
+from pathlib import Path
+from typing import Any, Iterator, NamedTuple, Optional
+
+# Regex constructs Oniguruma supports and `re` either rejects or interprets
+# differently. A grammar using any of these would invalidate this tokenizer.
+_FOREIGN_SYNTAX = {
+ "possessive quantifier": r"(?:[*+?}])\+",
+ "atomic group": r"\(\?>",
+ r"\G anchor": r"\\G",
+ r"\h horizontal space": r"\\[hH]",
+ "conditional": r"\(\?\(",
+ "named backreference": r"\\k<",
+ "variable-length lookbehind": r"\(\?<[=!][^)]*[*+]",
+}
+
+
+# Every key a TextMate grammar can put a regex under. `match`, `begin`, and
+# `end` are the three this tokenizer runs; an editor also runs the folding
+# markers, so they are checked for dialect even though nothing here reads them.
+_REGEX_KEYS = (
+ "match",
+ "begin",
+ "end",
+ "foldingStartMarker",
+ "foldingStopMarker",
+)
+
+
+class Token(NamedTuple):
+ """A scope assignment covering `[start, end)` of line `line`.
+
+ `scopes` runs outermost first, starting with the grammar's own scope name,
+ so the last entry is the most specific. A theme matches from there
+ outward, taking the first entry it has a color for.
+ """
+
+ line: int
+ start: int
+ end: int
+ scopes: list[str]
+
+
+class Rule(NamedTuple):
+ """An entry of the region stack: an open `begin`/`end`, or the grammar.
+
+ The grammar's own entry sits at the bottom with an `end_re` of `None`,
+ since nothing closes it.
+ """
+
+ scopes: list[str]
+ content_scopes: list[str]
+ end_re: Optional[re.Pattern[str]]
+ end_captures: dict[str, Any]
+ patterns: list[dict[str, Any]]
+
+
+class _RuleMatch(NamedTuple):
+ """A rule that matched, and where it matched.
+
+ `kind` names the grammar key the matching regex came from: `"end"` for
+ the open region's own `end`, in which case `pattern` is empty, or
+ `"match"` or `"begin"` for one of the rules in the region's `patterns`.
+ `match` is always a real match; it is what `kind` describes.
+ """
+
+ kind: str
+ pattern: dict[str, Any]
+ match: re.Match[str]
+
+
+class Grammar:
+ """A parsed TextMate grammar, ready to tokenize with."""
+
+ def __init__(self, raw: dict[str, Any]) -> None:
+ self.raw = raw
+ self.scope_name: str = raw.get("scopeName", "")
+ self._repository: dict[str, Any] = raw.get("repository", {})
+ self._compiled: dict[str, re.Pattern[str]] = {}
+
+ @staticmethod
+ def load(path: Path) -> "Grammar":
+ with path.open(encoding="utf-8") as f:
+ return Grammar(json.load(f))
+
+ def compile(self, regex: str) -> re.Pattern[str]:
+ compiled = self._compiled.get(regex)
+ if compiled is None:
+ compiled = self._compiled[regex] = re.compile(regex)
+ return compiled
+
+ def resolve(
+ self,
+ patterns: Optional[list[dict[str, Any]]],
+ active: frozenset[str] = frozenset(),
+ ) -> list[dict[str, Any]]:
+ """Flattens `include` directives into a list of concrete rules.
+
+ A `name` on the repository entry itself is deliberately dropped: VS
+ Code inlines an include-only rule's patterns without pushing its scope,
+ so honoring it here would disagree with the real tokenizer. `active`
+ breaks a cycle between mutually including entries.
+
+ An include of a different grammar, such as `source.cpp`, contributes
+ nothing: we tokenize Carbon, and embedded regions are left to whatever
+ scope their `contentName` assigns.
+ """
+ resolved: list[dict[str, Any]] = []
+ for pattern in patterns or []:
+ target = pattern.get("include")
+ if target is None:
+ resolved.append(pattern)
+ elif target.startswith("#"):
+ name = target[1:]
+ if name in active:
+ continue
+ entry = self._repository.get(name, {})
+ resolved.extend(
+ self.resolve(entry.get("patterns"), active | {name})
+ )
+ elif target == "$self":
+ resolved.extend(self.resolve(self.raw.get("patterns"), active))
+ return resolved
+
+ def initial_stack(self) -> list[Rule]:
+ base = [self.scope_name]
+ return [
+ Rule(base, base, None, {}, self.resolve(self.raw.get("patterns")))
+ ]
+
+ def all_regexes(self) -> Iterator[tuple[str, str]]:
+ """Yields every `(key, regex)` the grammar holds, for validation.
+
+ A grammar keeps regexes only under the keys in `_REGEX_KEYS`, so
+ finding them all means walking every dictionary in it and picking
+ those keys out. `key` says which one it was, which is all a message
+ needs to point at the right place.
+ """
+
+ def walk(node: Any) -> Iterator[tuple[str, str]]:
+ if isinstance(node, dict):
+ for key in _REGEX_KEYS:
+ value = node.get(key)
+ if isinstance(value, str):
+ yield key, value
+ for value in node.values():
+ yield from walk(value)
+ elif isinstance(node, list):
+ for value in node:
+ yield from walk(value)
+
+ yield from walk(self.raw)
+
+
+def check_regex_dialect(grammar: Grammar) -> list[str]:
+ """Returns a message per regex `re` cannot stand in for Oniguruma on."""
+ problems: list[str] = []
+ for key, regex in grammar.all_regexes():
+ # A `\N` backreference is substituted before compiling, so do the
+ # same here with an arbitrary value before validating.
+ probe = re.sub(r"\\\d", "x", regex)
+ try:
+ re.compile(probe)
+ except re.error as e:
+ problems.append(f"`{key}` regex does not compile: {regex}: {e}")
+ continue
+ for name, foreign in _FOREIGN_SYNTAX.items():
+ if re.search(foreign, regex):
+ problems.append(
+ f"`{key}` regex uses {name}, which `re` does not share "
+ f"with Oniguruma: {regex}"
+ )
+ return problems
+
+
+def _substitute_backrefs(regex: str, begin_match: re.Match[str]) -> str:
+ """Replaces `\\N` in an `end` with the text `begin` captured, as VS Code
+ does before compiling the `end` regex."""
+
+ def replace(ref: re.Match[str]) -> str:
+ return re.escape(begin_match.group(int(ref.group(1))) or "")
+
+ return re.sub(r"\\(\d)", replace, regex)
+
+
+def _append_token(
+ tokens: list[Token], linenum: int, start: int, end: int, scopes: list[str]
+) -> None:
+ """Appends a token, unless it would be empty.
+
+ Callers append the text between two things without first checking that
+ there is any, and about a quarter of the time there is none, so the check
+ lives here rather than at every call.
+ """
+ if end > start:
+ tokens.append(Token(linenum, start, end, scopes))
+
+
+def _append_capture_tokens(
+ tokens: list[Token],
+ linenum: int,
+ match: re.Match[str],
+ captures: dict[str, Any],
+ scopes: list[str],
+) -> None:
+ """Appends the tokens for one match, splitting it at its capture groups.
+
+ `captures` maps a group number, written as a string, to the scope the
+ group's text takes on top of `scopes`; group `"0"` is the whole match. The
+ tokens tile the match with no gaps: text not covered by a listed group is
+ still appended, under `scopes` alone.
+ """
+ if not captures:
+ _append_token(tokens, linenum, match.start(), match.end(), scopes)
+ return
+ pos = match.start()
+ for group in range((match.re.groups or 0) + 1):
+ spec = captures.get(str(group))
+ # Skip a group the grammar does not name, one that did not
+ # participate in the match (`start` is then -1), and one that lies
+ # behind text already appended, since tokens come out in order.
+ if spec is None or match.start(group) < pos:
+ continue
+ if match.start(group) == match.end(group):
+ continue
+ _append_token(tokens, linenum, pos, match.start(group), scopes)
+ _append_token(
+ tokens,
+ linenum,
+ match.start(group),
+ match.end(group),
+ scopes + [spec["name"]],
+ )
+ pos = match.end(group)
+ _append_token(tokens, linenum, pos, match.end(), scopes)
+
+
+def _push_scope(
+ scopes: list[str], pattern: dict[str, Any], key: str
+) -> list[str]:
+ """Adds the scope `pattern[key]` names, if it names one."""
+ name = pattern.get(key)
+ return scopes + [name] if name is not None else scopes
+
+
+def _find_earliest(
+ grammar: Grammar, rule: Rule, text: str, pos: int
+) -> Optional[_RuleMatch]:
+ """Finds which of `rule`'s regexes matches soonest at or after `pos`.
+
+ The open region's own `end` is tried first and ties are broken towards
+ whatever was tried earlier, which gives TextMate's two ordering rules: a
+ region that can close here closes here, and otherwise the first rule
+ listed in the grammar wins.
+ """
+ best: Optional[_RuleMatch] = None
+ if rule.end_re is not None:
+ match = rule.end_re.search(text, pos)
+ if match is not None:
+ best = _RuleMatch("end", {}, match)
+ for pattern in rule.patterns:
+ # The key a regex came from is what the rule does with it, so take
+ # both from the same lookup.
+ if (regex := pattern.get("match")) is not None:
+ kind = "match"
+ elif (regex := pattern.get("begin")) is not None:
+ kind = "begin"
+ else:
+ continue
+ match = grammar.compile(regex).search(text, pos)
+ if match is None:
+ continue
+ if best is not None and match.start() >= best.match.start():
+ continue
+ best = _RuleMatch(kind, pattern, match)
+ return best
+
+
+def tokenize_line(
+ grammar: Grammar, linenum: int, text: str, stack: list[Rule]
+) -> tuple[list[Token], list[Rule]]:
+ """Tokenizes one line, given the regions open when it starts.
+
+ `text` must carry its trailing newline: VS Code tokenizes a line together
+ with its terminator, so that newline gets a token of its own, and carrying
+ it is what keeps the two token streams identical. `stack` is the chain of
+ `begin`/`end` regions open at the start of the line, outermost first and
+ never empty: its first entry is the grammar itself. The stack returned is
+ the one open at the start of the next line, which is how a region spans
+ lines.
+ """
+ tokens: list[Token] = []
+ pos = 0
+ # Counts region transitions that consumed nothing, which make no progress.
+ stalls = 0
+ while pos <= len(text):
+ rule = stack[-1]
+ found = _find_earliest(grammar, rule, text, pos)
+ if found is None:
+ break
+ # Text before the match belongs to the region containing it.
+ _append_token(
+ tokens, linenum, pos, found.match.start(), rule.content_scopes
+ )
+ if found.kind == "end":
+ _append_capture_tokens(
+ tokens, linenum, found.match, rule.end_captures, rule.scopes
+ )
+ stack = stack[:-1]
+ else:
+ scopes = _push_scope(rule.content_scopes, found.pattern, "name")
+ captures = "captures" if found.kind == "match" else "beginCaptures"
+ _append_capture_tokens(
+ tokens,
+ linenum,
+ found.match,
+ found.pattern.get(captures, {}),
+ scopes,
+ )
+ if found.kind == "begin":
+ stack = stack + [
+ Rule(
+ scopes,
+ _push_scope(scopes, found.pattern, "contentName"),
+ grammar.compile(
+ _substitute_backrefs(
+ found.pattern["end"], found.match
+ )
+ ),
+ found.pattern.get("endCaptures", {}),
+ grammar.resolve(found.pattern.get("patterns")),
+ )
+ ]
+
+ if found.match.end() > found.match.start():
+ pos = found.match.end()
+ elif found.kind == "match":
+ # A `match` that consumed nothing would match there forever, so
+ # step over a character.
+ pos = found.match.start() + 1
+ else:
+ # Entering or leaving a region is progress in itself, and the
+ # character stays available to the region now on top: a lookahead
+ # `end` such as `(?=[\[\(])` depends on that.
+ pos = found.match.start()
+ stalls += 1
+ # The bound is arbitrary; it only has to exceed the transitions a
+ # real line could ask for. Reaching it means a rule opens and
+ # closes forever, which is a grammar bug: say so rather than
+ # silently truncating the line.
+ if stalls > len(text) + 64:
+ raise RuntimeError(
+ f"line {linenum} stopped making progress at offset {pos}: "
+ f"rule {found.pattern!r} neither consumes nor terminates"
+ )
+ # Whatever is left over, including the newline, belongs to the region the
+ # line ends inside of.
+ _append_token(tokens, linenum, pos, len(text), stack[-1].content_scopes)
+ return tokens, stack
+
+
+def split_lines(text: str) -> list[str]:
+ """Splits a file into lines the way an editor numbers them.
+
+ Terminators are not kept on the lines, and a file ending in a newline
+ yields an empty final line, which is the blank line an editor shows there.
+
+ Not `str.splitlines`: that also breaks on `\\f`, `\\v`, and a handful of
+ Unicode separators, which a TextMate grammar sees as ordinary characters
+ within a line, and it drops that final empty line.
+ """
+ return text.split("\n")
+
+
+def tokenize(grammar: Grammar, text: str) -> Iterator[Token]:
+ """Tokenizes a whole file."""
+ stack = grammar.initial_stack()
+ for linenum, line in enumerate(split_lines(text)):
+ tokens, stack = tokenize_line(grammar, linenum, line + "\n", stack)
+ yield from tokens