Files
carbon-lang/toolchain/lex/testdata/fail_char_literals_bad_encoding.carbon
T
Richard Smith 7727c62880 Enforce a couple of char literal restrictions from #1964: (#5960)
* `\x` escapes are not permitted in character literals
* ASCII control characters (U+0000 .. U+001F) are not permitted in
character literals unless specified with escape sequences.
2025-08-15 00:45:16 +00:00

118 lines
6.2 KiB
Plaintext

// Part of the Carbon Language project, under the Apache License v2.0 with LLVM
// Exceptions. See /LICENSE for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
// AUTOUPDATE
// TIP: To test this file alone, run:
// TIP: bazel test //toolchain/testing:file_test --test_arg=--file_tests=toolchain/lex/testdata/fail_char_literals_bad_encoding.carbon
// TIP: To dump output, run:
// TIP: bazel run //toolchain/testing:file_test -- --dump_output --file_tests=toolchain/lex/testdata/fail_char_literals_bad_encoding.carbon
// CHECK:STDOUT: - filename: fail_char_literals_bad_encoding.carbon
// CHECK:STDOUT: tokens:
// Be careful when operating on this file: it contains invalid UTF-8 sequences
// and Unicode control characters, and text editors are likely to corrupt it.
// The next literal contains 0x00.
// CHECK:STDERR: fail_char_literals_bad_encoding.carbon:[[@LINE+4]]:1: error: control character in character literal; specify as escape sequence `\u{00}` [CharLiteralControlCharacter]
// CHECK:STDERR: ''
// CHECK:STDERR: ^
// CHECK:STDERR:
''
// CHECK:STDOUT: - { index: 1, kind: "Error", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "'\x00'", has_leading_space: true }
// The next literal contains 0x01.
// CHECK:STDERR: fail_char_literals_bad_encoding.carbon:[[@LINE+4]]:1: error: control character in character literal; specify as escape sequence `\u{01}` [CharLiteralControlCharacter]
// CHECK:STDERR: ''
// CHECK:STDERR: ^
// CHECK:STDERR:
''
// CHECK:STDOUT: - { index: 2, kind: "Error", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "'\x01'", has_leading_space: true }
// The next literal contains 0x1F.
// CHECK:STDERR: fail_char_literals_bad_encoding.carbon:[[@LINE+4]]:1: error: control character in character literal; specify as escape sequence `\u{1F}` [CharLiteralControlCharacter]
// CHECK:STDERR: ''
// CHECK:STDERR: ^
// CHECK:STDERR:
''
// CHECK:STDOUT: - { index: 3, kind: "Error", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "'\x1F'", has_leading_space: true }
// The next literal contains 0x7F.
// CHECK:STDERR: fail_char_literals_bad_encoding.carbon:[[@LINE+4]]:1: error: control character in character literal; specify as escape sequence `\u{7F}` [CharLiteralControlCharacter]
// CHECK:STDERR: ''
// CHECK:STDERR: ^
// CHECK:STDERR:
''
// CHECK:STDOUT: - { index: 4, kind: "Error", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "'\x7F'", has_leading_space: true }
// The next literal contains 0xC2 0x80, which is U+0080.
// CHECK:STDERR: fail_char_literals_bad_encoding.carbon:[[@LINE+4]]:1: error: control character in character literal; specify as escape sequence `\u{80}` [CharLiteralControlCharacter]
// CHECK:STDERR: '€'
// CHECK:STDERR: ^
// CHECK:STDERR:
'€'
// CHECK:STDOUT: - { index: 5, kind: "Error", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "'\xC2\x80'", has_leading_space: true }
// The next literal contains 0xC2 0x9F, which is U+009F.
// CHECK:STDERR: fail_char_literals_bad_encoding.carbon:[[@LINE+4]]:1: error: control character in character literal; specify as escape sequence `\u{9F}` [CharLiteralControlCharacter]
// CHECK:STDERR: 'Ÿ'
// CHECK:STDERR: ^
// CHECK:STDERR:
'Ÿ'
// CHECK:STDOUT: - { index: 6, kind: "Error", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "'\xC2\x9F'", has_leading_space: true }
// The next literal contains 0xC3, which is invalid UTF-8 due to having a lead
// byte with no trail byte.
// CHECK:STDERR: fail_char_literals_bad_encoding.carbon:[[@LINE+4]]:1: error: incomplete UTF-8 [CharLiteralUnderflow]
// CHECK:STDERR: 'Ã'
// CHECK:STDERR: ^
// CHECK:STDERR:
'Ã'
// CHECK:STDOUT: - { index: 7, kind: "Error", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "'\xC3'", has_leading_space: true }
// The next literal contains 0xC3, which is invalid UTF-8 due to having two lead
// bytes in a row.
// CHECK:STDERR: fail_char_literals_bad_encoding.carbon:[[@LINE+4]]:1: error: invalid UTF-8 character [CharLiteralInvalidUTF8]
// CHECK:STDERR: 'Ãÿ'
// CHECK:STDERR: ^
// CHECK:STDERR:
'Ãÿ'
// CHECK:STDOUT: - { index: 8, kind: "Error", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "'\xC3\xFF'", has_leading_space: true }
// The next literal contains 0xFF 0x80 0x80 0x80 0x80 0x80 0x80 0x80, which is
// invalid UTF-8 due to being too large.
// CHECK:STDERR: fail_char_literals_bad_encoding.carbon:[[@LINE+4]]:1: error: invalid UTF-8 character [CharLiteralInvalidUTF8]
// CHECK:STDERR: 'ÿ€€€€€€€'
// CHECK:STDERR: ^
// CHECK:STDERR:
'ÿ€€€€€€€'
// CHECK:STDOUT: - { index: 9, kind: "Error", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "'\xFF\x80\x80\x80\x80\x80\x80\x80'", has_leading_space: true }
// The next literal contains 0xED 0xA0 0x80, which would be U+D800 but is
// invalid UTF-8 due to encoding a high surrogate value.
// CHECK:STDERR: fail_char_literals_bad_encoding.carbon:[[@LINE+4]]:1: error: invalid UTF-8 character [CharLiteralInvalidUTF8]
// CHECK:STDERR: 'í €'
// CHECK:STDERR: ^
// CHECK:STDERR:
'í €'
// CHECK:STDOUT: - { index: 10, kind: "Error", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "'\xED\xA0\x80'", has_leading_space: true }
// The next literal contains 0xED 0xBF 0xBF, which would be U+DFFF but is
// invalid UTF-8 due to encoding a low surrogate value.
// CHECK:STDERR: fail_char_literals_bad_encoding.carbon:[[@LINE+4]]:1: error: invalid UTF-8 character [CharLiteralInvalidUTF8]
// CHECK:STDERR: 'í¿¿'
// CHECK:STDERR: ^
// CHECK:STDERR:
'í¿¿'
// CHECK:STDOUT: - { index: 11, kind: "Error", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "'\xED\xBF\xBF'", has_leading_space: true }
// The next literal contains 0xED 0xA0 0xBD 0xED 0xBA 0x80. This would be
// U+D83D U+DE80, but those are both surrogates. Treated as UTF-16, that
// would in turn produce U+1F680. But of course we don't do that.
// CHECK:STDERR: fail_char_literals_bad_encoding.carbon:[[@LINE+4]]:1: error: invalid UTF-8 character [CharLiteralInvalidUTF8]
// CHECK:STDERR: '🚀'
// CHECK:STDERR: ^
// CHECK:STDERR:
'🚀'
// CHECK:STDOUT: - { index: 12, kind: "Error", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "'\xED\xA0\xBD\xED\xBA\x80'", has_leading_space: true }