Skip to content

Commit 0b8aa89

Browse files
authored
gh-153569: Simplify tokenizer state (#158124)
* Use diagnostic line for missing-brace errors * Keep implicit prepared newline metadata in the reader * Reuse parsed formatted string kind in lexer * Transfer owned tokenizer encoding strings * Remove unused tokenizer state and redundant branches * Avoid rescanning partial decoded tokenizer lines * Reuse prepared tokenizer line terminators for CR detection * Reuse interactive tokenizer chunks without carriage returns * Release tokenizer constructor resources on allocation failure * Narrow interactive newline normalization wrapper * Close duplicated tokenizer descriptors when fdopen fails * Preserve cached tokenizer lines when newline allocation fails * Collect cycles involving tokenizer readline callbacks * Preserve buffered text when decoding source files * Clear the transferred syntax error line before cleanup * Propagate tokenizer column conversion failures * Allocate parser repetition buffers on first match * Allocate type comment storage only when needed * Reuse decoded multiline text when counting token columns * Skip empty decoded tokenizer chunks * Check parser array growth before allocating * Name tokenizer bounds and derive byte lengths from their data * Set the tokenizer exception when file input size overflows * gh-153569: Send exact line endings in the REPL test * Share parser repetition buffer growth * Fold Unicode input coverage into the existing REPL test * Keep parser array growth policy private * Drop obsolete bitmap stress from source discard test * Preserve input failures while scanning string literals * fixup! Preserve input failures while scanning string literals * fixup! fixup! Preserve input failures while scanning string literals * gh-153569: Remove assertion-only lexer bookkeeping
1 parent 9112dae commit 0b8aa89

23 files changed

Lines changed: 775 additions & 1308 deletions

‎Lib/test/test_fstring.py‎

Lines changed: 14 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -645,6 +645,20 @@ def test_unclosed_multiline_replacement_field(self):
645645
f"{prefix[-1]}-string: expecting '}}' to close '{{' "
646646
f"on line {lineno}")
647647

648+
def test_unclosed_replacement_field_quote_line(self):
649+
for prefix in ('f', 't', 'rf', 'rt'):
650+
for quote in ('"', "'"):
651+
triple = quote * 3
652+
for suffix in ('', '\nx'):
653+
source = prefix + triple + '{1' + triple + suffix
654+
with self.subTest(source=source):
655+
with self.assertRaises(SyntaxError) as cm:
656+
compile(source, '<test>', 'exec')
657+
self.assertEqual(
658+
cm.exception.msg,
659+
f"{prefix[-1]}-string: expecting '}}'")
660+
self.assertEqual(cm.exception.lineno, 1)
661+
648662
@unittest.skipIf(support.is_wasi, "exhausts limited stack on WASI")
649663
def test_mismatched_parens(self):
650664
self.assertAllRaise(SyntaxError, r"closing parenthesis '\}' "

‎Lib/test/test_repl.py‎

Lines changed: 9 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -187,16 +187,19 @@ def read_until(marker, start=0):
187187
def test_lexer_buffer_realloc_with_null_start(self):
188188
# gh-144759: NULL pointer arithmetic when the lexer buffer grows
189189
# while parsing long input.
190-
long_value = "a" * 2000
190+
long_value = "é漢" * 2000
191191
user_input = dedent(f"""\
192192
x = f'{{{long_value!r}}}'
193193
print(x)
194194
""")
195-
p = spawn_repl()
196-
p.stdin.write(user_input)
197-
output = kill_python(p)
198-
self.assertEqual(p.returncode, 0)
199-
self.assertIn(long_value, output)
195+
for newline in ("\n", "\r\n"):
196+
with self.subTest(newline=newline):
197+
p = spawn_repl(encoding="utf-8")
198+
# Bypass Windows text-mode translation of CRLF to CRCRLF.
199+
p.stdin.buffer.write(user_input.replace("\n", newline).encode("utf-8"))
200+
output = kill_python(p)
201+
self.assertEqual(p.returncode, 0)
202+
self.assertIn(long_value, output)
200203

201204
@cpython_only
202205
def test_multiline_fstring_source_reallocation(self):

‎Lib/test/test_source_encoding.py‎

Lines changed: 29 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -29,6 +29,30 @@ def test_compilestring(self):
2929
exec(c, d)
3030
self.assertEqual(d['u'], '\xf3')
3131

32+
def test_compilestring_line_endings(self):
33+
for newline in ('\n', '\r', '\r\n', '\r\r\n'):
34+
expected = 'é\n\ntext' if newline == '\r\r\n' else 'é\ntext'
35+
for suffix in ('', newline):
36+
for mode in ('exec', 'eval'):
37+
source = f"'''é{newline}text'''{suffix}"
38+
if mode == 'exec':
39+
source = 'value = ' + source
40+
for encoding in (None, 'utf-8', 'latin-1'):
41+
with self.subTest(newline=newline, suffix=suffix,
42+
mode=mode, encoding=encoding):
43+
input = source
44+
if encoding is not None:
45+
input = (f'# coding: {encoding}{newline}'
46+
+ source).encode(encoding)
47+
code = compile(input, '<test>', mode)
48+
if mode == 'exec':
49+
namespace = {}
50+
exec(code, namespace)
51+
value = namespace['value']
52+
else:
53+
value = eval(code)
54+
self.assertEqual(value, expected)
55+
3256
def test_issue2301(self):
3357
try:
3458
compile(b"# coding: cp932\nprint '\x94\x4e'", "dummy", "exec")
@@ -135,6 +159,11 @@ def test_stateful_file_decoder_spans_lines(self):
135159
)
136160
self._assert_python_file_ok(source)
137161

162+
@support.requires_subprocess()
163+
def test_stateful_file_decoder_preserves_buffered_text(self):
164+
source = b"# coding: hz\nx~\ny = 1\nassert xy == 1\n"
165+
self._assert_python_file_ok(source)
166+
138167
@support.requires_subprocess()
139168
def test_stateful_file_decoder_finalizes_before_implicit_newline(self):
140169
source = b"# coding: hz\n# ~{1dA?"

‎Lib/test/test_tokenize.py‎

Lines changed: 33 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -8,6 +8,7 @@
88
import token
99
import tokenize
1010
import unittest
11+
import weakref
1112
from io import BytesIO, StringIO
1213
from textwrap import dedent
1314
from unittest import TestCase, mock
@@ -2268,6 +2269,18 @@ def check_tokenize(self, s, expected):
22682269
)
22692270
self.assertEqual(result, expected.rstrip().splitlines())
22702271

2272+
def test_readline_reference_cycle(self):
2273+
class Readline:
2274+
def __call__(self):
2275+
return ""
2276+
2277+
readline = Readline()
2278+
readline.iterator = _tokenize.TokenizerIter(readline, extra_tokens=True)
2279+
ref = weakref.ref(readline)
2280+
del readline
2281+
support.gc_collect()
2282+
self.assertIsNone(ref())
2283+
22712284
def test_encoding(self):
22722285
def readline(encoding):
22732286
yield "1+1".encode(encoding)
@@ -2341,6 +2354,20 @@ def test_utf8_decoder_spans_readline_calls(self):
23412354
tokenize.TokenInfo(token.ENDMARKER, "", (2, 0), (2, 0), ""),
23422355
])
23432356

2357+
def test_utf8_decoder_spans_many_readline_calls(self):
2358+
for prefix in (b"", b"previous\n"):
2359+
with self.subTest(prefix=prefix):
2360+
chunks = ([prefix + b"x\xc3"] + [b"\xa9\xc3"] * 100
2361+
+ [b"\xa9\n", b"z\xc3", b"\xa9\n", b""])
2362+
source = b"".join(chunks)
2363+
expected = list(_tokenize.TokenizerIter(
2364+
BytesIO(source).readline, encoding="utf-8", extra_tokens=True
2365+
))
2366+
tokens = list(_tokenize.TokenizerIter(
2367+
iter(chunks).__next__, encoding="utf-8", extra_tokens=True
2368+
))
2369+
self.assertEqual(tokens, expected)
2370+
23442371
def test_utf8_decoder_replaces_incomplete_input_at_eof(self):
23452372
expected = [
23462373
tokenize.TokenInfo(token.NAME, "x�", (1, 0), (1, 2), "x�"),
@@ -2396,6 +2423,12 @@ def test_multiline_readline_chunk_with_unterminated_tail(self):
23962423
)
23972424
self.assertEqual(readline.call_count, 2)
23982425

2426+
def test_readline_memory_error_in_string(self):
2427+
readline = mock.Mock(side_effect=['"""first\n', MemoryError])
2428+
iterator = _tokenize.TokenizerIter(readline, extra_tokens=True)
2429+
with self.assertRaises(MemoryError):
2430+
next(iterator)
2431+
23992432
def test_readline_callback_is_not_read_ahead(self):
24002433
readline = mock.Mock(side_effect=["x\n", "y\n", ""])
24012434
iterator = _tokenize.TokenizerIter(readline, extra_tokens=True)

‎Lib/test/test_type_comments.py‎

Lines changed: 10 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -338,6 +338,16 @@ def test_ignores(self):
338338
tree = self.classic_parse(ignores)
339339
self.assertEqual(tree.type_ignores, [])
340340

341+
def test_many_ignores(self):
342+
comment_count = 25
343+
tags = [f"[tag_{index}]" for index in range(comment_count)]
344+
source = "".join(f"pass # type: ignore{tag}\n" for tag in tags)
345+
for tree in self.parse_all(source):
346+
self.assertEqual(
347+
[(item.lineno, item.tag) for item in tree.type_ignores],
348+
list(enumerate(tags, start=1)),
349+
)
350+
341351
def test_longargs(self):
342352
for tree in self.parse_all(longargs, minver=8):
343353
for t in tree.body:
Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,4 @@
1+
Correct missing-brace error messages for multiline f-strings and t-strings.
2+
Improve tokenization of UTF-8 input split across many readline calls.
3+
Preserve source order when reading files with stateful encodings.
4+
Preserve input errors when scanning string literals.

‎Modules/_testinternalcapi/tokenizer.c‎

Lines changed: 41 additions & 26 deletions
Original file line numberDiff line numberDiff line change
@@ -38,6 +38,12 @@ static PyObject *
3838
test_tokenizer_source(PyObject *Py_UNUSED(module),
3939
PyObject *Py_UNUSED(args))
4040
{
41+
const char multiple_lines[] = "a\nb\n";
42+
const char first_line[] = "alpha\n";
43+
const char second_line[] = "\xce\xb2\n";
44+
const char expected[] = "alpha\n\xce\xb2\n";
45+
const char tail[] = "tail";
46+
const char terminated_line[] = "x\n";
4147
_PyTok_SourceText source;
4248
_PyTok_SourceInit(&source);
4349

@@ -46,23 +52,19 @@ test_tokenizer_source(PyObject *Py_UNUSED(module),
4652
}
4753

4854
if (check_system_error(
49-
_PyTok_SourceAppendLine(&source, "", 0, 0) < 0,
55+
_PyTok_SourceAppendLine(&source, "", 0) < 0,
5056
"accepted empty source line") < 0 ||
5157
check_system_error(
52-
_PyTok_SourceAppendLine(&source, "a\nb\n", 4, 0) < 0,
58+
_PyTok_SourceAppendLine(
59+
&source, multiple_lines, sizeof(multiple_lines) - 1) < 0,
5360
"accepted multiple source lines") < 0 ||
54-
check_system_error(
55-
_PyTok_SourceAppendLine(&source, "a", 1, 1) < 0,
56-
"accepted missing implicit newline") < 0 ||
5761
check(_PyTok_SourceAppendLine(
58-
&source, "alpha\n", 6, 0) == 0,
62+
&source, first_line, sizeof(first_line) - 1) == 0,
5963
"wrong first source offset") < 0 ||
6064
check(_PyTok_SourceAppendLine(
61-
&source, "\xce\xb2\n", 3, 1) == 6,
62-
"wrong second source offset") < 0 ||
63-
check(!_PyTok_SourceLineIsImplicit(&source, 1) &&
64-
_PyTok_SourceLineIsImplicit(&source, 2),
65-
"wrong implicit newline flags") < 0) {
65+
&source, second_line, sizeof(second_line) - 1) ==
66+
(Py_ssize_t)sizeof(first_line) - 1,
67+
"wrong second source offset") < 0) {
6668
goto error;
6769
}
6870

@@ -74,16 +76,17 @@ test_tokenizer_source(PyObject *Py_UNUSED(module),
7476
goto error;
7577
}
7678

77-
if (check(source.len == 9 &&
78-
memcmp(source.bytes, "alpha\n\xce\xb2\n", 10) == 0,
79+
if (check(source.len == (Py_ssize_t)sizeof(expected) - 1 &&
80+
memcmp(source.bytes, expected, sizeof(expected)) == 0,
7981
"wrong source contents") < 0) {
8082
goto error;
8183
}
8284

8385
_PyTok_SourceClear(&source);
84-
if (_PyTok_SourceAppendLine(&source, "tail", 4, 0) < 0 ||
86+
if (_PyTok_SourceAppendLine(&source, tail, sizeof(tail) - 1) < 0 ||
8587
check_system_error(
86-
_PyTok_SourceAppendLine(&source, "x\n", 2, 0) < 0,
88+
_PyTok_SourceAppendLine(
89+
&source, terminated_line, sizeof(terminated_line) - 1) < 0,
8790
"appended after unterminated source line") < 0) {
8891
goto error;
8992
}
@@ -104,27 +107,35 @@ static PyObject *
104107
test_tokenizer_source_discard(PyObject *Py_UNUSED(module),
105108
PyObject *Py_UNUSED(args))
106109
{
110+
enum { LINE_COUNT = 2 };
111+
const char first_line[] = "x\n";
112+
const char second_line[] = "y\n";
113+
const char tail[] = "tail";
114+
const char final_line[] = "z\n";
115+
const _PyTok_Off first_batch_len = LINE_COUNT * (sizeof(first_line) - 1);
116+
const _PyTok_Off second_batch_len = LINE_COUNT * (sizeof(second_line) - 1);
117+
const _PyTok_Off discarded_len = first_batch_len + second_batch_len;
107118
_PyTok_SourceText source;
108119
_PyTok_SourceInit(&source);
109-
for (int i = 0; i < 260; i++) {
110-
if (_PyTok_SourceAppendLine(&source, "x\n", 2, 1) < 0) {
120+
for (int i = 0; i < LINE_COUNT; i++) {
121+
if (_PyTok_SourceAppendLine(&source, first_line, sizeof(first_line) - 1) < 0) {
111122
goto error;
112123
}
113124
}
114125
char *bytes = source.bytes;
115126
_PyTok_Off capacity = source.cap;
116127
_PyTok_SourceDiscard(&source);
117-
if (check(source.base_offset == 520 && source.len == 0 &&
128+
if (check(source.base_offset == first_batch_len && source.len == 0 &&
118129
source.nlines == 0 && source.bytes == bytes &&
119130
source.cap == capacity && source.bytes[0] == '\0',
120131
"discard did not preserve source allocation") < 0) {
121132
goto error;
122133
}
123-
for (int i = 0; i < 260; i++) {
124-
if (check(_PyTok_SourceAppendLine(&source, "y\n", 2, 0) == 520 + 2 * i,
125-
"wrong source offset after discard") < 0 ||
126-
check(!_PyTok_SourceLineIsImplicit(&source, i + 1),
127-
"discard preserved implicit newline flag") < 0) {
134+
for (int i = 0; i < LINE_COUNT; i++) {
135+
if (check(_PyTok_SourceAppendLine(
136+
&source, second_line, sizeof(second_line) - 1) ==
137+
first_batch_len + ((Py_ssize_t)sizeof(second_line) - 1) * i,
138+
"wrong source offset after discard") < 0) {
128139
goto error;
129140
}
130141
}
@@ -133,18 +144,22 @@ test_tokenizer_source_discard(PyObject *Py_UNUSED(module),
133144
goto error;
134145
}
135146
_PyTok_SourceDiscard(&source);
136-
if (check(_PyTok_SourceAppendLine(&source, "tail", 4, 0) == 1040,
147+
if (check(_PyTok_SourceAppendLine(
148+
&source, tail, sizeof(tail) - 1) == discarded_len,
137149
"wrong source offset after repeated discard") < 0) {
138150
goto error;
139151
}
140152
_PyTok_SourceDiscard(&source);
141-
if (check(_PyTok_SourceAppendLine(&source, "z\n", 2, 0) == 1044,
153+
if (check(_PyTok_SourceAppendLine(
154+
&source, final_line, sizeof(final_line) - 1) ==
155+
discarded_len + (Py_ssize_t)sizeof(tail) - 1,
142156
"cannot append after discarding unterminated line") < 0) {
143157
goto error;
144158
}
145159
_PyTok_SourceDiscard(&source);
146160
source.base_offset = PY_SSIZE_T_MAX - 1;
147-
if (check(_PyTok_SourceAppendLine(&source, "z\n", 2, 0) < 0 &&
161+
if (check(_PyTok_SourceAppendLine(
162+
&source, final_line, sizeof(final_line) - 1) < 0 &&
148163
PyErr_ExceptionMatches(PyExc_MemoryError),
149164
"accepted overflowing logical source offset") < 0) {
150165
goto error;

‎Parser/lexer/lexer.c‎

Lines changed: 4 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -314,7 +314,10 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
314314

315315
// Handle valid f or t string creation:
316316
if (saw_f || saw_t) {
317-
return _PyLexer_scan_fstring_start(tok, token, c);
317+
ftstring_kind kind = saw_t
318+
? (saw_r ? RAW_TSTRING : TSTRING)
319+
: (saw_r ? RAW_FSTRING : FSTRING);
320+
return _PyLexer_scan_fstring_start(tok, token, c, kind);
318321
}
319322
return _PyLexer_scan_string(tok, token, c);
320323
}

‎Parser/lexer/lexer_internal.h‎

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -66,7 +66,8 @@ int _PyLexer_close_ftstring_expr(
6666
void _PyLexer_mark_ftstring_debug(struct tok_state *, ftstring_state *);
6767
int _PyLexer_check_string_prefixes(struct tok_state *, int, int, int, int, int);
6868
int _PyLexer_scan_number(struct tok_state *, struct token *, int, int);
69-
int _PyLexer_scan_fstring_start(struct tok_state *, struct token *, int);
69+
int _PyLexer_scan_fstring_start(
70+
struct tok_state *, struct token *, int, ftstring_kind);
7071
int _PyLexer_scan_string(struct tok_state *, struct token *, int);
7172
int _PyLexer_get_normal(struct tok_state *, ftstring_state *, struct token *);
7273
int _PyLexer_get_ftstring(struct tok_state *, ftstring_state *, struct token *);

0 commit comments

Comments
 (0)