Skip to content

Commit 28e15b4

Browse files
committed
gh-153569: report tokenizer diagnostics without rewinding the scanner
1 parent 68f376e commit 28e15b4

9 files changed

Lines changed: 117 additions & 46 deletions

File tree

‎Lib/test/test_codeop.py‎

Lines changed: 11 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -113,6 +113,17 @@ def test_valid(self, compiler):
113113
av("def f():\n pass\n#foo\n")
114114
av("@a.b.c\ndef f():\n pass\n")
115115

116+
@subTests('symbol', ('single', 'exec'))
117+
@subTests('prefix', ('', 'f', 't'))
118+
def test_incomplete_string_diagnostics(self, symbol, prefix):
119+
opening = f' á = {prefix}"""first\n'
120+
source = 'if True:\n' + opening + 'second'
121+
with self.assertRaises(_IncompleteInputError) as cm:
122+
Compile()(source, '<input>', symbol)
123+
text = opening + 'second' + ('\n' if symbol == 'exec' else '')
124+
self.assertEqual(cm.exception.args, (
125+
'incomplete input', ('<input>', 2, 9, text, 2, -1)))
126+
116127
@subTests('compiler', COMPILERS)
117128
def test_incomplete(self, compiler):
118129
ai = functools.partial(self.assertIncomplete, compiler=compiler)

‎Lib/test/test_source_encoding.py‎

Lines changed: 22 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -3,8 +3,8 @@
33
import unittest
44
from test import support
55
from test.support import script_helper
6-
from test.support.os_helper import TESTFN, unlink, rmtree
7-
from test.support.import_helper import unload
6+
from test.support.os_helper import TESTFN, TESTFN_ASCII, unlink, rmtree
7+
from test.support.import_helper import import_module, unload
88
import importlib
99
import os
1010
import sys
@@ -83,12 +83,30 @@ def test_truncated_utf8_at_eof(self):
8383
self.assertRaises(SyntaxError, compile, seq, '<test>', 'exec')
8484

8585
def test_invalid_utf8_offset_after_non_ascii(self):
86+
for name in ('é', 'éé', '𝒜'):
87+
with self.subTest(name=name):
88+
source = ('x = ' + name).encode() + b'\xff\n'
89+
with self.assertRaises(SyntaxError) as caught:
90+
compile(source, '<test>', 'exec')
91+
error = caught.exception
92+
self.assertEqual(
93+
(error.lineno, error.offset, error.end_lineno, error.end_offset),
94+
(1, 5 + len(name), 1, 5 + len(name)),
95+
)
96+
97+
@support.cpython_only
98+
def test_invalid_utf8_file_offset_after_non_ascii(self):
99+
_testcapi = import_module('_testcapi')
100+
self.addCleanup(unlink, TESTFN_ASCII)
101+
with open(TESTFN_ASCII, 'wb') as f:
102+
f.write(b'\nx = \xc3\xa9\xc3\xa9\xff\n')
86103
with self.assertRaises(SyntaxError) as caught:
87-
compile(b"x = \xc3\xa9\xff\n", "<test>", "exec")
104+
_testcapi.run_file(
105+
os.fsencode(TESTFN_ASCII), _testcapi.Py_file_input, {})
88106
error = caught.exception
89107
self.assertEqual(
90108
(error.lineno, error.offset, error.end_lineno, error.end_offset),
91-
(1, 6, 1, 6),
109+
(2, 7, 2, 7),
92110
)
93111

94112
def test_long_bom_conflict_message_is_not_truncated(self):

‎Lib/test/test_tstring.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -323,6 +323,8 @@ def test_nested_templates(self):
323323

324324
def test_syntax_errors(self):
325325
for case, err in (
326+
('t"""{(\n1\n)}\ntail', "unterminated triple-quoted t-string literal"),
327+
('f"""{(\n1\n)}\ntail', "unterminated triple-quoted f-string literal"),
326328
("t'", "unterminated t-string literal"),
327329
("t'''", "unterminated triple-quoted t-string literal"),
328330
("t''''", "unterminated triple-quoted t-string literal"),

‎Parser/lexer/lexer.c‎

Lines changed: 10 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -89,6 +89,7 @@ verify_identifier(struct tok_state *tok)
8989
assert(PyUnicode_GET_LENGTH(s) > 0);
9090
if (invalid < PyUnicode_GET_LENGTH(s)) {
9191
Py_UCS4 ch = PyUnicode_READ_CHAR(s, invalid);
92+
const char *error_cursor = tok->cur;
9293
if (invalid + 1 < PyUnicode_GET_LENGTH(s)) {
9394
/* Determine the offset in UTF-8 encoded input */
9495
Py_SETREF(s, PyUnicode_Substring(s, 0, invalid + 1));
@@ -99,14 +100,20 @@ verify_identifier(struct tok_state *tok)
99100
tok->done = E_ERROR;
100101
return 0;
101102
}
102-
tok->cur = tok->start + PyBytes_GET_SIZE(s);
103+
error_cursor = tok->start + PyBytes_GET_SIZE(s);
103104
}
104105
Py_DECREF(s);
105106
if (Py_UNICODE_ISPRINTABLE(ch)) {
106-
_PyTokenizer_syntaxerror(tok, "invalid character '%c' (U+%04X)", ch, ch);
107+
_PyTokenizer_syntaxerror_at(
108+
tok, tok->line_start,
109+
error_cursor - tok->line_start, tok->lineno, -1, -1,
110+
"invalid character '%c' (U+%04X)", ch, ch);
107111
}
108112
else {
109-
_PyTokenizer_syntaxerror(tok, "invalid non-printable character U+%04X", ch);
113+
_PyTokenizer_syntaxerror_at(
114+
tok, tok->line_start,
115+
error_cursor - tok->line_start, tok->lineno, -1, -1,
116+
"invalid non-printable character U+%04X", ch);
110117
}
111118
return 0;
112119
}

‎Parser/lexer/state.h‎

Lines changed: 9 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -72,6 +72,14 @@ typedef struct {
7272
indentation_level stack[MAXINDENT];
7373
} lexer_layout_state;
7474

75+
/* Supplemental source context for a terminal error. location is the reporting
76+
cursor, independent of the scanner cursor; lineno == 0 means absent.
77+
The text span may cover multiple physical lines. */
78+
typedef struct {
79+
_PyTok_Loc location;
80+
_PyTok_Span text_span;
81+
} _PyTokenizer_Diagnostic;
82+
7583
/* Tokenizer state */
7684
struct tok_state {
7785
_PyTok_Off buf_offset;
@@ -86,6 +94,7 @@ struct tok_state {
8694
lexer_layout_state layout;
8795
int lineno; /* Current line number */
8896
_PyTok_Loc start_loc;
97+
_PyTokenizer_Diagnostic diagnostic;
8998
int level; /* () [] {} Parentheses nesting level */
9099
/* Used to allow free continuations inside them */
91100
char parenstack[MAXLEVEL];

‎Parser/lexer/string.c‎

Lines changed: 51 additions & 32 deletions
Original file line numberDiff line numberDiff line change
@@ -7,13 +7,18 @@
77

88
#define MAKE_TOKEN(token_type) _PyLexer_token_setup(tok, token, token_type, p_start, p_end)
99

10-
static void
11-
rewind_to_string_start(struct tok_state *tok, _PyTok_Off start,
12-
_PyTok_Loc location)
10+
static int
11+
string_error_token(struct tok_state *tok, struct token *token,
12+
_PyTok_Off start, _PyTok_Loc location)
1313
{
14-
tok->cur = start + 1;
15-
tok->line_start = start - location.byte_col;
16-
tok->lineno = location.lineno;
14+
tok->diagnostic = (_PyTokenizer_Diagnostic){
15+
.location = {location.lineno, location.byte_col + 1},
16+
.text_span = _PyTok_SpanFromBounds(start - location.byte_col, tok->inp),
17+
};
18+
int type = _PyLexer_token_setup(tok, token, ERRORTOKEN, -1, -1);
19+
token->start_loc = location;
20+
token->end_loc = (_PyTok_Loc){location.lineno, -1};
21+
return type;
1722
}
1823

1924
int
@@ -351,7 +356,9 @@ _PyLexer_scan_string(struct tok_state *tok, struct token *token, int c)
351356
}
352357
if (c == EOF || (quote_size == 1 && c == '\n')) {
353358
int end_lineno = tok->lineno;
354-
rewind_to_string_start(tok, tok->start, tok->start_loc);
359+
_PyTok_Loc location = tok->start_loc;
360+
const char *line = tok->start - location.byte_col;
361+
Py_ssize_t cursor_offset = (Py_ssize_t)location.byte_col + 1;
355362

356363
const ftstring_state *state = _PyLexer_CurrentFTString(tok);
357364
if (state != NULL) {
@@ -364,41 +371,49 @@ _PyLexer_scan_string(struct tok_state *tok, struct token *token, int c)
364371
assert(tok->parenstack[level] == '{');
365372
int lineno = tok->parenlinenostack[level];
366373
if (lineno != tok->lineno) {
367-
return MAKE_TOKEN(_PyTokenizer_syntaxerror(tok,
374+
_PyTokenizer_syntaxerror_at(
375+
tok, line, cursor_offset, location.lineno, -1, -1,
368376
"%c-string: expecting '}' to close '{' on line %d",
369-
_PyLexer_StringPrefix(state->kind), lineno));
377+
_PyLexer_StringPrefix(state->kind), lineno);
378+
}
379+
else {
380+
_PyTokenizer_syntaxerror_at(
381+
tok, line, cursor_offset, location.lineno, -1, -1,
382+
"%c-string: expecting '}'",
383+
_PyLexer_StringPrefix(state->kind));
370384
}
371-
return MAKE_TOKEN(_PyTokenizer_syntaxerror(tok,
372-
"%c-string: expecting '}'",
373-
_PyLexer_StringPrefix(state->kind)));
385+
return string_error_token(tok, token, tok->start, location);
374386
}
375387
}
376388

377389
if (quote_size == 3) {
378-
_PyTokenizer_syntaxerror(tok, "unterminated triple-quoted string literal"
379-
" (detected at line %d)", end_lineno);
390+
_PyTokenizer_syntaxerror_at(
391+
tok, line, cursor_offset, location.lineno, -1, -1,
392+
"unterminated triple-quoted string literal"
393+
" (detected at line %d)", end_lineno);
380394
if (c != '\n') {
381395
tok->done = E_EOFS;
382396
}
383-
return MAKE_TOKEN(ERRORTOKEN);
397+
return string_error_token(tok, token, tok->start, location);
384398
}
385399
else {
386400
if (has_escaped_quote) {
387-
_PyTokenizer_syntaxerror(
388-
tok,
401+
_PyTokenizer_syntaxerror_at(
402+
tok, line, cursor_offset, location.lineno, -1, -1,
389403
"unterminated string literal (detected at line %d); "
390404
"perhaps you escaped the end quote?",
391405
end_lineno
392406
);
393407
} else {
394-
_PyTokenizer_syntaxerror(
395-
tok, "unterminated string literal (detected at line %d)", end_lineno
408+
_PyTokenizer_syntaxerror_at(
409+
tok, line, cursor_offset, location.lineno, -1, -1,
410+
"unterminated string literal (detected at line %d)", end_lineno
396411
);
397412
}
398413
if (c != '\n') {
399414
tok->done = E_EOLS;
400415
}
401-
return MAKE_TOKEN(ERRORTOKEN);
416+
return string_error_token(tok, token, tok->start, location);
402417
}
403418
}
404419
if (c == quote) {
@@ -462,25 +477,29 @@ _PyLexer_get_ftstring(struct tok_state *tok, ftstring_state *current, struct tok
462477
}
463478

464479
int end_lineno = tok->lineno;
465-
rewind_to_string_start(tok,
466-
current->start,
467-
current->start_loc);
480+
_PyTok_Loc location = current->start_loc;
481+
const char *line = _PyLexer_BufferPointer(tok, current->start) - location.byte_col;
482+
Py_ssize_t cursor_offset = (Py_ssize_t)location.byte_col + 1;
468483

469484
if (quote_size == 3) {
470-
_PyTokenizer_syntaxerror(tok,
471-
"unterminated triple-quoted %c-string literal"
472-
" (detected at line %d)",
473-
_PyLexer_StringPrefix(current->kind), end_lineno);
485+
_PyTokenizer_syntaxerror_at(
486+
tok, line, cursor_offset, location.lineno, -1, -1,
487+
"unterminated triple-quoted %c-string literal"
488+
" (detected at line %d)",
489+
_PyLexer_StringPrefix(current->kind), end_lineno);
474490
if (c != '\n') {
475491
tok->done = E_EOFS;
476492
}
477-
return MAKE_TOKEN(ERRORTOKEN);
493+
return string_error_token(tok, token,
494+
_PyLexer_BufferPointer(tok, current->start), location);
478495
}
479496
else {
480-
return MAKE_TOKEN(_PyTokenizer_syntaxerror(tok,
481-
"unterminated %c-string literal (detected at"
482-
" line %d)",
483-
_PyLexer_StringPrefix(current->kind), end_lineno));
497+
_PyTokenizer_syntaxerror_at(
498+
tok, line, cursor_offset, location.lineno, -1, -1,
499+
"unterminated %c-string literal (detected at line %d)",
500+
_PyLexer_StringPrefix(current->kind), end_lineno);
501+
return string_error_token(tok, token,
502+
_PyLexer_BufferPointer(tok, current->start), location);
484503
}
485504
}
486505

‎Parser/tokenizer/decoder.c‎

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -264,7 +264,8 @@ _PyTok_DetectEncoding(struct tok_state *tok, const _PyTok_Chunk *first,
264264
end_col--;
265265
}
266266
_PyTokenizer_syntaxerror_at(
267-
tok, line_data, 0, cookie_line, 0, end_col, "encoding problem: %s with BOM", cookie);
267+
tok, line_data, 0, cookie_line, 0, end_col,
268+
"encoding problem: %s with BOM", cookie);
268269
PyMem_Free(cookie);
269270
return _PYTOK_ENCODING_ERROR;
270271
}

‎Parser/tokenizer/helpers.c‎

Lines changed: 3 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -333,12 +333,9 @@ _PyTokenizer_ensure_utf8(const char *line, struct tok_state *tok, int lineno)
333333
}
334334
}
335335
if (badchar) {
336-
tok->lineno = lineno;
337-
tok->line_start = _PyLexer_BufferOffset(tok, line_start);
338-
tok->cur = _PyLexer_BufferOffset(tok, badchar);
339-
_PyTokenizer_syntaxerror_known_range(tok,
340-
(int)(badchar - line_start) + 1,
341-
(int)(badchar - line_start) + 1,
336+
_PyTokenizer_syntaxerror_at(
337+
tok, line_start, badchar - line_start + 1, lineno,
338+
-1, -1,
342339
"Non-UTF-8 code starting with '\\x%.2x'"
343340
"%s%V on line %i, "
344341
"but no encoding declared; "

‎Parser/tokenizer/helpers.h‎

Lines changed: 7 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -8,10 +8,17 @@
88
int _PyTokenizer_syntaxerror_at(struct tok_state *, const char *,
99
Py_ssize_t, int, int, int, const char *, ...);
1010
int _PyTokenizer_syntaxerror(struct tok_state *tok, const char *format, ...);
11+
/* Positive range columns are 1-based byte columns. A start column of -1
12+
derives the character column from the reporting cursor; an end column of
13+
-1 uses the start column. */
1114
int _PyTokenizer_syntaxerror_known_range(struct tok_state *tok, int col_offset, int end_col_offset, const char *format, ...);
15+
int _PyTokenizer_syntaxerror_at(
16+
struct tok_state *tok, const char *line_start, Py_ssize_t cursor_offset,
17+
int lineno, int col_offset, int end_col_offset, const char *format, ...);
1218
int _PyTokenizer_indenterror(struct tok_state *tok);
1319
int _PyTokenizer_warn_invalid_escape_sequence(struct tok_state *tok, int first_invalid_escape_char);
1420
int _PyTokenizer_parser_warn(struct tok_state *tok, PyObject *category, const char *format, ...);
21+
1522
void _PyTokenizer_raise_init_error(PyObject *filename);
1623

1724
int _PyTokenizer_ensure_utf8(const char *line, struct tok_state *tok, int lineno);

0 commit comments

Comments
 (0)