77
88#define MAKE_TOKEN (token_type ) _PyLexer_token_setup(tok, token, token_type, p_start, p_end)
99
10- static void
11- rewind_to_string_start (struct tok_state * tok , _PyTok_Off start ,
12- _PyTok_Loc location )
10+ static int
11+ string_error_token (struct tok_state * tok , struct token * token ,
12+ _PyTok_Off start , _PyTok_Loc location )
1313{
14- tok -> cur = start + 1 ;
15- tok -> line_start = start - location .byte_col ;
16- tok -> lineno = location .lineno ;
14+ tok -> diagnostic = (_PyTokenizer_Diagnostic ){
15+ .location = {location .lineno , location .byte_col + 1 },
16+ .text_span = _PyTok_SpanFromBounds (start - location .byte_col , tok -> inp ),
17+ };
18+ int type = _PyLexer_token_setup (tok , token , ERRORTOKEN , -1 , -1 );
19+ token -> start_loc = location ;
20+ token -> end_loc = (_PyTok_Loc ){location .lineno , -1 };
21+ return type ;
1722}
1823
1924int
@@ -351,7 +356,9 @@ _PyLexer_scan_string(struct tok_state *tok, struct token *token, int c)
351356 }
352357 if (c == EOF || (quote_size == 1 && c == '\n' )) {
353358 int end_lineno = tok -> lineno ;
354- rewind_to_string_start (tok , tok -> start , tok -> start_loc );
359+ _PyTok_Loc location = tok -> start_loc ;
360+ const char * line = tok -> start - location .byte_col ;
361+ Py_ssize_t cursor_offset = (Py_ssize_t )location .byte_col + 1 ;
355362
356363 const ftstring_state * state = _PyLexer_CurrentFTString (tok );
357364 if (state != NULL ) {
@@ -364,41 +371,49 @@ _PyLexer_scan_string(struct tok_state *tok, struct token *token, int c)
364371 assert (tok -> parenstack [level ] == '{' );
365372 int lineno = tok -> parenlinenostack [level ];
366373 if (lineno != tok -> lineno ) {
367- return MAKE_TOKEN (_PyTokenizer_syntaxerror (tok ,
374+ _PyTokenizer_syntaxerror_at (
375+ tok , line , cursor_offset , location .lineno , -1 , -1 ,
368376 "%c-string: expecting '}' to close '{' on line %d" ,
369- _PyLexer_StringPrefix (state -> kind ), lineno ));
377+ _PyLexer_StringPrefix (state -> kind ), lineno );
378+ }
379+ else {
380+ _PyTokenizer_syntaxerror_at (
381+ tok , line , cursor_offset , location .lineno , -1 , -1 ,
382+ "%c-string: expecting '}'" ,
383+ _PyLexer_StringPrefix (state -> kind ));
370384 }
371- return MAKE_TOKEN (_PyTokenizer_syntaxerror (tok ,
372- "%c-string: expecting '}'" ,
373- _PyLexer_StringPrefix (state -> kind )));
385+ return string_error_token (tok , token , tok -> start , location );
374386 }
375387 }
376388
377389 if (quote_size == 3 ) {
378- _PyTokenizer_syntaxerror (tok , "unterminated triple-quoted string literal"
379- " (detected at line %d)" , end_lineno );
390+ _PyTokenizer_syntaxerror_at (
391+ tok , line , cursor_offset , location .lineno , -1 , -1 ,
392+ "unterminated triple-quoted string literal"
393+ " (detected at line %d)" , end_lineno );
380394 if (c != '\n' ) {
381395 tok -> done = E_EOFS ;
382396 }
383- return MAKE_TOKEN ( ERRORTOKEN );
397+ return string_error_token ( tok , token , tok -> start , location );
384398 }
385399 else {
386400 if (has_escaped_quote ) {
387- _PyTokenizer_syntaxerror (
388- tok ,
401+ _PyTokenizer_syntaxerror_at (
402+ tok , line , cursor_offset , location . lineno , -1 , -1 ,
389403 "unterminated string literal (detected at line %d); "
390404 "perhaps you escaped the end quote?" ,
391405 end_lineno
392406 );
393407 } else {
394- _PyTokenizer_syntaxerror (
395- tok , "unterminated string literal (detected at line %d)" , end_lineno
408+ _PyTokenizer_syntaxerror_at (
409+ tok , line , cursor_offset , location .lineno , -1 , -1 ,
410+ "unterminated string literal (detected at line %d)" , end_lineno
396411 );
397412 }
398413 if (c != '\n' ) {
399414 tok -> done = E_EOLS ;
400415 }
401- return MAKE_TOKEN ( ERRORTOKEN );
416+ return string_error_token ( tok , token , tok -> start , location );
402417 }
403418 }
404419 if (c == quote ) {
@@ -462,25 +477,29 @@ _PyLexer_get_ftstring(struct tok_state *tok, ftstring_state *current, struct tok
462477 }
463478
464479 int end_lineno = tok -> lineno ;
465- rewind_to_string_start ( tok ,
466- current -> start ,
467- current -> start_loc ) ;
480+ _PyTok_Loc location = current -> start_loc ;
481+ const char * line = _PyLexer_BufferPointer ( tok , current -> start ) - location . byte_col ;
482+ Py_ssize_t cursor_offset = ( Py_ssize_t ) location . byte_col + 1 ;
468483
469484 if (quote_size == 3 ) {
470- _PyTokenizer_syntaxerror (tok ,
471- "unterminated triple-quoted %c-string literal"
472- " (detected at line %d)" ,
473- _PyLexer_StringPrefix (current -> kind ), end_lineno );
485+ _PyTokenizer_syntaxerror_at (
486+ tok , line , cursor_offset , location .lineno , -1 , -1 ,
487+ "unterminated triple-quoted %c-string literal"
488+ " (detected at line %d)" ,
489+ _PyLexer_StringPrefix (current -> kind ), end_lineno );
474490 if (c != '\n' ) {
475491 tok -> done = E_EOFS ;
476492 }
477- return MAKE_TOKEN (ERRORTOKEN );
493+ return string_error_token (tok , token ,
494+ _PyLexer_BufferPointer (tok , current -> start ), location );
478495 }
479496 else {
480- return MAKE_TOKEN (_PyTokenizer_syntaxerror (tok ,
481- "unterminated %c-string literal (detected at"
482- " line %d)" ,
483- _PyLexer_StringPrefix (current -> kind ), end_lineno ));
497+ _PyTokenizer_syntaxerror_at (
498+ tok , line , cursor_offset , location .lineno , -1 , -1 ,
499+ "unterminated %c-string literal (detected at line %d)" ,
500+ _PyLexer_StringPrefix (current -> kind ), end_lineno );
501+ return string_error_token (tok , token ,
502+ _PyLexer_BufferPointer (tok , current -> start ), location );
484503 }
485504 }
486505
0 commit comments