1313 tokenizing. */
1414static const char * type_comment_prefix = "# type: " ;
1515
16- static inline int
17- contains_null_bytes (const char * str , size_t size )
18- {
19- return memchr (str , 0 , size ) != NULL ;
20- }
21-
2216int
2317_PyLexer_refill (struct tok_state * tok )
2418{
@@ -40,8 +34,8 @@ _PyLexer_refill(struct tok_state *tok)
4034 return 0 ;
4135 }
4236 tok -> line_start = tok -> cur ;
43- if ( contains_null_bytes ( _PyTok_SourcePointer ( & tok -> source , tok -> line_start ),
44- tok -> inp - tok -> line_start ) ) {
37+ _PyTok_Span line = { tok -> line_start , tok -> inp };
38+ if ( _PyTok_SourceFindByte ( & tok -> source , line , '\0' ) >= 0 ) {
4539 _PyTokenizer_syntaxerror (tok , "source code cannot contain null bytes" );
4640 tok -> cur = tok -> inp ;
4741 return 0 ;
@@ -57,8 +51,7 @@ _PyLexer_backup(struct tok_state *tok, int c)
5751 if (-- tok -> cur < tok -> buf_offset ) {
5852 Py_FatalError ("tokenizer beginning of buffer" );
5953 }
60- const char * cur = _PyTok_SourcePointer (& tok -> source , tok -> cur );
61- if ((int )(unsigned char )* cur != Py_CHARMASK (c )) {
54+ if (_PyTok_SourceByte (& tok -> source , tok -> cur ) != Py_CHARMASK (c )) {
6255 Py_FatalError ("tok_backup: wrong character" );
6356 }
6457 }
@@ -175,10 +168,6 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
175168 /* Skip comment, unless it's a type comment */
176169 if (c == '#' ) {
177170
178- const char * p = NULL ;
179- const char * prefix , * type_start ;
180- int current_starting_col_offset ;
181-
182171 while (c != EOF && c != '\n' && c != '\r' ) {
183172 c = tok_nextc (tok );
184173 }
@@ -195,23 +184,20 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
195184 }
196185 }
197186
198- if (tok -> tok_extra_tokens ) {
199- p = _PyTok_SourcePointer (& tok -> source , tok -> start );
200- }
201-
187+ _PyTok_Off comment = tok -> start ;
202188 if (tok -> type_comments ) {
203- p = _PyTok_SourcePointer (& tok -> source , tok -> start );
204- current_starting_col_offset = tok -> start_loc .byte_col ;
205- prefix = type_comment_prefix ;
206- while (* prefix && p < _PyTok_SourcePointer (& tok -> source , tok -> cur )) {
189+ const char * prefix = type_comment_prefix ;
190+ while (* prefix && comment < tok -> cur ) {
207191 if (* prefix == ' ' ) {
208- while (* p == ' ' || * p == '\t' ) {
209- p ++ ;
210- current_starting_col_offset ++ ;
192+ while (comment < tok -> cur ) {
193+ int ch = _PyTok_SourceByte (& tok -> source , comment );
194+ if (ch != ' ' && ch != '\t' ) {
195+ break ;
196+ }
197+ comment ++ ;
211198 }
212- } else if (* prefix == * p ) {
213- p ++ ;
214- current_starting_col_offset ++ ;
199+ } else if (* prefix == _PyTok_SourceByte (& tok -> source , comment )) {
200+ comment ++ ;
215201 } else {
216202 break ;
217203 }
@@ -221,36 +207,26 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
221207
222208 /* This is a type comment if we matched all of type_comment_prefix. */
223209 if (!* prefix ) {
224- int is_type_ignore = 1 ;
225- // +6 in order to skip the word 'ignore'
226- const char * ignore_end = p + 6 ;
227- const int ignore_end_col_offset = current_starting_col_offset + 6 ;
228210 tok_backup (tok , c ); /* don't eat the newline or EOF */
229-
230- type_start = p ;
231-
232211 /* A TYPE_IGNORE is "type: ignore" followed by the end of the token
233212 * or anything ASCII and non-alphanumeric. */
234- is_type_ignore = (
235- _PyTok_SourcePointer (& tok -> source , tok -> cur ) >= ignore_end
236- && memcmp (p , "ignore" , 6 ) == 0
237- && !(_PyTok_SourcePointer (& tok -> source , tok -> cur ) > ignore_end
238- && ((unsigned char )ignore_end [0 ] >= 128 || Py_ISALNUM (ignore_end [0 ]))));
213+ int is_type_ignore = tok -> cur - comment >= 6 &&
214+ memcmp (_PyTok_SourcePointer (& tok -> source , comment ),
215+ "ignore" , 6 ) == 0 ;
216+ if (is_type_ignore && comment + 6 < tok -> cur ) {
217+ int ch = _PyTok_SourceByte (& tok -> source , comment + 6 );
218+ is_type_ignore = ch < 128 && !Py_ISALNUM (ch );
219+ }
239220
240221 int type = is_type_ignore ? TYPE_IGNORE : TYPE_COMMENT ;
241- int start_col_offset = is_type_ignore
242- ? ignore_end_col_offset : current_starting_col_offset ;
222+ p_start = comment + (is_type_ignore ? 6 : 0 );
243223 p_end = tok -> cur ;
244- if (is_type_ignore ) {
245- p_start = _PyTok_SourceOffset (& tok -> source , ignore_end );
246-
247- /* If this type ignore is the only thing on the line, consume the newline also. */
248- if (blankline ) {
249- tok_nextc (tok );
250- tok -> layout .at_bol = 1 ;
251- }
252- } else {
253- p_start = _PyTok_SourceOffset (& tok -> source , type_start );
224+ int start_col_offset = tok -> start_loc .byte_col +
225+ (int )(p_start - tok -> start );
226+ /* Consume the newline after a standalone type ignore. */
227+ if (is_type_ignore && blankline ) {
228+ tok_nextc (tok );
229+ tok -> layout .at_bol = 1 ;
254230 }
255231 _PyLexer_token_setup (tok , token , type , p_start , p_end );
256232 token -> start_loc = (_PyTok_Loc ){tok -> lineno , start_col_offset };
@@ -261,7 +237,7 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
261237 }
262238 if (tok -> tok_extra_tokens ) {
263239 tok_backup (tok , c ); /* don't eat the newline or EOF */
264- p_start = _PyTok_SourceOffset ( & tok -> source , p ) ;
240+ p_start = comment ;
265241 p_end = tok -> cur ;
266242 tok -> layout .comment_newline = blankline ;
267243 return MAKE_TOKEN (COMMENT );
0 commit comments