Skip to content

Commit f36c44b

Browse files
committed
gh-153569: Keep tokenizer source scans and copies on offsets and spans
1 parent 723a678 commit f36c44b

8 files changed

Lines changed: 102 additions & 96 deletions

File tree

‎Lib/test/test_compile.py‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -726,6 +726,8 @@ def test_single_statement(self):
726726
self.compile_single("1 + 2\t\t\n")
727727
self.compile_single("1 + 2\t\t\n ")
728728
self.compile_single("1 + 2 # one plus two")
729+
self.compile_single("1 + 2\n# trailing comment")
730+
self.compile_single("1 + 2\n \t\f# trailing comment\n\t")
729731
self.compile_single("1; 2")
730732
self.compile_single("import sys; sys")
731733
self.compile_single("def f():\n pass")
@@ -742,6 +744,7 @@ def test_bad_single_statement(self):
742744
self.assertInvalidSingle('del x\ndel y')
743745
self.assertInvalidSingle('f()\ng()')
744746
self.assertInvalidSingle('f()\n# blah\nblah()')
747+
self.assertInvalidSingle('f()\n \t\f# comment\n\tg()')
745748
self.assertInvalidSingle('f()\nxy # blah\nblah()')
746749
self.assertInvalidSingle('x = 5 # comment\nx = 6\n')
747750
self.assertInvalidSingle("c = '''\nd=1\n'''\na = 1\n\nb = 2\n")

‎Lib/test/test_type_comments.py‎

Lines changed: 17 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -348,6 +348,23 @@ def test_many_ignores(self):
348348
list(enumerate(tags, start=1)),
349349
)
350350

351+
def test_type_comment_suffixes(self):
352+
for prefix in ("# type: ", "#\ttype:\t", "#type:"):
353+
for ending in ("", "\n"):
354+
for text, tag in (("", None), ("i", None), ("ignor", None),
355+
("ignore", ""), ("ignore_tag", "_tag"),
356+
("ignore[tag]", "[tag]"), ("ignore0", None),
357+
("ignoreé", None)):
358+
with self.subTest(prefix=prefix, ending=ending, text=text):
359+
source = f"pass\né = 1 {prefix}{text}{ending}"
360+
tree = ast.parse(source, type_comments=True)
361+
self.assertEqual(tree.body[1].type_comment,
362+
text if tag is None else None)
363+
self.assertEqual(
364+
[(item.lineno, item.tag) for item in tree.type_ignores],
365+
[] if tag is None else [(2, tag)],
366+
)
367+
351368
def test_longargs(self):
352369
for tree in self.parse_all(longargs, minver=8):
353370
for t in tree.body:

‎Modules/_testinternalcapi/tokenizer.c‎

Lines changed: 7 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -130,11 +130,17 @@ test_tokenizer_source_views(PyObject *Py_UNUSED(module),
130130
const char *view = _PyTok_SourceSpanView(&source, span, &len);
131131
if (check(len == (Py_ssize_t)sizeof(line) - 1 && memcmp(view, line, len) == 0,
132132
"source span changed after growth") < 0 ||
133-
check(_PyTok_SourceOffset(&source, view) == start &&
133+
check(_PyTok_SourcePointer(&source, start) == view &&
134134
_PyTok_SourcePointer(&source, span.end) == view + len,
135135
"source view lost its logical offset") < 0) {
136136
goto error;
137137
}
138+
if (check(_PyTok_SourceByte(&source, start) == 0xce &&
139+
_PyTok_SourceFindByte(&source, span, '\n') == span.end - 1 &&
140+
_PyTok_SourceFindByte(&source, span, '?') == -1,
141+
"wrong byte lookup after source growth") < 0) {
142+
goto error;
143+
}
138144
_PyTok_Off end = source.base_offset + source.len;
139145
view = _PyTok_SourceSpanView(&source, (_PyTok_Span){end, end}, &len);
140146
if (check(len == 0 && view == source.bytes + source.len,

‎Parser/lexer/lexer.c‎

Lines changed: 29 additions & 53 deletions
Original file line numberDiff line numberDiff line change
@@ -13,12 +13,6 @@
1313
tokenizing. */
1414
static const char* type_comment_prefix = "# type: ";
1515

16-
static inline int
17-
contains_null_bytes(const char* str, size_t size)
18-
{
19-
return memchr(str, 0, size) != NULL;
20-
}
21-
2216
int
2317
_PyLexer_refill(struct tok_state *tok)
2418
{
@@ -40,8 +34,8 @@ _PyLexer_refill(struct tok_state *tok)
4034
return 0;
4135
}
4236
tok->line_start = tok->cur;
43-
if (contains_null_bytes(_PyTok_SourcePointer(&tok->source, tok->line_start),
44-
tok->inp - tok->line_start)) {
37+
_PyTok_Span line = {tok->line_start, tok->inp};
38+
if (_PyTok_SourceFindByte(&tok->source, line, '\0') >= 0) {
4539
_PyTokenizer_syntaxerror(tok, "source code cannot contain null bytes");
4640
tok->cur = tok->inp;
4741
return 0;
@@ -57,8 +51,7 @@ _PyLexer_backup(struct tok_state *tok, int c)
5751
if (--tok->cur < tok->buf_offset) {
5852
Py_FatalError("tokenizer beginning of buffer");
5953
}
60-
const char *cur = _PyTok_SourcePointer(&tok->source, tok->cur);
61-
if ((int)(unsigned char)*cur != Py_CHARMASK(c)) {
54+
if (_PyTok_SourceByte(&tok->source, tok->cur) != Py_CHARMASK(c)) {
6255
Py_FatalError("tok_backup: wrong character");
6356
}
6457
}
@@ -175,10 +168,6 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
175168
/* Skip comment, unless it's a type comment */
176169
if (c == '#') {
177170

178-
const char* p = NULL;
179-
const char *prefix, *type_start;
180-
int current_starting_col_offset;
181-
182171
while (c != EOF && c != '\n' && c != '\r') {
183172
c = tok_nextc(tok);
184173
}
@@ -195,23 +184,20 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
195184
}
196185
}
197186

198-
if (tok->tok_extra_tokens) {
199-
p = _PyTok_SourcePointer(&tok->source, tok->start);
200-
}
201-
187+
_PyTok_Off comment = tok->start;
202188
if (tok->type_comments) {
203-
p = _PyTok_SourcePointer(&tok->source, tok->start);
204-
current_starting_col_offset = tok->start_loc.byte_col;
205-
prefix = type_comment_prefix;
206-
while (*prefix && p < _PyTok_SourcePointer(&tok->source, tok->cur)) {
189+
const char *prefix = type_comment_prefix;
190+
while (*prefix && comment < tok->cur) {
207191
if (*prefix == ' ') {
208-
while (*p == ' ' || *p == '\t') {
209-
p++;
210-
current_starting_col_offset++;
192+
while (comment < tok->cur) {
193+
int ch = _PyTok_SourceByte(&tok->source, comment);
194+
if (ch != ' ' && ch != '\t') {
195+
break;
196+
}
197+
comment++;
211198
}
212-
} else if (*prefix == *p) {
213-
p++;
214-
current_starting_col_offset++;
199+
} else if (*prefix == _PyTok_SourceByte(&tok->source, comment)) {
200+
comment++;
215201
} else {
216202
break;
217203
}
@@ -221,36 +207,26 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
221207

222208
/* This is a type comment if we matched all of type_comment_prefix. */
223209
if (!*prefix) {
224-
int is_type_ignore = 1;
225-
// +6 in order to skip the word 'ignore'
226-
const char *ignore_end = p + 6;
227-
const int ignore_end_col_offset = current_starting_col_offset + 6;
228210
tok_backup(tok, c); /* don't eat the newline or EOF */
229-
230-
type_start = p;
231-
232211
/* A TYPE_IGNORE is "type: ignore" followed by the end of the token
233212
* or anything ASCII and non-alphanumeric. */
234-
is_type_ignore = (
235-
_PyTok_SourcePointer(&tok->source, tok->cur) >= ignore_end
236-
&& memcmp(p, "ignore", 6) == 0
237-
&& !(_PyTok_SourcePointer(&tok->source, tok->cur) > ignore_end
238-
&& ((unsigned char)ignore_end[0] >= 128 || Py_ISALNUM(ignore_end[0]))));
213+
int is_type_ignore = tok->cur - comment >= 6 &&
214+
memcmp(_PyTok_SourcePointer(&tok->source, comment),
215+
"ignore", 6) == 0;
216+
if (is_type_ignore && comment + 6 < tok->cur) {
217+
int ch = _PyTok_SourceByte(&tok->source, comment + 6);
218+
is_type_ignore = ch < 128 && !Py_ISALNUM(ch);
219+
}
239220

240221
int type = is_type_ignore ? TYPE_IGNORE : TYPE_COMMENT;
241-
int start_col_offset = is_type_ignore
242-
? ignore_end_col_offset : current_starting_col_offset;
222+
p_start = comment + (is_type_ignore ? 6 : 0);
243223
p_end = tok->cur;
244-
if (is_type_ignore) {
245-
p_start = _PyTok_SourceOffset(&tok->source, ignore_end);
246-
247-
/* If this type ignore is the only thing on the line, consume the newline also. */
248-
if (blankline) {
249-
tok_nextc(tok);
250-
tok->layout.at_bol = 1;
251-
}
252-
} else {
253-
p_start = _PyTok_SourceOffset(&tok->source, type_start);
224+
int start_col_offset = tok->start_loc.byte_col +
225+
(int)(p_start - tok->start);
226+
/* Consume the newline after a standalone type ignore. */
227+
if (is_type_ignore && blankline) {
228+
tok_nextc(tok);
229+
tok->layout.at_bol = 1;
254230
}
255231
_PyLexer_token_setup(tok, token, type, p_start, p_end);
256232
token->start_loc = (_PyTok_Loc){tok->lineno, start_col_offset};
@@ -261,7 +237,7 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
261237
}
262238
if (tok->tok_extra_tokens) {
263239
tok_backup(tok, c); /* don't eat the newline or EOF */
264-
p_start = _PyTok_SourceOffset(&tok->source, p);
240+
p_start = comment;
265241
p_end = tok->cur;
266242
tok->layout.comment_newline = blankline;
267243
return MAKE_TOKEN(COMMENT);

‎Parser/lexer/string.c‎

Lines changed: 15 additions & 16 deletions
Original file line numberDiff line numberDiff line change
@@ -76,13 +76,10 @@ finish_ftstring_expr(struct tok_state *tok, ftstring_state *state,
7676
if (!(state->debug_expr || tstring_interpolation) || token->metadata) {
7777
return 0;
7878
}
79-
Py_ssize_t expr_len;
80-
const char *expr = _PyTok_SourceSpanView(
81-
&tok->source, state->expr_span, &expr_len);
8279
tokenizer_comments *comments = state->comments;
8380
PyObject *res;
8481
if (comments != NULL && comments->count > 0) {
85-
Py_ssize_t stripped_size = expr_len;
82+
Py_ssize_t stripped_size = state->expr_span.end - state->expr_span.start;
8683
Py_ssize_t comment_count = 0;
8784
for (Py_ssize_t i = 0; i < comments->count; i++) {
8885
_PyTok_Span comment = comments->spans[i];
@@ -103,24 +100,26 @@ finish_ftstring_expr(struct tok_state *tok, ftstring_state *state,
103100
}
104101
_PyTok_Off copied_to = state->expr_span.start;
105102
Py_ssize_t stripped_len = 0;
106-
for (Py_ssize_t i = 0; i < comment_count; i++) {
107-
_PyTok_Span comment = comments->spans[i];
108-
Py_ssize_t length = comment.start - copied_to;
109-
memcpy(stripped + stripped_len,
110-
expr + copied_to - state->expr_span.start,
111-
(size_t)length);
103+
for (Py_ssize_t i = 0; i <= comment_count; i++) {
104+
_PyTok_Span span = {
105+
copied_to,
106+
i < comment_count ? comments->spans[i].start : state->expr_span.end,
107+
};
108+
Py_ssize_t length;
109+
const char *text = _PyTok_SourceSpanView(&tok->source, span, &length);
110+
memcpy(stripped + stripped_len, text, (size_t)length);
112111
stripped_len += length;
113-
copied_to = comment.end;
112+
if (i < comment_count) {
113+
copied_to = comments->spans[i].end;
114+
}
114115
}
115-
Py_ssize_t length = state->expr_span.end - copied_to;
116-
memcpy(stripped + stripped_len,
117-
expr + copied_to - state->expr_span.start,
118-
(size_t)length);
119-
stripped_len += length;
120116
res = PyUnicode_DecodeUTF8(stripped, stripped_len, NULL);
121117
PyMem_Free(stripped);
122118
}
123119
else {
120+
Py_ssize_t expr_len;
121+
const char *expr = _PyTok_SourceSpanView(
122+
&tok->source, state->expr_span, &expr_len);
124123
res = PyUnicode_DecodeUTF8(expr, expr_len, NULL);
125124
}
126125

‎Parser/tokenizer/api.c‎

Lines changed: 15 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -107,22 +107,28 @@ _PyTokenizer_ImplyDedents(struct tok_state *tok)
107107
int
108108
_PyTokenizer_HasTrailingStatement(const struct tok_state *tok)
109109
{
110-
const char *cur = _PyTok_SourcePointer(&tok->source, tok->cur);
111-
char c = *cur;
112-
for (;;) {
113-
while (c == ' ' || c == '\t' || c == '\n' || c == '\014') {
114-
c = *++cur;
115-
}
116-
if (!c) {
110+
_PyTok_Off cur = tok->cur;
111+
_PyTok_Off end = tok->source.base_offset + tok->source.len;
112+
while (cur < end) {
113+
int c = _PyTok_SourceByte(&tok->source, cur++);
114+
if (c == '\0') {
117115
return 0;
118116
}
117+
if (c == ' ' || c == '\t' || c == '\n' || c == '\014') {
118+
continue;
119+
}
119120
if (c != '#') {
120121
return 1;
121122
}
122-
while (c && c != '\n') {
123-
c = *++cur;
123+
while (cur < end) {
124+
c = _PyTok_SourceByte(&tok->source, cur);
125+
if (c == '\0' || c == '\n') {
126+
break;
127+
}
128+
cur++;
124129
}
125130
}
131+
return 0;
126132
}
127133

128134
int

‎Parser/tokenizer/reader.c‎

Lines changed: 6 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -175,13 +175,11 @@ next_prepared(struct tok_state *tok, _PyTok_Chunk *chunk)
175175
return _PYTOK_READ_EOF;
176176
}
177177
_PyTok_Span tail = {tok->inp, tok->source.base_offset + tok->source.len};
178-
Py_ssize_t remaining;
179-
const char *start = _PyTok_SourceSpanView(&tok->source, tail, &remaining);
180-
const char *newline = memchr(start, '\n', remaining);
181-
chunk->data = (char *)start;
182-
chunk->len = newline != NULL ? newline - start + 1 : remaining;
178+
_PyTok_Off newline = _PyTok_SourceFindByte(&tok->source, tail, '\n');
179+
_PyTok_Span line = {tail.start, newline >= 0 ? newline + 1 : tail.end};
180+
chunk->data = (char *)_PyTok_SourceSpanView(&tok->source, line, &chunk->len);
183181
chunk->ownership = _PYTOK_CHUNK_BORROWED;
184-
chunk->implicit_newline = chunk->len == remaining &&
182+
chunk->implicit_newline = line.end == tail.end &&
185183
tok->reader->prepared_final_newline_is_implicit;
186184
return _PYTOK_READ_LINE;
187185
}
@@ -667,11 +665,10 @@ _PyTok_ReaderUnderflow(struct tok_state *tok)
667665
tok->inp = source_start + chunk.len;
668666
}
669667
else {
670-
_PyTok_Off source_start = _PyTok_SourceOffset(&tok->source, chunk.data);
671668
if (tok->start < 0 && _PyLexer_CurrentFTString(tok) == NULL) {
672-
tok->buf_offset = source_start;
669+
tok->buf_offset = tok->inp;
673670
}
674-
tok->inp = source_start + chunk.len;
671+
tok->inp += chunk.len;
675672
}
676673
tok->implicit_newline = chunk.implicit_newline;
677674

‎Parser/tokenizer/source.h‎

Lines changed: 10 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -21,14 +21,6 @@ _PyTok_SourceData(const _PyTok_SourceText *source)
2121

2222
/* Convert positions within the retained source window. Pointers and views
2323
are borrowed; append, discard, and clear invalidate them. */
24-
static inline _PyTok_Off
25-
_PyTok_SourceOffset(const _PyTok_SourceText *source, const char *position)
26-
{
27-
const char *base = _PyTok_SourceData(source);
28-
assert(position >= base && position <= base + source->len);
29-
return source->base_offset + (position - base);
30-
}
31-
3224
static inline const char *
3325
_PyTok_SourcePointer(const _PyTok_SourceText *source, _PyTok_Off offset)
3426
{
@@ -56,6 +48,16 @@ _PyTok_SourceSpanView(const _PyTok_SourceText *source, _PyTok_Span span,
5648
return _PyTok_SourcePointer(source, span.start);
5749
}
5850

51+
/* Return the first matching offset within span, or -1 if absent. */
52+
static inline _PyTok_Off
53+
_PyTok_SourceFindByte(const _PyTok_SourceText *source, _PyTok_Span span, int byte)
54+
{
55+
Py_ssize_t length;
56+
const char *data = _PyTok_SourceSpanView(source, span, &length);
57+
const char *found = memchr(data, byte, length);
58+
return found != NULL ? span.start + (found - data) : -1;
59+
}
60+
5961
PyAPI_FUNC(void) _PyTok_SourceInit(_PyTok_SourceText *);
6062
/* Clear invalidates all spans and views for the source. */
6163
PyAPI_FUNC(void) _PyTok_SourceClear(_PyTok_SourceText *);

0 commit comments

Comments
 (0)