Skip to content

Commit 400de53

Browse files
committed
gh-153569: Simplify tokenizer source views and reader ownership
1 parent 1643525 commit 400de53

15 files changed

Lines changed: 198 additions & 108 deletions

File tree

‎Lib/test/test_capi/test_tokenizer.py‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -12,6 +12,9 @@ def test_source(self):
1212
def test_source_discard(self):
1313
_testinternalcapi.test_tokenizer_source_discard()
1414

15+
def test_source_views(self):
16+
_testinternalcapi.test_tokenizer_source_views()
17+
1518

1619
if __name__ == "__main__":
1720
unittest.main()

‎Lib/test/test_tokenize.py‎

Lines changed: 28 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -2498,6 +2498,34 @@ def test_fstring_offsets_survive_buffer_reallocation(self):
24982498
chunks.__next__, extra_tokens=extra_tokens))
24992499
self.assertEqual(tokens, expected)
25002500

2501+
def test_multiline_unicode_columns_after_source_discard(self):
2502+
# The first line is discarded before the multiline token is read.
2503+
# Its saved offsets must remain valid when the source buffer grows.
2504+
padding = "é" * 9000
2505+
source = f"pass\né = '''{padding}\n漢''' + ö\nß = 1\n"
2506+
expected = [
2507+
(token.NAME, "pass", (1, 0), (1, 4)),
2508+
(token.NAME, "é", (2, 0), (2, 1)),
2509+
(token.STRING, f"'''{padding}\n漢'''", (2, 4), (3, 4)),
2510+
(token.NAME, "ö", (3, 7), (3, 8)),
2511+
(token.NAME, "ß", (4, 0), (4, 1)),
2512+
]
2513+
for extra_tokens in (False, True):
2514+
for encoding in (None, "utf-8"):
2515+
with self.subTest(extra_tokens=extra_tokens, encoding=encoding):
2516+
stream = (StringIO(source) if encoding is None
2517+
else BytesIO(source.encode(encoding)))
2518+
tokens = list(tokenize._generate_tokens_from_c_tokenizer(
2519+
stream.readline, extra_tokens=extra_tokens,
2520+
encoding=encoding))
2521+
self.assertEqual(
2522+
[t[:4] for t in tokens
2523+
if t.type in (token.NAME, token.STRING)],
2524+
expected,
2525+
)
2526+
string = next(t for t in tokens if t.type == token.STRING)
2527+
self.assertEqual(string.line, f"é = '''{padding}\n漢''' + ö\n")
2528+
25012529
def test_extra_tokens_relaxes_lexer_errors(self):
25022530
cases = [
25032531
(

‎Modules/_testinternalcapi/tokenizer.c‎

Lines changed: 47 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -103,6 +103,52 @@ test_tokenizer_source(PyObject *Py_UNUSED(module),
103103
return NULL;
104104
}
105105

106+
static PyObject *
107+
test_tokenizer_source_views(PyObject *Py_UNUSED(module),
108+
PyObject *Py_UNUSED(args))
109+
{
110+
const char line[] = "\xce\xb2\n";
111+
_PyTok_SourceText source;
112+
_PyTok_SourceInit(&source);
113+
if (_PyTok_SourceAppendLine(&source, line, sizeof(line) - 1) < 0) {
114+
goto error;
115+
}
116+
_PyTok_SourceDiscard(&source);
117+
_PyTok_Off start = _PyTok_SourceAppendLine(&source, line, sizeof(line) - 1);
118+
if (start < 0) {
119+
goto error;
120+
}
121+
_PyTok_Span span = {start, start + (Py_ssize_t)sizeof(line) - 1};
122+
// Grow storage after saving a span with a nonzero logical base.
123+
Py_ssize_t capacity = source.cap;
124+
while (source.cap == capacity) {
125+
if (_PyTok_SourceAppendLine(&source, line, sizeof(line) - 1) < 0) {
126+
goto error;
127+
}
128+
}
129+
Py_ssize_t len;
130+
const char *view = _PyTok_SourceSpanView(&source, span, &len);
131+
if (check(len == (Py_ssize_t)sizeof(line) - 1 && memcmp(view, line, len) == 0,
132+
"source span changed after growth") < 0 ||
133+
check(_PyTok_SourceOffset(&source, view) == start &&
134+
_PyTok_SourcePointer(&source, span.end) == view + len,
135+
"source view lost its logical offset") < 0) {
136+
goto error;
137+
}
138+
_PyTok_Off end = source.base_offset + source.len;
139+
view = _PyTok_SourceSpanView(&source, (_PyTok_Span){end, end}, &len);
140+
if (check(len == 0 && view == source.bytes + source.len,
141+
"wrong empty view at source end") < 0) {
142+
goto error;
143+
}
144+
_PyTok_SourceClear(&source);
145+
Py_RETURN_NONE;
146+
147+
error:
148+
_PyTok_SourceClear(&source);
149+
return NULL;
150+
}
151+
106152
static PyObject *
107153
test_tokenizer_source_discard(PyObject *Py_UNUSED(module),
108154
PyObject *Py_UNUSED(args))
@@ -180,6 +226,7 @@ test_tokenizer_source_discard(PyObject *Py_UNUSED(module),
180226

181227
static PyMethodDef test_methods[] = {
182228
{"test_tokenizer_source", test_tokenizer_source, METH_NOARGS},
229+
{"test_tokenizer_source_views", test_tokenizer_source_views, METH_NOARGS},
183230
{"test_tokenizer_source_discard", test_tokenizer_source_discard, METH_NOARGS},
184231
{NULL},
185232
};

‎Parser/lexer/lexer.c‎

Lines changed: 19 additions & 15 deletions
Original file line numberDiff line numberDiff line change
@@ -29,8 +29,9 @@ _PyLexer_refill(struct tok_state *tok)
2929
#if defined(Py_DEBUG)
3030
if (tok->debug) {
3131
fprintf(stderr, "line[%d] = ", tok->lineno);
32-
_PyTokenizer_print_escape(stderr, _PyLexer_BufferPointer(tok, tok->cur),
33-
tok->inp - tok->cur);
32+
_PyTokenizer_print_escape(
33+
stderr, _PyTok_SourcePointer(&tok->source, tok->cur),
34+
tok->inp - tok->cur);
3435
fprintf(stderr, " tok->done = %d\n", tok->done);
3536
}
3637
#endif
@@ -39,7 +40,7 @@ _PyLexer_refill(struct tok_state *tok)
3940
return 0;
4041
}
4142
tok->line_start = tok->cur;
42-
if (contains_null_bytes(_PyLexer_BufferPointer(tok, tok->line_start),
43+
if (contains_null_bytes(_PyTok_SourcePointer(&tok->source, tok->line_start),
4344
tok->inp - tok->line_start)) {
4445
_PyTokenizer_syntaxerror(tok, "source code cannot contain null bytes");
4546
tok->cur = tok->inp;
@@ -56,7 +57,8 @@ _PyLexer_backup(struct tok_state *tok, int c)
5657
if (--tok->cur < tok->buf_offset) {
5758
Py_FatalError("tokenizer beginning of buffer");
5859
}
59-
if ((int)(unsigned char)*_PyLexer_BufferPointer(tok, tok->cur) != Py_CHARMASK(c)) {
60+
const char *cur = _PyTok_SourcePointer(&tok->source, tok->cur);
61+
if ((int)(unsigned char)*cur != Py_CHARMASK(c)) {
6062
Py_FatalError("tok_backup: wrong character");
6163
}
6264
}
@@ -74,7 +76,8 @@ verify_identifier(struct tok_state *tok)
7476
PyObject *s;
7577
if (tok_failed(tok))
7678
return 0;
77-
s = PyUnicode_DecodeUTF8(_PyLexer_BufferPointer(tok, tok->start), tok->cur - tok->start, NULL);
79+
s = PyUnicode_DecodeUTF8(_PyTok_SourcePointer(&tok->source, tok->start),
80+
tok->cur - tok->start, NULL);
7881
if (s == NULL) {
7982
if (PyErr_ExceptionMatches(PyExc_UnicodeDecodeError)) {
8083
tok->done = E_DECODE;
@@ -105,13 +108,13 @@ verify_identifier(struct tok_state *tok)
105108
Py_DECREF(s);
106109
if (Py_UNICODE_ISPRINTABLE(ch)) {
107110
_PyTokenizer_syntaxerror_at(
108-
tok, _PyLexer_BufferPointer(tok, tok->line_start),
111+
tok, _PyTok_SourcePointer(&tok->source, tok->line_start),
109112
error_cursor - tok->line_start, tok->lineno, -1, -1,
110113
"invalid character '%c' (U+%04X)", ch, ch);
111114
}
112115
else {
113116
_PyTokenizer_syntaxerror_at(
114-
tok, _PyLexer_BufferPointer(tok, tok->line_start),
117+
tok, _PyTok_SourcePointer(&tok->source, tok->line_start),
115118
error_cursor - tok->line_start, tok->lineno, -1, -1,
116119
"invalid non-printable character U+%04X", ch);
117120
}
@@ -193,14 +196,14 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
193196
}
194197

195198
if (tok->tok_extra_tokens) {
196-
p = _PyLexer_BufferPointer(tok, tok->start);
199+
p = _PyTok_SourcePointer(&tok->source, tok->start);
197200
}
198201

199202
if (tok->type_comments) {
200-
p = _PyLexer_BufferPointer(tok, tok->start);
203+
p = _PyTok_SourcePointer(&tok->source, tok->start);
201204
current_starting_col_offset = tok->start_loc.byte_col;
202205
prefix = type_comment_prefix;
203-
while (*prefix && p < _PyLexer_BufferPointer(tok, tok->cur)) {
206+
while (*prefix && p < _PyTok_SourcePointer(&tok->source, tok->cur)) {
204207
if (*prefix == ' ') {
205208
while (*p == ' ' || *p == '\t') {
206209
p++;
@@ -229,24 +232,25 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
229232
/* A TYPE_IGNORE is "type: ignore" followed by the end of the token
230233
* or anything ASCII and non-alphanumeric. */
231234
is_type_ignore = (
232-
_PyLexer_BufferPointer(tok, tok->cur) >= ignore_end && memcmp(p, "ignore", 6) == 0
233-
&& !(_PyLexer_BufferPointer(tok, tok->cur) > ignore_end
235+
_PyTok_SourcePointer(&tok->source, tok->cur) >= ignore_end
236+
&& memcmp(p, "ignore", 6) == 0
237+
&& !(_PyTok_SourcePointer(&tok->source, tok->cur) > ignore_end
234238
&& ((unsigned char)ignore_end[0] >= 128 || Py_ISALNUM(ignore_end[0]))));
235239

236240
int type = is_type_ignore ? TYPE_IGNORE : TYPE_COMMENT;
237241
int start_col_offset = is_type_ignore
238242
? ignore_end_col_offset : current_starting_col_offset;
239243
p_end = tok->cur;
240244
if (is_type_ignore) {
241-
p_start = _PyLexer_BufferOffset(tok, ignore_end);
245+
p_start = _PyTok_SourceOffset(&tok->source, ignore_end);
242246

243247
/* If this type ignore is the only thing on the line, consume the newline also. */
244248
if (blankline) {
245249
tok_nextc(tok);
246250
tok->layout.at_bol = 1;
247251
}
248252
} else {
249-
p_start = _PyLexer_BufferOffset(tok, type_start);
253+
p_start = _PyTok_SourceOffset(&tok->source, type_start);
250254
}
251255
_PyLexer_token_setup(tok, token, type, p_start, p_end);
252256
token->start_loc = (_PyTok_Loc){tok->lineno, start_col_offset};
@@ -257,7 +261,7 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
257261
}
258262
if (tok->tok_extra_tokens) {
259263
tok_backup(tok, c); /* don't eat the newline or EOF */
260-
p_start = _PyLexer_BufferOffset(tok, p);
264+
p_start = _PyTok_SourceOffset(&tok->source, p);
261265
p_end = tok->cur;
262266
tok->layout.comment_newline = blankline;
263267
return MAKE_TOKEN(COMMENT);

‎Parser/lexer/state.c‎

Lines changed: 0 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -58,9 +58,6 @@ _PyLexer_PopFTString(struct tok_state *tok)
5858
void
5959
_PyTokenizer_Free(struct tok_state *tok)
6060
{
61-
if (tok->encoding != NULL) {
62-
PyMem_Free(tok->encoding);
63-
}
6461
Py_XDECREF(tok->filename);
6562
Py_XDECREF(tok->module);
6663
_PyTok_ReaderFree(tok);

‎Parser/lexer/state.h‎

Lines changed: 0 additions & 31 deletions
Original file line numberDiff line numberDiff line change
@@ -80,7 +80,6 @@ struct tok_state {
8080
_PyTok_SourceText source;
8181
int done; /* E_OK normally, E_EOF at EOF, otherwise error code */
8282
/* NB If done != E_OK, cur must be == inp!!! */
83-
FILE *fp; /* Rest of input; NULL if tokenizing a string */
8483
lexer_layout_state layout;
8584
int lineno; /* Current line number */
8685
_PyTok_Loc start_loc;
@@ -92,9 +91,6 @@ struct tok_state {
9291
int parencolstack[MAXLEVEL];
9392
PyObject *filename;
9493
PyObject *module;
95-
/* Stuff for PEP 0263 */
96-
char *encoding; /* Source encoding. */
97-
9894
struct _PyTok_Reader *reader;
9995

10096
int type_comments; /* Whether to look for type comments */
@@ -135,33 +131,6 @@ _PyLexer_FTStringBracketDepth(const struct tok_state *tok,
135131
return tok->level - state->paren_level;
136132
}
137133

138-
static inline _PyTok_Off
139-
_PyLexer_BufferOffset(const struct tok_state *tok, const char *position)
140-
{
141-
const char *base = _PyTok_SourceData(&tok->source);
142-
assert(position >= base && position <= base + tok->source.len);
143-
return tok->source.base_offset + (position - base);
144-
}
145-
146-
static inline const char *
147-
_PyLexer_BufferPointer(const struct tok_state *tok, _PyTok_Off offset)
148-
{
149-
assert(offset >= tok->source.base_offset);
150-
assert(offset - tok->source.base_offset <= tok->source.len);
151-
return _PyTok_SourceData(&tok->source) + (offset - tok->source.base_offset);
152-
}
153-
154-
static inline const char *
155-
_PyLexer_BufferSpanView(const struct tok_state *tok, _PyTok_Span span,
156-
Py_ssize_t *length)
157-
{
158-
assert(length != NULL);
159-
assert(_PyTok_SpanIsValid(span));
160-
*length = span.end - span.start;
161-
(void)_PyLexer_BufferPointer(tok, span.end);
162-
return _PyLexer_BufferPointer(tok, span.start);
163-
}
164-
165134
static inline int
166135
_PyLexer_ByteColumn(const struct tok_state *tok)
167136
{

‎Parser/lexer/string.c‎

Lines changed: 6 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -77,8 +77,8 @@ finish_ftstring_expr(struct tok_state *tok, ftstring_state *state,
7777
return 0;
7878
}
7979
Py_ssize_t expr_len;
80-
const char *expr = _PyLexer_BufferSpanView(
81-
tok, state->expr_span, &expr_len);
80+
const char *expr = _PyTok_SourceSpanView(
81+
&tok->source, state->expr_span, &expr_len);
8282
tokenizer_comments *comments = state->comments;
8383
PyObject *res;
8484
if (comments != NULL && comments->count > 0) {
@@ -339,7 +339,8 @@ _PyLexer_scan_string(struct tok_state *tok, struct token *token, int c)
339339
}
340340
int end_lineno = tok->lineno;
341341
_PyTok_Loc location = tok->start_loc;
342-
const char *line = _PyLexer_BufferPointer(tok, tok->start) - location.byte_col;
342+
const char *line = _PyTok_SourcePointer(
343+
&tok->source, tok->start - location.byte_col);
343344
Py_ssize_t cursor_offset = (Py_ssize_t)location.byte_col + 1;
344345

345346
const ftstring_state *state = _PyLexer_CurrentFTString(tok);
@@ -460,7 +461,8 @@ _PyLexer_get_ftstring(struct tok_state *tok, ftstring_state *current, struct tok
460461

461462
int end_lineno = tok->lineno;
462463
_PyTok_Loc location = current->start_loc;
463-
const char *line = _PyLexer_BufferPointer(tok, current->start) - location.byte_col;
464+
const char *line = _PyTok_SourcePointer(
465+
&tok->source, current->start - location.byte_col);
464466
Py_ssize_t cursor_offset = (Py_ssize_t)location.byte_col + 1;
465467

466468
if (quote_size == 3) {

‎Parser/tokenizer/api.c‎

Lines changed: 15 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -4,6 +4,7 @@
44

55
#include "tokenizer.h"
66
#include "reader.h"
7+
#include "reader_internal.h"
78
#include "../lexer/state.h"
89

910
_PyTokenizer_Info
@@ -21,10 +22,10 @@ _PyTokenizer_GetInfo(const struct tok_state *tok)
2122
.delimiter_loc = {-1, -1},
2223
.in_formatted_string = tok->ftstring_depth != 0,
2324
.is_interactive = _PyTok_ReaderIsInteractive(tok),
24-
.is_file = tok->fp != NULL && tok->fp != stdin,
25+
.is_file = tok->reader->fp != NULL && tok->reader->fp != stdin,
2526
.filename = tok->filename,
2627
.module = tok->module,
27-
.encoding = tok->encoding,
28+
.encoding = tok->reader->encoding,
2829
};
2930
if (tok->level > 0) {
3031
int level = tok->level - 1;
@@ -45,14 +46,14 @@ _PyToken_TextView(const struct tok_state *tok, const struct token *token,
4546
*length = 0;
4647
return "";
4748
}
48-
return _PyLexer_BufferSpanView(tok, token->span, length);
49+
return _PyTok_SourceSpanView(&tok->source, token->span, length);
4950
}
5051

5152
const char *
5253
_PyTokenizer_SpanView(const struct tok_state *tok, _PyTok_Span span,
5354
Py_ssize_t *length)
5455
{
55-
return _PyLexer_BufferSpanView(tok, span, length);
56+
return _PyTok_SourceSpanView(&tok->source, span, length);
5657
}
5758

5859
void
@@ -63,16 +64,18 @@ _PyToken_GetView(const struct tok_state *tok, const struct token *token,
6364
assert((token->span.start == -1 && token->span.end == -1) ||
6465
_PyTok_SpanIsValid(token->span));
6566
if (token->span.start >= 0) {
66-
(void)_PyLexer_BufferPointer(tok, token->span.end);
67+
(void)_PyTok_SourcePointer(&tok->source, token->span.end);
6768
}
6869
view->text = token->span.start < 0
69-
? NULL : _PyLexer_BufferPointer(tok, token->span.start);
70+
? NULL : _PyTok_SourcePointer(&tok->source, token->span.start);
7071
view->length = token->span.end - token->span.start;
71-
view->end_line = _PyLexer_BufferPointer(tok, tok->line_start);
72-
view->line = ISSTRINGLIT(token->type)
73-
? view->text - token->start_loc.byte_col : view->end_line;
74-
view->line_length = tok->inp - tok->line_start +
75-
(view->end_line - view->line);
72+
view->end_line_start = tok->line_start;
73+
view->line_span = (_PyTok_Span){
74+
ISSTRINGLIT(token->type)
75+
? token->span.start - token->start_loc.byte_col : tok->line_start,
76+
tok->inp,
77+
};
78+
view->line = _PyTok_SourcePointer(&tok->source, view->line_span.start);
7679
view->implicit_newline = tok->implicit_newline;
7780
view->at_eof = tok->done == E_EOF;
7881
}
@@ -111,7 +114,7 @@ _PyTokenizer_ImplyDedents(struct tok_state *tok)
111114
int
112115
_PyTokenizer_HasTrailingStatement(const struct tok_state *tok)
113116
{
114-
const char *cur = _PyLexer_BufferPointer(tok, tok->cur);
117+
const char *cur = _PyTok_SourcePointer(&tok->source, tok->cur);
115118
char c = *cur;
116119
for (;;) {
117120
while (c == ' ' || c == '\t' || c == '\n' || c == '\014') {

0 commit comments

Comments
 (0)