Skip to content

Commit 3e32141

Browse files
committed
gh-153569: Move tokenizer source access into the source API
1 parent 1643525 commit 3e32141

10 files changed

Lines changed: 138 additions & 74 deletions

File tree

‎Lib/test/test_capi/test_tokenizer.py‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -12,6 +12,9 @@ def test_source(self):
1212
def test_source_discard(self):
1313
_testinternalcapi.test_tokenizer_source_discard()
1414

15+
def test_source_views(self):
16+
_testinternalcapi.test_tokenizer_source_views()
17+
1518

1619
if __name__ == "__main__":
1720
unittest.main()

‎Modules/_testinternalcapi/tokenizer.c‎

Lines changed: 47 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -103,6 +103,52 @@ test_tokenizer_source(PyObject *Py_UNUSED(module),
103103
return NULL;
104104
}
105105

106+
static PyObject *
107+
test_tokenizer_source_views(PyObject *Py_UNUSED(module),
108+
PyObject *Py_UNUSED(args))
109+
{
110+
const char line[] = "\xce\xb2\n";
111+
_PyTok_SourceText source;
112+
_PyTok_SourceInit(&source);
113+
if (_PyTok_SourceAppendLine(&source, line, sizeof(line) - 1) < 0) {
114+
goto error;
115+
}
116+
_PyTok_SourceDiscard(&source);
117+
_PyTok_Off start = _PyTok_SourceAppendLine(&source, line, sizeof(line) - 1);
118+
if (start < 0) {
119+
goto error;
120+
}
121+
_PyTok_Span span = {start, start + (Py_ssize_t)sizeof(line) - 1};
122+
// Grow storage after saving a span with a nonzero logical base.
123+
Py_ssize_t capacity = source.cap;
124+
while (source.cap == capacity) {
125+
if (_PyTok_SourceAppendLine(&source, line, sizeof(line) - 1) < 0) {
126+
goto error;
127+
}
128+
}
129+
Py_ssize_t len;
130+
const char *view = _PyTok_SourceSpanView(&source, span, &len);
131+
if (check(len == (Py_ssize_t)sizeof(line) - 1 && memcmp(view, line, len) == 0,
132+
"source span changed after growth") < 0 ||
133+
check(_PyTok_SourceOffset(&source, view) == start &&
134+
_PyTok_SourcePointer(&source, span.end) == view + len,
135+
"source view lost its logical offset") < 0) {
136+
goto error;
137+
}
138+
_PyTok_Off end = source.base_offset + source.len;
139+
view = _PyTok_SourceSpanView(&source, (_PyTok_Span){end, end}, &len);
140+
if (check(len == 0 && view == source.bytes + source.len,
141+
"wrong empty view at source end") < 0) {
142+
goto error;
143+
}
144+
_PyTok_SourceClear(&source);
145+
Py_RETURN_NONE;
146+
147+
error:
148+
_PyTok_SourceClear(&source);
149+
return NULL;
150+
}
151+
106152
static PyObject *
107153
test_tokenizer_source_discard(PyObject *Py_UNUSED(module),
108154
PyObject *Py_UNUSED(args))
@@ -180,6 +226,7 @@ test_tokenizer_source_discard(PyObject *Py_UNUSED(module),
180226

181227
static PyMethodDef test_methods[] = {
182228
{"test_tokenizer_source", test_tokenizer_source, METH_NOARGS},
229+
{"test_tokenizer_source_views", test_tokenizer_source_views, METH_NOARGS},
183230
{"test_tokenizer_source_discard", test_tokenizer_source_discard, METH_NOARGS},
184231
{NULL},
185232
};

‎Parser/lexer/lexer.c‎

Lines changed: 19 additions & 15 deletions
Original file line numberDiff line numberDiff line change
@@ -29,8 +29,9 @@ _PyLexer_refill(struct tok_state *tok)
2929
#if defined(Py_DEBUG)
3030
if (tok->debug) {
3131
fprintf(stderr, "line[%d] = ", tok->lineno);
32-
_PyTokenizer_print_escape(stderr, _PyLexer_BufferPointer(tok, tok->cur),
33-
tok->inp - tok->cur);
32+
_PyTokenizer_print_escape(
33+
stderr, _PyTok_SourcePointer(&tok->source, tok->cur),
34+
tok->inp - tok->cur);
3435
fprintf(stderr, " tok->done = %d\n", tok->done);
3536
}
3637
#endif
@@ -39,7 +40,7 @@ _PyLexer_refill(struct tok_state *tok)
3940
return 0;
4041
}
4142
tok->line_start = tok->cur;
42-
if (contains_null_bytes(_PyLexer_BufferPointer(tok, tok->line_start),
43+
if (contains_null_bytes(_PyTok_SourcePointer(&tok->source, tok->line_start),
4344
tok->inp - tok->line_start)) {
4445
_PyTokenizer_syntaxerror(tok, "source code cannot contain null bytes");
4546
tok->cur = tok->inp;
@@ -56,7 +57,8 @@ _PyLexer_backup(struct tok_state *tok, int c)
5657
if (--tok->cur < tok->buf_offset) {
5758
Py_FatalError("tokenizer beginning of buffer");
5859
}
59-
if ((int)(unsigned char)*_PyLexer_BufferPointer(tok, tok->cur) != Py_CHARMASK(c)) {
60+
const char *cur = _PyTok_SourcePointer(&tok->source, tok->cur);
61+
if ((int)(unsigned char)*cur != Py_CHARMASK(c)) {
6062
Py_FatalError("tok_backup: wrong character");
6163
}
6264
}
@@ -74,7 +76,8 @@ verify_identifier(struct tok_state *tok)
7476
PyObject *s;
7577
if (tok_failed(tok))
7678
return 0;
77-
s = PyUnicode_DecodeUTF8(_PyLexer_BufferPointer(tok, tok->start), tok->cur - tok->start, NULL);
79+
s = PyUnicode_DecodeUTF8(_PyTok_SourcePointer(&tok->source, tok->start),
80+
tok->cur - tok->start, NULL);
7881
if (s == NULL) {
7982
if (PyErr_ExceptionMatches(PyExc_UnicodeDecodeError)) {
8083
tok->done = E_DECODE;
@@ -105,13 +108,13 @@ verify_identifier(struct tok_state *tok)
105108
Py_DECREF(s);
106109
if (Py_UNICODE_ISPRINTABLE(ch)) {
107110
_PyTokenizer_syntaxerror_at(
108-
tok, _PyLexer_BufferPointer(tok, tok->line_start),
111+
tok, _PyTok_SourcePointer(&tok->source, tok->line_start),
109112
error_cursor - tok->line_start, tok->lineno, -1, -1,
110113
"invalid character '%c' (U+%04X)", ch, ch);
111114
}
112115
else {
113116
_PyTokenizer_syntaxerror_at(
114-
tok, _PyLexer_BufferPointer(tok, tok->line_start),
117+
tok, _PyTok_SourcePointer(&tok->source, tok->line_start),
115118
error_cursor - tok->line_start, tok->lineno, -1, -1,
116119
"invalid non-printable character U+%04X", ch);
117120
}
@@ -193,14 +196,14 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
193196
}
194197

195198
if (tok->tok_extra_tokens) {
196-
p = _PyLexer_BufferPointer(tok, tok->start);
199+
p = _PyTok_SourcePointer(&tok->source, tok->start);
197200
}
198201

199202
if (tok->type_comments) {
200-
p = _PyLexer_BufferPointer(tok, tok->start);
203+
p = _PyTok_SourcePointer(&tok->source, tok->start);
201204
current_starting_col_offset = tok->start_loc.byte_col;
202205
prefix = type_comment_prefix;
203-
while (*prefix && p < _PyLexer_BufferPointer(tok, tok->cur)) {
206+
while (*prefix && p < _PyTok_SourcePointer(&tok->source, tok->cur)) {
204207
if (*prefix == ' ') {
205208
while (*p == ' ' || *p == '\t') {
206209
p++;
@@ -229,24 +232,25 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
229232
/* A TYPE_IGNORE is "type: ignore" followed by the end of the token
230233
* or anything ASCII and non-alphanumeric. */
231234
is_type_ignore = (
232-
_PyLexer_BufferPointer(tok, tok->cur) >= ignore_end && memcmp(p, "ignore", 6) == 0
233-
&& !(_PyLexer_BufferPointer(tok, tok->cur) > ignore_end
235+
_PyTok_SourcePointer(&tok->source, tok->cur) >= ignore_end
236+
&& memcmp(p, "ignore", 6) == 0
237+
&& !(_PyTok_SourcePointer(&tok->source, tok->cur) > ignore_end
234238
&& ((unsigned char)ignore_end[0] >= 128 || Py_ISALNUM(ignore_end[0]))));
235239

236240
int type = is_type_ignore ? TYPE_IGNORE : TYPE_COMMENT;
237241
int start_col_offset = is_type_ignore
238242
? ignore_end_col_offset : current_starting_col_offset;
239243
p_end = tok->cur;
240244
if (is_type_ignore) {
241-
p_start = _PyLexer_BufferOffset(tok, ignore_end);
245+
p_start = _PyTok_SourceOffset(&tok->source, ignore_end);
242246

243247
/* If this type ignore is the only thing on the line, consume the newline also. */
244248
if (blankline) {
245249
tok_nextc(tok);
246250
tok->layout.at_bol = 1;
247251
}
248252
} else {
249-
p_start = _PyLexer_BufferOffset(tok, type_start);
253+
p_start = _PyTok_SourceOffset(&tok->source, type_start);
250254
}
251255
_PyLexer_token_setup(tok, token, type, p_start, p_end);
252256
token->start_loc = (_PyTok_Loc){tok->lineno, start_col_offset};
@@ -257,7 +261,7 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
257261
}
258262
if (tok->tok_extra_tokens) {
259263
tok_backup(tok, c); /* don't eat the newline or EOF */
260-
p_start = _PyLexer_BufferOffset(tok, p);
264+
p_start = _PyTok_SourceOffset(&tok->source, p);
261265
p_end = tok->cur;
262266
tok->layout.comment_newline = blankline;
263267
return MAKE_TOKEN(COMMENT);

‎Parser/lexer/lexer_internal.h‎

Lines changed: 1 addition & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -40,14 +40,11 @@ tok_nextc(struct tok_state *tok)
4040
return EOF;
4141
}
4242
}
43-
assert(tok->cur >= tok->source.base_offset);
44-
assert(tok->cur - tok->source.base_offset < tok->source.len);
4543
if (tok->cur - tok->line_start >= INT_MAX) {
4644
tok->done = E_COLUMNOVERFLOW;
4745
return EOF;
4846
}
49-
return Py_CHARMASK(
50-
tok->source.bytes[tok->cur++ - tok->source.base_offset]);
47+
return _PyTok_SourceByte(&tok->source, tok->cur++);
5148
}
5249

5350
/* Return -1 on error, otherwise whether the line is blank. */

‎Parser/lexer/state.h‎

Lines changed: 0 additions & 27 deletions
Original file line numberDiff line numberDiff line change
@@ -135,33 +135,6 @@ _PyLexer_FTStringBracketDepth(const struct tok_state *tok,
135135
return tok->level - state->paren_level;
136136
}
137137

138-
static inline _PyTok_Off
139-
_PyLexer_BufferOffset(const struct tok_state *tok, const char *position)
140-
{
141-
const char *base = _PyTok_SourceData(&tok->source);
142-
assert(position >= base && position <= base + tok->source.len);
143-
return tok->source.base_offset + (position - base);
144-
}
145-
146-
static inline const char *
147-
_PyLexer_BufferPointer(const struct tok_state *tok, _PyTok_Off offset)
148-
{
149-
assert(offset >= tok->source.base_offset);
150-
assert(offset - tok->source.base_offset <= tok->source.len);
151-
return _PyTok_SourceData(&tok->source) + (offset - tok->source.base_offset);
152-
}
153-
154-
static inline const char *
155-
_PyLexer_BufferSpanView(const struct tok_state *tok, _PyTok_Span span,
156-
Py_ssize_t *length)
157-
{
158-
assert(length != NULL);
159-
assert(_PyTok_SpanIsValid(span));
160-
*length = span.end - span.start;
161-
(void)_PyLexer_BufferPointer(tok, span.end);
162-
return _PyLexer_BufferPointer(tok, span.start);
163-
}
164-
165138
static inline int
166139
_PyLexer_ByteColumn(const struct tok_state *tok)
167140
{

‎Parser/lexer/string.c‎

Lines changed: 6 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -77,8 +77,8 @@ finish_ftstring_expr(struct tok_state *tok, ftstring_state *state,
7777
return 0;
7878
}
7979
Py_ssize_t expr_len;
80-
const char *expr = _PyLexer_BufferSpanView(
81-
tok, state->expr_span, &expr_len);
80+
const char *expr = _PyTok_SourceSpanView(
81+
&tok->source, state->expr_span, &expr_len);
8282
tokenizer_comments *comments = state->comments;
8383
PyObject *res;
8484
if (comments != NULL && comments->count > 0) {
@@ -339,7 +339,8 @@ _PyLexer_scan_string(struct tok_state *tok, struct token *token, int c)
339339
}
340340
int end_lineno = tok->lineno;
341341
_PyTok_Loc location = tok->start_loc;
342-
const char *line = _PyLexer_BufferPointer(tok, tok->start) - location.byte_col;
342+
const char *line = _PyTok_SourcePointer(
343+
&tok->source, tok->start - location.byte_col);
343344
Py_ssize_t cursor_offset = (Py_ssize_t)location.byte_col + 1;
344345

345346
const ftstring_state *state = _PyLexer_CurrentFTString(tok);
@@ -460,7 +461,8 @@ _PyLexer_get_ftstring(struct tok_state *tok, ftstring_state *current, struct tok
460461

461462
int end_lineno = tok->lineno;
462463
_PyTok_Loc location = current->start_loc;
463-
const char *line = _PyLexer_BufferPointer(tok, current->start) - location.byte_col;
464+
const char *line = _PyTok_SourcePointer(
465+
&tok->source, current->start - location.byte_col);
464466
Py_ssize_t cursor_offset = (Py_ssize_t)location.byte_col + 1;
465467

466468
if (quote_size == 3) {

‎Parser/tokenizer/api.c‎

Lines changed: 6 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -45,14 +45,14 @@ _PyToken_TextView(const struct tok_state *tok, const struct token *token,
4545
*length = 0;
4646
return "";
4747
}
48-
return _PyLexer_BufferSpanView(tok, token->span, length);
48+
return _PyTok_SourceSpanView(&tok->source, token->span, length);
4949
}
5050

5151
const char *
5252
_PyTokenizer_SpanView(const struct tok_state *tok, _PyTok_Span span,
5353
Py_ssize_t *length)
5454
{
55-
return _PyLexer_BufferSpanView(tok, span, length);
55+
return _PyTok_SourceSpanView(&tok->source, span, length);
5656
}
5757

5858
void
@@ -63,12 +63,12 @@ _PyToken_GetView(const struct tok_state *tok, const struct token *token,
6363
assert((token->span.start == -1 && token->span.end == -1) ||
6464
_PyTok_SpanIsValid(token->span));
6565
if (token->span.start >= 0) {
66-
(void)_PyLexer_BufferPointer(tok, token->span.end);
66+
(void)_PyTok_SourcePointer(&tok->source, token->span.end);
6767
}
6868
view->text = token->span.start < 0
69-
? NULL : _PyLexer_BufferPointer(tok, token->span.start);
69+
? NULL : _PyTok_SourcePointer(&tok->source, token->span.start);
7070
view->length = token->span.end - token->span.start;
71-
view->end_line = _PyLexer_BufferPointer(tok, tok->line_start);
71+
view->end_line = _PyTok_SourcePointer(&tok->source, tok->line_start);
7272
view->line = ISSTRINGLIT(token->type)
7373
? view->text - token->start_loc.byte_col : view->end_line;
7474
view->line_length = tok->inp - tok->line_start +
@@ -111,7 +111,7 @@ _PyTokenizer_ImplyDedents(struct tok_state *tok)
111111
int
112112
_PyTokenizer_HasTrailingStatement(const struct tok_state *tok)
113113
{
114-
const char *cur = _PyLexer_BufferPointer(tok, tok->cur);
114+
const char *cur = _PyTok_SourcePointer(&tok->source, tok->cur);
115115
char c = *cur;
116116
for (;;) {
117117
while (c == ' ' || c == '\t' || c == '\n' || c == '\014') {

‎Parser/tokenizer/helpers.c‎

Lines changed: 7 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -91,9 +91,9 @@ _PyTokenizer_syntaxerror(struct tok_state *tok, const char *format, ...)
9191
// These errors are cleaned on startup. Todo: Fix it.
9292
va_list vargs;
9393
va_start(vargs, format);
94-
int ret = _syntaxerror_range(tok, _PyLexer_BufferPointer(tok, tok->line_start),
95-
tok->cur - tok->line_start, tok->lineno,
96-
format, -1, -1, vargs);
94+
int ret = _syntaxerror_range(
95+
tok, _PyTok_SourcePointer(&tok->source, tok->line_start),
96+
tok->cur - tok->line_start, tok->lineno, format, -1, -1, vargs);
9797
va_end(vargs);
9898
return ret;
9999
}
@@ -105,9 +105,10 @@ _PyTokenizer_syntaxerror_known_range(struct tok_state *tok,
105105
{
106106
va_list vargs;
107107
va_start(vargs, format);
108-
int ret = _syntaxerror_range(tok, _PyLexer_BufferPointer(tok, tok->line_start),
109-
tok->cur - tok->line_start, tok->lineno,
110-
format, col_offset, end_col_offset, vargs);
108+
int ret = _syntaxerror_range(
109+
tok, _PyTok_SourcePointer(&tok->source, tok->line_start),
110+
tok->cur - tok->line_start, tok->lineno,
111+
format, col_offset, end_col_offset, vargs);
111112
va_end(vargs);
112113
return ret;
113114
}

‎Parser/tokenizer/reader.c‎

Lines changed: 12 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -173,15 +173,14 @@ next_prepared(struct tok_state *tok, _PyTok_Chunk *chunk)
173173
if (tok->lineno >= tok->source.nlines) {
174174
return _PYTOK_READ_EOF;
175175
}
176-
const char *start = _PyLexer_BufferPointer(tok, tok->inp);
177-
const char *newline = memchr(
178-
start, '\n', tok->source.bytes + tok->source.len - start);
179-
_PyTok_Off end = newline != NULL
180-
? newline - tok->source.bytes + 1 : tok->source.len;
176+
_PyTok_Span tail = {tok->inp, tok->source.base_offset + tok->source.len};
177+
Py_ssize_t remaining;
178+
const char *start = _PyTok_SourceSpanView(&tok->source, tail, &remaining);
179+
const char *newline = memchr(start, '\n', remaining);
181180
chunk->data = (char *)start;
182-
chunk->len = tok->source.bytes + end - start;
181+
chunk->len = newline != NULL ? newline - start + 1 : remaining;
183182
chunk->ownership = _PYTOK_CHUNK_BORROWED;
184-
chunk->implicit_newline = end == tok->source.len &&
183+
chunk->implicit_newline = chunk->len == remaining &&
185184
tok->reader->prepared_final_newline_is_implicit;
186185
return _PYTOK_READ_LINE;
187186
}
@@ -665,19 +664,20 @@ _PyTok_ReaderUnderflow(struct tok_state *tok)
665664
}
666665
tok->inp = source_start + chunk.len;
667666
}
668-
if (prepared) {
667+
else {
668+
_PyTok_Off source_start = _PyTok_SourceOffset(&tok->source, chunk.data);
669669
if (tok->start < 0 && _PyLexer_CurrentFTString(tok) == NULL) {
670-
tok->buf_offset = tok->source.base_offset +
671-
(chunk.data - tok->source.bytes);
670+
tok->buf_offset = source_start;
672671
}
673-
tok->inp = _PyLexer_BufferOffset(tok, chunk.data) + chunk.len;
672+
tok->inp = source_start + chunk.len;
674673
}
675674
tok->implicit_newline = chunk.implicit_newline;
676675

677676
tok->lineno++;
678677
if (kind == _PYTOK_READER_FILE &&
679678
(tok->encoding == NULL || strcmp(tok->encoding, "utf-8") == 0) &&
680-
!_PyTokenizer_ensure_utf8(_PyLexer_BufferPointer(tok, tok->cur), tok, tok->lineno)) {
679+
!_PyTokenizer_ensure_utf8(
680+
_PyTok_SourcePointer(&tok->source, tok->cur), tok, tok->lineno)) {
681681
_PyTok_ChunkClear(&chunk);
682682
return 0;
683683
}

0 commit comments

Comments
 (0)