Skip to content

Commit b516b52

Browse files
committed
gh-153569: Consolidate tokenizer source access and token views
1 parent 400de53 commit b516b52

6 files changed

Lines changed: 25 additions & 37 deletions

File tree

‎Parser/lexer/lexer_internal.h‎

Lines changed: 1 addition & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -40,14 +40,11 @@ tok_nextc(struct tok_state *tok)
4040
return EOF;
4141
}
4242
}
43-
assert(tok->cur >= tok->source.base_offset);
44-
assert(tok->cur - tok->source.base_offset < tok->source.len);
4543
if (tok->cur - tok->line_start >= INT_MAX) {
4644
tok->done = E_COLUMNOVERFLOW;
4745
return EOF;
4846
}
49-
return Py_CHARMASK(
50-
tok->source.bytes[tok->cur++ - tok->source.base_offset]);
47+
return _PyTok_SourceByte(&tok->source, tok->cur++);
5148
}
5249

5350
/* Return -1 on error, otherwise whether the line is blank. */

‎Parser/tokenizer/api.c‎

Lines changed: 1 addition & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -61,14 +61,7 @@ _PyToken_GetView(const struct tok_state *tok, const struct token *token,
6161
_PyToken_View *view)
6262
{
6363
assert(view != NULL);
64-
assert((token->span.start == -1 && token->span.end == -1) ||
65-
_PyTok_SpanIsValid(token->span));
66-
if (token->span.start >= 0) {
67-
(void)_PyTok_SourcePointer(&tok->source, token->span.end);
68-
}
69-
view->text = token->span.start < 0
70-
? NULL : _PyTok_SourcePointer(&tok->source, token->span.start);
71-
view->length = token->span.end - token->span.start;
64+
view->text = _PyToken_TextView(tok, token, &view->length);
7265
view->end_line_start = tok->line_start;
7366
view->line_span = (_PyTok_Span){
7467
ISSTRINGLIT(token->type)

‎Parser/tokenizer/reader.c‎

Lines changed: 10 additions & 11 deletions
Original file line numberDiff line numberDiff line change
@@ -174,15 +174,14 @@ next_prepared(struct tok_state *tok, _PyTok_Chunk *chunk)
174174
if (tok->lineno >= tok->source.nlines) {
175175
return _PYTOK_READ_EOF;
176176
}
177-
const char *start = _PyTok_SourcePointer(&tok->source, tok->inp);
178-
const char *newline = memchr(
179-
start, '\n', tok->source.bytes + tok->source.len - start);
180-
_PyTok_Off end = newline != NULL
181-
? newline - tok->source.bytes + 1 : tok->source.len;
177+
_PyTok_Span tail = {tok->inp, tok->source.base_offset + tok->source.len};
178+
Py_ssize_t remaining;
179+
const char *start = _PyTok_SourceSpanView(&tok->source, tail, &remaining);
180+
const char *newline = memchr(start, '\n', remaining);
182181
chunk->data = (char *)start;
183-
chunk->len = tok->source.bytes + end - start;
182+
chunk->len = newline != NULL ? newline - start + 1 : remaining;
184183
chunk->ownership = _PYTOK_CHUNK_BORROWED;
185-
chunk->implicit_newline = end == tok->source.len &&
184+
chunk->implicit_newline = chunk->len == remaining &&
186185
tok->reader->prepared_final_newline_is_implicit;
187186
return _PYTOK_READ_LINE;
188187
}
@@ -667,12 +666,12 @@ _PyTok_ReaderUnderflow(struct tok_state *tok)
667666
}
668667
tok->inp = source_start + chunk.len;
669668
}
670-
if (prepared) {
669+
else {
670+
_PyTok_Off source_start = _PyTok_SourceOffset(&tok->source, chunk.data);
671671
if (tok->start < 0 && _PyLexer_CurrentFTString(tok) == NULL) {
672-
tok->buf_offset = tok->source.base_offset +
673-
(chunk.data - tok->source.bytes);
672+
tok->buf_offset = source_start;
674673
}
675-
tok->inp = _PyTok_SourceOffset(&tok->source, chunk.data) + chunk.len;
674+
tok->inp = source_start + chunk.len;
676675
}
677676
tok->implicit_newline = chunk.implicit_newline;
678677

‎Parser/tokenizer/source.h‎

Lines changed: 8 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -37,6 +37,14 @@ _PyTok_SourcePointer(const _PyTok_SourceText *source, _PyTok_Off offset)
3737
return _PyTok_SourceData(source) + (offset - source->base_offset);
3838
}
3939

40+
static inline unsigned char
41+
_PyTok_SourceByte(const _PyTok_SourceText *source, _PyTok_Off offset)
42+
{
43+
assert(offset >= source->base_offset);
44+
assert(offset - source->base_offset < source->len);
45+
return (unsigned char)source->bytes[offset - source->base_offset];
46+
}
47+
4048
static inline const char *
4149
_PyTok_SourceSpanView(const _PyTok_SourceText *source, _PyTok_Span span,
4250
Py_ssize_t *length)

‎Parser/tokenizer/tokenizer.h‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -73,7 +73,7 @@ _PyTokenizer_Info _PyTokenizer_GetInfo(const struct tok_state *);
7373
/* An absent token span has a nonnull empty text view. */
7474
const char *_PyToken_TextView(
7575
const struct tok_state *, const struct token *, Py_ssize_t *);
76-
/* Use the token from the most recent Get. text is NULL for an absent span;
76+
/* Use the token from the most recent Get. An absent span has nonnull empty text.
7777
line contains the bytes of line_span, the token's complete physical line
7878
range. end_line_start is the logical offset of its final physical line. */
7979
void _PyToken_GetView(

‎Python/Python-tokenize.c‎

Lines changed: 4 additions & 13 deletions
Original file line numberDiff line numberDiff line change
@@ -185,7 +185,7 @@ _get_col_offsets(tokenizeriterobject *it, const struct token *token,
185185
Py_ssize_t lineno = token->start_loc.lineno;
186186
Py_ssize_t end_lineno = token->end_loc.lineno;
187187
Py_ssize_t byte_offset = -1;
188-
if (token_start >= 0 && token_start >= view->line_span.start) {
188+
if (token_start >= view->line_span.start) {
189189
byte_offset = token_start - view->line_span.start;
190190
if (line_changed) {
191191
*col_offset = _PyPegen_byte_offset_to_character_offset_line(line, 0, byte_offset);
@@ -199,7 +199,7 @@ _get_col_offsets(tokenizeriterobject *it, const struct token *token,
199199
}
200200
}
201201

202-
if (token_end >= 0 && token_end >= view->end_line_start) {
202+
if (token_end >= view->end_line_start) {
203203
Py_ssize_t end_byte_offset = token_end - view->end_line_start;
204204
if (lineno == end_lineno) {
205205
// Avoid rescanning the prefix of a very long line.
@@ -251,15 +251,7 @@ tokenizeriter_next(PyObject *op)
251251
}
252252
_PyToken_View view;
253253
_PyToken_GetView(it->tok, &token, &view);
254-
const char *token_start = view.text;
255-
PyObject *str;
256-
if (token.span.start < 0) {
257-
assert(token.span.start == -1 && token.span.end == -1);
258-
str = Py_GetConstant(Py_CONSTANT_EMPTY_STR);
259-
}
260-
else {
261-
str = PyUnicode_FromStringAndSize(token_start, view.length);
262-
}
254+
PyObject *str = PyUnicode_FromStringAndSize(view.text, view.length);
263255
if (str == NULL) {
264256
goto exit;
265257
}
@@ -307,8 +299,7 @@ tokenizeriter_next(PyObject *op)
307299
else if (type == NEWLINE) {
308300
if (!view.implicit_newline) {
309301
Py_DECREF(str);
310-
assert(token_start != NULL);
311-
if (token_start[0] == '\r') {
302+
if (view.text[0] == '\r') {
312303
str = PyUnicode_FromString("\r\n");
313304
} else {
314305
str = PyUnicode_FromString("\n");

0 commit comments

Comments
 (0)