Skip to content

Commit ffb781d

Browse files
committed
gh-153569: Remove redundant tokenizer views and simplify reader branching
1 parent e60d602 commit ffb781d

6 files changed

Lines changed: 34 additions & 33 deletions

File tree

‎Parser/lexer/string.c‎

Lines changed: 13 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -79,7 +79,8 @@ finish_ftstring_expr(struct tok_state *tok, ftstring_state *state,
7979
tokenizer_comments *comments = state->comments;
8080
PyObject *res;
8181
if (comments != NULL && comments->count > 0) {
82-
Py_ssize_t stripped_size = state->expr_span.end - state->expr_span.start;
82+
Py_ssize_t stripped_size =
83+
state->expr_span.end - state->expr_span.start;
8384
Py_ssize_t comment_count = 0;
8485
for (Py_ssize_t i = 0; i < comments->count; i++) {
8586
_PyTok_Span comment = comments->spans[i];
@@ -98,22 +99,22 @@ finish_ftstring_expr(struct tok_state *tok, ftstring_state *state,
9899
PyErr_NoMemory();
99100
return -1;
100101
}
101-
_PyTok_Off copied_to = state->expr_span.start;
102+
_PyTok_Span kept = {state->expr_span.start, state->expr_span.start};
102103
Py_ssize_t stripped_len = 0;
103-
for (Py_ssize_t i = 0; i <= comment_count; i++) {
104-
_PyTok_Span span = {
105-
copied_to,
106-
i < comment_count ? comments->spans[i].start : state->expr_span.end,
107-
};
104+
for (Py_ssize_t i = 0; i < comment_count; i++) {
105+
kept.end = comments->spans[i].start;
108106
Py_ssize_t length;
109-
const char *text = _PyTok_SourceSpanView(&tok->source, span, &length);
107+
const char *text = _PyTok_SourceSpanView(
108+
&tok->source, kept, &length);
110109
memcpy(stripped + stripped_len, text, (size_t)length);
111110
stripped_len += length;
112-
if (i < comment_count) {
113-
copied_to = comments->spans[i].end;
114-
}
111+
kept.start = comments->spans[i].end;
115112
}
116-
res = PyUnicode_DecodeUTF8(stripped, stripped_len, NULL);
113+
kept.end = state->expr_span.end;
114+
Py_ssize_t length;
115+
const char *text = _PyTok_SourceSpanView(&tok->source, kept, &length);
116+
memcpy(stripped + stripped_len, text, (size_t)length);
117+
res = PyUnicode_DecodeUTF8(stripped, stripped_size, NULL);
117118
PyMem_Free(stripped);
118119
}
119120
else {

‎Parser/tokenizer/api.c‎

Lines changed: 0 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -68,7 +68,6 @@ _PyToken_GetView(const struct tok_state *tok, const struct token *token,
6868
? token->span.start - token->start_loc.byte_col : tok->line_start,
6969
tok->inp,
7070
};
71-
view->line = _PyTok_SourcePointer(&tok->source, view->line_span.start);
7271
view->implicit_newline = tok->implicit_newline;
7372
view->at_eof = tok->done == E_EOF;
7473
}

‎Parser/tokenizer/reader.c‎

Lines changed: 8 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -177,7 +177,8 @@ next_prepared(struct tok_state *tok, _PyTok_Chunk *chunk)
177177
_PyTok_Span tail = {tok->inp, tok->source.base_offset + tok->source.len};
178178
_PyTok_Off newline = _PyTok_SourceFindByte(&tok->source, tail, '\n');
179179
_PyTok_Span line = {tail.start, newline >= 0 ? newline + 1 : tail.end};
180-
chunk->data = (char *)_PyTok_SourceSpanView(&tok->source, line, &chunk->len);
180+
chunk->data = (char *)_PyTok_SourceSpanView(
181+
&tok->source, line, &chunk->len);
181182
chunk->ownership = _PYTOK_CHUNK_BORROWED;
182183
chunk->implicit_newline = line.end == tail.end &&
183184
tok->reader->prepared_final_newline_is_implicit;
@@ -607,7 +608,7 @@ _PyTok_ReaderUnderflow(struct tok_state *tok)
607608
_PyTok_ReaderKind kind = reader->kind;
608609
int prepared = kind == _PYTOK_READER_PREPARED;
609610
int streaming = reader_is_streaming(kind);
610-
int reset_buffer = !prepared && tok->start < 0 &&
611+
int reset_buffer = tok->start < 0 &&
611612
_PyLexer_CurrentFTString(tok) == NULL;
612613

613614
_PyTok_Chunk chunk;
@@ -660,12 +661,11 @@ _PyTok_ReaderUnderflow(struct tok_state *tok)
660661
tok->cur = source_start;
661662
tok->buf_offset = source_start;
662663
tok->line_start = tok->buf_offset;
663-
tok->start = -1;
664664
}
665665
tok->inp = source_start + chunk.len;
666666
}
667667
else {
668-
if (tok->start < 0 && _PyLexer_CurrentFTString(tok) == NULL) {
668+
if (reset_buffer) {
669669
tok->buf_offset = tok->inp;
670670
}
671671
tok->inp += chunk.len;
@@ -674,9 +674,11 @@ _PyTok_ReaderUnderflow(struct tok_state *tok)
674674

675675
tok->lineno++;
676676
if (kind == _PYTOK_READER_FILE &&
677-
(reader->encoding == NULL || strcmp(reader->encoding, "utf-8") == 0) &&
677+
(reader->encoding == NULL ||
678+
strcmp(reader->encoding, "utf-8") == 0) &&
678679
!_PyTokenizer_ensure_utf8(
679-
_PyTok_SourcePointer(&tok->source, tok->cur), tok, tok->lineno)) {
680+
_PyTok_SourcePointer(&tok->source, tok->cur),
681+
tok, tok->lineno)) {
680682
_PyTok_ChunkClear(&chunk);
681683
return 0;
682684
}

‎Parser/tokenizer/source.h‎

Lines changed: 3 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -43,14 +43,15 @@ _PyTok_SourceSpanView(const _PyTok_SourceText *source, _PyTok_Span span,
4343
{
4444
assert(length != NULL);
4545
assert(_PyTok_SpanIsValid(span));
46+
assert(span.end - source->base_offset <= source->len);
4647
*length = span.end - span.start;
47-
(void)_PyTok_SourcePointer(source, span.end);
4848
return _PyTok_SourcePointer(source, span.start);
4949
}
5050

5151
/* Return the first matching offset within span, or -1 if absent. */
5252
static inline _PyTok_Off
53-
_PyTok_SourceFindByte(const _PyTok_SourceText *source, _PyTok_Span span, int byte)
53+
_PyTok_SourceFindByte(const _PyTok_SourceText *source, _PyTok_Span span,
54+
int byte)
5455
{
5556
Py_ssize_t length;
5657
const char *data = _PyTok_SourceSpanView(source, span, &length);

‎Parser/tokenizer/tokenizer.h‎

Lines changed: 2 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -21,7 +21,6 @@ struct token {
2121
typedef struct {
2222
const char *text;
2323
Py_ssize_t length;
24-
const char *line;
2524
_PyTok_Span line_span;
2625
_PyTok_Off end_line_start;
2726
int implicit_newline;
@@ -74,8 +73,8 @@ _PyTokenizer_Info _PyTokenizer_GetInfo(const struct tok_state *);
7473
const char *_PyToken_TextView(
7574
const struct tok_state *, const struct token *, Py_ssize_t *);
7675
/* Use the token from the most recent Get. An absent span has nonnull empty text.
77-
line contains the bytes of line_span, the token's complete physical line
78-
range. end_line_start is the logical offset of its final physical line. */
76+
line_span covers the token's complete physical line range. end_line_start
77+
is the logical offset of its final physical line. */
7978
void _PyToken_GetView(
8079
const struct tok_state *tok, const struct token *token,
8180
_PyToken_View *view);

‎Python/Python-tokenize.c‎

Lines changed: 8 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -158,12 +158,16 @@ _tokenizer_error(tokenizeriterobject *it)
158158

159159
static PyObject *
160160
_get_current_line(tokenizeriterobject *it, int current_lineno,
161-
const char *line_start, Py_ssize_t size, int *line_changed)
161+
const _PyToken_View *view, int *line_changed)
162162
{
163163
_Py_CRITICAL_SECTION_ASSERT_OBJECT_LOCKED(it);
164164
if (current_lineno != it->last_lineno) {
165-
// Line has changed since last token, so we fetch the new line and cache it
166-
// in the iter object.
165+
Py_ssize_t size;
166+
const char *line_start = _PyTokenizer_SpanView(
167+
it->tok, view->line_span, &size);
168+
if (size > 0 && view->implicit_newline) {
169+
size--;
170+
}
167171
Py_XDECREF(it->last_line);
168172
it->last_line = PyUnicode_DecodeUTF8(line_start, size, "replace");
169173
it->byte_col_offset_diff = 0;
@@ -263,13 +267,8 @@ tokenizeriter_next(PyObject *op)
263267
if (it->extra_tokens && is_trailing_token) {
264268
line = Py_GetConstant(Py_CONSTANT_EMPTY_STR);
265269
} else {
266-
Py_ssize_t size = view.line_span.end - view.line_span.start;
267-
if (size >= 1 && view.implicit_newline) {
268-
size -= 1;
269-
}
270-
271270
line = _get_current_line(
272-
it, token.end_loc.lineno, view.line, size, &line_changed);
271+
it, token.end_loc.lineno, &view, &line_changed);
273272
}
274273
if (line == NULL) {
275274
Py_DECREF(str);

0 commit comments

Comments
 (0)