Skip to content

Commit 1cfbc93

Browse files
committed
gh-153569: Move tokenizer input state and encoding ownership into the reader
1 parent 2b46f42 commit 1cfbc93

6 files changed

Lines changed: 30 additions & 31 deletions

File tree

‎Parser/lexer/state.c‎

Lines changed: 0 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -58,9 +58,6 @@ _PyLexer_PopFTString(struct tok_state *tok)
5858
void
5959
_PyTokenizer_Free(struct tok_state *tok)
6060
{
61-
if (tok->encoding != NULL) {
62-
PyMem_Free(tok->encoding);
63-
}
6461
Py_XDECREF(tok->filename);
6562
Py_XDECREF(tok->module);
6663
_PyTok_ReaderFree(tok);

‎Parser/lexer/state.h‎

Lines changed: 0 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -80,7 +80,6 @@ struct tok_state {
8080
_PyTok_SourceText source;
8181
int done; /* E_OK normally, E_EOF at EOF, otherwise error code */
8282
/* NB If done != E_OK, cur must be == inp!!! */
83-
FILE *fp; /* Rest of input; NULL if tokenizing a string */
8483
lexer_layout_state layout;
8584
int lineno; /* Current line number */
8685
_PyTok_Loc start_loc;
@@ -92,9 +91,6 @@ struct tok_state {
9291
int parencolstack[MAXLEVEL];
9392
PyObject *filename;
9493
PyObject *module;
95-
/* Stuff for PEP 0263 */
96-
char *encoding; /* Source encoding. */
97-
9894
struct _PyTok_Reader *reader;
9995

10096
int type_comments; /* Whether to look for type comments */

‎Parser/tokenizer/api.c‎

Lines changed: 3 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -4,6 +4,7 @@
44

55
#include "tokenizer.h"
66
#include "reader.h"
7+
#include "reader_internal.h"
78
#include "../lexer/state.h"
89

910
_PyTokenizer_Info
@@ -21,10 +22,10 @@ _PyTokenizer_GetInfo(const struct tok_state *tok)
2122
.delimiter_loc = {-1, -1},
2223
.in_formatted_string = tok->ftstring_depth != 0,
2324
.is_interactive = _PyTok_ReaderIsInteractive(tok),
24-
.is_file = tok->fp != NULL && tok->fp != stdin,
25+
.is_file = tok->reader->fp != NULL && tok->reader->fp != stdin,
2526
.filename = tok->filename,
2627
.module = tok->module,
27-
.encoding = tok->encoding,
28+
.encoding = tok->reader->encoding,
2829
};
2930
if (tok->level > 0) {
3031
int level = tok->level - 1;

‎Parser/tokenizer/decoder.c‎

Lines changed: 11 additions & 10 deletions
Original file line numberDiff line numberDiff line change
@@ -117,8 +117,8 @@ _PyTok_SetEncoding(struct tok_state *tok, const char *encoding)
117117
tok->done = E_NOMEM;
118118
return -1;
119119
}
120-
PyMem_Free(tok->encoding);
121-
tok->encoding = copy;
120+
PyMem_Free(tok->reader->encoding);
121+
tok->reader->encoding = copy;
122122
return 0;
123123
}
124124

@@ -246,8 +246,8 @@ _PyTok_DetectEncoding(struct tok_state *tok, const _PyTok_Chunk *first,
246246
PyMem_Free(cookie);
247247
return _PYTOK_ENCODING_ERROR;
248248
}
249-
PyMem_Free(tok->encoding);
250-
tok->encoding = cookie;
249+
PyMem_Free(tok->reader->encoding);
250+
tok->reader->encoding = cookie;
251251
return _PYTOK_ENCODING_DONE;
252252
}
253253

@@ -393,9 +393,10 @@ _PyTok_PrepareString(struct tok_state *tok, const char *input, int utf8_only,
393393
.len = raw_len,
394394
.ownership = _PYTOK_CHUNK_BORROWED,
395395
};
396-
if (tok->encoding != NULL && strcmp(tok->encoding, "utf-8") != 0) {
396+
const char *encoding = tok->reader->encoding;
397+
if (encoding != NULL && strcmp(encoding, "utf-8") != 0) {
397398
if (_PyTok_DecodeOnce(
398-
tok, &decoded, tok->encoding, NULL) < 0) {
399+
tok, &decoded, encoding, NULL) < 0) {
399400
return -1;
400401
}
401402
}
@@ -407,7 +408,7 @@ _PyTok_PrepareString(struct tok_state *tok, const char *input, int utf8_only,
407408
return -1;
408409
}
409410
if (!utf8_only &&
410-
(tok->encoding == NULL || strcmp(tok->encoding, "utf-8") == 0) &&
411+
(encoding == NULL || strcmp(encoding, "utf-8") == 0) &&
411412
!_PyTokenizer_ensure_utf8(_PyTok_SourceData(&tok->source), tok, 1)) {
412413
return -1;
413414
}
@@ -418,15 +419,15 @@ int
418419
_PyTok_StartDecoder(struct tok_state *tok, const char *errors)
419420
{
420421
_PyTok_Reader *reader = tok->reader;
421-
if (tok->encoding == NULL || reader->decoder != NULL) {
422+
if (reader->encoding == NULL || reader->decoder != NULL) {
422423
return 0;
423424
}
424425
if (reader->kind == _PYTOK_READER_FILE &&
425-
strcmp(tok->encoding, "utf-8") == 0) {
426+
strcmp(reader->encoding, "utf-8") == 0) {
426427
return 0;
427428
}
428429

429-
PyObject *codec = _PyCodec_LookupTextEncoding(tok->encoding, NULL);
430+
PyObject *codec = _PyCodec_LookupTextEncoding(reader->encoding, NULL);
430431
if (codec != NULL) {
431432
PyObject *factory = PyObject_GetAttrString(codec, "incrementaldecoder");
432433
Py_DECREF(codec);

‎Parser/tokenizer/reader.c‎

Lines changed: 14 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -34,6 +34,7 @@ _PyTok_ReaderFree(struct tok_state *tok)
3434
i < (int)Py_ARRAY_LENGTH(reader->prefetched_lines); i++) {
3535
_PyTok_ChunkClear(&reader->prefetched_lines[i]);
3636
}
37+
PyMem_Free(reader->encoding);
3738
PyMem_Free(reader->file_buffer);
3839
PyMem_Free(reader->decoded);
3940
PyMem_Free(reader);
@@ -202,7 +203,7 @@ read_file_line(struct tok_state *tok, _PyTok_Chunk *chunk)
202203
int available = (int)Py_MIN(reader->file_buffer_cap - len, INT_MAX);
203204
size_t read = 0;
204205
char *result = _Py_UniversalNewlineFgetsWithSize(
205-
reader->file_buffer + len, available, tok->fp, NULL, &read);
206+
reader->file_buffer + len, available, reader->fp, NULL, &read);
206207
if (result == NULL) {
207208
if (len == 0) {
208209
return _PYTOK_READ_EOF;
@@ -227,7 +228,7 @@ initialize_file(struct tok_state *tok)
227228
{
228229
_PyTok_Reader *reader = tok->reader;
229230
reader->file_initialized = 1;
230-
if (tok->encoding != NULL) {
231+
if (reader->encoding != NULL) {
231232
return _PyTok_StartDecoder(tok, "strict");
232233
}
233234

@@ -410,7 +411,7 @@ next_readline(struct tok_state *tok, _PyTok_Chunk *chunk)
410411
}
411412

412413
_PyTok_Chunk input = {0};
413-
if (tok->encoding != NULL) {
414+
if (reader->encoding != NULL) {
414415
if (!PyBytes_Check(raw)) {
415416
PyErr_SetString(PyExc_TypeError,
416417
"readline() returned a non-bytes object");
@@ -433,7 +434,7 @@ next_readline(struct tok_state *tok, _PyTok_Chunk *chunk)
433434
input.ownership = _PYTOK_CHUNK_PYOBJECT;
434435
int decoded;
435436
if (reader->decoder == NULL &&
436-
strcmp(tok->encoding, "utf-8") == 0 &&
437+
strcmp(reader->encoding, "utf-8") == 0 &&
437438
chunk_is_line(&input)) {
438439
decoded = _PyTok_DecodeOnce(
439440
tok, &input, "utf-8", "replace");
@@ -509,7 +510,7 @@ next_interactive(struct tok_state *tok, _PyTok_Chunk *chunk)
509510
return _PYTOK_READ_STOPPED;
510511
}
511512
char *input = PyOS_Readline(
512-
tok->fp != NULL ? tok->fp : stdin, stdout, reader->prompt);
513+
reader->fp != NULL ? reader->fp : stdin, stdout, reader->prompt);
513514
if (reader->nextprompt != NULL) {
514515
reader->prompt = reader->nextprompt;
515516
}
@@ -526,9 +527,9 @@ next_interactive(struct tok_state *tok, _PyTok_Chunk *chunk)
526527
.len = len,
527528
.ownership = _PYTOK_CHUNK_PYMEM,
528529
};
529-
if (tok->encoding != NULL &&
530+
if (reader->encoding != NULL &&
530531
_PyTok_DecodeOnce(
531-
tok, &decoded, tok->encoding, NULL) < 0) {
532+
tok, &decoded, reader->encoding, NULL) < 0) {
532533
_PyTok_ChunkClear(&decoded);
533534
return _PYTOK_READ_ERROR;
534535
}
@@ -604,7 +605,8 @@ int
604605
_PyTok_ReaderUnderflow(struct tok_state *tok)
605606
{
606607
assert(tok->cur >= tok->buf_offset && tok->cur <= tok->inp);
607-
_PyTok_ReaderKind kind = tok->reader->kind;
608+
_PyTok_Reader *reader = tok->reader;
609+
_PyTok_ReaderKind kind = reader->kind;
608610
int prepared = kind == _PYTOK_READER_PREPARED;
609611
int streaming = reader_is_streaming(kind);
610612
int reset_buffer = !prepared && tok->start < 0 &&
@@ -675,7 +677,7 @@ _PyTok_ReaderUnderflow(struct tok_state *tok)
675677

676678
tok->lineno++;
677679
if (kind == _PYTOK_READER_FILE &&
678-
(tok->encoding == NULL || strcmp(tok->encoding, "utf-8") == 0) &&
680+
(reader->encoding == NULL || strcmp(reader->encoding, "utf-8") == 0) &&
679681
!_PyTokenizer_ensure_utf8(
680682
_PyTok_SourcePointer(&tok->source, tok->cur), tok, tok->lineno)) {
681683
_PyTok_ChunkClear(&chunk);
@@ -779,7 +781,7 @@ _PyTokenizer_FromFile(FILE *fp, const char *encoding,
779781
_PyTokenizer_Free(tok);
780782
return NULL;
781783
}
782-
tok->fp = fp;
784+
tok->reader->fp = fp;
783785
tok->reader->prompt = ps1;
784786
tok->reader->nextprompt = ps2;
785787
return tok;
@@ -840,8 +842,8 @@ _PyTokenizer_FindEncodingFilename(int fd, PyObject *filename)
840842
tok->filename = Py_NewRef(filename != NULL ? filename : &_Py_STR(anon_string));
841843
char *encoding = NULL;
842844
if (initialize_file(tok) == 0) {
843-
encoding = tok->encoding;
844-
tok->encoding = NULL;
845+
encoding = tok->reader->encoding;
846+
tok->reader->encoding = NULL;
845847
}
846848
fclose(fp);
847849
_PyTokenizer_Free(tok);

‎Parser/tokenizer/reader_internal.h‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -41,6 +41,8 @@ typedef struct {
4141
} _PyTok_Chunk;
4242

4343
typedef struct _PyTok_Reader {
44+
FILE *fp; // Borrowed input stream; NULL for string and readline input.
45+
char *encoding; // Owned source encoding.
4446
PyObject *readline;
4547
PyObject *decoder;
4648
const char *prompt;

0 commit comments

Comments
 (0)