From be46fa5403ae45cffac103efa0fcb6a684caf12c Mon Sep 17 00:00:00 2001 From: Pablo Galindo Salgado Date: Thu, 27 Aug 2026 15:33:35 +0100 Subject: [PATCH 1/5] gh-153569: replace tokenizer input readers --- Lib/test/test_cmd_line_script.py | 19 + Lib/test/test_fstring.py | 2 + Lib/test/test_repl.py | 30 + Lib/test/test_source_encoding.py | 40 ++ Lib/test/test_tokenize.py | 154 +++++ Makefile.pre.in | 8 +- PCbuild/_freeze_module.vcxproj | 7 +- PCbuild/_freeze_module.vcxproj.filters | 9 +- PCbuild/pythoncore.vcxproj | 8 +- PCbuild/pythoncore.vcxproj.filters | 16 +- Parser/lexer/buffer.c | 14 - Parser/lexer/lexer.c | 10 +- Parser/lexer/state.c | 29 +- Parser/lexer/state.h | 23 +- Parser/lexer/string.c | 2 +- Parser/myreadline.c | 2 - Parser/tokenizer/decoder.c | 518 +++++++++++++++++ Parser/tokenizer/file_tokenizer.c | 494 ---------------- Parser/tokenizer/helpers.c | 269 --------- Parser/tokenizer/helpers.h | 11 - Parser/tokenizer/reader.c | 767 +++++++++++++++++++++++++ Parser/tokenizer/reader.h | 9 + Parser/tokenizer/reader_internal.h | 80 +++ Parser/tokenizer/readline_tokenizer.c | 135 ----- Parser/tokenizer/string_tokenizer.c | 148 ----- Parser/tokenizer/tokenizer.h | 5 +- Parser/tokenizer/utf8_tokenizer.c | 55 -- Python/Python-tokenize.c | 2 +- Python/errors.c | 5 +- Python/traceback.c | 4 +- Tools/c-analyzer/TODO | 2 - Tools/peg_generator/pegen/build.py | 6 +- 32 files changed, 1669 insertions(+), 1214 deletions(-) create mode 100644 Parser/tokenizer/decoder.c delete mode 100644 Parser/tokenizer/file_tokenizer.c create mode 100644 Parser/tokenizer/reader.c create mode 100644 Parser/tokenizer/reader.h create mode 100644 Parser/tokenizer/reader_internal.h delete mode 100644 Parser/tokenizer/readline_tokenizer.c delete mode 100644 Parser/tokenizer/string_tokenizer.c delete mode 100644 Parser/tokenizer/utf8_tokenizer.c diff --git a/Lib/test/test_cmd_line_script.py b/Lib/test/test_cmd_line_script.py index 73b1f671c58555f..3802ed62ce7c26b 100644 --- a/Lib/test/test_cmd_line_script.py +++ b/Lib/test/test_cmd_line_script.py @@ -9,6 +9,7 @@ import os import os.path import py_compile +import select import subprocess import io @@ -168,6 +169,24 @@ def test_stdin_loader(self): expected = repr(importlib.machinery.BuiltinImporter).encode("utf-8") self.assertIn(expected, out) + @unittest.skipIf(sys.platform == "win32", "select() cannot wait for pipes") + def test_stdin_syntax_error_does_not_read_ahead(self): + process = spawn_python() + try: + process.stdin.write(b")\n") + process.stdin.flush() + output = b"" + while b"SyntaxError" not in output: + ready, _, _ = select.select( + [process.stdout], [], [], support.SHORT_TIMEOUT + ) + self.assertTrue(ready, output) + data = os.read(process.stdout.fileno(), 4096) + self.assertTrue(data, output) + output += data + finally: + kill_python(process) + @contextlib.contextmanager def interactive_python(self, separate_stderr=False): if separate_stderr: diff --git a/Lib/test/test_fstring.py b/Lib/test/test_fstring.py index 45890f8fb35dcb7..c1ef1a73f05c204 100644 --- a/Lib/test/test_fstring.py +++ b/Lib/test/test_fstring.py @@ -1603,6 +1603,8 @@ def test_debug_conversion(self): self.assertEqual(f'''{ 3 =}''', '\n3\n=3') + x = 1 + self.assertEqual(eval('f"""{(\nx\n)=}"""'), '(\nx\n)=1') # Since = is handled specially, make sure all existing uses of # it still work. diff --git a/Lib/test/test_repl.py b/Lib/test/test_repl.py index 0f5009c32124902..ea0f72e0e8d4568 100644 --- a/Lib/test/test_repl.py +++ b/Lib/test/test_repl.py @@ -152,6 +152,36 @@ def test_multiline_string_parsing(self): output = kill_python(p) self.assertEqual(p.returncode, 0) + @unittest.skipIf(sys.platform == "win32", "select() cannot wait for pipes") + def test_secondary_prompt_is_not_read_ahead(self): + process = spawn_repl() + output = "" + + def read_until(marker, start=0): + nonlocal output + while marker not in output[start:]: + ready, _, _ = select.select( + [process.stdout], [], [], SHORT_TIMEOUT + ) + self.assertTrue(ready, output) + data = os.read(process.stdout.fileno(), 4096) + self.assertTrue(data, output) + output += data.decode() + + try: + read_until(">>> ") + process.stdin.write("(\n") + process.stdin.flush() + read_until("... ") + after_secondary_prompt = len(output) + + process.stdin.write("1)\n") + process.stdin.flush() + read_until(">>> ", after_secondary_prompt) + self.assertEqual(output[after_secondary_prompt:], "1\n>>> ") + finally: + kill_python(process) + @cpython_only def test_lexer_buffer_realloc_with_null_start(self): # gh-144759: NULL pointer arithmetic in the lexer when start and diff --git a/Lib/test/test_source_encoding.py b/Lib/test/test_source_encoding.py index 53fffe7cfb56d09..862a20a058be75a 100644 --- a/Lib/test/test_source_encoding.py +++ b/Lib/test/test_source_encoding.py @@ -82,6 +82,46 @@ def test_truncated_utf8_at_eof(self): with self.subTest(seq=seq): self.assertRaises(SyntaxError, compile, seq, '', 'exec') + def test_invalid_utf8_offset_after_non_ascii(self): + with self.assertRaises(SyntaxError) as caught: + compile(b"x = \xc3\xa9\xff\n", "", "exec") + error = caught.exception + self.assertEqual( + (error.lineno, error.offset, error.end_lineno, error.end_offset), + (1, 6, 1, 6), + ) + + def test_long_bom_conflict_message_is_not_truncated(self): + encoding = "x" * 400 + source = b"\xef\xbb\xbf# coding:" + encoding.encode() + b"\n" + with self.assertRaises(SyntaxError) as caught: + compile(source, "", "exec") + self.assertEqual( + caught.exception.msg, + f"encoding problem: {encoding} with BOM", + ) + + def _assert_python_file_ok(self, source): + with tempfile.TemporaryDirectory() as directory: + filename = script_helper.make_script(directory, "source", source) + script_helper.assert_python_ok(filename) + + @support.requires_subprocess() + def test_stateful_file_decoder_spans_lines(self): + encoded_name = "変数".encode("iso2022_jp") + payload = encoded_name[3:-3] + source = ( + b"# coding: iso2022_jp\n" + b"# \x1b$B" + payload + b"\n" + + payload + b"\x1b(B = 1\n" + ) + self._assert_python_file_ok(source) + + @support.requires_subprocess() + def test_stateful_file_decoder_finalizes_before_implicit_newline(self): + source = b"# coding: hz\n# ~{1dA?" + self._assert_python_file_ok(source) + @support.requires_subprocess() def test_20731(self): sub = subprocess.Popen([sys.executable, diff --git a/Lib/test/test_tokenize.py b/Lib/test/test_tokenize.py index e2db09d61f409bd..7e02191db86be5a 100644 --- a/Lib/test/test_tokenize.py +++ b/Lib/test/test_tokenize.py @@ -1,4 +1,5 @@ import contextlib +import _tokenize import itertools import os import re @@ -2273,6 +2274,159 @@ def readline(encoding): )) self.assertEqual(tokens, expected) + def test_stateful_decoder_spans_readline_calls(self): + encoded_name = "変数".encode("iso2022_jp") + payload = encoded_name[3:-3] + lines = iter([ + b"# \x1b$B" + payload + b"\n", + payload + b"\x1b(B\n", + b"", + ]) + tokens = list(tokenize._generate_tokens_from_c_tokenizer( + lines.__next__, + extra_tokens=True, + encoding="iso2022_jp", + )) + self.assertEqual(tokens, [ + tokenize.TokenInfo( + token.COMMENT, "# 変数", (1, 0), (1, 4), "# 変数\n" + ), + tokenize.TokenInfo(token.NL, "\n", (1, 4), (1, 5), "# 変数\n"), + tokenize.TokenInfo(token.NAME, "変数", (2, 0), (2, 2), "変数\n"), + tokenize.TokenInfo(token.NEWLINE, "\n", (2, 2), (2, 3), "変数\n"), + tokenize.TokenInfo(token.ENDMARKER, "", (3, 0), (3, 0), ""), + ]) + + def test_utf16_bom_in_each_readline_chunk(self): + lines = iter([ + "x\n".encode("utf-16"), + "y\n".encode("utf-16"), + b"", + ]) + tokens = _tokenize.TokenizerIter( + lines.__next__, encoding="utf-16", extra_tokens=True + ) + self.assertEqual(list(tokens), [ + (token.NAME, "x", (1, 0), (1, 1), "x\n"), + (token.NEWLINE, "\n", (1, 1), (1, 2), "x\n"), + (token.NAME, "y", (2, 0), (2, 1), "y\n"), + (token.NEWLINE, "\n", (2, 1), (2, 2), "y\n"), + (token.ENDMARKER, "", (3, 0), (3, 0), ""), + ]) + + def test_utf8_decoder_spans_readline_calls(self): + lines = iter([b"x\xc3", b"\xa9\n", b""]) + tokens = list(tokenize._generate_tokens_from_c_tokenizer( + lines.__next__, + extra_tokens=True, + encoding="utf-8", + )) + self.assertEqual(tokens, [ + tokenize.TokenInfo(token.NAME, "xé", (1, 0), (1, 2), "xé\n"), + tokenize.TokenInfo(token.NEWLINE, "\n", (1, 2), (1, 3), "xé\n"), + tokenize.TokenInfo(token.ENDMARKER, "", (2, 0), (2, 0), ""), + ]) + + def test_utf8_decoder_replaces_incomplete_input_at_eof(self): + expected = [ + tokenize.TokenInfo(token.NAME, "x�", (1, 0), (1, 2), "x�"), + tokenize.TokenInfo(token.NEWLINE, "", (1, 2), (1, 3), "x�"), + tokenize.TokenInfo(token.ENDMARKER, "", (2, 0), (2, 0), ""), + ] + for chunks in ([b"x\xe9", b""], [b"x\xe9"]): + with self.subTest(chunks=chunks): + lines = iter(chunks) + tokens = list(tokenize._generate_tokens_from_c_tokenizer( + lines.__next__, + extra_tokens=True, + encoding="utf-8", + )) + self.assertEqual(tokens, expected) + + def test_multiline_readline_chunk(self): + expected = [ + tokenize.TokenInfo(token.NAME, "x", (1, 0), (1, 1), "x=1\n"), + tokenize.TokenInfo(token.OP, "=", (1, 1), (1, 2), "x=1\n"), + tokenize.TokenInfo(token.NUMBER, "1", (1, 2), (1, 3), "x=1\n"), + tokenize.TokenInfo(token.NEWLINE, "\n", (1, 3), (1, 4), "x=1\n"), + tokenize.TokenInfo(token.NAME, "y", (2, 0), (2, 1), "y=2\n"), + tokenize.TokenInfo(token.OP, "=", (2, 1), (2, 2), "y=2\n"), + tokenize.TokenInfo(token.NUMBER, "2", (2, 2), (2, 3), "y=2\n"), + tokenize.TokenInfo(token.NEWLINE, "\n", (2, 3), (2, 4), "y=2\n"), + tokenize.TokenInfo(token.ENDMARKER, "", (3, 0), (3, 0), ""), + ] + lines = iter([b"x=1\ny=2\n", b""]) + tokens = list(tokenize._generate_tokens_from_c_tokenizer( + lines.__next__, + extra_tokens=True, + encoding="utf-8", + )) + self.assertEqual(tokens, expected) + + def test_multiline_readline_chunk_with_unterminated_tail(self): + readline = mock.Mock(side_effect=["x\nz", ""]) + iterator = _tokenize.TokenizerIter(readline, extra_tokens=True) + expected = [ + (token.NAME, "x", (1, 0), (1, 1), "x\n"), + (token.NEWLINE, "\n", (1, 1), (1, 2), "x\n"), + (token.NAME, "z", (2, 0), (2, 1), "z"), + (token.NEWLINE, "", (2, 1), (2, 2), "z"), + ] + self.assertEqual(readline.call_count, 0) + for token_info in expected: + self.assertEqual(next(iterator), token_info) + self.assertEqual(readline.call_count, 1) + self.assertEqual( + next(iterator), + (token.ENDMARKER, "", (3, 0), (3, 0), ""), + ) + self.assertEqual(readline.call_count, 2) + + def test_readline_callback_is_not_read_ahead(self): + readline = mock.Mock(side_effect=["x\n", "y\n", ""]) + iterator = _tokenize.TokenizerIter(readline, extra_tokens=True) + expected = [ + ((token.NAME, "x", (1, 0), (1, 1), "x\n"), 1), + ((token.NEWLINE, "\n", (1, 1), (1, 2), "x\n"), 1), + ((token.NAME, "y", (2, 0), (2, 1), "y\n"), 2), + ] + self.assertEqual(readline.call_count, 0) + for token_info, calls in expected: + self.assertEqual(next(iterator), token_info) + self.assertEqual(readline.call_count, calls) + + def test_encoded_readline_replaces_invalid_bytes(self): + lines = iter([b"\xff\n", b""]) + tokens = list(tokenize._generate_tokens_from_c_tokenizer( + lines.__next__, + extra_tokens=True, + encoding="utf-8", + )) + self.assertEqual(tokens, [ + tokenize.TokenInfo(token.NAME, "�", (1, 0), (1, 1), "�\n"), + tokenize.TokenInfo(token.NEWLINE, "\n", (1, 1), (1, 2), "�\n"), + tokenize.TokenInfo(token.ENDMARKER, "", (2, 0), (2, 0), ""), + ]) + + def test_stop_iteration_skips_encoded_readline_codec_lookup(self): + iterator = _tokenize.TokenizerIter( + lambda: b"", + extra_tokens=True, + encoding="missing-tokenizer-codec", + ) + with self.assertRaises(LookupError): + next(iterator) + + iterator = _tokenize.TokenizerIter( + iter(()).__next__, + extra_tokens=True, + encoding="missing-tokenizer-codec", + ) + self.assertEqual( + next(iterator), + (token.ENDMARKER, "", (1, 0), (1, 0), ""), + ) + def test_extra_tokens_relaxes_lexer_errors(self): cases = [ ( diff --git a/Makefile.pre.in b/Makefile.pre.in index 97e3dcdf67202be..adcfe4c5259eb22 100644 --- a/Makefile.pre.in +++ b/Makefile.pre.in @@ -400,11 +400,9 @@ TOKENIZER_OBJS= \ Parser/lexer/state.o \ Parser/lexer/string.o \ Parser/tokenizer/cursor.o \ - Parser/tokenizer/file_tokenizer.o \ - Parser/tokenizer/readline_tokenizer.o \ + Parser/tokenizer/decoder.o \ + Parser/tokenizer/reader.o \ Parser/tokenizer/source.o \ - Parser/tokenizer/string_tokenizer.o \ - Parser/tokenizer/utf8_tokenizer.o \ Parser/tokenizer/helpers.o PEGEN_HEADERS= \ @@ -418,6 +416,8 @@ TOKENIZER_HEADERS= \ Parser/lexer/lexer_internal.h \ Parser/lexer/state.h \ Parser/tokenizer/cursor.h \ + Parser/tokenizer/reader.h \ + Parser/tokenizer/reader_internal.h \ Parser/tokenizer/source.h \ Parser/tokenizer/tokenizer.h \ Parser/tokenizer/helpers.h diff --git a/PCbuild/_freeze_module.vcxproj b/PCbuild/_freeze_module.vcxproj index 3cd930e6d8e66d9..469fd77cc8be9dc 100644 --- a/PCbuild/_freeze_module.vcxproj +++ b/PCbuild/_freeze_module.vcxproj @@ -186,10 +186,9 @@ - - - - + + + diff --git a/PCbuild/_freeze_module.vcxproj.filters b/PCbuild/_freeze_module.vcxproj.filters index 72f36691f4a143d..976c99b7d24bdfd 100644 --- a/PCbuild/_freeze_module.vcxproj.filters +++ b/PCbuild/_freeze_module.vcxproj.filters @@ -475,16 +475,13 @@ Source Files - + Source Files - + Source Files - - Source Files - - + Source Files diff --git a/PCbuild/pythoncore.vcxproj b/PCbuild/pythoncore.vcxproj index 8451ffb6422109b..33647ec284061f1 100644 --- a/PCbuild/pythoncore.vcxproj +++ b/PCbuild/pythoncore.vcxproj @@ -424,6 +424,8 @@ + + @@ -593,10 +595,8 @@ - - - - + + diff --git a/PCbuild/pythoncore.vcxproj.filters b/PCbuild/pythoncore.vcxproj.filters index 81e736dd6dcc818..434dd13267fe934 100644 --- a/PCbuild/pythoncore.vcxproj.filters +++ b/PCbuild/pythoncore.vcxproj.filters @@ -336,6 +336,12 @@ Parser + + Parser + + + Parser + Parser @@ -1361,16 +1367,10 @@ Parser - - Parser - - - Parser - - + Parser - + Parser diff --git a/Parser/lexer/buffer.c b/Parser/lexer/buffer.c index e122fd0d9878ea2..cd6885a7d01040a 100644 --- a/Parser/lexer/buffer.c +++ b/Parser/lexer/buffer.c @@ -32,20 +32,6 @@ _PyLexer_restore_fstring_buffers(struct tok_state *tok) } } -/* Read a line of text from TOK into S, using the stream in TOK. - Return NULL on failure, else S. - - On entry, tok->decoding_buffer will be one of: - 1) NULL: need to call tok->decoding_readline to get a new line - 2) PyUnicodeObject *: decoding_feof has called tok->decoding_readline and - stored the result in tok->decoding_buffer - 3) PyByteArrayObject *: previous call to tok_readline_recode did not have enough room - (in the s buffer) to copy entire contents of the line read - by tok->decoding_readline. tok->decoding_buffer has the overflow. - In this case, tok_readline_recode is called in a loop (with an expanded buffer) - until the buffer ends with a '\n' (or until the end of the file is - reached): see tok_nextc and its calls to tok_reserve_buf. -*/ int _PyLexer_tok_reserve_buf(struct tok_state *tok, Py_ssize_t size) { diff --git a/Parser/lexer/lexer.c b/Parser/lexer/lexer.c index 41f03a408cfc756..a96362c8961023a 100644 --- a/Parser/lexer/lexer.c +++ b/Parser/lexer/lexer.c @@ -5,6 +5,7 @@ #include "lexer_internal.h" #include "../tokenizer/helpers.h" +#include "../tokenizer/reader.h" /* Alternate tab spacing */ #define ALTTABSIZE 1 @@ -41,7 +42,7 @@ _PyLexer_nextc(struct tok_state *tok) if (tok->done != E_OK) { return EOF; } - rc = tok->underflow(tok); + rc = _PyTok_ReaderUnderflow(tok); #if defined(Py_DEBUG) if (tok->debug) { fprintf(stderr, "line[%d] = ", tok->lineno); @@ -89,7 +90,7 @@ verify_identifier(struct tok_state *tok) return 1; } PyObject *s; - if (tok->decoding_erred) + if (tok->input_error) return 0; s = PyUnicode_DecodeUTF8(tok->start, tok->cur - tok->start, NULL); if (s == NULL) { @@ -483,7 +484,6 @@ _PyLexer_get_normal_mode(struct tok_state *tok, tokenizer_mode* current_tok, str } p_start = tok->start; p_end = tok->cur - 1; /* Leave '\n' out of the string */ - tok->cont_line = 0; return MAKE_TOKEN(NEWLINE); } @@ -528,7 +528,6 @@ _PyLexer_get_normal_mode(struct tok_state *tok, tokenizer_mode* current_tok, str if ((c = tok_continuation_line(tok)) == -1) { return MAKE_TOKEN(ERRORTOKEN); } - tok->cont_line = 1; goto again; /* Read next line */ } @@ -687,9 +686,8 @@ int _PyTokenizer_Get(struct tok_state *tok, struct token *token) { int result = tok_get(tok, token); - if (tok->decoding_erred) { + if (tok->input_error) { result = ERRORTOKEN; - tok->done = E_DECODE; } return result; } diff --git a/Parser/lexer/state.c b/Parser/lexer/state.c index 5cf9b4d768c3ebb..2a6408bef927a36 100644 --- a/Parser/lexer/state.c +++ b/Parser/lexer/state.c @@ -4,6 +4,7 @@ #include "errcode.h" #include "state.h" +#include "../tokenizer/reader.h" /* Never change this */ #define TABSIZE 8 @@ -28,36 +29,30 @@ _PyTokenizer_tok_new(void) tok->end = NULL; tok->done = E_OK; tok->fp = NULL; - tok->input = NULL; tok->tabsize = TABSIZE; tok->indent = 0; tok->indstack[0] = 0; tok->atbol = 1; tok->pendin = 0; - tok->prompt = tok->nextprompt = NULL; + tok->prompt = NULL; tok->lineno = 0; tok->starting_col_offset = -1; tok->col_offset = -1; tok->level = 0; tok->altindstack[0] = 0; - tok->decoding_state = STATE_INIT; - tok->decoding_erred = 0; - tok->enc = NULL; + tok->input_error = 0; tok->encoding = NULL; - tok->cont_line = 0; tok->filename = NULL; tok->module = NULL; - tok->decoding_readline = NULL; - tok->decoding_buffer = NULL; - tok->readline = NULL; tok->type_comments = 0; tok->interactive_underflow = IUNDERFLOW_NORMAL; - tok->underflow = NULL; tok->str = NULL; tok->report_warnings = 1; tok->tok_extra_tokens = 0; tok->comment_newline = 0; tok->implicit_newline = 0; + _PyTok_SourceInit(&tok->source); + tok->reader = NULL; tok->tok_mode_stack[0] = (tokenizer_mode){.kind =TOK_REGULAR_MODE, .quote='\0', .quote_size = 0, .in_debug=0}; tok->tok_mode_stack_index = 0; #ifdef Py_DEBUG @@ -91,20 +86,10 @@ _PyTokenizer_Free(struct tok_state *tok) if (tok->encoding != NULL) { PyMem_Free(tok->encoding); } - Py_XDECREF(tok->decoding_readline); - Py_XDECREF(tok->decoding_buffer); - Py_XDECREF(tok->readline); Py_XDECREF(tok->filename); Py_XDECREF(tok->module); - if ((tok->readline != NULL || tok->fp != NULL ) && tok->buf != NULL) { - PyMem_Free(tok->buf); - } - if (tok->input) { - PyMem_Free(tok->input); - } - if (tok->interactive_src_start != NULL) { - PyMem_Free(tok->interactive_src_start); - } + _PyTok_ReaderFree(tok); + _PyTok_SourceClear(&tok->source); free_fstring_expressions(tok); PyMem_Free(tok); } diff --git a/Parser/lexer/state.h b/Parser/lexer/state.h index 9cd196a114c7cb1..0824785195491ee 100644 --- a/Parser/lexer/state.h +++ b/Parser/lexer/state.h @@ -2,6 +2,7 @@ #define _PY_LEXER_H_ #include "object.h" +#include "../tokenizer/source.h" #define MAXINDENT 100 /* Max indentation level */ #define MAXLEVEL 200 /* Max parentheses level */ @@ -12,12 +13,6 @@ #define INSIDE_FSTRING_EXPR_AT_TOP(tok) \ (tok->curly_bracket_depth - tok->curly_bracket_expr_start_depth == 1) -enum decoding_state { - STATE_INIT, - STATE_SEEK_CODING, - STATE_NORMAL -}; - enum interactive_underflow_t { /* Normal mode of operation: return a new token when asked in interactive mode */ IUNDERFLOW_NORMAL, @@ -90,7 +85,7 @@ struct tok_state { int indstack[MAXINDENT]; /* Stack of indents */ int atbol; /* Nonzero if at begin of new line */ int pendin; /* Pending indents (if > 0) or dedents (if < 0) */ - const char *prompt, *nextprompt; /* For interactive prompting */ + const char *prompt; /* For interactive prompting */ int lineno; /* Current line number */ int first_lineno; /* First line of a single line or multi line string expression (cf. issue 16806) */ @@ -106,27 +101,21 @@ struct tok_state { /* Stuff for checking on different tab sizes */ int altindstack[MAXINDENT]; /* Stack of alternate indents */ /* Stuff for PEP 0263 */ - enum decoding_state decoding_state; - int decoding_erred; /* whether erred in decoding */ + int input_error; char *encoding; /* Source encoding. */ - int cont_line; /* whether we are in a continuation line. */ const char* line_start; /* pointer to start of current line */ const char* multi_line_start; /* pointer to start of first line of a single line or multi line string expression (cf. issue 16806) */ - PyObject *decoding_readline; /* open(...).readline */ - PyObject *decoding_buffer; - PyObject *readline; /* readline() function */ - const char* enc; /* Encoding for the current str. */ char* str; /* Source string being tokenized (if tokenizing from a string)*/ - char* input; /* Tokenizer's newline translated copy of the string. */ + + _PyTok_SourceText source; + struct _PyTok_Reader *reader; int type_comments; /* Whether to look for type comments */ /* How to proceed when asked for a new token in interactive mode */ enum interactive_underflow_t interactive_underflow; - int (*underflow)(struct tok_state *); /* Function to call when buffer is empty and we need to refill it*/ - int report_warnings; // TODO: Factor this into its own thing tokenizer_mode tok_mode_stack[MAXFSTRINGLEVEL]; diff --git a/Parser/lexer/string.c b/Parser/lexer/string.c index e546954ba5c0d82..d67c48f7f678eda 100644 --- a/Parser/lexer/string.c +++ b/Parser/lexer/string.c @@ -489,7 +489,7 @@ _PyLexer_get_fstring_mode(struct tok_state *tok, tokenizer_mode* current_tok, st ); if (c == EOF || (current_tok->quote_size == 1 && c == '\n')) { - if (tok->decoding_erred) { + if (tok->input_error) { return MAKE_TOKEN(ERRORTOKEN); } diff --git a/Parser/myreadline.c b/Parser/myreadline.c index ee77479ba7bdccb..457b9598a0e61a5 100644 --- a/Parser/myreadline.c +++ b/Parser/myreadline.c @@ -366,8 +366,6 @@ PyOS_StdioReadline(FILE *sys_stdin, FILE *sys_stdout, const char *prompt) char *(*PyOS_ReadlineFunctionPointer)(FILE *, FILE *, const char *) = NULL; -/* Interface used by file_tokenizer.c and bltinmodule.c */ - char * PyOS_Readline(FILE *sys_stdin, FILE *sys_stdout, const char *prompt) { diff --git a/Parser/tokenizer/decoder.c b/Parser/tokenizer/decoder.c new file mode 100644 index 000000000000000..6cddd7bac414066 --- /dev/null +++ b/Parser/tokenizer/decoder.c @@ -0,0 +1,518 @@ +#include "Python.h" +#include "pycore_codecs.h" +#include "pycore_global_strings.h" +#include "pycore_runtime.h" +#include "errcode.h" + +#include "reader_internal.h" +#include "helpers.h" +#include "../lexer/state.h" + +char * +_PyTok_CopyBytes(const char *data, Py_ssize_t len) +{ + if (len < 0 || len == PY_SSIZE_T_MAX) { + PyErr_NoMemory(); + return NULL; + } + char *copy = PyMem_Malloc((size_t)len + 1); + if (copy == NULL) { + PyErr_NoMemory(); + return NULL; + } + memcpy(copy, data, (size_t)len); + copy[len] = '\0'; + return copy; +} + +static void +chunk_release_data(_PyTok_Chunk *chunk) +{ + switch (chunk->ownership) { + case _PYTOK_CHUNK_BORROWED: + break; + case _PYTOK_CHUNK_PYMEM: + PyMem_Free(chunk->data); + break; + case _PYTOK_CHUNK_PYOBJECT: + Py_DECREF(chunk->owner); + break; + } +} + +void +_PyTok_ChunkClear(_PyTok_Chunk *chunk) +{ + chunk_release_data(chunk); + *chunk = (_PyTok_Chunk){0}; +} + +static int +chunk_set_unicode(struct tok_state *tok, _PyTok_Chunk *chunk, + PyObject *unicode, int strip_bom) +{ + Py_ssize_t utf8_len; + const char *utf8 = PyUnicode_AsUTF8AndSize(unicode, &utf8_len); + if (utf8 == NULL) { + Py_DECREF(unicode); + tok->done = PyErr_ExceptionMatches(PyExc_MemoryError) + ? E_NOMEM : E_DECODE; + return -1; + } + if (strip_bom && PyUnicode_GET_LENGTH(unicode) > 0 && + PyUnicode_ReadChar(unicode, 0) == 0xFEFF) { + utf8 += 3; + utf8_len -= 3; + } + chunk_release_data(chunk); + chunk->owner = unicode; + chunk->data = (char *)utf8; + chunk->len = utf8_len; + chunk->ownership = _PYTOK_CHUNK_PYOBJECT; + return 0; +} + +char * +_PyTok_NormalizeNewlines(const char *data, Py_ssize_t len, int preserve_crlf, + int add_final_newline, Py_ssize_t *out_len, + int *implicit_newline) +{ + if (len > PY_SSIZE_T_MAX - 2) { + PyErr_NoMemory(); + return NULL; + } + char *result = PyMem_Malloc((size_t)len + 2); + if (result == NULL) { + PyErr_NoMemory(); + return NULL; + } + Py_ssize_t write = 0; + for (Py_ssize_t read = 0; read < len; read++) { + char c = data[read]; + if (!preserve_crlf && c == '\r') { + if (read + 1 < len && data[read + 1] == '\n') { + read++; + } + c = '\n'; + } + result[write++] = c; + } + int implicit = add_final_newline && write > 0 && result[write - 1] != '\n'; + if (implicit) { + result[write++] = '\n'; + } + result[write] = '\0'; + *out_len = write; + *implicit_newline = implicit; + return result; +} + +int +_PyTok_SetEncoding(struct tok_state *tok, const char *encoding) +{ + char *copy = _PyTok_CopyBytes(encoding, strlen(encoding)); + if (copy == NULL) { + tok->done = E_NOMEM; + return -1; + } + PyMem_Free(tok->encoding); + tok->encoding = copy; + return 0; +} + +static int +find_cookie(const char *line, Py_ssize_t len, char **encoding, int *scan_next) +{ + Py_ssize_t i = 0; + *encoding = NULL; + *scan_next = 1; + for (; i < len; i++) { + if (line[i] == '#') { + break; + } + if (line[i] == '\n' || line[i] == '\r') { + return 0; + } + if (line[i] != ' ' && line[i] != '\t' && line[i] != '\f') { + *scan_next = 0; + return 0; + } + } + for (; i + 6 < len; i++) { + if (memcmp(line + i, "coding", 6) != 0) { + continue; + } + const char *cursor = line + i + 6; + if (*cursor != ':' && *cursor != '=') { + continue; + } + do { + cursor++; + } while (cursor < line + len && + (*cursor == ' ' || *cursor == '\t')); + const char *start = cursor; + while (cursor < line + len && + (Py_ISALNUM(*cursor) || *cursor == '-' || + *cursor == '_' || *cursor == '.')) { + cursor++; + } + if (cursor == start) { + continue; + } + char *found = _PyTok_CopyBytes(start, cursor - start); + if (found == NULL) { + return -1; + } + char normalized[13]; + int n; + for (n = 0; n < 12 && found[n] != '\0'; n++) { + normalized[n] = found[n] == '_' ? '-' : Py_TOLOWER(found[n]); + } + normalized[n] = '\0'; + const char *canonical = found; + if (strcmp(normalized, "utf-8") == 0 || + strncmp(normalized, "utf-8-", 6) == 0) { + canonical = "utf-8"; + } + else if (strcmp(normalized, "latin-1") == 0 || + strcmp(normalized, "iso-8859-1") == 0 || + strcmp(normalized, "iso-latin-1") == 0 || + strncmp(normalized, "latin-1-", 8) == 0 || + strncmp(normalized, "iso-8859-1-", 11) == 0 || + strncmp(normalized, "iso-latin-1-", 12) == 0) { + canonical = "iso-8859-1"; + } + if (canonical != found) { + PyMem_Free(found); + found = _PyTok_CopyBytes(canonical, strlen(canonical)); + if (found == NULL) { + return -1; + } + } + *encoding = found; + *scan_next = 0; + return 0; + } + return 0; +} + +_PyTok_EncodingResult +_PyTok_DetectEncoding(struct tok_state *tok, const _PyTok_Chunk *first, + const _PyTok_Chunk *second, int final, + Py_ssize_t *bom_len) +{ + int bom = first->len >= 3 && + (unsigned char)first->data[0] == 0xEF && + (unsigned char)first->data[1] == 0xBB && + (unsigned char)first->data[2] == 0xBF; + *bom_len = bom ? 3 : 0; + + char *cookie = NULL; + int scan_next = 0; + int cookie_line = 1; + const char *first_data = first->data + (bom ? 3 : 0); + Py_ssize_t first_len = first->len - (bom ? 3 : 0); + if (find_cookie(first_data, first_len, &cookie, &scan_next) < 0) { + return _PYTOK_ENCODING_ERROR; + } + if (cookie == NULL && scan_next && second != NULL) { + if (find_cookie(second->data, second->len, &cookie, &scan_next) < 0) { + return _PYTOK_ENCODING_ERROR; + } + cookie_line = 2; + } + else if (cookie == NULL && scan_next && !final) { + return _PYTOK_ENCODING_NEED_SECOND_LINE; + } + + if (bom) { + if (_PyTok_SetEncoding(tok, "utf-8") < 0) { + PyMem_Free(cookie); + return _PYTOK_ENCODING_ERROR; + } + } + if (cookie == NULL) { + return _PYTOK_ENCODING_DONE; + } + if (bom && strcmp(cookie, "utf-8") != 0) { + const _PyTok_Chunk *line = cookie_line == 2 ? second : first; + const char *line_data = line->data + (cookie_line == 1 ? 3 : 0); + Py_ssize_t line_len = line->len - (cookie_line == 1 ? 3 : 0); + const char *saved_line_start = tok->line_start; + char *saved_cur = tok->cur; + int saved_lineno = tok->lineno; + tok->line_start = line_data; + tok->cur = (char *)line_data; + tok->lineno = cookie_line; + int end_col = (int)Py_MIN(line_len, INT_MAX); + if (end_col > 0 && (line_data[end_col - 1] == '\n' || + line_data[end_col - 1] == '\r')) { + end_col--; + } + _PyTokenizer_syntaxerror_known_range( + tok, 0, end_col, "encoding problem: %s with BOM", cookie); + tok->line_start = saved_line_start; + tok->cur = saved_cur; + tok->lineno = saved_lineno; + PyMem_Free(cookie); + return _PYTOK_ENCODING_ERROR; + } + if (!bom && _PyTok_SetEncoding(tok, cookie) < 0) { + PyMem_Free(cookie); + return _PYTOK_ENCODING_ERROR; + } + PyMem_Free(cookie); + return _PYTOK_ENCODING_DONE; +} + +int +_PyTok_DecodeOnce(struct tok_state *tok, _PyTok_Chunk *chunk, + const char *encoding) +{ + PyObject *unicode = PyUnicode_Decode( + chunk->data, chunk->len, encoding, NULL); + if (unicode == NULL) { + tok->done = PyErr_ExceptionMatches(PyExc_MemoryError) + ? E_NOMEM : E_DECODE; + return -1; + } + return chunk_set_unicode(tok, chunk, unicode, 0); +} + +static Py_ssize_t +raw_line_length(const char *data, Py_ssize_t len) +{ + for (Py_ssize_t i = 0; i < len; i++) { + if (data[i] == '\n') { + return i + 1; + } + if (data[i] == '\r') { + return i + 1 < len && data[i + 1] == '\n' ? i + 2 : i + 1; + } + } + return len; +} + +static int +store_prepared_source(struct tok_state *tok, const char *data, Py_ssize_t len, + int preserve_crlf, int add_final_newline) +{ + Py_ssize_t pos = 0; + while (pos < len) { + Py_ssize_t raw_line_len; + if (preserve_crlf) { + const char *newline = memchr(data + pos, '\n', len - pos); + raw_line_len = newline == NULL + ? len - pos : newline - data - pos + 1; + } + else { + raw_line_len = raw_line_length(data + pos, len - pos); + } + int terminated = preserve_crlf + ? data[pos + raw_line_len - 1] == '\n' + : data[pos + raw_line_len - 1] == '\n' || + data[pos + raw_line_len - 1] == '\r'; + int add_newline = add_final_newline && + pos + raw_line_len == len && !terminated; + int normalize = add_newline || + (!preserve_crlf && + memchr(data + pos, '\r', raw_line_len) != NULL); + + const char *line = data + pos; + Py_ssize_t line_len = raw_line_len; + char *normalized = NULL; + int implicit = 0; + if (normalize) { + normalized = _PyTok_NormalizeNewlines( + line, line_len, preserve_crlf, add_newline, + &line_len, &implicit); + if (normalized == NULL) { + tok->done = E_NOMEM; + return -1; + } + line = normalized; + } + _PyTok_Off appended = _PyTok_SourceAppendLine( + &tok->source, line, line_len, implicit); + PyMem_Free(normalized); + if (appended < 0) { + tok->done = PyErr_ExceptionMatches(PyExc_MemoryError) + ? E_NOMEM : E_ERROR; + return -1; + } + pos += raw_line_len; + } + return 0; +} + +int +_PyTok_PrepareString(struct tok_state *tok, const char *input, int utf8_only, + int exec_input, int preserve_crlf) +{ + Py_ssize_t raw_len = strlen(input); + char *raw = (char *)input; + + if (utf8_only) { + if (_PyTok_SetEncoding(tok, "utf-8") < 0) { + return -1; + } + } + else { + Py_ssize_t first_original_len = raw_line_length(raw, raw_len); + _PyTok_Chunk first = { + .data = raw, + .len = first_original_len, + .ownership = _PYTOK_CHUNK_BORROWED, + }; + _PyTok_Chunk second = {0}; + int have_second = first_original_len < raw_len; + if (have_second) { + second.data = raw + first_original_len; + second.len = raw_line_length(second.data, + raw_len - first_original_len); + } + Py_ssize_t bom_len; + _PyTok_EncodingResult detection = _PyTok_DetectEncoding( + tok, &first, have_second ? &second : NULL, 1, &bom_len); + if (detection == _PYTOK_ENCODING_ERROR) { + return -1; + } + raw += bom_len; + raw_len -= bom_len; + } + + _PyTok_Chunk decoded = { + .data = raw, + .len = raw_len, + .ownership = _PYTOK_CHUNK_BORROWED, + }; + if (tok->encoding != NULL && strcmp(tok->encoding, "utf-8") != 0) { + if (_PyTok_DecodeOnce(tok, &decoded, tok->encoding) < 0) { + return -1; + } + } + + int stored = store_prepared_source( + tok, decoded.data, decoded.len, preserve_crlf, exec_input); + _PyTok_ChunkClear(&decoded); + if (stored < 0) { + return -1; + } + tok->str = tok->source.bytes != NULL ? tok->source.bytes : (char *)""; + if (!utf8_only && + (tok->encoding == NULL || strcmp(tok->encoding, "utf-8") == 0) && + !_PyTokenizer_ensure_utf8(tok->str, tok, 1)) { + return -1; + } + return 0; +} + +int +_PyTok_StartDecoder(struct tok_state *tok, const char *errors) +{ + _PyTok_Reader *reader = tok->reader; + if (tok->encoding == NULL || reader->decoder != NULL) { + return 0; + } + if (reader->kind == _PYTOK_READER_FILE && + strcmp(tok->encoding, "utf-8") == 0) { + return 0; + } + + PyObject *codec = _PyCodec_LookupTextEncoding(tok->encoding, NULL); + if (codec != NULL) { + PyObject *factory = PyObject_GetAttrString(codec, "incrementaldecoder"); + Py_DECREF(codec); + if (factory != NULL) { + reader->decoder = PyObject_CallFunction(factory, "s", errors); + Py_DECREF(factory); + } + } + if (reader->decoder == NULL) { + tok->done = PyErr_ExceptionMatches(PyExc_MemoryError) + ? E_NOMEM : E_DECODE; + if (reader->kind == _PYTOK_READER_FILE) { + _PyTokenizer_raise_init_error( + tok->filename != NULL ? tok->filename : Py_None); + } + return -1; + } + return 0; +} + +int +_PyTok_DecodeChunk(struct tok_state *tok, _PyTok_Chunk *chunk, int final) +{ + _PyTok_Reader *reader = tok->reader; + if (reader->decoder == NULL) { + return 0; + } + int strip_bom = reader->kind == _PYTOK_READER_READLINE && + chunk->len >= 2 && + (((unsigned char)chunk->data[0] == 0xFF && + (unsigned char)chunk->data[1] == 0xFE) || + ((unsigned char)chunk->data[0] == 0xFE && + (unsigned char)chunk->data[1] == 0xFF)); + PyObject *input; + if (chunk->ownership == _PYTOK_CHUNK_PYOBJECT && + PyBytes_Check(chunk->owner) && + chunk->data == PyBytes_AS_STRING(chunk->owner)) { + input = Py_NewRef(chunk->owner); + } + else { + input = PyBytes_FromStringAndSize(chunk->data, chunk->len); + } + if (input == NULL) { + tok->done = E_NOMEM; + return -1; + } + PyObject *unicode = PyObject_CallMethodObjArgs( + reader->decoder, &_Py_ID(decode), input, + final ? Py_True : Py_False, NULL); + Py_DECREF(input); + if (unicode == NULL) { + tok->done = PyErr_ExceptionMatches(PyExc_MemoryError) + ? E_NOMEM : E_DECODE; + if (reader->kind == _PYTOK_READER_FILE) { + _PyTokenizer_raise_init_error( + tok->filename != NULL ? tok->filename : Py_None); + } + return -1; + } + if (!PyUnicode_Check(unicode)) { + PyErr_Format(PyExc_TypeError, + "decoder should return a string result, not '%.200s'", + Py_TYPE(unicode)->tp_name); + Py_DECREF(unicode); + tok->done = E_DECODE; + return -1; + } + return chunk_set_unicode(tok, chunk, unicode, strip_bom); +} + +int +_PyTok_DecoderHasBufferedInput(struct tok_state *tok) +{ + if (tok->reader->decoder == NULL) { + return 0; + } + PyObject *state = PyObject_CallMethodNoArgs( + tok->reader->decoder, &_Py_ID(getstate)); + if (state == NULL) { + tok->done = PyErr_ExceptionMatches(PyExc_MemoryError) + ? E_NOMEM : E_DECODE; + return -1; + } + if (!PyTuple_Check(state) || PyTuple_GET_SIZE(state) != 2 || + !PyBytes_Check(PyTuple_GET_ITEM(state, 0)) || + !PyLong_Check(PyTuple_GET_ITEM(state, 1))) { + Py_DECREF(state); + PyErr_SetString(PyExc_TypeError, + "incremental decoder getstate() must return (bytes, int)"); + tok->done = E_DECODE; + return -1; + } + int pending = PyBytes_GET_SIZE(PyTuple_GET_ITEM(state, 0)) != 0; + Py_DECREF(state); + return pending; +} diff --git a/Parser/tokenizer/file_tokenizer.c b/Parser/tokenizer/file_tokenizer.c deleted file mode 100644 index a11702557a07af3..000000000000000 --- a/Parser/tokenizer/file_tokenizer.c +++ /dev/null @@ -1,494 +0,0 @@ -#include "Python.h" -#include "pycore_call.h" // _PyObject_CallNoArgs() -#include "pycore_fileutils.h" // _Py_UniversalNewlineFgetsWithSize() -#include "pycore_runtime.h" // _Py_ID() - -#include "errcode.h" // E_NOMEM - -#ifdef HAVE_UNISTD_H -# include // lseek(), read() -#endif - -#include "helpers.h" -#include "../lexer/state.h" -#include "../lexer/lexer.h" -#include "../lexer/buffer.h" - - -static int -tok_concatenate_interactive_new_line(struct tok_state *tok, const char *line) { - assert(tok->fp_interactive); - - if (!line) { - return 0; - } - - Py_ssize_t current_size = tok->interactive_src_end - tok->interactive_src_start; - Py_ssize_t line_size = strlen(line); - char last_char = line[line_size > 0 ? line_size - 1 : line_size]; - if (last_char != '\n') { - line_size += 1; - } - char* new_str = tok->interactive_src_start; - - new_str = PyMem_Realloc(new_str, current_size + line_size + 1); - if (!new_str) { - if (tok->interactive_src_start) { - PyMem_Free(tok->interactive_src_start); - } - tok->interactive_src_start = NULL; - tok->interactive_src_end = NULL; - tok->done = E_NOMEM; - return -1; - } - strcpy(new_str + current_size, line); - tok->implicit_newline = 0; - if (last_char != '\n') { - /* Last line does not end in \n, fake one */ - new_str[current_size + line_size - 1] = '\n'; - new_str[current_size + line_size] = '\0'; - tok->implicit_newline = 1; - } - tok->interactive_src_start = new_str; - tok->interactive_src_end = new_str + current_size + line_size; - return 0; -} - -static int -tok_readline_raw(struct tok_state *tok) -{ - do { - if (!_PyLexer_tok_reserve_buf(tok, BUFSIZ)) { - return 0; - } - int n_chars = (int)(tok->end - tok->inp); - size_t line_size = 0; - char *line = _Py_UniversalNewlineFgetsWithSize(tok->inp, n_chars, tok->fp, NULL, &line_size); - if (line == NULL) { - return 1; - } - if (tok->fp_interactive && - tok_concatenate_interactive_new_line(tok, line) == -1) { - return 0; - } - tok->inp += line_size; - if (tok->inp == tok->buf) { - return 0; - } - } while (tok->inp[-1] != '\n'); - return 1; -} - -static int -tok_readline_recode(struct tok_state *tok) { - PyObject *line; - const char *buf; - Py_ssize_t buflen; - line = tok->decoding_buffer; - if (line == NULL) { - line = PyObject_CallNoArgs(tok->decoding_readline); - if (line == NULL) { - _PyTokenizer_error_ret(tok); - goto error; - } - } - else { - tok->decoding_buffer = NULL; - } - buf = PyUnicode_AsUTF8AndSize(line, &buflen); - if (buf == NULL) { - _PyTokenizer_error_ret(tok); - goto error; - } - // Make room for the null terminator *and* potentially - // an extra newline character that we may need to artificially - // add. - size_t buffer_size = buflen + 2; - if (!_PyLexer_tok_reserve_buf(tok, buffer_size)) { - goto error; - } - memcpy(tok->inp, buf, buflen); - tok->inp += buflen; - *tok->inp = '\0'; - if (tok->fp_interactive && - tok_concatenate_interactive_new_line(tok, buf) == -1) { - goto error; - } - Py_DECREF(line); - return 1; -error: - Py_XDECREF(line); - return 0; -} - -/* Fetch the next byte from TOK. */ -static int fp_getc(struct tok_state *tok) { - return getc(tok->fp); -} - -/* Unfetch the last byte back into TOK. */ -static void fp_ungetc(int c, struct tok_state *tok) { - ungetc(c, tok->fp); -} - -/* Set the readline function for TOK to a StreamReader's - readline function. The StreamReader is named ENC. - - This function is called from _PyTokenizer_check_bom and _PyTokenizer_check_coding_spec. - - ENC is usually identical to the future value of tok->encoding, - except for the (currently unsupported) case of UTF-16. - - Return 1 on success, 0 on failure. */ -static int -fp_setreadl(struct tok_state *tok, const char* enc) -{ - PyObject *readline, *open, *stream; - int fd; - long pos; - - fd = fileno(tok->fp); - /* Due to buffering the file offset for fd can be different from the file - * position of tok->fp. If tok->fp was opened in text mode on Windows, - * its file position counts CRLF as one char and can't be directly mapped - * to the file offset for fd. Instead we step back one byte and read to - * the end of line.*/ - pos = ftell(tok->fp); - if (pos == -1 || - lseek(fd, (off_t)(pos > 0 ? pos - 1 : pos), SEEK_SET) == (off_t)-1) { - PyErr_SetFromErrnoWithFilename(PyExc_OSError, NULL); - return 0; - } - - open = PyImport_ImportModuleAttrString("io", "open"); - if (open == NULL) { - return 0; - } - stream = PyObject_CallFunction(open, "isisOOO", - fd, "r", -1, enc, Py_None, Py_None, Py_False); - Py_DECREF(open); - if (stream == NULL) { - return 0; - } - - readline = PyObject_GetAttr(stream, &_Py_ID(readline)); - Py_DECREF(stream); - if (readline == NULL) { - return 0; - } - Py_XSETREF(tok->decoding_readline, readline); - - if (pos > 0) { - PyObject *bufobj = _PyObject_CallNoArgs(readline); - if (bufobj == NULL) { - return 0; - } - Py_DECREF(bufobj); - } - - return 1; -} - -static int -tok_underflow_interactive(struct tok_state *tok) { - if (tok->interactive_underflow == IUNDERFLOW_STOP) { - tok->done = E_INTERACT_STOP; - return 1; - } - char *newtok = PyOS_Readline(tok->fp ? tok->fp : stdin, stdout, tok->prompt); - if (newtok != NULL) { - char *translated = _PyTokenizer_translate_newlines(newtok, 0, 0, tok); - PyMem_Free(newtok); - if (translated == NULL) { - return 0; - } - newtok = translated; - } - if (tok->encoding && newtok && *newtok) { - /* Recode to UTF-8 */ - Py_ssize_t buflen; - const char* buf; - PyObject *u = _PyTokenizer_translate_into_utf8(newtok, tok->encoding); - PyMem_Free(newtok); - if (u == NULL) { - tok->done = E_DECODE; - return 0; - } - buflen = PyBytes_GET_SIZE(u); - buf = PyBytes_AS_STRING(u); - newtok = PyMem_Malloc(buflen+1); - if (newtok == NULL) { - Py_DECREF(u); - tok->done = E_NOMEM; - return 0; - } - strcpy(newtok, buf); - Py_DECREF(u); - } - if (tok->fp_interactive && - tok_concatenate_interactive_new_line(tok, newtok) == -1) { - PyMem_Free(newtok); - return 0; - } - if (tok->nextprompt != NULL) { - tok->prompt = tok->nextprompt; - } - if (newtok == NULL) { - tok->done = E_INTR; - } - else if (*newtok == '\0') { - PyMem_Free(newtok); - tok->done = E_EOF; - } - else if (tok->start != NULL) { - Py_ssize_t cur_multi_line_start = tok->multi_line_start - tok->buf; - _PyLexer_remember_fstring_buffers(tok); - size_t size = strlen(newtok); - ADVANCE_LINENO(); - if (!_PyLexer_tok_reserve_buf(tok, size + 1)) { - PyMem_Free(tok->buf); - tok->buf = NULL; - PyMem_Free(newtok); - return 0; - } - memcpy(tok->cur, newtok, size + 1); - PyMem_Free(newtok); - tok->inp += size; - tok->multi_line_start = tok->buf + cur_multi_line_start; - _PyLexer_restore_fstring_buffers(tok); - } - else { - _PyLexer_remember_fstring_buffers(tok); - ADVANCE_LINENO(); - PyMem_Free(tok->buf); - tok->buf = newtok; - tok->cur = tok->buf; - tok->line_start = tok->buf; - tok->inp = strchr(tok->buf, '\0'); - tok->end = tok->inp + 1; - _PyLexer_restore_fstring_buffers(tok); - } - if (tok->done != E_OK) { - if (tok->prompt != NULL) { - PySys_WriteStderr("\n"); - } - return 0; - } - - if (tok->tok_mode_stack_index && !_PyLexer_update_ftstring_expr(tok, 0)) { - return 0; - } - return 1; -} - -static int -tok_underflow_file(struct tok_state *tok) -{ - if (tok->decoding_state == STATE_INIT) { - /* We have not yet determined the encoding. - If an encoding is found, use the file-pointer - reader functions from now on. */ - if (!_PyTokenizer_check_bom(fp_getc, fp_ungetc, fp_setreadl, tok)) { - _PyTokenizer_error_ret(tok); - return 0; - } - assert(tok->decoding_state != STATE_INIT); - } - int raw = tok->decoding_readline == NULL; - if (raw && tok->decoding_state != STATE_NORMAL) { - /* Keep the first line in the buffer to validate it later if - * the encoding has not yet been determined. */ - } - else if (tok->start == NULL && !INSIDE_FSTRING(tok)) { - tok->cur = tok->inp = tok->buf; - } - /* Read until '\n' or EOF */ - if (!raw) { - /* We already have a codec associated with this input. */ - if (!tok_readline_recode(tok)) { - return 0; - } - } - else { - /* We want a 'raw' read. */ - if (!tok_readline_raw(tok)) { - return 0; - } - } - if (tok->inp == tok->cur) { - tok->done = E_EOF; - return 0; - } - tok->implicit_newline = 0; - if (tok->inp[-1] != '\n') { - assert(tok->inp + 1 < tok->end); - /* Last line does not end in \n, fake one */ - *tok->inp++ = '\n'; - *tok->inp = '\0'; - tok->implicit_newline = 1; - } - - if (tok->tok_mode_stack_index && !_PyLexer_update_ftstring_expr(tok, 0)) { - return 0; - } - - ADVANCE_LINENO(); - if (tok->decoding_state != STATE_NORMAL) { - if (!_PyTokenizer_check_coding_spec(tok->cur, strlen(tok->cur), - tok, fp_setreadl)) - { - return 0; - } - if (tok->lineno >= 2) { - tok->decoding_state = STATE_NORMAL; - } - } - if (raw && tok->decoding_state == STATE_NORMAL) { - const char *line = tok->lineno <= 2 ? tok->buf : tok->cur; - int lineno = tok->lineno <= 2 ? 1 : tok->lineno; - if (!tok->encoding) { - /* The default encoding is UTF-8, so make sure we don't have any - non-UTF-8 sequences in it. */ - if (!_PyTokenizer_ensure_utf8(line, tok, lineno)) { - _PyTokenizer_error_ret(tok); - return 0; - } - } - else { - PyObject *tmp = PyUnicode_Decode(line, strlen(line), - tok->encoding, NULL); - if (tmp == NULL) { - _PyTokenizer_error_ret(tok); - return 0; - } - Py_DECREF(tmp); - } - } - assert(tok->done == E_OK); - return tok->done == E_OK; -} - -/* Set up tokenizer for file */ -struct tok_state * -_PyTokenizer_FromFile(FILE *fp, const char* enc, - const char *ps1, const char *ps2) -{ - struct tok_state *tok = _PyTokenizer_tok_new(); - if (tok == NULL) - return NULL; - if ((tok->buf = (char *)PyMem_Malloc(BUFSIZ)) == NULL) { - _PyTokenizer_Free(tok); - PyErr_NoMemory(); - return NULL; - } - tok->cur = tok->inp = tok->buf; - tok->end = tok->buf + BUFSIZ; - tok->fp = fp; - tok->prompt = ps1; - tok->nextprompt = ps2; - if (ps1 || ps2) { - tok->underflow = &tok_underflow_interactive; - } else { - tok->underflow = &tok_underflow_file; - } - if (enc != NULL) { - /* Must copy encoding declaration since it - gets copied into the parse tree. */ - tok->encoding = _PyTokenizer_new_string(enc, strlen(enc), tok); - if (!tok->encoding) { - _PyTokenizer_Free(tok); - return NULL; - } - tok->decoding_state = STATE_NORMAL; - } - return tok; -} - -#if defined(__wasi__) || (defined(__EMSCRIPTEN__) && (__EMSCRIPTEN_major__ >= 3)) -// fdopen() with borrowed fd. WASI does not provide dup() and Emscripten's -// dup() emulation with open() is slow. -typedef union { - void *cookie; - int fd; -} borrowed; - -static ssize_t -borrow_read(void *cookie, char *buf, size_t size) -{ - borrowed b = {.cookie = cookie}; - return read(b.fd, (void *)buf, size); -} - -static FILE * -fdopen_borrow(int fd) { - // supports only reading. seek fails. close and write are no-ops. - cookie_io_functions_t io_cb = {borrow_read, NULL, NULL, NULL}; - borrowed b = {.fd = fd}; - return fopencookie(b.cookie, "r", io_cb); -} -#else -static FILE * -fdopen_borrow(int fd) { - fd = _Py_dup(fd); - if (fd < 0) { - return NULL; - } - return fdopen(fd, "r"); -} -#endif - -/* Get the encoding of a Python file. Check for the coding cookie and check if - the file starts with a BOM. - - _PyTokenizer_FindEncodingFilename() returns NULL when it can't find the - encoding in the first or second line of the file (in which case the encoding - should be assumed to be UTF-8). - - The char* returned is malloc'ed via PyMem_Malloc() and thus must be freed - by the caller. */ -char * -_PyTokenizer_FindEncodingFilename(int fd, PyObject *filename) -{ - struct tok_state *tok; - FILE *fp; - char *encoding = NULL; - - fp = fdopen_borrow(fd); - if (fp == NULL) { - return NULL; - } - tok = _PyTokenizer_FromFile(fp, NULL, NULL, NULL); - if (tok == NULL) { - fclose(fp); - return NULL; - } - if (filename != NULL) { - tok->filename = Py_NewRef(filename); - } - else { - tok->filename = PyUnicode_FromString(""); - if (tok->filename == NULL) { - fclose(fp); - _PyTokenizer_Free(tok); - return encoding; - } - } - struct token token; - // We don't want to report warnings here because it could cause infinite recursion - // if fetching the encoding shows a warning. - tok->report_warnings = 0; - while (tok->lineno < 2 && tok->done == E_OK) { - _PyToken_Init(&token); - _PyTokenizer_Get(tok, &token); - _PyToken_Free(&token); - } - fclose(fp); - if (tok->encoding) { - encoding = (char *)PyMem_Malloc(strlen(tok->encoding) + 1); - if (encoding) { - strcpy(encoding, tok->encoding); - } - } - _PyTokenizer_Free(tok); - return encoding; -} diff --git a/Parser/tokenizer/helpers.c b/Parser/tokenizer/helpers.c index 62b0971d418c396..bbd64760a18a664 100644 --- a/Parser/tokenizer/helpers.c +++ b/Parser/tokenizer/helpers.c @@ -95,20 +95,6 @@ _PyTokenizer_indenterror(struct tok_state *tok) return ERRORTOKEN; } -char * -_PyTokenizer_error_ret(struct tok_state *tok) /* XXX */ -{ - tok->decoding_erred = 1; - if ((tok->fp != NULL || tok->readline != NULL) && tok->buf != NULL) {/* see _PyTokenizer_Free */ - PyMem_Free(tok->buf); - } - tok->buf = tok->cur = tok->inp = NULL; - tok->start = NULL; - tok->end = NULL; - tok->done = E_DECODE; - return NULL; /* as if it were EOF */ -} - int _PyTokenizer_warn_invalid_escape_sequence(struct tok_state *tok, int first_invalid_escape_char) { @@ -234,261 +220,6 @@ _PyTokenizer_parser_warn(struct tok_state *tok, PyObject *category, const char * } -/* ############## STRING MANIPULATION ############## */ - -char * -_PyTokenizer_new_string(const char *s, Py_ssize_t len, struct tok_state *tok) -{ - char* result = (char *)PyMem_Malloc(len + 1); - if (!result) { - tok->done = E_NOMEM; - PyErr_NoMemory(); - return NULL; - } - memcpy(result, s, len); - result[len] = '\0'; - return result; -} - -PyObject * -_PyTokenizer_translate_into_utf8(const char* str, const char* enc) { - PyObject *utf8; - PyObject* buf = PyUnicode_Decode(str, strlen(str), enc, NULL); - if (buf == NULL) - return NULL; - utf8 = PyUnicode_AsUTF8String(buf); - Py_DECREF(buf); - return utf8; -} - -char * -_PyTokenizer_translate_newlines(const char *s, int exec_input, int preserve_crlf, - struct tok_state *tok) { - int skip_next_lf = 0; - size_t needed_length = strlen(s) + 2, final_length; - char *buf, *current; - char c = '\0'; - buf = PyMem_Malloc(needed_length); - if (buf == NULL) { - tok->done = E_NOMEM; - PyErr_NoMemory(); - return NULL; - } - for (current = buf; *s; s++, current++) { - c = *s; - if (skip_next_lf) { - skip_next_lf = 0; - if (c == '\n') { - c = *++s; - if (!c) - break; - } - } - if (!preserve_crlf && c == '\r') { - skip_next_lf = 1; - c = '\n'; - } - *current = c; - } - /* If this is exec input, add a newline to the end of the string if - there isn't one already. */ - if (exec_input && c != '\n' && c != '\0') { - *current = '\n'; - current++; - } - *current = '\0'; - final_length = current - buf + 1; - if (final_length < needed_length && final_length) { - /* should never fail */ - char* result = PyMem_Realloc(buf, final_length); - if (result == NULL) { - PyMem_Free(buf); - } - buf = result; - } - return buf; -} - -/* ############## ENCODING STUFF ############## */ - - -/* See whether the file starts with a BOM. If it does, - invoke the set_readline function with the new encoding. - Return 1 on success, 0 on failure. */ -int -_PyTokenizer_check_bom(int get_char(struct tok_state *), - void unget_char(int, struct tok_state *), - int set_readline(struct tok_state *, const char *), - struct tok_state *tok) -{ - int ch1, ch2, ch3; - ch1 = get_char(tok); - tok->decoding_state = STATE_SEEK_CODING; - if (ch1 == EOF) { - return 1; - } else if (ch1 == 0xEF) { - ch2 = get_char(tok); - if (ch2 != 0xBB) { - unget_char(ch2, tok); - unget_char(ch1, tok); - return 1; - } - ch3 = get_char(tok); - if (ch3 != 0xBF) { - unget_char(ch3, tok); - unget_char(ch2, tok); - unget_char(ch1, tok); - return 1; - } - } else { - unget_char(ch1, tok); - return 1; - } - if (tok->encoding != NULL) - PyMem_Free(tok->encoding); - tok->encoding = _PyTokenizer_new_string("utf-8", 5, tok); - if (!tok->encoding) - return 0; - /* No need to set_readline: input is already utf-8 */ - return 1; -} - -static const char * -get_normal_name(const char *s) /* for utf-8 and latin-1 */ -{ - char buf[13]; - int i; - for (i = 0; i < 12; i++) { - int c = s[i]; - if (c == '\0') - break; - else if (c == '_') - buf[i] = '-'; - else - buf[i] = Py_TOLOWER(c); - } - buf[i] = '\0'; - if (strcmp(buf, "utf-8") == 0 || - strncmp(buf, "utf-8-", 6) == 0) - return "utf-8"; - else if (strcmp(buf, "latin-1") == 0 || - strcmp(buf, "iso-8859-1") == 0 || - strcmp(buf, "iso-latin-1") == 0 || - strncmp(buf, "latin-1-", 8) == 0 || - strncmp(buf, "iso-8859-1-", 11) == 0 || - strncmp(buf, "iso-latin-1-", 12) == 0) - return "iso-8859-1"; - else - return s; -} - -/* Return the coding spec in S, or NULL if none is found. */ -static int -get_coding_spec(const char *s, char **spec, Py_ssize_t size, struct tok_state *tok) -{ - Py_ssize_t i; - *spec = NULL; - /* Coding spec must be in a comment, and that comment must be - * the only statement on the source code line. */ - for (i = 0; i < size - 6; i++) { - if (s[i] == '#') - break; - if (s[i] != ' ' && s[i] != '\t' && s[i] != '\014') - return 1; - } - for (; i < size - 6; i++) { /* XXX inefficient search */ - const char* t = s + i; - if (memcmp(t, "coding", 6) == 0) { - const char* begin = NULL; - t += 6; - if (t[0] != ':' && t[0] != '=') - continue; - do { - t++; - } while (t[0] == ' ' || t[0] == '\t'); - - begin = t; - while (Py_ISALNUM(t[0]) || - t[0] == '-' || t[0] == '_' || t[0] == '.') - t++; - - if (begin < t) { - char* r = _PyTokenizer_new_string(begin, t - begin, tok); - const char* q; - if (!r) - return 0; - q = get_normal_name(r); - if (r != q) { - PyMem_Free(r); - r = _PyTokenizer_new_string(q, strlen(q), tok); - if (!r) - return 0; - } - *spec = r; - break; - } - } - } - return 1; -} - -/* Check whether the line contains a coding spec. If it does, - invoke the set_readline function for the new encoding. - This function receives the tok_state and the new encoding. - Return 1 on success, 0 on failure. */ -int -_PyTokenizer_check_coding_spec(const char* line, Py_ssize_t size, struct tok_state *tok, - int set_readline(struct tok_state *, const char *)) -{ - char *cs; - if (tok->cont_line) { - /* It's a continuation line, so it can't be a coding spec. */ - tok->decoding_state = STATE_NORMAL; - return 1; - } - if (!get_coding_spec(line, &cs, size, tok)) { - return 0; - } - if (!cs) { - Py_ssize_t i; - for (i = 0; i < size; i++) { - if (line[i] == '#' || line[i] == '\n' || line[i] == '\r') - break; - if (line[i] != ' ' && line[i] != '\t' && line[i] != '\014') { - /* Stop checking coding spec after a line containing - * anything except a comment. */ - tok->decoding_state = STATE_NORMAL; - break; - } - } - return 1; - } - tok->decoding_state = STATE_NORMAL; - if (tok->encoding == NULL) { - assert(tok->decoding_readline == NULL); - if (strcmp(cs, "utf-8") != 0 && !set_readline(tok, cs)) { - _PyTokenizer_raise_init_error(tok->filename); - _PyTokenizer_error_ret(tok); - PyMem_Free(cs); - return 0; - } - tok->encoding = cs; - } else { /* then, compare cs with BOM */ - if (strcmp(tok->encoding, cs) != 0) { - tok->line_start = line; - tok->cur = (char *)line; - assert(size <= INT_MAX); - _PyTokenizer_syntaxerror_known_range(tok, 0, (int)size, - "encoding problem: %s with BOM", cs); - PyMem_Free(cs); - _PyTokenizer_error_ret(tok); - return 0; - } - PyMem_Free(cs); - } - return 1; -} - /* Check whether the characters at s start a valid UTF-8 sequence. Return the number of characters forming the sequence if yes, 0 if not. The special cases match diff --git a/Parser/tokenizer/helpers.h b/Parser/tokenizer/helpers.h index 34303999a60aff7..5edf5a3dfd2e0d1 100644 --- a/Parser/tokenizer/helpers.h +++ b/Parser/tokenizer/helpers.h @@ -14,19 +14,8 @@ int _PyTokenizer_syntaxerror_known_range(struct tok_state *tok, int col_offset, int _PyTokenizer_indenterror(struct tok_state *tok); int _PyTokenizer_warn_invalid_escape_sequence(struct tok_state *tok, int first_invalid_escape_char); int _PyTokenizer_parser_warn(struct tok_state *tok, PyObject *category, const char *format, ...); -char *_PyTokenizer_error_ret(struct tok_state *tok); void _PyTokenizer_raise_init_error(PyObject *filename); -char *_PyTokenizer_new_string(const char *s, Py_ssize_t len, struct tok_state *tok); -char *_PyTokenizer_translate_newlines(const char *s, int exec_input, int preserve_crlf, struct tok_state *tok); -PyObject *_PyTokenizer_translate_into_utf8(const char* str, const char* enc); - -int _PyTokenizer_check_bom(int get_char(struct tok_state *), - void unget_char(int, struct tok_state *), - int set_readline(struct tok_state *, const char *), - struct tok_state *tok); -int _PyTokenizer_check_coding_spec(const char* line, Py_ssize_t size, struct tok_state *tok, - int set_readline(struct tok_state *, const char *)); int _PyTokenizer_ensure_utf8(const char *line, struct tok_state *tok, int lineno); #ifdef Py_DEBUG diff --git a/Parser/tokenizer/reader.c b/Parser/tokenizer/reader.c new file mode 100644 index 000000000000000..15a19f50f9a103b --- /dev/null +++ b/Parser/tokenizer/reader.c @@ -0,0 +1,767 @@ +#include "Python.h" +#include "pycore_fileutils.h" + +#include "errcode.h" +#include "helpers.h" +#include "reader.h" +#include "reader_internal.h" +#include "../lexer/buffer.h" +#include "../lexer/lexer.h" +#include "../lexer/state.h" + +#ifdef HAVE_UNISTD_H +# include +#endif + +void +_PyTok_ReaderFree(struct tok_state *tok) +{ + _PyTok_Reader *reader = tok->reader; + if (reader == NULL) { + return; + } + Py_XDECREF(reader->readline); + Py_XDECREF(reader->decoder); + for (int i = 0; + i < (int)Py_ARRAY_LENGTH(reader->prefetched_lines); i++) { + _PyTok_ChunkClear(&reader->prefetched_lines[i]); + } + PyMem_Free(reader->file_buffer); + PyMem_Free(reader->decoded); + if (reader->kind != _PYTOK_READER_PREPARED) { + PyMem_Free(tok->buf); + tok->buf = NULL; + } + PyMem_Free(reader); + tok->reader = NULL; +} + +static int +reserve_buffer(char **buffer, Py_ssize_t *capacity, Py_ssize_t needed) +{ + if (needed <= *capacity) { + return 0; + } + Py_ssize_t cap = *capacity > 0 ? *capacity : BUFSIZ; + while (cap < needed) { + if (cap > PY_SSIZE_T_MAX / 2) { + cap = needed; + break; + } + cap *= 2; + } + char *resized = PyMem_Realloc(*buffer, cap); + if (resized == NULL) { + PyErr_NoMemory(); + return -1; + } + *buffer = resized; + *capacity = cap; + return 0; +} + +static int +append_decoded(_PyTok_Reader *reader, const char *data, Py_ssize_t len) +{ + if (reader->decoded_pos > 0) { + Py_ssize_t remaining = reader->decoded_len - reader->decoded_pos; + memmove(reader->decoded, reader->decoded + reader->decoded_pos, + (size_t)remaining); + reader->decoded_pos = 0; + reader->decoded_len = remaining; + } + if (len < 0 || reader->decoded_len > PY_SSIZE_T_MAX - len - 1 || + reserve_buffer(&reader->decoded, &reader->decoded_cap, + reader->decoded_len + len + 1) < 0) { + PyErr_NoMemory(); + return -1; + } + memcpy(reader->decoded + reader->decoded_len, data, (size_t)len); + reader->decoded_len += len; + reader->decoded[reader->decoded_len] = '\0'; + return 0; +} + +static int +append_implicit_newline(_PyTok_Reader *reader) +{ + if (reader->decoded_pos == reader->decoded_len || + reader->decoded[reader->decoded_len - 1] == '\n') { + return 0; + } + if (append_decoded(reader, "\n", 1) < 0) { + return -1; + } + reader->decoded_tail_is_implicit = 1; + return 0; +} + +static int +pop_decoded_line(_PyTok_Reader *reader, _PyTok_Chunk *chunk) +{ + if (reader->decoded_pos == reader->decoded_len) { + return 0; + } + char *start = reader->decoded + reader->decoded_pos; + char *newline = memchr(start, '\n', + reader->decoded_len - reader->decoded_pos); + if (newline == NULL) { + return 0; + } + Py_ssize_t len = newline - start + 1; + chunk->data = start; + chunk->len = len; + chunk->ownership = _PYTOK_CHUNK_BORROWED; + reader->decoded_pos += len; + chunk->implicit_newline = reader->decoded_pos == reader->decoded_len && + reader->decoded_tail_is_implicit; + if (reader->decoded_pos == reader->decoded_len) { + reader->decoded_pos = reader->decoded_len = 0; + reader->decoded_tail_is_implicit = 0; + } + return 1; +} + +static int +chunk_is_line(const _PyTok_Chunk *chunk) +{ + if (chunk->len == 0 || chunk->data[chunk->len - 1] != '\n') { + return 0; + } + return memchr(chunk->data, '\n', chunk->len - 1) == NULL; +} + +static _PyTok_ReadResult +next_prepared(struct tok_state *tok, _PyTok_Chunk *chunk) +{ + int lineno = tok->lineno + 1; + if (lineno > tok->source.nlines) { + return _PYTOK_READ_EOF; + } + const char *start = tok->inp; + const char *newline = memchr( + start, '\n', tok->source.bytes + tok->source.len - start); + _PyTok_Off end = newline != NULL + ? newline - tok->source.bytes + 1 : tok->source.len; + chunk->data = (char *)start; + chunk->len = tok->source.bytes + end - start; + chunk->ownership = _PYTOK_CHUNK_BORROWED; + chunk->implicit_newline = _PyTok_SourceLineIsImplicit( + &tok->source, lineno); + return _PYTOK_READ_LINE; +} + +static _PyTok_ReadResult +read_file_line(struct tok_state *tok, _PyTok_Chunk *chunk) +{ + _PyTok_Reader *reader = tok->reader; + Py_ssize_t len = 0; + for (;;) { + if (len > PY_SSIZE_T_MAX - BUFSIZ || + reserve_buffer(&reader->file_buffer, &reader->file_buffer_cap, + len + BUFSIZ) < 0) { + return _PYTOK_READ_ERROR; + } + int available = (int)Py_MIN(reader->file_buffer_cap - len, INT_MAX); + size_t read = 0; + char *result = _Py_UniversalNewlineFgetsWithSize( + reader->file_buffer + len, available, tok->fp, NULL, &read); + if (result == NULL) { + if (len == 0) { + return _PYTOK_READ_EOF; + } + break; + } + len += (Py_ssize_t)read; + if (len > 0 && reader->file_buffer[len - 1] == '\n') { + break; + } + } + int implicit = len == 0 || reader->file_buffer[len - 1] != '\n'; + chunk->data = reader->file_buffer; + chunk->len = len; + chunk->implicit_newline = implicit; + chunk->ownership = _PYTOK_CHUNK_BORROWED; + return _PYTOK_READ_LINE; +} + +static int +initialize_file(struct tok_state *tok) +{ + _PyTok_Reader *reader = tok->reader; + reader->file_initialized = 1; + if (tok->encoding != NULL) { + return _PyTok_StartDecoder(tok, "strict"); + } + + _PyTok_ReadResult result = read_file_line( + tok, &reader->prefetched_lines[0]); + if (result == _PYTOK_READ_EOF) { + reader->file_eof = 1; + return 0; + } + if (result != _PYTOK_READ_LINE) { + return -1; + } + reader->prefetched_count = 1; + Py_ssize_t bom_len; + _PyTok_EncodingResult detection = _PyTok_DetectEncoding( + tok, &reader->prefetched_lines[0], NULL, 0, &bom_len); + if (detection == _PYTOK_ENCODING_ERROR) { + return -1; + } + if (detection == _PYTOK_ENCODING_NEED_SECOND_LINE) { + char *first = _PyTok_CopyBytes( + reader->prefetched_lines[0].data, + reader->prefetched_lines[0].len); + if (first == NULL) { + tok->done = E_NOMEM; + return -1; + } + reader->prefetched_lines[0].data = first; + reader->prefetched_lines[0].ownership = _PYTOK_CHUNK_PYMEM; + result = read_file_line(tok, &reader->prefetched_lines[1]); + if (result == _PYTOK_READ_LINE) { + reader->prefetched_count = 2; + } + else if (result == _PYTOK_READ_EOF) { + reader->file_eof = 1; + } + else { + return -1; + } + _PyTok_Chunk *second = reader->prefetched_count == 2 + ? &reader->prefetched_lines[1] : NULL; + detection = _PyTok_DetectEncoding( + tok, &reader->prefetched_lines[0], second, 1, &bom_len); + if (detection == _PYTOK_ENCODING_ERROR) { + return -1; + } + } + if (bom_len != 0) { + _PyTok_Chunk *first = &reader->prefetched_lines[0]; + if (first->ownership == _PYTOK_CHUNK_PYMEM) { + memmove(first->data, first->data + bom_len, + (size_t)(first->len - bom_len)); + first->data[first->len - bom_len] = '\0'; + } + else { + first->data += bom_len; + } + first->len -= bom_len; + } + if (_PyTok_StartDecoder(tok, "strict") < 0) { + return -1; + } + return 0; +} + +static int +finalize_decoding(struct tok_state *tok) +{ + _PyTok_Reader *reader = tok->reader; + if (reader->decoder_finalized) { + return 0; + } + reader->decoder_finalized = 1; + if (reader->decoder != NULL) { + _PyTok_Chunk input = { + .data = "", + .ownership = _PYTOK_CHUNK_BORROWED, + }; + int decoded = _PyTok_DecodeChunk(tok, &input, 1); + if (decoded == 0 && + append_decoded(reader, input.data, input.len) < 0) { + tok->done = E_NOMEM; + decoded = -1; + } + _PyTok_ChunkClear(&input); + if (decoded < 0) { + return -1; + } + } + if (reader->decoded_pos < reader->decoded_len && + append_implicit_newline(reader) < 0) { + tok->done = E_NOMEM; + return -1; + } + return 0; +} + +static _PyTok_ReadResult +next_file(struct tok_state *tok, _PyTok_Chunk *chunk) +{ + _PyTok_Reader *reader = tok->reader; + if (!reader->file_initialized && initialize_file(tok) < 0) { + return _PYTOK_READ_ERROR; + } + for (;;) { + if (pop_decoded_line(reader, chunk)) { + return _PYTOK_READ_LINE; + } + _PyTok_Chunk input = {0}; + if (reader->prefetched_index < reader->prefetched_count) { + input = reader->prefetched_lines[reader->prefetched_index]; + reader->prefetched_lines[reader->prefetched_index++] = + (_PyTok_Chunk){0}; + } + else if (!reader->file_eof) { + _PyTok_ReadResult result = read_file_line(tok, &input); + if (result == _PYTOK_READ_ERROR) { + return result; + } + if (result == _PYTOK_READ_EOF) { + reader->file_eof = 1; + } + } + if (input.data != NULL) { + int implicit = input.implicit_newline; + if (reader->decoder == NULL && !implicit) { + *chunk = input; + return _PYTOK_READ_LINE; + } + int decoded = _PyTok_DecodeChunk(tok, &input, 0); + if (decoded == 0 && chunk_is_line(&input)) { + *chunk = input; + return _PYTOK_READ_LINE; + } + if (decoded == 0 && + append_decoded(reader, input.data, input.len) < 0) { + tok->done = E_NOMEM; + decoded = -1; + } + if (decoded == 0 && implicit) { + reader->decoded_tail_is_implicit = 1; + } + _PyTok_ChunkClear(&input); + if (decoded < 0) { + return _PYTOK_READ_ERROR; + } + continue; + } + if (!reader->decoder_finalized) { + if (finalize_decoding(tok) < 0) { + return _PYTOK_READ_ERROR; + } + continue; + } + return _PYTOK_READ_EOF; + } +} + +static _PyTok_ReadResult +next_readline(struct tok_state *tok, _PyTok_Chunk *chunk) +{ + _PyTok_Reader *reader = tok->reader; + for (;;) { + if (pop_decoded_line(reader, chunk)) { + return _PYTOK_READ_LINE; + } + if (reader->decoder_finalized) { + return _PYTOK_READ_EOF; + } + + PyObject *raw = PyObject_CallNoArgs(reader->readline); + if (raw == NULL) { + if (PyErr_ExceptionMatches(PyExc_StopIteration)) { + PyErr_Clear(); + if (finalize_decoding(tok) < 0) { + return _PYTOK_READ_ERROR; + } + continue; + } + return _PYTOK_READ_ERROR; + } + + _PyTok_Chunk input = {0}; + if (tok->encoding != NULL) { + if (!PyBytes_Check(raw)) { + PyErr_SetString(PyExc_TypeError, + "readline() returned a non-bytes object"); + Py_DECREF(raw); + return _PYTOK_READ_ERROR; + } + if (_PyTok_StartDecoder(tok, "replace") < 0) { + Py_XDECREF(raw); + return _PYTOK_READ_ERROR; + } + if (PyBytes_GET_SIZE(raw) == 0) { + Py_DECREF(raw); + if (finalize_decoding(tok) < 0) { + return _PYTOK_READ_ERROR; + } + continue; + } + input.owner = raw; + input.data = PyBytes_AS_STRING(raw); + input.len = PyBytes_GET_SIZE(raw); + input.ownership = _PYTOK_CHUNK_PYOBJECT; + if (_PyTok_DecodeChunk(tok, &input, 0) < 0) { + _PyTok_ChunkClear(&input); + return _PYTOK_READ_ERROR; + } + } + else { + if (!PyUnicode_Check(raw)) { + PyErr_SetString(PyExc_TypeError, + "readline() returned a non-string object"); + Py_DECREF(raw); + return _PYTOK_READ_ERROR; + } + Py_ssize_t utf8_len; + const char *utf8 = PyUnicode_AsUTF8AndSize(raw, &utf8_len); + if (utf8 == NULL) { + Py_DECREF(raw); + return _PYTOK_READ_ERROR; + } + input.owner = raw; + input.data = (char *)utf8; + input.len = utf8_len; + input.ownership = _PYTOK_CHUNK_PYOBJECT; + if (input.len == 0) { + _PyTok_ChunkClear(&input); + if (finalize_decoding(tok) < 0) { + return _PYTOK_READ_ERROR; + } + continue; + } + } + + if (reader->decoded_pos == reader->decoded_len && + chunk_is_line(&input)) { + *chunk = input; + return _PYTOK_READ_LINE; + } + + if (append_decoded(reader, input.data, input.len) < 0) { + _PyTok_ChunkClear(&input); + tok->done = E_NOMEM; + return _PYTOK_READ_ERROR; + } + _PyTok_ChunkClear(&input); + if (reader->decoded_pos < reader->decoded_len && + reader->decoded[reader->decoded_len - 1] != '\n') { + int pending = _PyTok_DecoderHasBufferedInput(tok); + if (pending < 0) { + return _PYTOK_READ_ERROR; + } + if (!pending && append_implicit_newline(reader) < 0) { + tok->done = E_NOMEM; + return _PYTOK_READ_ERROR; + } + } + if (pop_decoded_line(reader, chunk)) { + return _PYTOK_READ_LINE; + } + } +} + +static _PyTok_ReadResult +next_interactive(struct tok_state *tok, _PyTok_Chunk *chunk) +{ + _PyTok_Reader *reader = tok->reader; + if (tok->interactive_underflow == IUNDERFLOW_STOP) { + return _PYTOK_READ_STOPPED; + } + char *input = PyOS_Readline( + tok->fp != NULL ? tok->fp : stdin, stdout, tok->prompt); + if (reader->nextprompt != NULL) { + tok->prompt = reader->nextprompt; + } + if (input == NULL) { + return _PYTOK_READ_INTERRUPT; + } + Py_ssize_t len = strlen(input); + if (len == 0) { + PyMem_Free(input); + return _PYTOK_READ_EOF; + } + _PyTok_Chunk decoded = { + .data = input, + .len = len, + .ownership = _PYTOK_CHUNK_PYMEM, + }; + if (tok->encoding != NULL && + _PyTok_DecodeOnce(tok, &decoded, tok->encoding) < 0) { + _PyTok_ChunkClear(&decoded); + return _PYTOK_READ_ERROR; + } + chunk->data = _PyTok_NormalizeNewlines( + decoded.data, decoded.len, 0, 1, + &chunk->len, &chunk->implicit_newline); + _PyTok_ChunkClear(&decoded); + if (chunk->data == NULL) { + PyErr_NoMemory(); + tok->done = E_NOMEM; + return _PYTOK_READ_ERROR; + } + chunk->ownership = _PYTOK_CHUNK_PYMEM; + return _PYTOK_READ_LINE; +} + +static _PyTok_ReadResult +reader_next(struct tok_state *tok, _PyTok_Chunk *chunk) +{ + *chunk = (_PyTok_Chunk){0}; + switch (tok->reader->kind) { + case _PYTOK_READER_PREPARED: + return next_prepared(tok, chunk); + case _PYTOK_READER_FILE: + return next_file(tok, chunk); + case _PYTOK_READER_READLINE: + return next_readline(tok, chunk); + case _PYTOK_READER_INTERACTIVE: + return next_interactive(tok, chunk); + } + Py_UNREACHABLE(); +} + +int +_PyTok_ReaderUnderflow(struct tok_state *tok) +{ + int prepared = tok->reader->kind == _PYTOK_READER_PREPARED; + int reset_buffer = !prepared && tok->start == NULL && !INSIDE_FSTRING(tok); + + if (reset_buffer && tok->reader->kind != _PYTOK_READER_INTERACTIVE) { + tok->cur = tok->inp = tok->buf; + } + + _PyTok_Chunk chunk; + _PyTok_ReadResult result = reader_next(tok, &chunk); + if (result != _PYTOK_READ_LINE) { + if (result == _PYTOK_READ_EOF) { + tok->done = E_EOF; + } + else if (result == _PYTOK_READ_STOPPED) { + tok->done = E_INTERACT_STOP; + } + else if (result == _PYTOK_READ_INTERRUPT) { + tok->done = E_INTR; + } + else { + tok->input_error = 1; + if (tok->done == E_OK) { + tok->done = PyErr_ExceptionMatches(PyExc_MemoryError) + ? E_NOMEM : E_ERROR; + } + } + if (tok->reader->kind == _PYTOK_READER_INTERACTIVE && + result != _PYTOK_READ_STOPPED) { + PySys_WriteStderr("\n"); + } + return 0; + } + + Py_ssize_t copy_len = chunk.len; + if (tok->reader->kind == _PYTOK_READER_INTERACTIVE && + chunk.implicit_newline) { + copy_len--; + } + if (reset_buffer && tok->reader->kind == _PYTOK_READER_INTERACTIVE) { + tok->cur = tok->inp = tok->buf; + } + if (!prepared && !_PyLexer_tok_reserve_buf(tok, copy_len + 1)) { + _PyTok_ChunkClear(&chunk); + tok->input_error = 1; + return 0; + } + if (!prepared && + _PyTok_SourceAppendLine(&tok->source, chunk.data, chunk.len, + chunk.implicit_newline) < 0) { + _PyTok_ChunkClear(&chunk); + tok->done = PyErr_ExceptionMatches(PyExc_MemoryError) + ? E_NOMEM : E_ERROR; + tok->input_error = 1; + return 0; + } + if (tok->fp_interactive) { + tok->interactive_src_start = tok->source.bytes; + tok->interactive_src_end = tok->source.bytes + tok->source.len; + } + if (prepared) { + if (tok->start == NULL) { + tok->buf = tok->cur; + } + tok->inp = chunk.data + chunk.len; + } + else { + memcpy(tok->inp, chunk.data, (size_t)copy_len); + tok->inp += copy_len; + *tok->inp = '\0'; + } + tok->implicit_newline = chunk.implicit_newline; + + if (!prepared && tok->tok_mode_stack_index && + !_PyLexer_update_ftstring_expr(tok, 0)) { + _PyTok_ChunkClear(&chunk); + tok->input_error = 1; + return 0; + } + ADVANCE_LINENO(); + if (tok->reader->kind == _PYTOK_READER_FILE && + (tok->encoding == NULL || strcmp(tok->encoding, "utf-8") == 0) && + !_PyTokenizer_ensure_utf8(tok->cur, tok, tok->lineno)) { + _PyTok_ChunkClear(&chunk); + tok->input_error = 1; + return 0; + } + _PyTok_ChunkClear(&chunk); + return 1; +} + +static struct tok_state * +tokenizer_new_with_reader(_PyTok_ReaderKind kind) +{ + struct tok_state *tok = _PyTokenizer_tok_new(); + if (tok == NULL) { + return NULL; + } + tok->reader = PyMem_Calloc(1, sizeof(*tok->reader)); + if (tok->reader == NULL) { + PyErr_NoMemory(); + _PyTokenizer_Free(tok); + return NULL; + } + tok->reader->kind = kind; + if (kind == _PYTOK_READER_PREPARED) { + return tok; + } + tok->buf = PyMem_Malloc(BUFSIZ); + if (tok->buf == NULL) { + PyErr_NoMemory(); + _PyTokenizer_Free(tok); + return NULL; + } + tok->cur = tok->inp = tok->buf; + tok->end = tok->buf + BUFSIZ; + return tok; +} + +static struct tok_state * +tokenizer_from_string(const char *input, int utf8_only, int exec_input, + int preserve_crlf) +{ + struct tok_state *tok = tokenizer_new_with_reader(_PYTOK_READER_PREPARED); + if (tok == NULL) { + return NULL; + } + if (_PyTok_PrepareString( + tok, input, utf8_only, exec_input, preserve_crlf) < 0) { + _PyTokenizer_Free(tok); + return NULL; + } + tok->buf = tok->cur = tok->inp = tok->str; + tok->end = tok->buf; + return tok; +} + +struct tok_state * +_PyTokenizer_FromString(const char *input, int exec_input, int preserve_crlf) +{ + return tokenizer_from_string(input, 0, exec_input, preserve_crlf); +} + +struct tok_state * +_PyTokenizer_FromUTF8(const char *input, int exec_input, int preserve_crlf) +{ + return tokenizer_from_string(input, 1, exec_input, preserve_crlf); +} + +struct tok_state * +_PyTokenizer_FromReadline(PyObject *readline, const char *encoding) +{ + struct tok_state *tok = tokenizer_new_with_reader(_PYTOK_READER_READLINE); + if (tok == NULL) { + return NULL; + } + if (encoding != NULL && _PyTok_SetEncoding(tok, encoding) < 0) { + _PyTokenizer_Free(tok); + return NULL; + } + tok->reader->readline = Py_NewRef(readline); + return tok; +} + +struct tok_state * +_PyTokenizer_FromFile(FILE *fp, const char *encoding, + const char *ps1, const char *ps2) +{ + _PyTok_ReaderKind kind = ps1 != NULL || ps2 != NULL + ? _PYTOK_READER_INTERACTIVE : _PYTOK_READER_FILE; + struct tok_state *tok = tokenizer_new_with_reader(kind); + if (tok == NULL) { + return NULL; + } + if (encoding != NULL && _PyTok_SetEncoding(tok, encoding) < 0) { + _PyTokenizer_Free(tok); + return NULL; + } + tok->fp = fp; + tok->prompt = ps1; + tok->reader->nextprompt = ps2; + return tok; +} + +#if defined(__wasi__) || (defined(__EMSCRIPTEN__) && (__EMSCRIPTEN_major__ >= 3)) +/* WASI has no dup(), and Emscripten's emulation is slow. */ +typedef union { + void *cookie; + int fd; +} borrowed_fd; + +static ssize_t +borrow_read(void *cookie, char *buffer, size_t size) +{ + borrowed_fd borrowed = {.cookie = cookie}; + return read(borrowed.fd, buffer, size); +} + +static FILE * +fdopen_borrow(int fd) +{ + cookie_io_functions_t callbacks = {borrow_read, NULL, NULL, NULL}; + borrowed_fd borrowed = {.fd = fd}; + return fopencookie(borrowed.cookie, "r", callbacks); +} +#else +static FILE * +fdopen_borrow(int fd) +{ + int copy = _Py_dup(fd); + return copy < 0 ? NULL : fdopen(copy, "r"); +} +#endif + +char * +_PyTokenizer_FindEncodingFilename(int fd, PyObject *filename) +{ + FILE *fp = fdopen_borrow(fd); + if (fp == NULL) { + return NULL; + } + struct tok_state *tok = _PyTokenizer_FromFile(fp, NULL, NULL, NULL); + if (tok == NULL) { + fclose(fp); + return NULL; + } + tok->filename = filename != NULL + ? Py_NewRef(filename) : PyUnicode_FromString(""); + if (tok->filename == NULL) { + fclose(fp); + _PyTokenizer_Free(tok); + return NULL; + } + /* Reporting a warning here could recursively ask for the encoding. */ + tok->report_warnings = 0; + while (tok->lineno < 2 && tok->done == E_OK) { + struct token token; + _PyToken_Init(&token); + _PyTokenizer_Get(tok, &token); + _PyToken_Free(&token); + } + fclose(fp); + char *encoding = tok->encoding == NULL + ? NULL : _PyTok_CopyBytes(tok->encoding, strlen(tok->encoding)); + _PyTokenizer_Free(tok); + return encoding; +} diff --git a/Parser/tokenizer/reader.h b/Parser/tokenizer/reader.h new file mode 100644 index 000000000000000..c27bc2aa3fb8197 --- /dev/null +++ b/Parser/tokenizer/reader.h @@ -0,0 +1,9 @@ +#ifndef Py_TOKENIZER_READER_H +#define Py_TOKENIZER_READER_H + +struct tok_state; + +void _PyTok_ReaderFree(struct tok_state *); +int _PyTok_ReaderUnderflow(struct tok_state *); + +#endif diff --git a/Parser/tokenizer/reader_internal.h b/Parser/tokenizer/reader_internal.h new file mode 100644 index 000000000000000..544d1d679f6b608 --- /dev/null +++ b/Parser/tokenizer/reader_internal.h @@ -0,0 +1,80 @@ +#ifndef Py_TOKENIZER_READER_INTERNAL_H +#define Py_TOKENIZER_READER_INTERNAL_H + +#include "Python.h" + +typedef enum { + _PYTOK_READER_PREPARED, + _PYTOK_READER_FILE, + _PYTOK_READER_READLINE, + _PYTOK_READER_INTERACTIVE, +} _PyTok_ReaderKind; + +typedef enum { + _PYTOK_READ_LINE, + _PYTOK_READ_EOF, + _PYTOK_READ_STOPPED, + _PYTOK_READ_INTERRUPT, + _PYTOK_READ_ERROR, +} _PyTok_ReadResult; + +typedef enum { + _PYTOK_ENCODING_ERROR = -1, + _PYTOK_ENCODING_DONE, + _PYTOK_ENCODING_NEED_SECOND_LINE, +} _PyTok_EncodingResult; + +typedef enum { + _PYTOK_CHUNK_BORROWED, + _PYTOK_CHUNK_PYMEM, + _PYTOK_CHUNK_PYOBJECT, +} _PyTok_ChunkOwnership; + +typedef struct { + char *data; + Py_ssize_t len; + int implicit_newline; + PyObject *owner; + _PyTok_ChunkOwnership ownership; +} _PyTok_Chunk; + +typedef struct _PyTok_Reader { + _PyTok_ReaderKind kind; + PyObject *readline; + PyObject *decoder; + const char *nextprompt; + + char *file_buffer; + Py_ssize_t file_buffer_cap; + _PyTok_Chunk prefetched_lines[2]; + int prefetched_index; + int prefetched_count; + + char *decoded; + Py_ssize_t decoded_pos; + Py_ssize_t decoded_len; + Py_ssize_t decoded_cap; + int decoded_tail_is_implicit; + + int file_initialized; + int file_eof; + int decoder_finalized; +} _PyTok_Reader; + +struct tok_state; + +char *_PyTok_CopyBytes(const char *, Py_ssize_t); +int _PyTok_DecodeOnce(struct tok_state *, _PyTok_Chunk *, const char *); +char *_PyTok_NormalizeNewlines( + const char *, Py_ssize_t, int, int, Py_ssize_t *, int *); +void _PyTok_ChunkClear(_PyTok_Chunk *); +int _PyTok_SetEncoding(struct tok_state *, const char *); +_PyTok_EncodingResult _PyTok_DetectEncoding( + struct tok_state *, const _PyTok_Chunk *, const _PyTok_Chunk *, int, + Py_ssize_t *); +int _PyTok_PrepareString(struct tok_state *, const char *, int, int, int); +int _PyTok_StartDecoder(struct tok_state *, const char *); +int _PyTok_DecodeChunk(struct tok_state *, _PyTok_Chunk *, int); +int _PyTok_DecoderHasBufferedInput(struct tok_state *); + +#endif diff --git a/Parser/tokenizer/readline_tokenizer.c b/Parser/tokenizer/readline_tokenizer.c deleted file mode 100644 index 917f7b40cfbbfed..000000000000000 --- a/Parser/tokenizer/readline_tokenizer.c +++ /dev/null @@ -1,135 +0,0 @@ -#include "Python.h" -#include "errcode.h" - -#include "helpers.h" -#include "../lexer/lexer.h" -#include "../lexer/state.h" -#include "../lexer/buffer.h" - -static int -tok_readline_string(struct tok_state* tok) { - PyObject* line = NULL; - PyObject* raw_line = PyObject_CallNoArgs(tok->readline); - if (raw_line == NULL) { - if (PyErr_ExceptionMatches(PyExc_StopIteration)) { - PyErr_Clear(); - return 1; - } - _PyTokenizer_error_ret(tok); - goto error; - } - if(tok->encoding != NULL) { - if (!PyBytes_Check(raw_line)) { - PyErr_Format(PyExc_TypeError, "readline() returned a non-bytes object"); - _PyTokenizer_error_ret(tok); - goto error; - } - line = PyUnicode_Decode(PyBytes_AS_STRING(raw_line), PyBytes_GET_SIZE(raw_line), - tok->encoding, "replace"); - Py_CLEAR(raw_line); - if (line == NULL) { - _PyTokenizer_error_ret(tok); - goto error; - } - } else { - if(!PyUnicode_Check(raw_line)) { - PyErr_Format(PyExc_TypeError, "readline() returned a non-string object"); - _PyTokenizer_error_ret(tok); - goto error; - } - line = raw_line; - raw_line = NULL; - } - Py_ssize_t buflen; - const char* buf = PyUnicode_AsUTF8AndSize(line, &buflen); - if (buf == NULL) { - _PyTokenizer_error_ret(tok); - goto error; - } - - // Make room for the null terminator *and* potentially - // an extra newline character that we may need to artificially - // add. - size_t buffer_size = buflen + 2; - if (!_PyLexer_tok_reserve_buf(tok, buffer_size)) { - goto error; - } - memcpy(tok->inp, buf, buflen); - tok->inp += buflen; - *tok->inp = '\0'; - - tok->line_start = tok->cur; - Py_DECREF(line); - return 1; -error: - Py_XDECREF(raw_line); - Py_XDECREF(line); - return 0; -} - -static int -tok_underflow_readline(struct tok_state* tok) { - assert(tok->decoding_state == STATE_NORMAL); - assert(tok->fp == NULL && tok->input == NULL && tok->decoding_readline == NULL); - if (tok->start == NULL && !INSIDE_FSTRING(tok)) { - tok->cur = tok->inp = tok->buf; - } - if (!tok_readline_string(tok)) { - return 0; - } - if (tok->inp == tok->cur) { - tok->done = E_EOF; - return 0; - } - tok->implicit_newline = 0; - if (tok->inp[-1] != '\n') { - assert(tok->inp + 1 < tok->end); - /* Last line does not end in \n, fake one */ - *tok->inp++ = '\n'; - *tok->inp = '\0'; - tok->implicit_newline = 1; - } - - if (tok->tok_mode_stack_index && !_PyLexer_update_ftstring_expr(tok, 0)) { - return 0; - } - - ADVANCE_LINENO(); - /* The default encoding is UTF-8, so make sure we don't have any - non-UTF-8 sequences in it. */ - if (!tok->encoding && !_PyTokenizer_ensure_utf8(tok->cur, tok, tok->lineno)) { - _PyTokenizer_error_ret(tok); - return 0; - } - assert(tok->done == E_OK); - return tok->done == E_OK; -} - -struct tok_state * -_PyTokenizer_FromReadline(PyObject* readline, const char* enc, - int exec_input, int preserve_crlf) -{ - struct tok_state *tok = _PyTokenizer_tok_new(); - if (tok == NULL) - return NULL; - if ((tok->buf = (char *)PyMem_Malloc(BUFSIZ)) == NULL) { - _PyTokenizer_Free(tok); - PyErr_NoMemory(); - return NULL; - } - tok->cur = tok->inp = tok->buf; - tok->end = tok->buf + BUFSIZ; - tok->fp = NULL; - if (enc != NULL) { - tok->encoding = _PyTokenizer_new_string(enc, strlen(enc), tok); - if (!tok->encoding) { - _PyTokenizer_Free(tok); - return NULL; - } - } - tok->decoding_state = STATE_NORMAL; - tok->underflow = &tok_underflow_readline; - Py_INCREF(readline); - tok->readline = readline; - return tok; -} diff --git a/Parser/tokenizer/string_tokenizer.c b/Parser/tokenizer/string_tokenizer.c deleted file mode 100644 index 7f07cca37ee0191..000000000000000 --- a/Parser/tokenizer/string_tokenizer.c +++ /dev/null @@ -1,148 +0,0 @@ -#include "Python.h" -#include "errcode.h" - -#include "helpers.h" -#include "../lexer/state.h" - -static int -tok_underflow_string(struct tok_state *tok) { - char *end = strchr(tok->inp, '\n'); - if (end != NULL) { - end++; - } - else { - end = strchr(tok->inp, '\0'); - if (end == tok->inp) { - tok->done = E_EOF; - return 0; - } - } - if (tok->start == NULL) { - tok->buf = tok->cur; - } - tok->line_start = tok->cur; - ADVANCE_LINENO(); - tok->inp = end; - return 1; -} - -/* Fetch a byte from TOK, using the string buffer. */ -static int -buf_getc(struct tok_state *tok) { - return Py_CHARMASK(*tok->str++); -} - -/* Unfetch a byte from TOK, using the string buffer. */ -static void -buf_ungetc(int c, struct tok_state *tok) { - tok->str--; - assert(Py_CHARMASK(*tok->str) == c); /* tok->cur may point to read-only segment */ -} - -/* Set the readline function for TOK to ENC. For the string-based - tokenizer, this means to just record the encoding. */ -static int -buf_setreadl(struct tok_state *tok, const char* enc) { - tok->enc = enc; - return 1; -} - -/* Decode a byte string STR for use as the buffer of TOK. - Look for encoding declarations inside STR, and record them - inside TOK. */ -static char * -decode_str(const char *input, int single, struct tok_state *tok, int preserve_crlf) -{ - PyObject* utf8 = NULL; - char *str; - const char *s; - const char *newl[2] = {NULL, NULL}; - int lineno = 0; - tok->input = str = _PyTokenizer_translate_newlines(input, single, preserve_crlf, tok); - if (str == NULL) - return NULL; - tok->enc = NULL; - tok->str = str; - if (!_PyTokenizer_check_bom(buf_getc, buf_ungetc, buf_setreadl, tok)) - return _PyTokenizer_error_ret(tok); - str = tok->str; /* string after BOM if any */ - assert(str); - if (tok->enc != NULL) { - utf8 = _PyTokenizer_translate_into_utf8(str, tok->enc); - if (utf8 == NULL) - return _PyTokenizer_error_ret(tok); - str = PyBytes_AsString(utf8); - } - for (s = str;; s++) { - if (*s == '\0') break; - else if (*s == '\n') { - assert(lineno < 2); - newl[lineno] = s; - lineno++; - if (lineno == 2) break; - } - } - tok->enc = NULL; - /* need to check line 1 and 2 separately since check_coding_spec - assumes a single line as input */ - if (newl[0]) { - tok->lineno = 1; - if (!_PyTokenizer_check_coding_spec(str, newl[0] - str, tok, buf_setreadl)) { - return NULL; - } - if (tok->enc == NULL && tok->decoding_state != STATE_NORMAL && newl[1]) { - tok->lineno = 2; - if (!_PyTokenizer_check_coding_spec(newl[0]+1, newl[1] - newl[0], - tok, buf_setreadl)) - return NULL; - } - } - tok->lineno = 0; - if (tok->enc != NULL) { - assert(utf8 == NULL); - utf8 = _PyTokenizer_translate_into_utf8(str, tok->enc); - if (utf8 == NULL) - return _PyTokenizer_error_ret(tok); - str = PyBytes_AS_STRING(utf8); - } - else if (!_PyTokenizer_ensure_utf8(str, tok, 1)) { - return _PyTokenizer_error_ret(tok); - } - if (utf8 != NULL) { - char *translated = _PyTokenizer_translate_newlines( - str, single, preserve_crlf, tok); - if (translated == NULL) { - Py_DECREF(utf8); - return _PyTokenizer_error_ret(tok); - } - PyMem_Free(tok->input); - tok->input = translated; - str = translated; - Py_CLEAR(utf8); - } - tok->str = str; - assert(tok->decoding_buffer == NULL); - tok->decoding_buffer = utf8; /* CAUTION */ - return str; -} - -/* Set up tokenizer for string */ -struct tok_state * -_PyTokenizer_FromString(const char *str, int exec_input, int preserve_crlf) -{ - struct tok_state *tok = _PyTokenizer_tok_new(); - char *decoded; - - if (tok == NULL) - return NULL; - decoded = decode_str(str, exec_input, tok, preserve_crlf); - if (decoded == NULL) { - _PyTokenizer_Free(tok); - return NULL; - } - - tok->buf = tok->cur = tok->inp = decoded; - tok->end = decoded; - tok->underflow = &tok_underflow_string; - return tok; -} diff --git a/Parser/tokenizer/tokenizer.h b/Parser/tokenizer/tokenizer.h index 8fbeb2d6ae6df1d..d8c889115cfd73a 100644 --- a/Parser/tokenizer/tokenizer.h +++ b/Parser/tokenizer/tokenizer.h @@ -5,9 +5,12 @@ struct tok_state *_PyTokenizer_FromString(const char *, int, int); struct tok_state *_PyTokenizer_FromUTF8(const char *, int, int); -struct tok_state *_PyTokenizer_FromReadline(PyObject*, const char*, int, int); +struct tok_state *_PyTokenizer_FromReadline(PyObject *, const char *); struct tok_state *_PyTokenizer_FromFile(FILE *, const char*, const char *, const char *); +/* Return the declared encoding in PyMem-allocated storage, or NULL. + An exception is set on error. */ +char *_PyTokenizer_FindEncodingFilename(int, PyObject *); #define tok_dump _Py_tok_dump diff --git a/Parser/tokenizer/utf8_tokenizer.c b/Parser/tokenizer/utf8_tokenizer.c deleted file mode 100644 index 1a925f445400faa..000000000000000 --- a/Parser/tokenizer/utf8_tokenizer.c +++ /dev/null @@ -1,55 +0,0 @@ -#include "Python.h" -#include "errcode.h" - -#include "helpers.h" -#include "../lexer/state.h" - -static int -tok_underflow_string(struct tok_state *tok) { - char *end = strchr(tok->inp, '\n'); - if (end != NULL) { - end++; - } - else { - end = strchr(tok->inp, '\0'); - if (end == tok->inp) { - tok->done = E_EOF; - return 0; - } - } - if (tok->start == NULL) { - tok->buf = tok->cur; - } - tok->line_start = tok->cur; - ADVANCE_LINENO(); - tok->inp = end; - return 1; -} - -/* Set up tokenizer for UTF-8 string */ -struct tok_state * -_PyTokenizer_FromUTF8(const char *str, int exec_input, int preserve_crlf) -{ - struct tok_state *tok = _PyTokenizer_tok_new(); - char *translated; - if (tok == NULL) - return NULL; - tok->input = translated = _PyTokenizer_translate_newlines(str, exec_input, preserve_crlf, tok); - if (translated == NULL) { - _PyTokenizer_Free(tok); - return NULL; - } - tok->decoding_state = STATE_NORMAL; - tok->enc = NULL; - tok->str = translated; - tok->encoding = _PyTokenizer_new_string("utf-8", 5, tok); - if (!tok->encoding) { - _PyTokenizer_Free(tok); - return NULL; - } - - tok->buf = tok->cur = tok->inp = translated; - tok->end = translated; - tok->underflow = &tok_underflow_string; - return tok; -} diff --git a/Python/Python-tokenize.c b/Python/Python-tokenize.c index e6d39e4c7dc8235..762b7b3e4c8d71d 100644 --- a/Python/Python-tokenize.c +++ b/Python/Python-tokenize.c @@ -66,7 +66,7 @@ tokenizeriter_new_impl(PyTypeObject *type, PyObject *readline, if (filename == NULL) { return NULL; } - self->tok = _PyTokenizer_FromReadline(readline, encoding, 1, 1); + self->tok = _PyTokenizer_FromReadline(readline, encoding); if (self->tok == NULL) { Py_DECREF(filename); return NULL; diff --git a/Python/errors.c b/Python/errors.c index 48b03e5fd714b18..c75cfec30b99a5a 100644 --- a/Python/errors.c +++ b/Python/errors.c @@ -13,6 +13,8 @@ #include "pycore_traceback.h" // _PyTraceBack_FromFrame() #include "pycore_unicodeobject.h" // _PyUnicode_Equal() +#include "Parser/tokenizer/tokenizer.h" + #ifdef MS_WINDOWS # include # include @@ -2050,9 +2052,6 @@ PyErr_ProgramText(const char *filename, int lineno) return res; } -/* Function from Parser/tokenizer/file_tokenizer.c */ -extern char* _PyTokenizer_FindEncodingFilename(int, PyObject *); - PyObject * _PyErr_ProgramDecodedTextObject(PyObject *filename, int lineno, const char* encoding) { diff --git a/Python/traceback.c b/Python/traceback.c index fe6a465bc64cc94..87328b411928b7f 100644 --- a/Python/traceback.c +++ b/Python/traceback.c @@ -12,6 +12,7 @@ #include "pycore_traceback.h" // EXCEPTION_TB_HEADER #include "frameobject.h" // PyFrame_New() +#include "Parser/tokenizer/tokenizer.h" #include "osdefs.h" // SEP #ifdef HAVE_UNISTD_H @@ -57,9 +58,6 @@ #define MAX_FRAME_DEPTH 100 #define DEFAULT_MAX_NTHREADS 100 -/* Function from Parser/tokenizer/file_tokenizer.c */ -extern char* _PyTokenizer_FindEncodingFilename(int, PyObject *); - /*[clinic input] class traceback "PyTracebackObject *" "&PyTraceback_Type" [clinic start generated code]*/ diff --git a/Tools/c-analyzer/TODO b/Tools/c-analyzer/TODO index 2077534ccf41286..ec07ea6aa9ff7dc 100644 --- a/Tools/c-analyzer/TODO +++ b/Tools/c-analyzer/TODO @@ -427,8 +427,6 @@ Objects/typeobject.c:type_new():PyId___slots__ _Py_IDENTIFIER( Objects/unicodeobject.c:unicodeiter_reduce():PyId_iter _Py_IDENTIFIER(iter) Objects/weakrefobject.c:proxy_bytes():PyId___bytes__ _Py_IDENTIFIER(__bytes__) Objects/weakrefobject.c:weakref_repr():PyId___name__ _Py_IDENTIFIER(__name__) -Parser/tokenizer/file_tokenizer.c:fp_setreadl():PyId_open _Py_IDENTIFIER(open) -Parser/tokenizer/file_tokenizer.c:fp_setreadl():PyId_readline _Py_IDENTIFIER(readline) Python/Python-ast.c:ast_type_reduce():PyId___dict__ _Py_IDENTIFIER(__dict__) Python/Python-ast.c:make_type():PyId___module__ _Py_IDENTIFIER(__module__) Python/_warnings.c:PyId_stderr _Py_IDENTIFIER(stderr) diff --git a/Tools/peg_generator/pegen/build.py b/Tools/peg_generator/pegen/build.py index 37883afeffcb472..bfd8e43c6912e86 100644 --- a/Tools/peg_generator/pegen/build.py +++ b/Tools/peg_generator/pegen/build.py @@ -129,10 +129,8 @@ def compile_c_extension( str(MOD_DIR.parent.parent.parent / "Parser" / "lexer" / "state.c"), str(MOD_DIR.parent.parent.parent / "Parser" / "lexer" / "string.c"), str(MOD_DIR.parent.parent.parent / "Parser" / "lexer" / "buffer.c"), - str(MOD_DIR.parent.parent.parent / "Parser" / "tokenizer" / "string_tokenizer.c"), - str(MOD_DIR.parent.parent.parent / "Parser" / "tokenizer" / "file_tokenizer.c"), - str(MOD_DIR.parent.parent.parent / "Parser" / "tokenizer" / "utf8_tokenizer.c"), - str(MOD_DIR.parent.parent.parent / "Parser" / "tokenizer" / "readline_tokenizer.c"), + str(MOD_DIR.parent.parent.parent / "Parser" / "tokenizer" / "decoder.c"), + str(MOD_DIR.parent.parent.parent / "Parser" / "tokenizer" / "reader.c"), str(MOD_DIR.parent.parent.parent / "Parser" / "tokenizer" / "helpers.c"), str(MOD_DIR.parent.parent.parent / "Parser" / "pegen.c"), str(MOD_DIR.parent.parent.parent / "Parser" / "pegen_errors.c"), From adf62b2607fc34a0cb8eb9edd8d9638c98639ea3 Mon Sep 17 00:00:00 2001 From: Pablo Galindo Salgado Date: Thu, 27 Aug 2026 17:25:06 +0100 Subject: [PATCH 2/5] gh-153569: preserve unterminated interactive input --- Parser/tokenizer/reader.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Parser/tokenizer/reader.c b/Parser/tokenizer/reader.c index 15a19f50f9a103b..121cba501f9fecf 100644 --- a/Parser/tokenizer/reader.c +++ b/Parser/tokenizer/reader.c @@ -487,7 +487,7 @@ next_interactive(struct tok_state *tok, _PyTok_Chunk *chunk) return _PYTOK_READ_ERROR; } chunk->data = _PyTok_NormalizeNewlines( - decoded.data, decoded.len, 0, 1, + decoded.data, decoded.len, 0, 0, &chunk->len, &chunk->implicit_newline); _PyTok_ChunkClear(&decoded); if (chunk->data == NULL) { From 19d3ba4f3e9ff3d7487fb452a290a37ebe492c46 Mon Sep 17 00:00:00 2001 From: Pablo Galindo Salgado Date: Thu, 27 Aug 2026 18:31:43 +0100 Subject: [PATCH 3/5] gh-153569: fix tokenizer includes in external builds --- Python/errors.c | 2 +- Python/traceback.c | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/Python/errors.c b/Python/errors.c index c75cfec30b99a5a..eb148998fc4652d 100644 --- a/Python/errors.c +++ b/Python/errors.c @@ -13,7 +13,7 @@ #include "pycore_traceback.h" // _PyTraceBack_FromFrame() #include "pycore_unicodeobject.h" // _PyUnicode_Equal() -#include "Parser/tokenizer/tokenizer.h" +#include "../Parser/tokenizer/tokenizer.h" #ifdef MS_WINDOWS # include diff --git a/Python/traceback.c b/Python/traceback.c index 87328b411928b7f..c8b3dbaeb452376 100644 --- a/Python/traceback.c +++ b/Python/traceback.c @@ -12,7 +12,7 @@ #include "pycore_traceback.h" // EXCEPTION_TB_HEADER #include "frameobject.h" // PyFrame_New() -#include "Parser/tokenizer/tokenizer.h" +#include "../Parser/tokenizer/tokenizer.h" #include "osdefs.h" // SEP #ifdef HAVE_UNISTD_H From dea35e4109b4d70723bd8916507e372695953090 Mon Sep 17 00:00:00 2001 From: Pablo Galindo Salgado Date: Thu, 27 Aug 2026 19:39:57 +0100 Subject: [PATCH 4/5] gh-153569: avoid retaining duplicate reader input --- Parser/tokenizer/reader.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Parser/tokenizer/reader.c b/Parser/tokenizer/reader.c index 121cba501f9fecf..2663e5a9192da0e 100644 --- a/Parser/tokenizer/reader.c +++ b/Parser/tokenizer/reader.c @@ -565,7 +565,7 @@ _PyTok_ReaderUnderflow(struct tok_state *tok) tok->input_error = 1; return 0; } - if (!prepared && + if (tok->reader->kind == _PYTOK_READER_INTERACTIVE && _PyTok_SourceAppendLine(&tok->source, chunk.data, chunk.len, chunk.implicit_newline) < 0) { _PyTok_ChunkClear(&chunk); From 12a9a357440bba9d3f6247f9dcec90fd842e25b2 Mon Sep 17 00:00:00 2001 From: Pablo Galindo Salgado Date: Thu, 27 Aug 2026 20:33:50 +0100 Subject: [PATCH 5/5] gh-153569: speed up UTF-8 readline decoding --- Parser/tokenizer/decoder.c | 7 ++++--- Parser/tokenizer/reader.c | 25 +++++++++++++++++++------ Parser/tokenizer/reader_internal.h | 3 ++- 3 files changed, 25 insertions(+), 10 deletions(-) diff --git a/Parser/tokenizer/decoder.c b/Parser/tokenizer/decoder.c index 6cddd7bac414066..af17b8b63235f51 100644 --- a/Parser/tokenizer/decoder.c +++ b/Parser/tokenizer/decoder.c @@ -267,10 +267,10 @@ _PyTok_DetectEncoding(struct tok_state *tok, const _PyTok_Chunk *first, int _PyTok_DecodeOnce(struct tok_state *tok, _PyTok_Chunk *chunk, - const char *encoding) + const char *encoding, const char *errors) { PyObject *unicode = PyUnicode_Decode( - chunk->data, chunk->len, encoding, NULL); + chunk->data, chunk->len, encoding, errors); if (unicode == NULL) { tok->done = PyErr_ExceptionMatches(PyExc_MemoryError) ? E_NOMEM : E_DECODE; @@ -387,7 +387,8 @@ _PyTok_PrepareString(struct tok_state *tok, const char *input, int utf8_only, .ownership = _PYTOK_CHUNK_BORROWED, }; if (tok->encoding != NULL && strcmp(tok->encoding, "utf-8") != 0) { - if (_PyTok_DecodeOnce(tok, &decoded, tok->encoding) < 0) { + if (_PyTok_DecodeOnce( + tok, &decoded, tok->encoding, NULL) < 0) { return -1; } } diff --git a/Parser/tokenizer/reader.c b/Parser/tokenizer/reader.c index 2663e5a9192da0e..82b824f56374fcf 100644 --- a/Parser/tokenizer/reader.c +++ b/Parser/tokenizer/reader.c @@ -381,12 +381,11 @@ next_readline(struct tok_state *tok, _PyTok_Chunk *chunk) Py_DECREF(raw); return _PYTOK_READ_ERROR; } - if (_PyTok_StartDecoder(tok, "replace") < 0) { - Py_XDECREF(raw); - return _PYTOK_READ_ERROR; - } if (PyBytes_GET_SIZE(raw) == 0) { Py_DECREF(raw); + if (_PyTok_StartDecoder(tok, "replace") < 0) { + return _PYTOK_READ_ERROR; + } if (finalize_decoding(tok) < 0) { return _PYTOK_READ_ERROR; } @@ -396,7 +395,20 @@ next_readline(struct tok_state *tok, _PyTok_Chunk *chunk) input.data = PyBytes_AS_STRING(raw); input.len = PyBytes_GET_SIZE(raw); input.ownership = _PYTOK_CHUNK_PYOBJECT; - if (_PyTok_DecodeChunk(tok, &input, 0) < 0) { + int decoded; + if (reader->decoder == NULL && + strcmp(tok->encoding, "utf-8") == 0 && + chunk_is_line(&input)) { + decoded = _PyTok_DecodeOnce( + tok, &input, "utf-8", "replace"); + } + else { + decoded = _PyTok_StartDecoder(tok, "replace"); + if (decoded == 0) { + decoded = _PyTok_DecodeChunk(tok, &input, 0); + } + } + if (decoded < 0) { _PyTok_ChunkClear(&input); return _PYTOK_READ_ERROR; } @@ -482,7 +494,8 @@ next_interactive(struct tok_state *tok, _PyTok_Chunk *chunk) .ownership = _PYTOK_CHUNK_PYMEM, }; if (tok->encoding != NULL && - _PyTok_DecodeOnce(tok, &decoded, tok->encoding) < 0) { + _PyTok_DecodeOnce( + tok, &decoded, tok->encoding, NULL) < 0) { _PyTok_ChunkClear(&decoded); return _PYTOK_READ_ERROR; } diff --git a/Parser/tokenizer/reader_internal.h b/Parser/tokenizer/reader_internal.h index 544d1d679f6b608..121d0f96f6698a2 100644 --- a/Parser/tokenizer/reader_internal.h +++ b/Parser/tokenizer/reader_internal.h @@ -64,7 +64,8 @@ typedef struct _PyTok_Reader { struct tok_state; char *_PyTok_CopyBytes(const char *, Py_ssize_t); -int _PyTok_DecodeOnce(struct tok_state *, _PyTok_Chunk *, const char *); +int _PyTok_DecodeOnce( + struct tok_state *, _PyTok_Chunk *, const char *, const char *); char *_PyTok_NormalizeNewlines( const char *, Py_ssize_t, int, int, Py_ssize_t *, int *); void _PyTok_ChunkClear(_PyTok_Chunk *);