From a2cd5b755d405064defd301ce9e5ec6ffed175a6 Mon Sep 17 00:00:00 2001 From: Ivan Levkivskyi Date: Tue, 29 Sep 2026 01:48:44 +0100 Subject: [PATCH 1/3] Quick fix for non UTF-8 encodings with new parser --- mypy/build.py | 11 +++++-- mypy/nativeparse.py | 43 ++++++++++++++++++------- mypy/test/test_nativeparse.py | 59 +++++++++++++++++------------------ 3 files changed, 68 insertions(+), 45 deletions(-) diff --git a/mypy/build.py b/mypy/build.py index a55fe4eb92d5b..e68928ba66693 100644 --- a/mypy/build.py +++ b/mypy/build.py @@ -3231,9 +3231,14 @@ def get_source(self) -> str: def parse_file_inner(self, source: str | None, raw_data: FileRawData | None = None) -> None: t0 = time_ref() - self.tree = self.manager.parse_file( - self.id, self.xpath, source, options=self.options, raw_data=raw_data - ) + try: + self.tree = self.manager.parse_file( + self.id, self.xpath, source, options=self.options, raw_data=raw_data + ) + except (UnicodeDecodeError, DecodeError) as decodeerr: + # Convert a decode error to a standard-looking blocker. + err = f"{self.path}: error: Cannot decode file: {str(decodeerr)}" + raise CompileError([err], module_with_blocker=self.id) from decodeerr self.time_spent_us += time_spent_us(t0) def parse_file(self, *, temporary: bool = False, raw_data: FileRawData | None = None) -> None: diff --git a/mypy/nativeparse.py b/mypy/nativeparse.py index d81c2878698a3..e7bbb4a61908b 100644 --- a/mypy/nativeparse.py +++ b/mypy/nativeparse.py @@ -46,6 +46,7 @@ read_str_opt, read_tag, ) +from mypy.errors import CompileError from mypy.nodes import ( ARG_KINDS, ARG_POS, @@ -157,7 +158,7 @@ UnionType, UnpackType, ) -from mypy.util import unnamed_function +from mypy.util import decode_python_encoding, unnamed_function TypeIgnores = list[tuple[int, list[str]]] @@ -297,16 +298,36 @@ def parse_to_binary_ast( source: str | bytes | None = None, skip_function_bodies: bool = False, ) -> tuple[bytes, list[ParseError], TypeIgnores, bytes, bool, bool, str, list[tuple[int, str]]]: - ast_bytes, errors, ignores, import_bytes, ast_data = ast_serialize.parse( - filename, - source, - skip_function_bodies=skip_function_bodies, - python_version=options.python_version, - platform=options.platform, - always_true=options.always_true, - always_false=options.always_false, - cache_version=5, - ) + try: + ast_bytes, errors, ignores, import_bytes, ast_data = ast_serialize.parse( + filename, + source, + skip_function_bodies=skip_function_bodies, + python_version=options.python_version, + platform=options.platform, + always_true=options.always_true, + always_false=options.always_false, + cache_version=5, + ) + except ValueError as exc: + if source is None: + # Try reading/encoding manually, since native parser only supports UTF-8. + # If we still cannot decode, decode error will buble up to caller. + with open(filename, "rb") as f: + source = decode_python_encoding(f.read()) + ast_bytes, errors, ignores, import_bytes, ast_data = ast_serialize.parse( + filename, + source, + skip_function_bodies=skip_function_bodies, + python_version=options.python_version, + platform=options.platform, + always_true=options.always_true, + always_false=options.always_false, + cache_version=5, + ) + else: + # Convert everything unexpected to a standard-looking blocker. + raise CompileError([f"{filename}: error: Cannot parse file: {exc}"]) return ( ast_bytes, errors, diff --git a/mypy/test/test_nativeparse.py b/mypy/test/test_nativeparse.py index d74ff99d5e5e4..efe3be63c97c8 100644 --- a/mypy/test/test_nativeparse.py +++ b/mypy/test/test_nativeparse.py @@ -1,8 +1,4 @@ -"""Tests for the experimental native mypy parser. - -To run these, you will need to manually install ast_serialize from -https://github.com/mypyc/ast_serialize first (see the README for the details). -""" +"""Tests for the native mypy parser.""" from __future__ import annotations @@ -12,6 +8,7 @@ import unittest from collections.abc import Iterator +import pytest from librt.internal import ReadBuffer from mypy import defaults, nodes @@ -27,36 +24,24 @@ ) from mypy.config_parser import parse_mypy_comments from mypy.errors import CompileError +from mypy.nativeparse import ( + State, + deserialize_imports, + native_parse, + parse_to_binary_ast, + read_statements, +) from mypy.nodes import MypyFile, ParseError from mypy.options import Options from mypy.test.data import DataDrivenTestCase, DataSuite from mypy.test.helpers import assert_string_arrays_equal from mypy.util import get_mypy_comments -# If the experimental ast_serialize module isn't installed, the following import will fail -# and we won't run any native parser tests. -try: - from mypy.nativeparse import ( - State, - deserialize_imports, - native_parse, - parse_to_binary_ast, - read_statements, - ) - - has_nativeparse = True -except ImportError: - has_nativeparse = False - class NativeParserSuite(DataSuite): required_out_section = True base_path = "." - files = ( - ["native-parser.test", "native-parser-python311.test", "native-parser-python312.test"] - if has_nativeparse - else [] - ) + files = ["native-parser.test", "native-parser-python311.test", "native-parser-python312.test"] def run_case(self, testcase: DataDrivenTestCase) -> None: test_parser(testcase) @@ -65,7 +50,7 @@ def run_case(self, testcase: DataDrivenTestCase) -> None: class NativeParserImportsSuite(DataSuite): required_out_section = True base_path = "." - files = ["native-parser-imports.test"] if has_nativeparse else [] + files = ["native-parser-imports.test"] def run_case(self, testcase: DataDrivenTestCase) -> None: test_parser_imports(testcase) @@ -238,7 +223,6 @@ def format_reachable_imports(node: MypyFile) -> list[str]: return output -@unittest.skipUnless(has_nativeparse, "nativeparse not available") class TestNativeParserBinaryFormat(unittest.TestCase): def _assert_trivial_binary_data(self, b: bytes, /) -> None: # A quick sanity check to ensure the serialized data looks as expected. Only covers @@ -296,14 +280,27 @@ def test_trivial_binary_data_from_bytes_source(self) -> None: self._assert_trivial_binary_data(b) def test_invalid_bytes_raises(self) -> None: - with self.assertRaises(UnicodeDecodeError): + with self.assertRaises(CompileError): parse_to_binary_ast("", Options(), b"\xff") +class TestNativeParserCustomEncoding(unittest.TestCase): + def test_latin1(self) -> None: + source = "# coding: latin1\nJérôme = False" + with temp_source(source, encoding="latin1") as fnam: + parse_to_binary_ast(fnam, Options()) + + def test_latin1_broken(self) -> None: + source = "# coding: ascii\nJérôme = False" + with temp_source(source, encoding="latin1") as fnam: + with pytest.raises(UnicodeDecodeError): + parse_to_binary_ast(fnam, Options()) + + @contextlib.contextmanager -def temp_source(text: str) -> Iterator[str]: +def temp_source(text: str, encoding: str = "utf-8") -> Iterator[str]: with tempfile.TemporaryDirectory() as temp_dir: temp_path = os.path.join(temp_dir, "t.py") - with open(temp_path, "w") as f: - f.write(text) + with open(temp_path, "wb") as f: + f.write(text.encode(encoding)) yield temp_path From 40ddc5683228bb6f9af81c0cfc5051029f56fa6a Mon Sep 17 00:00:00 2001 From: Ivan Levkivskyi Date: Tue, 29 Sep 2026 01:56:54 +0100 Subject: [PATCH 2/3] Whatever --- mypy/nativeparse.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mypy/nativeparse.py b/mypy/nativeparse.py index e7bbb4a61908b..feb8c04f19e46 100644 --- a/mypy/nativeparse.py +++ b/mypy/nativeparse.py @@ -327,7 +327,7 @@ def parse_to_binary_ast( ) else: # Convert everything unexpected to a standard-looking blocker. - raise CompileError([f"{filename}: error: Cannot parse file: {exc}"]) + raise CompileError([f"{filename}: error: Cannot parse file: {exc}"]) from exc return ( ast_bytes, errors, From 5f17b397e2d225495fef92719643034aa7d70e51 Mon Sep 17 00:00:00 2001 From: Ivan Levkivskyi Date: Tue, 29 Sep 2026 18:23:08 +0100 Subject: [PATCH 3/3] Address CR; unify error handling --- mypy/build.py | 54 +++++++++++++-------------- mypy/nativeparse.py | 10 +++-- mypy/test/data.py | 10 ++++- mypy/test/helpers.py | 10 ++++- test-data/unit/check-incremental.test | 20 +++++++++- test-data/unit/cmdline.test | 6 +-- test-data/unit/fine-grained.test | 4 +- 7 files changed, 75 insertions(+), 39 deletions(-) diff --git a/mypy/build.py b/mypy/build.py index e68928ba66693..09952f37aa280 100644 --- a/mypy/build.py +++ b/mypy/build.py @@ -3191,31 +3191,10 @@ def get_source(self) -> str: with self.wrap_context(): source = self.source if self.path and source is None: - try: + with self.handle_file_read_errors(): path = manager.maybe_swap_for_shadow_path(self.path) source = decode_python_encoding(manager.fscache.read(path)) self.source_hash = manager.fscache.hash_digest(path) - except OSError as ioerr: - # ioerr.strerror differs for os.stat failures between Windows and - # other systems, but os.strerror(ioerr.errno) does not, so we use that. - # (We want the error messages to be platform-independent so that the - # tests have predictable output.) - assert ioerr.errno is not None - raise CompileError( - [ - "mypy: error: Cannot read file '{}': {}".format( - self.path.replace(os.getcwd() + os.sep, ""), - os.strerror(ioerr.errno), - ) - ], - module_with_blocker=self.id, - ) from ioerr - except (UnicodeDecodeError, DecodeError) as decodeerr: - if self.path.endswith(".pyd"): - err = f"{self.path}: error: Stubgen does not support .pyd files" - else: - err = f"{self.path}: error: Cannot decode file: {str(decodeerr)}" - raise CompileError([err], module_with_blocker=self.id) from decodeerr elif self.path and manager.fscache.isdir(self.path): source = "" self.source_hash = "" @@ -3229,16 +3208,37 @@ def get_source(self) -> str: self.time_spent_us += time_spent_us(t0) return source + @contextlib.contextmanager + def handle_file_read_errors(self) -> Iterator[None]: + try: + yield + except OSError as ioerr: + # ioerr.strerror differs for os.stat failures between Windows and + # other systems, but os.strerror(ioerr.errno) does not, so we use that. + # (We want the error messages to be platform-independent so that the + # tests have predictable output.) + assert ioerr.errno is not None + err_path = self.manager.errors.simplify_path(self.xpath) + raise CompileError( + [f"{err_path}: error: Cannot read file: {os.strerror(ioerr.errno)}"], + module_with_blocker=self.id, + ) from ioerr + except (UnicodeDecodeError, DecodeError) as decodeerr: + err_path = self.manager.errors.simplify_path(self.xpath) + if err_path.endswith(".pyd"): + err = f"{err_path}: error: Stubgen does not support .pyd files" + else: + err = f"{err_path}: error: Cannot decode file: {str(decodeerr)}" + raise CompileError([err], module_with_blocker=self.id) from decodeerr + def parse_file_inner(self, source: str | None, raw_data: FileRawData | None = None) -> None: t0 = time_ref() - try: + # Error handling here matches get_source(), since in case of an error in + # the new parser we fall back to reading the file manually ourselves. + with self.handle_file_read_errors(): self.tree = self.manager.parse_file( self.id, self.xpath, source, options=self.options, raw_data=raw_data ) - except (UnicodeDecodeError, DecodeError) as decodeerr: - # Convert a decode error to a standard-looking blocker. - err = f"{self.path}: error: Cannot decode file: {str(decodeerr)}" - raise CompileError([err], module_with_blocker=self.id) from decodeerr self.time_spent_us += time_spent_us(t0) def parse_file(self, *, temporary: bool = False, raw_data: FileRawData | None = None) -> None: diff --git a/mypy/nativeparse.py b/mypy/nativeparse.py index feb8c04f19e46..622b199227bd2 100644 --- a/mypy/nativeparse.py +++ b/mypy/nativeparse.py @@ -158,7 +158,7 @@ UnionType, UnpackType, ) -from mypy.util import decode_python_encoding, unnamed_function +from mypy.util import decode_python_encoding, hash_digest, unnamed_function TypeIgnores = list[tuple[int, list[str]]] @@ -312,9 +312,10 @@ def parse_to_binary_ast( except ValueError as exc: if source is None: # Try reading/encoding manually, since native parser only supports UTF-8. - # If we still cannot decode, decode error will buble up to caller. + # If we still cannot decode, decode error will bubble up to caller. with open(filename, "rb") as f: - source = decode_python_encoding(f.read()) + source_bytes = f.read() + source = decode_python_encoding(source_bytes) ast_bytes, errors, ignores, import_bytes, ast_data = ast_serialize.parse( filename, source, @@ -325,6 +326,9 @@ def parse_to_binary_ast( always_false=options.always_false, cache_version=5, ) + # Compute hash using original non-UTF-8 bytes, so that + # incremental mode works correctly. + ast_data["source_hash"] = hash_digest(source_bytes) else: # Convert everything unexpected to a standard-looking blocker. raise CompileError([f"{filename}: error: Cannot parse file: {exc}"]) from exc diff --git a/mypy/test/data.py b/mypy/test/data.py index 574a343fa3f36..627f0bc861089 100644 --- a/mypy/test/data.py +++ b/mypy/test/data.py @@ -379,7 +379,7 @@ def setup(self) -> None: # Write the first incremental steps dir = os.path.dirname(path) os.makedirs(dir, exist_ok=True) - with open(path, "w", encoding="utf8") as f: + with open(path, "w", encoding=choose_file_encoding(path)) as f: f.write(content) for num, paths in self.deleted_paths.items(): @@ -831,6 +831,14 @@ def has_stable_flags(testcase: DataDrivenTestCase) -> bool: return True +def choose_file_encoding(path: str) -> str: + base_name, _ = os.path.splitext(path) + if base_name.endswith("_latin1"): + return "latin-1" + else: + return "utf-8" + + class DataSuite: # option fields - class variables files: list[str] diff --git a/mypy/test/helpers.py b/mypy/test/helpers.py index 60fea32a637fc..203e444e9a14b 100644 --- a/mypy/test/helpers.py +++ b/mypy/test/helpers.py @@ -27,7 +27,13 @@ from mypy.main import process_options from mypy.options import Options from mypy.test.config import test_data_prefix, test_temp_dir -from mypy.test.data import DataDrivenTestCase, DeleteFile, UpdateFile, fix_cobertura_filename +from mypy.test.data import ( + DataDrivenTestCase, + DeleteFile, + UpdateFile, + choose_file_encoding, + fix_cobertura_filename, +) skip = pytest.mark.skip @@ -440,7 +446,7 @@ def write_and_fudge_mtime(content: str, target_path: str) -> None: dir = os.path.dirname(target_path) os.makedirs(dir, exist_ok=True) - with open(target_path, "w", encoding="utf-8") as target: + with open(target_path, "w", encoding=choose_file_encoding(target_path)) as target: target.write(content) if new_time: diff --git a/test-data/unit/check-incremental.test b/test-data/unit/check-incremental.test index cb66c05fb5c8c..71ec6e9736853 100644 --- a/test-data/unit/check-incremental.test +++ b/test-data/unit/check-incremental.test @@ -2334,7 +2334,7 @@ tmp/c.py:1: error: Module "d" has no attribute "x" [delete nonexistent.py.2] [out] [out2] -mypy: error: Cannot read file 'tmp/nonexistent.py': No such file or directory +tmp/nonexistent.py: error: Cannot read file: No such file or directory [case testSerializeAbstractPropertyIncremental] from abc import abstractmethod @@ -8275,3 +8275,21 @@ value: str = "x" tmp/a.py:9: note: Revealed type is "builtins.list[builtins.int]" [out2] tmp/a.py:9: note: Revealed type is "builtins.list[builtins.int]" + +[case testIncrementalWorksCorrectlyWithNonUtf8] +import a +[file a.py] +import b_latin1 +[file a.py.2] +import b_latin1 +# touch +[file b_latin1.py] +# coding: latin-1 +Jérôme = False +[file b_latin1.py.2] +# coding: latin-1 +Jérôme = False +[rechecked a] +[stale] +[out] +[out2] diff --git a/test-data/unit/cmdline.test b/test-data/unit/cmdline.test index 9928f64acd116..45e7da1845128 100644 --- a/test-data/unit/cmdline.test +++ b/test-data/unit/cmdline.test @@ -422,7 +422,7 @@ int_pow.py:11: note: Revealed type is "Any" [case testMissingFile] # cmd: mypy nope.py [out] -mypy: error: Cannot read file 'nope.py': No such file or directory +nope.py: error: Cannot read file: No such file or directory == Return code: 2 [case testModulesAndPackages] @@ -673,7 +673,7 @@ c.py:1: error: Name "fail" is not defined \[mypy] files = config.py [out] -mypy: error: Cannot read file 'override.py': No such file or directory +override.py: error: Cannot read file: No such file or directory == Return code: 2 [case testErrorSummaryOnSuccess] @@ -730,7 +730,7 @@ Found 2 errors in 2 files (checked 2 source files) [case testErrorSummaryOnBadUsage] # cmd: mypy --error-summary missing.py [out] -mypy: error: Cannot read file 'missing.py': No such file or directory +missing.py: error: Cannot read file: No such file or directory Found 1 error in 1 file (errors prevented further checking) == Return code: 2 diff --git a/test-data/unit/fine-grained.test b/test-data/unit/fine-grained.test index 6fd975076cc89..bbf23a947dc87 100644 --- a/test-data/unit/fine-grained.test +++ b/test-data/unit/fine-grained.test @@ -6145,9 +6145,9 @@ a.py:1: error: "int" not callable [file a.py.2] 1() [out] -mypy: error: Cannot read file 'tmp/nonexistent.py': No such file or directory +nonexistent.py: error: Cannot read file: No such file or directory == -mypy: error: Cannot read file 'tmp/nonexistent.py': No such file or directory +nonexistent.py: error: Cannot read file: No such file or directory [case testNonExistentFileOnCommandLine2] # cmd: mypy a.py