diff --git a/mypy/build.py b/mypy/build.py index a55fe4eb92d5..09952f37aa28 100644 --- a/mypy/build.py +++ b/mypy/build.py @@ -3191,31 +3191,10 @@ def get_source(self) -> str: with self.wrap_context(): source = self.source if self.path and source is None: - try: + with self.handle_file_read_errors(): path = manager.maybe_swap_for_shadow_path(self.path) source = decode_python_encoding(manager.fscache.read(path)) self.source_hash = manager.fscache.hash_digest(path) - except OSError as ioerr: - # ioerr.strerror differs for os.stat failures between Windows and - # other systems, but os.strerror(ioerr.errno) does not, so we use that. - # (We want the error messages to be platform-independent so that the - # tests have predictable output.) - assert ioerr.errno is not None - raise CompileError( - [ - "mypy: error: Cannot read file '{}': {}".format( - self.path.replace(os.getcwd() + os.sep, ""), - os.strerror(ioerr.errno), - ) - ], - module_with_blocker=self.id, - ) from ioerr - except (UnicodeDecodeError, DecodeError) as decodeerr: - if self.path.endswith(".pyd"): - err = f"{self.path}: error: Stubgen does not support .pyd files" - else: - err = f"{self.path}: error: Cannot decode file: {str(decodeerr)}" - raise CompileError([err], module_with_blocker=self.id) from decodeerr elif self.path and manager.fscache.isdir(self.path): source = "" self.source_hash = "" @@ -3229,11 +3208,37 @@ def get_source(self) -> str: self.time_spent_us += time_spent_us(t0) return source + @contextlib.contextmanager + def handle_file_read_errors(self) -> Iterator[None]: + try: + yield + except OSError as ioerr: + # ioerr.strerror differs for os.stat failures between Windows and + # other systems, but os.strerror(ioerr.errno) does not, so we use that. + # (We want the error messages to be platform-independent so that the + # tests have predictable output.) + assert ioerr.errno is not None + err_path = self.manager.errors.simplify_path(self.xpath) + raise CompileError( + [f"{err_path}: error: Cannot read file: {os.strerror(ioerr.errno)}"], + module_with_blocker=self.id, + ) from ioerr + except (UnicodeDecodeError, DecodeError) as decodeerr: + err_path = self.manager.errors.simplify_path(self.xpath) + if err_path.endswith(".pyd"): + err = f"{err_path}: error: Stubgen does not support .pyd files" + else: + err = f"{err_path}: error: Cannot decode file: {str(decodeerr)}" + raise CompileError([err], module_with_blocker=self.id) from decodeerr + def parse_file_inner(self, source: str | None, raw_data: FileRawData | None = None) -> None: t0 = time_ref() - self.tree = self.manager.parse_file( - self.id, self.xpath, source, options=self.options, raw_data=raw_data - ) + # Error handling here matches get_source(), since in case of an error in + # the new parser we fall back to reading the file manually ourselves. + with self.handle_file_read_errors(): + self.tree = self.manager.parse_file( + self.id, self.xpath, source, options=self.options, raw_data=raw_data + ) self.time_spent_us += time_spent_us(t0) def parse_file(self, *, temporary: bool = False, raw_data: FileRawData | None = None) -> None: diff --git a/mypy/nativeparse.py b/mypy/nativeparse.py index d81c2878698a..622b199227bd 100644 --- a/mypy/nativeparse.py +++ b/mypy/nativeparse.py @@ -46,6 +46,7 @@ read_str_opt, read_tag, ) +from mypy.errors import CompileError from mypy.nodes import ( ARG_KINDS, ARG_POS, @@ -157,7 +158,7 @@ UnionType, UnpackType, ) -from mypy.util import unnamed_function +from mypy.util import decode_python_encoding, hash_digest, unnamed_function TypeIgnores = list[tuple[int, list[str]]] @@ -297,16 +298,40 @@ def parse_to_binary_ast( source: str | bytes | None = None, skip_function_bodies: bool = False, ) -> tuple[bytes, list[ParseError], TypeIgnores, bytes, bool, bool, str, list[tuple[int, str]]]: - ast_bytes, errors, ignores, import_bytes, ast_data = ast_serialize.parse( - filename, - source, - skip_function_bodies=skip_function_bodies, - python_version=options.python_version, - platform=options.platform, - always_true=options.always_true, - always_false=options.always_false, - cache_version=5, - ) + try: + ast_bytes, errors, ignores, import_bytes, ast_data = ast_serialize.parse( + filename, + source, + skip_function_bodies=skip_function_bodies, + python_version=options.python_version, + platform=options.platform, + always_true=options.always_true, + always_false=options.always_false, + cache_version=5, + ) + except ValueError as exc: + if source is None: + # Try reading/encoding manually, since native parser only supports UTF-8. + # If we still cannot decode, decode error will bubble up to caller. + with open(filename, "rb") as f: + source_bytes = f.read() + source = decode_python_encoding(source_bytes) + ast_bytes, errors, ignores, import_bytes, ast_data = ast_serialize.parse( + filename, + source, + skip_function_bodies=skip_function_bodies, + python_version=options.python_version, + platform=options.platform, + always_true=options.always_true, + always_false=options.always_false, + cache_version=5, + ) + # Compute hash using original non-UTF-8 bytes, so that + # incremental mode works correctly. + ast_data["source_hash"] = hash_digest(source_bytes) + else: + # Convert everything unexpected to a standard-looking blocker. + raise CompileError([f"{filename}: error: Cannot parse file: {exc}"]) from exc return ( ast_bytes, errors, diff --git a/mypy/test/data.py b/mypy/test/data.py index 574a343fa3f3..627f0bc86108 100644 --- a/mypy/test/data.py +++ b/mypy/test/data.py @@ -379,7 +379,7 @@ def setup(self) -> None: # Write the first incremental steps dir = os.path.dirname(path) os.makedirs(dir, exist_ok=True) - with open(path, "w", encoding="utf8") as f: + with open(path, "w", encoding=choose_file_encoding(path)) as f: f.write(content) for num, paths in self.deleted_paths.items(): @@ -831,6 +831,14 @@ def has_stable_flags(testcase: DataDrivenTestCase) -> bool: return True +def choose_file_encoding(path: str) -> str: + base_name, _ = os.path.splitext(path) + if base_name.endswith("_latin1"): + return "latin-1" + else: + return "utf-8" + + class DataSuite: # option fields - class variables files: list[str] diff --git a/mypy/test/helpers.py b/mypy/test/helpers.py index 60fea32a637f..203e444e9a14 100644 --- a/mypy/test/helpers.py +++ b/mypy/test/helpers.py @@ -27,7 +27,13 @@ from mypy.main import process_options from mypy.options import Options from mypy.test.config import test_data_prefix, test_temp_dir -from mypy.test.data import DataDrivenTestCase, DeleteFile, UpdateFile, fix_cobertura_filename +from mypy.test.data import ( + DataDrivenTestCase, + DeleteFile, + UpdateFile, + choose_file_encoding, + fix_cobertura_filename, +) skip = pytest.mark.skip @@ -440,7 +446,7 @@ def write_and_fudge_mtime(content: str, target_path: str) -> None: dir = os.path.dirname(target_path) os.makedirs(dir, exist_ok=True) - with open(target_path, "w", encoding="utf-8") as target: + with open(target_path, "w", encoding=choose_file_encoding(target_path)) as target: target.write(content) if new_time: diff --git a/mypy/test/test_nativeparse.py b/mypy/test/test_nativeparse.py index d74ff99d5e5e..efe3be63c97c 100644 --- a/mypy/test/test_nativeparse.py +++ b/mypy/test/test_nativeparse.py @@ -1,8 +1,4 @@ -"""Tests for the experimental native mypy parser. - -To run these, you will need to manually install ast_serialize from -https://github.com/mypyc/ast_serialize first (see the README for the details). -""" +"""Tests for the native mypy parser.""" from __future__ import annotations @@ -12,6 +8,7 @@ import unittest from collections.abc import Iterator +import pytest from librt.internal import ReadBuffer from mypy import defaults, nodes @@ -27,36 +24,24 @@ ) from mypy.config_parser import parse_mypy_comments from mypy.errors import CompileError +from mypy.nativeparse import ( + State, + deserialize_imports, + native_parse, + parse_to_binary_ast, + read_statements, +) from mypy.nodes import MypyFile, ParseError from mypy.options import Options from mypy.test.data import DataDrivenTestCase, DataSuite from mypy.test.helpers import assert_string_arrays_equal from mypy.util import get_mypy_comments -# If the experimental ast_serialize module isn't installed, the following import will fail -# and we won't run any native parser tests. -try: - from mypy.nativeparse import ( - State, - deserialize_imports, - native_parse, - parse_to_binary_ast, - read_statements, - ) - - has_nativeparse = True -except ImportError: - has_nativeparse = False - class NativeParserSuite(DataSuite): required_out_section = True base_path = "." - files = ( - ["native-parser.test", "native-parser-python311.test", "native-parser-python312.test"] - if has_nativeparse - else [] - ) + files = ["native-parser.test", "native-parser-python311.test", "native-parser-python312.test"] def run_case(self, testcase: DataDrivenTestCase) -> None: test_parser(testcase) @@ -65,7 +50,7 @@ def run_case(self, testcase: DataDrivenTestCase) -> None: class NativeParserImportsSuite(DataSuite): required_out_section = True base_path = "." - files = ["native-parser-imports.test"] if has_nativeparse else [] + files = ["native-parser-imports.test"] def run_case(self, testcase: DataDrivenTestCase) -> None: test_parser_imports(testcase) @@ -238,7 +223,6 @@ def format_reachable_imports(node: MypyFile) -> list[str]: return output -@unittest.skipUnless(has_nativeparse, "nativeparse not available") class TestNativeParserBinaryFormat(unittest.TestCase): def _assert_trivial_binary_data(self, b: bytes, /) -> None: # A quick sanity check to ensure the serialized data looks as expected. Only covers @@ -296,14 +280,27 @@ def test_trivial_binary_data_from_bytes_source(self) -> None: self._assert_trivial_binary_data(b) def test_invalid_bytes_raises(self) -> None: - with self.assertRaises(UnicodeDecodeError): + with self.assertRaises(CompileError): parse_to_binary_ast("", Options(), b"\xff") +class TestNativeParserCustomEncoding(unittest.TestCase): + def test_latin1(self) -> None: + source = "# coding: latin1\nJérôme = False" + with temp_source(source, encoding="latin1") as fnam: + parse_to_binary_ast(fnam, Options()) + + def test_latin1_broken(self) -> None: + source = "# coding: ascii\nJérôme = False" + with temp_source(source, encoding="latin1") as fnam: + with pytest.raises(UnicodeDecodeError): + parse_to_binary_ast(fnam, Options()) + + @contextlib.contextmanager -def temp_source(text: str) -> Iterator[str]: +def temp_source(text: str, encoding: str = "utf-8") -> Iterator[str]: with tempfile.TemporaryDirectory() as temp_dir: temp_path = os.path.join(temp_dir, "t.py") - with open(temp_path, "w") as f: - f.write(text) + with open(temp_path, "wb") as f: + f.write(text.encode(encoding)) yield temp_path diff --git a/test-data/unit/check-incremental.test b/test-data/unit/check-incremental.test index cb66c05fb5c8..71ec6e973685 100644 --- a/test-data/unit/check-incremental.test +++ b/test-data/unit/check-incremental.test @@ -2334,7 +2334,7 @@ tmp/c.py:1: error: Module "d" has no attribute "x" [delete nonexistent.py.2] [out] [out2] -mypy: error: Cannot read file 'tmp/nonexistent.py': No such file or directory +tmp/nonexistent.py: error: Cannot read file: No such file or directory [case testSerializeAbstractPropertyIncremental] from abc import abstractmethod @@ -8275,3 +8275,21 @@ value: str = "x" tmp/a.py:9: note: Revealed type is "builtins.list[builtins.int]" [out2] tmp/a.py:9: note: Revealed type is "builtins.list[builtins.int]" + +[case testIncrementalWorksCorrectlyWithNonUtf8] +import a +[file a.py] +import b_latin1 +[file a.py.2] +import b_latin1 +# touch +[file b_latin1.py] +# coding: latin-1 +Jérôme = False +[file b_latin1.py.2] +# coding: latin-1 +Jérôme = False +[rechecked a] +[stale] +[out] +[out2] diff --git a/test-data/unit/cmdline.test b/test-data/unit/cmdline.test index 9928f64acd11..45e7da184512 100644 --- a/test-data/unit/cmdline.test +++ b/test-data/unit/cmdline.test @@ -422,7 +422,7 @@ int_pow.py:11: note: Revealed type is "Any" [case testMissingFile] # cmd: mypy nope.py [out] -mypy: error: Cannot read file 'nope.py': No such file or directory +nope.py: error: Cannot read file: No such file or directory == Return code: 2 [case testModulesAndPackages] @@ -673,7 +673,7 @@ c.py:1: error: Name "fail" is not defined \[mypy] files = config.py [out] -mypy: error: Cannot read file 'override.py': No such file or directory +override.py: error: Cannot read file: No such file or directory == Return code: 2 [case testErrorSummaryOnSuccess] @@ -730,7 +730,7 @@ Found 2 errors in 2 files (checked 2 source files) [case testErrorSummaryOnBadUsage] # cmd: mypy --error-summary missing.py [out] -mypy: error: Cannot read file 'missing.py': No such file or directory +missing.py: error: Cannot read file: No such file or directory Found 1 error in 1 file (errors prevented further checking) == Return code: 2 diff --git a/test-data/unit/fine-grained.test b/test-data/unit/fine-grained.test index 6fd975076cc8..bbf23a947dc8 100644 --- a/test-data/unit/fine-grained.test +++ b/test-data/unit/fine-grained.test @@ -6145,9 +6145,9 @@ a.py:1: error: "int" not callable [file a.py.2] 1() [out] -mypy: error: Cannot read file 'tmp/nonexistent.py': No such file or directory +nonexistent.py: error: Cannot read file: No such file or directory == -mypy: error: Cannot read file 'tmp/nonexistent.py': No such file or directory +nonexistent.py: error: Cannot read file: No such file or directory [case testNonExistentFileOnCommandLine2] # cmd: mypy a.py