diff --git a/Lib/test/libregrtest/main.py b/Lib/test/libregrtest/main.py index 2e8397a8a91d324..20b94eff8aa3fdd 100644 --- a/Lib/test/libregrtest/main.py +++ b/Lib/test/libregrtest/main.py @@ -1,3 +1,4 @@ +import html import os import random import re @@ -6,7 +7,7 @@ import sysconfig import time import trace -from _colorize import get_colors # type: ignore[import-not-found] +from _colorize import decolor, get_colors # type: ignore[import-not-found] from typing import NoReturn from test.support import os_helper, MS_WINDOWS, flush_std_streams @@ -395,6 +396,7 @@ def run_test( result = run_single_test(test_name, runtests) self.results.accumulate_result(result, runtests) + result.print_github_annotation(runtests) return result @@ -480,6 +482,42 @@ def finalize_tests(self, coverage: trace.CoverageResults | None) -> None: if self.junit_filename: self.results.write_junit(self.junit_filename) + self.write_github_summary() + + def write_github_summary(self) -> None: + filename = os.environ.get("GITHUB_STEP_SUMMARY") + # Tests which failed in the last run (the re-run, if any) + failed = self.results.rerun_results + # Only write a summary on failure: a CI run has many test jobs + exitcode = self.results.get_exitcode(self.fail_env_changed, + self.fail_rerun) + if not filename or not failed or not exitcode: + return + cases = [(result.errors or []) + (result.failures or []) + for result in failed] + ncases = sum(map(len, cases)) + with open(filename, "a", encoding="utf-8") as fp: + def write(text: str) -> None: + # Separate Markdown blocks with an empty line + print(text, end="\n\n", file=fp) + + write(f"## {decolor(self.get_state())}: " + f"{count(len(failed), 'test file')} and " + f"{count(ncases, 'test case')} failed") + for result, result_cases in zip(failed, cases): + write(f"### {decolor(str(result))}") + if result.env_changed_reasons: + write("\n".join(f"- {reason}" + for reason in result.env_changed_reasons)) + for name, traceback in result_cases: + # Expand the tracebacks if they fit on a screen: up to 5 + # test cases with a traceback of less than 30 lines + is_open = ncases <= 5 and traceback.count("\n") < 30 + write(f"" + f"{html.escape(name)}") + write(f"```pytb\n{decolor(traceback).rstrip()}\n```") + write("") + def display_summary(self) -> None: if self.first_runtests is None: raise ValueError( @@ -579,10 +617,16 @@ def _run_tests(self, selected: TestTuple, tests: TestList | None) -> int: if use_load_tracker: self.logger.start_load_tracker() try: - if self.num_workers: - self._run_tests_mp(runtests, self.num_workers) - else: - self.run_tests_sequentially(runtests) + with os_helper.EnvironmentVarGuard() as env: + # In GitHub Actions, only annotate failures of the last run: + # not the first run if failed tests will be re-run (with + # --python, they are not: see rerun_failed_tests()) + if self.want_rerun and not self.python_cmd: + env.unset("GITHUB_STEP_SUMMARY") + if self.num_workers: + self._run_tests_mp(runtests, self.num_workers) + else: + self.run_tests_sequentially(runtests) coverage = self.results.get_coverage_results() self.display_result(runtests) diff --git a/Lib/test/libregrtest/result.py b/Lib/test/libregrtest/result.py index 5324a46bb8cb1c5..e1d05777598f2e9 100644 --- a/Lib/test/libregrtest/result.py +++ b/Lib/test/libregrtest/result.py @@ -1,11 +1,15 @@ import dataclasses import json +import os from _colorize import get_colors # type: ignore[import-not-found] from typing import Any +from .findtests import findtestdir +from .runtests import RunTests from .utils import ( StrJSON, TestName, FilterTuple, - format_duration, normalize_test_name, print_warning) + format_duration, normalize_test_name, print_warning, + print_github_annotation) @dataclasses.dataclass(slots=True) @@ -178,6 +182,25 @@ def __str__(self) -> str: def has_meaningful_duration(self): return State.has_meaningful_duration(self.state) + def print_github_annotation(self, runtests: RunTests) -> None: + """Annotate a failed test without test case failures. + + For example: crash, timeout, env changed. Locate the annotation in the + test file. Test case failures are annotated by the test runner, where + they are reported (see RegressionTestResult.printErrorList()). + """ + if (self.is_failed(runtests.fail_env_changed) + and not self.errors and not self.failures): + message = "\n".join([str(self), *(self.env_changed_reasons or ())]) + # Test file: "test_x" is test_x.py or test_x/__init__.py, + # "test.test_x.test_y" is test_x/test_y.py + name = self.test_name.removeprefix("test.") + path = os.path.join(findtestdir(runtests.test_dir), *name.split(".")) + filename = path + ".py" + if not os.path.exists(filename): + filename = os.path.join(path, "__init__.py") + print_github_annotation(self.test_name, message, filename) + def set_env_changed(self, *reasons): if self.state is None or self.state == State.PASSED: self.state = State.ENV_CHANGED diff --git a/Lib/test/libregrtest/run_workers.py b/Lib/test/libregrtest/run_workers.py index c6db78e0882ba31..995543d676876cc 100644 --- a/Lib/test/libregrtest/run_workers.py +++ b/Lib/test/libregrtest/run_workers.py @@ -624,6 +624,8 @@ def _process_result(self, item: QueueOutput) -> TestResult: stdout = mp_result.worker_stdout if stdout: print(stdout, flush=True) + # Annotate after the output: env changed warnings, crash traceback + result.print_github_annotation(self.runtests) return result diff --git a/Lib/test/libregrtest/setup.py b/Lib/test/libregrtest/setup.py index d62194acd9c29e5..8909225d0265bf5 100644 --- a/Lib/test/libregrtest/setup.py +++ b/Lib/test/libregrtest/setup.py @@ -113,8 +113,8 @@ def setup_tests(runtests: RunTests) -> None: set_match_tests(runtests.match_tests) if runtests.use_junit: - support.junit_xml_list = [] from .testresult import RegressionTestResult + support.junit_xml_list = [] RegressionTestResult.USE_XML = True else: support.junit_xml_list = None diff --git a/Lib/test/libregrtest/testresult.py b/Lib/test/libregrtest/testresult.py index 605f1f4e6a89fb6..bdce67226fff3bf 100644 --- a/Lib/test/libregrtest/testresult.py +++ b/Lib/test/libregrtest/testresult.py @@ -9,7 +9,17 @@ import traceback import unittest from test import support -from test.libregrtest.utils import sanitize_xml +from test.libregrtest.utils import print_github_annotation, sanitize_xml + +def test_to_filename(test): + test = getattr(test, 'test_case', test) # subTest() + if dt_test := getattr(test, '_dt_test', None): # doctest.DocTestCase + return dt_test.filename + module = type(test).__module__ + # Fixture errors (setUpClass, setUpModule) are not supported + if module.startswith('unittest.'): + return None + return getattr(sys.modules.get(module), '__file__', None) class RegressionTestResult(unittest.TextTestResult): USE_XML = False @@ -130,6 +140,15 @@ def addUnexpectedSuccess(self, test): self._add_result(test, outcome='UNEXPECTED_SUCCESS') super().addUnexpectedSuccess(test) + def printErrorList(self, flavour, errors): + for test, err in errors: + super().printErrorList(flavour, [(test, err)]) + # Annotate just after the failure report, so that the annotation + # links to the report in the job log + if filename := test_to_filename(test): + print_github_annotation(str(test), err, filename, + file=self.stream) + def get_xml_element(self): if not self.USE_XML: raise ValueError("USE_XML is false") diff --git a/Lib/test/libregrtest/utils.py b/Lib/test/libregrtest/utils.py index 83e0575619c99a0..50de4ae9ab6d674 100644 --- a/Lib/test/libregrtest/utils.py +++ b/Lib/test/libregrtest/utils.py @@ -3,6 +3,7 @@ import locale import math import os.path +import pathlib import platform import random import re @@ -13,6 +14,7 @@ import tempfile import textwrap import types +from _colorize import decolor # type: ignore[import-not-found] from collections.abc import Callable _winapi: types.ModuleType | None try: @@ -140,6 +142,60 @@ def print_warning(msg: str) -> None: orig_unraisablehook: Callable[..., None] | None = None +# Stdlib directory: Lib/ in a source checkout or an installed lib/python3.X/ +STDLIB_DIR = pathlib.Path(os.__file__).parent +# Stdlib directory relative to the repository +REPOSITORY_LIB_DIR = pathlib.Path("Lib") + + +def repository_path(filename: str) -> str: + """Path of a stdlib file relative to the repository. + + Map STDLIB_DIR to Lib/. Return other paths unchanged. + """ + path = pathlib.Path(filename) + if not path.is_relative_to(STDLIB_DIR): + return filename + return (REPOSITORY_LIB_DIR / path.relative_to(STDLIB_DIR)).as_posix() + + +# Escape the data and the properties of GitHub Actions workflow commands +GITHUB_ESCAPE_DATA = str.maketrans({"%": "%25", "\r": "%0D", "\n": "%0A"}) +GITHUB_ESCAPE_PROPERTY = GITHUB_ESCAPE_DATA | str.maketrans({":": "%3A", + ",": "%2C"}) + + +def print_github_annotation(title: str, message: str, filename: str | None, + file=None) -> None: + """Print a GitHub Actions error annotation, if run in GitHub Actions. + + message is a traceback or a failure description. Only keep the exception + which ends the traceback: the job log has the full traceback. Locate the + annotation in filename, the test file, at the last frame of the traceback + in this file, if any. + """ + if not os.environ.get("GITHUB_STEP_SUMMARY"): + return + if not filename: + raise ValueError(f"missing test file of annotation {title!r}") + message = decolor(message) + props = {"file": repository_path(filename)} + # Line of the last traceback frame (' File "filename", line 123') or + # failed doctest example (same, but not indented) in filename + if lines := re.findall(rf'^ *File "{re.escape(filename)}", line (\d+)', + message, re.MULTILINE): + props["line"] = lines[-1] + props["title"] = title + # Only keep the exception: strip the traceback frames, each one a + # ' File' line followed by its indented source lines + message = re.split(r'^ File .*\n(?: .*\n)*', message, + flags=re.MULTILINE)[-1].strip() + props_text = ",".join(f"{key}={value.translate(GITHUB_ESCAPE_PROPERTY)}" + for key, value in props.items()) + print(f"::error {props_text}::{message.translate(GITHUB_ESCAPE_DATA)}", + file=file, flush=True) + + def regrtest_unraisable_hook(unraisable) -> None: global orig_unraisablehook support.set_environment_altered( diff --git a/Lib/test/test_bool.py b/Lib/test/test_bool.py index dcdf7bdce03b800..2214e7408aa5603 100644 --- a/Lib/test/test_bool.py +++ b/Lib/test/test_bool.py @@ -4,6 +4,7 @@ from test.support import os_helper import os +import sys class BoolTest(unittest.TestCase): @@ -25,6 +26,11 @@ def test_repr(self): self.assertIs(eval(repr(True)), True) def test_str(self): + if sys.platform == "linux": + # XXX: Intentional failure to test GitHub Actions annotations + self.assertEqual( + str(True), + 'Yes') self.assertEqual(str(False), 'False') self.assertEqual(str(True), 'True') diff --git a/Lib/test/test_json/test_decode.py b/Lib/test/test_json/test_decode.py index 7bff16ddf2155e9..d01368254f6701e 100644 --- a/Lib/test/test_json/test_decode.py +++ b/Lib/test/test_json/test_decode.py @@ -1,4 +1,5 @@ import decimal +import sys import unittest.mock from io import StringIO from collections import OrderedDict @@ -9,6 +10,9 @@ class TestDecode: def test_decimal(self): rval = self.loads('1.1', parse_float=decimal.Decimal) + if sys.platform == "linux": + # XXX: Intentional failure to test GitHub Actions annotations + self.loads('{"bad": }') self.assertIsInstance(rval, decimal.Decimal) self.assertEqual(rval, decimal.Decimal('1.1')) diff --git a/Lib/test/test_math.py b/Lib/test/test_math.py index a1250c668223e7f..40dc0d9acc75fff 100644 --- a/Lib/test/test_math.py +++ b/Lib/test/test_math.py @@ -513,6 +513,11 @@ def testFabs(self): def testFloor(self): self.assertRaises(TypeError, math.floor) + if sys.platform == "linux": + # XXX: Intentional failure to test GitHub Actions annotations + for x in (0.5, 1.5): + with self.subTest(x=x): + self.assertEqual(math.floor(x), x) self.assertEqual(int, type(math.floor(0.5))) self.assertEqual(math.floor(0.5), 0) self.assertEqual(math.floor(1.0), 1) diff --git a/Lib/test/test_os/test_os.py b/Lib/test/test_os/test_os.py index 81b3043eb7e75bc..04321264e6f121d 100644 --- a/Lib/test/test_os/test_os.py +++ b/Lib/test/test_os/test_os.py @@ -4080,7 +4080,8 @@ def test_memfd_create(self): self.assertFalse(os.get_inheritable(fd)) with open(fd, "wb", closefd=False) as f: f.write(b'memfd_create') - self.assertEqual(f.tell(), 12) + # XXX: Intentional failure to test GitHub Actions job summaries + self.assertEqual(f.tell(), 13) fd2 = os.memfd_create("Hi") self.addCleanup(os.close, fd2) diff --git a/Lib/test/test_profiling/test_sampling_profiler/test_live_collector_ui.py b/Lib/test/test_profiling/test_sampling_profiler/test_live_collector_ui.py index ae4ee969f34d8b3..57598715f8b2880 100644 --- a/Lib/test/test_profiling/test_sampling_profiler/test_live_collector_ui.py +++ b/Lib/test/test_profiling/test_sampling_profiler/test_live_collector_ui.py @@ -11,7 +11,8 @@ import time import unittest from unittest import mock -from test.support import requires, requires_remote_subprocess_debugging +from test.support import ( + os_helper, requires, requires_remote_subprocess_debugging) from test.support.import_helper import import_module # Only run these tests if curses is available @@ -856,6 +857,9 @@ def mock_init_curses_side_effect(self, n_times, mock_self, stdscr): def test_run_failed_module_live(self): """Test that running a existing module that fails exits with clean error.""" + # Don't write to the GitHub Actions job summary of the test suite + env = self.enterContext(os_helper.EnvironmentVarGuard()) + env.unset('GITHUB_STEP_SUMMARY') args = [ "profiling.sampling.cli", "run", "--live", "-m", "test", diff --git a/Lib/test/test_regrtest.py b/Lib/test/test_regrtest.py index 53f0a2863dc4e45..fa569207bee3366 100644 --- a/Lib/test/test_regrtest.py +++ b/Lib/test/test_regrtest.py @@ -624,6 +624,9 @@ class BaseTestCase(unittest.TestCase): def setUp(self): self.testdir = os.path.realpath(os.path.dirname(__file__)) + # Don't annotate the GitHub Actions job running the test suite + env = self.enterContext(os_helper.EnvironmentVarGuard()) + env.unset('GITHUB_STEP_SUMMARY') self.tmptestdir = tempfile.mkdtemp() self.addCleanup(os_helper.rmtree, self.tmptestdir) @@ -1639,6 +1642,198 @@ def test_env_changed(self): success=True), stats=2) + def create_marked_test(self, name, code): + # Create a test: return its name and the line number of each of its + # "# marker" comments + code = textwrap.dedent(code) + name = self.create_test(name, code) + markers = {match[1]: str(lineno) + for lineno, text in enumerate(code.splitlines(), 1) + if (match := re.search(r'# (\w+)$', text))} + return name, markers + + def run_tests_github(self, *args, exitcode=0): + # Run tests as in GitHub Actions: return the output and the job + # summary (None if regrtest didn't write it) + filename = os.path.join(self.tmptestdir, 'github_step_summary.md') + self.addCleanup(os_helper.unlink, filename) + os_helper.unlink(filename) + env = dict(os.environ) + env.pop('SOURCE_DATE_EPOCH', None) + env['GITHUB_STEP_SUMMARY'] = filename + output = self.run_tests(*args, env=env, exitcode=exitcode) + try: + with open(filename, encoding='utf-8') as fp: + return output, fp.read() + except FileNotFoundError: + return output, None + + def test_github_annotations(self): + # Every test failure is annotated once in the GitHub Actions job log, + # located in the test file, at the failing line of the test file if + # there is a traceback. The job summary lists the failures. + cases, lines = self.create_marked_test('github_cases', """ + import doctest, json, sys, unittest + + def load_tests(loader, tests, pattern): + tests.addTests(doctest.DocTestSuite(sys.modules[__name__])) + return tests + + def doctest_fail(): + ''' + >>> 1 + 1 # doctest + 3 + ''' + + class Tests(unittest.TestCase): + def test_error(self): + json.loads("{") # error + + def test_fail(self): + self.assertEqual(1, 2) # fail + + def test_subtest(self): + with self.subTest(x=1.5): + self.fail("subtest") # subtest + + class SetUpClassTests(unittest.TestCase): + @classmethod + def setUpClass(cls): + raise ValueError("setUpClass") + + def test_never(self): + pass + """) + setup_module, _ = self.create_marked_test( + 'github_setup_module', """ + import unittest + + def setUpModule(): + raise ValueError("setUpModule") + + class Tests(unittest.TestCase): + def test_never(self): + pass + """) + env_changed, _ = self.create_marked_test('github_env_changed', """ + import os, unittest + + class Tests(unittest.TestCase): + def test_env_changed(self): + os.environ["REGRTEST_GITHUB_ENV_CHANGED"] = "1" + """) + crash, _ = self.create_marked_test('github_crash', """ + import os, unittest + + class Tests(unittest.TestCase): + def test_crash(self): + os._exit(1) + """) + + # Annotated test cases: (title, file, line) + test_cases = [ + (f'test_error ({cases}.Tests.test_error)', + f'{cases}.py', lines['error']), + (f'test_fail ({cases}.Tests.test_fail)', + f'{cases}.py', lines['fail']), + (f'test_subtest ({cases}.Tests.test_subtest) (x=1.5)', + f'{cases}.py', lines['subtest']), + # Doctest examples are subtests: "[0]" is the example index + (f'doctest_fail ({cases}) [0]', + f'{cases}.py', lines['doctest']), + ] + # Test cases only listed in the job summary: fixture errors + # (setUpClass, setUpModule) are not annotated + failed_titles = [title for title, _, _ in test_cases] + [ + f'setUpClass ({cases}.SetUpClassTests)', + f'setUpModule ({setup_module})', + ] + # Test files: heading in the job summary + test_files = { + cases: f'### {cases} failed (2 errors, 3 failures)', + setup_module: f'### {setup_module} failed (1 error)', + env_changed: f'### {env_changed} failed (env changed)', + crash: f'### {crash} worker non-zero exit code', + } + + # Representative command lines of the CI jobs + command_lines = { + # "make ci", Windows, macOS, installed Python; the JIT jobs use + # the equivalent "-j0 --verbose2 --verbose3". Only failures of + # the re-run are annotated, not the reported first failures. + 'fast-ci': ['--fast-ci', '-j2'], + # Sanitizers and Hypothesis: env changed is not a failure + 'parallel': ['-j2', '-W'], + # iOS and WASI run tests in the main process: a crash kills it + 'single-process': ['--fast-ci', '--single-process'], + # Profile task of PGO and BOLT builds: failures stop the build + 'pgo': ['--pgo'], + } + for name, args in command_lines.items(): + with self.subTest(name): + tests = list(test_files) + if '--single-process' in args or '--pgo' in args: + tests.remove(crash) + output, summary = self.run_tests_github( + *args, *tests, exitcode=EXITCODE_BAD_TEST) + + expected = list(test_cases) + if '--fast-ci' in args: + expected.append((env_changed, f'{env_changed}.py', None)) + if crash in tests: + expected.append((crash, f'{crash}.py', None)) + + # "::error file=...,line=...,title=...::message" + annotations = [] + output_lines = output.splitlines() + for i, line in enumerate(output_lines): + if not line.startswith('::error '): + continue + _, props, message = line.split('::', 2) + props = props.removeprefix('error ').split(',') + props = dict(prop.split('=', 1) for prop in props) + title = props['title'] + annotations.append((title, os.path.basename(props['file']), + props.get('line'))) + before = output_lines[:i] + if title == env_changed: + # Right after the env changed warnings + self.assertStartsWith(before[-1], 'Warning -- ') + elif title != crash: + # Right after the failure report of the test, which + # ends with the exception and an empty line + report = next( + text for text in reversed(before) + if text.startswith(('ERROR: ', 'FAIL: '))) + self.assertEqual(report.split(': ', 1)[1], title) + message = message.replace('%0A', '\n').splitlines() + self.assertEqual(before[-len(message) - 1:], + message + ['']) + self.assertCountEqual(annotations, expected, output) + + # The job summary lists the failed tests in completion order + summary_lines = summary.splitlines() + self.assertEqual(summary_lines[0], + f'## FAILURE: {len(tests)} test files and ' + f'{len(failed_titles)} test cases failed') + self.assertCountEqual( + [line for line in summary_lines + if line.startswith('### ')], + [test_files[name] for name in tests]) + self.assertIn('- os.environ was modified', summary_lines) + self.assertCountEqual( + re.findall(r'(.*)', summary), + failed_titles) + self.assertEqual(summary.count('```pytb\n'), + len(failed_titles)) + + def test_github_summary_success(self): + # No annotation and no job summary when all tests pass + testname = self.create_test() + output, summary = self.run_tests_github(testname) + self.assertNotIn('::error', output) + self.assertIsNone(summary) + def test_rerun_fail(self): # FAILURE then FAILURE code = textwrap.dedent(""" diff --git a/Lib/test/test_textwrap.py b/Lib/test/test_textwrap.py index aca1f427656bb50..c1d670c978c9702 100644 --- a/Lib/test/test_textwrap.py +++ b/Lib/test/test_textwrap.py @@ -8,6 +8,8 @@ # $Id$ # +import os +import sys import unittest from textwrap import TextWrapper, wrap, fill, dedent, indent, shorten @@ -66,6 +68,9 @@ def test_simple(self): self.check_wrap(text, 80, [text]) def test_empty_string(self): + if sys.platform == "linux": + # XXX: Intentional failure to test GitHub Actions annotations + os.environ["GHA_ANNOTATION_TEST"] = "1" # Check that wrapping the empty string returns an empty list. self.check_wrap("", 6, []) self.check_wrap("", 6, [], drop_whitespace=False) diff --git a/Misc/NEWS.d/next/Tests/2026-10-05-22-36-06.gh-issue-158895.3Dmc4J.rst b/Misc/NEWS.d/next/Tests/2026-10-05-22-36-06.gh-issue-158895.3Dmc4J.rst new file mode 100644 index 000000000000000..a1c9e648e5c744e --- /dev/null +++ b/Misc/NEWS.d/next/Tests/2026-10-05-22-36-06.gh-issue-158895.3Dmc4J.rst @@ -0,0 +1,3 @@ +When run in GitHub Actions, :mod:`test.regrtest` now annotates each test +failure in the job log and writes a summary of the failures to the job +summary.