Skip to content

Commit 4d6a6ee

Browse files
committed
Merge branch 'main' into dict_delitem_common_optimization
2 parents ee77d72 + 05302c0 commit 4d6a6ee

116 files changed

Lines changed: 3952 additions & 2387 deletions

File tree

Some content is hidden

Large Commits have some content hidden by default. Use the searchbox below for content that may be hidden.

‎.github/workflows/build.yml‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -547,6 +547,9 @@ jobs:
547547
- check-name: Undefined behavior
548548
sanitizer: UBSan
549549
free-threading: false
550+
- check-name: Memory
551+
sanitizer: MSan
552+
free-threading: false
550553
uses: ./.github/workflows/reusable-san.yml
551554
with:
552555
sanitizer: ${{ matrix.sanitizer }}

‎.github/workflows/reusable-san.yml‎

Lines changed: 20 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -60,7 +60,7 @@ jobs:
6060
|| ''
6161
}}
6262
- name: UBSan option setup
63-
if: inputs.sanitizer != 'TSan'
63+
if: inputs.sanitizer == 'UBSan'
6464
run: >-
6565
echo
6666
"UBSAN_OPTIONS=${SAN_LOG_OPTION}
@@ -69,6 +69,20 @@ jobs:
6969
>> "$GITHUB_ENV"
7070
env:
7171
SAN_LOG_OPTION: log_path=${{ github.workspace }}/san_log
72+
- name: MSan option setup
73+
if: inputs.sanitizer == 'MSan'
74+
run: |
75+
echo "MSAN_OPTIONS=${SAN_LOG_OPTION} allocator_may_return_null=1 handle_segv=0" >> "$GITHUB_ENV"
76+
# MSan reports false positives for memory initialized by libraries
77+
# that are not built with MSan, so disable modules that use them.
78+
# _remote_debugging links to libzstd directly, but we unpoision the memory.
79+
{
80+
echo '*disabled*'
81+
echo '_bz2 _ctypes _curses _curses_panel _dbm _decimal _gdbm _hashlib'
82+
echo '_lzma _sqlite3 _ssl _tkinter _uuid _zstd readline zlib'
83+
} > Modules/Setup.local
84+
env:
85+
SAN_LOG_OPTION: log_path=${{ github.workspace }}/san_log
7286
- name: Add ccache to PATH
7387
run: |
7488
echo "PATH=/usr/lib/ccache:$PATH" >> "$GITHUB_ENV"
@@ -93,6 +107,8 @@ jobs:
93107
# gh-157958: -O2 instead of the pydebug default -Og to avoid a clang 21
94108
# compile-time blowup on some interpreter files.
95109
# (https://github.com/llvm/llvm-project/issues/179695)
110+
# MSan uses --with-assertions instead of --with-pydebug because its
111+
# hooks on the Python memory allocators hide uninitialized reads.
96112
- name: Configure CPython
97113
run: >-
98114
./configure
@@ -101,9 +117,11 @@ jobs:
101117
${{
102118
inputs.sanitizer == 'TSan'
103119
&& '--with-thread-sanitizer'
120+
|| inputs.sanitizer == 'MSan'
121+
&& '--with-memory-sanitizer'
104122
|| '--with-undefined-behavior-sanitizer --with-strict-overflow'
105123
}}
106-
--with-pydebug
124+
${{ inputs.sanitizer == 'MSan' && '--with-assertions' || '--with-pydebug' }}
107125
${{ inputs.sanitizer == 'TSan' && '--with-openssl="$OPENSSL_DIR" --with-openssl-rpath=auto' || '' }}
108126
${{ inputs.free-threading && '--disable-gil' || '' }}
109127
- name: Build CPython

‎Doc/c-api/sys.rst‎

Lines changed: 23 additions & 20 deletions
Original file line numberDiff line numberDiff line change
@@ -152,28 +152,29 @@ Operating System Utilities
152152
<c-preinit>` and so that the LC_CTYPE locale is properly configured: see
153153
the :c:func:`Py_PreInitialize` function.
154154
155-
Decode a byte string from the :term:`filesystem encoding and error handler`.
156-
If the error handler is :ref:`surrogateescape error handler
157-
<surrogateescape>`, undecodable bytes are decoded as characters in range
158-
U+DC80..U+DCFF; and if a byte sequence can be decoded as a surrogate
159-
character, the bytes are escaped using the surrogateescape error handler
160-
instead of decoding them.
155+
Decode a byte string from the :term:`filesystem encoding <filesystem
156+
encoding and error handler>` with the :ref:`surrogateescape error handler
157+
<surrogateescape>`.
158+
159+
Undecodable bytes are decoded as characters in range U+DC80..U+DCFF. If a
160+
byte sequence can be decoded as a surrogate character, escape the bytes
161+
using the surrogateescape error handler instead of decoding them.
161162
162163
Return a pointer to a newly allocated wide character string, use
163164
:c:func:`PyMem_RawFree` to free the memory. If size is not ``NULL``, write
164165
the number of wide characters excluding the null character into ``*size``
165166
166-
Return ``NULL`` on decoding error or memory allocation error. If *size* is
167-
not ``NULL``, ``*size`` is set to ``(size_t)-1`` on memory error or set to
168-
``(size_t)-2`` on decoding error.
167+
On memory allocation failure, set *\*size* to ``(size_t)-1`` and return
168+
``NULL``.
169+
170+
On decode error, set *\*size* to ``(size_t)-2`` and return ``NULL``.
171+
Decoding errors should never happen, unless there is a bug in the C
172+
library.
169173
170174
The :term:`filesystem encoding and error handler` are selected by
171175
:c:func:`PyConfig_Read`: see :c:member:`~PyConfig.filesystem_encoding` and
172176
:c:member:`~PyConfig.filesystem_errors` members of :c:type:`PyConfig`.
173177
174-
Decoding errors should never happen, unless there is a bug in the C
175-
library.
176-
177178
Use the :c:func:`Py_EncodeLocale` function to encode the character string
178179
back to a byte string.
179180
@@ -195,17 +196,19 @@ Operating System Utilities
195196
196197
.. c:function:: char* Py_EncodeLocale(const wchar_t *text, size_t *error_pos)
197198
198-
Encode a wide character string to the :term:`filesystem encoding and error
199-
handler`. If the error handler is :ref:`surrogateescape error handler
200-
<surrogateescape>`, surrogate characters in the range U+DC80..U+DCFF are
201-
converted to bytes 0x80..0xFF.
199+
Encode a wide character string to the :term:`filesystem encoding <filesystem
200+
encoding and error handler>` with the :ref:`surrogateescape error handler
201+
<surrogateescape>`. Surrogate characters in the range U+DC80..U+DCFF are
202+
encoded to bytes 0x80..0xFF.
202203
203204
Return a pointer to a newly allocated byte string, use :c:func:`PyMem_Free`
204-
to free the memory. Return ``NULL`` on encoding error or memory allocation
205-
error.
205+
to free the memory.
206+
207+
On memory allocation failure, set *\*error_pos* to ``(size_t)-1`` and return
208+
``NULL``.
206209
207-
If error_pos is not ``NULL``, ``*error_pos`` is set to ``(size_t)-1`` on
208-
success, or set to the index of the invalid character on encoding error.
210+
On encoding error, set *\*error_pos* to the index of the first unencodable
211+
character and return ``NULL``.
209212
210213
The :term:`filesystem encoding and error handler` are selected by
211214
:c:func:`PyConfig_Read`: see :c:member:`~PyConfig.filesystem_encoding` and

‎Doc/c-api/unicode.rst‎

Lines changed: 40 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -168,8 +168,14 @@ access to internal read-only data of Unicode objects:
168168
The function performs no checks for any of its requirements,
169169
and is intended for usage in loops.
170170
171+
While :class:`str` objects are usually immutable in Python, this special C API allows
172+
mutating a fresh :class:`str` object if the string has not been "used" yet.
173+
171174
.. versionadded:: 3.3
172175
176+
.. soft-deprecated:: next
177+
Use the :c:type:`PyUnicodeWriter` API instead.
178+
173179
174180
.. c:function:: Py_UCS4 PyUnicode_READ(int kind, void *data, Py_ssize_t index)
175181
@@ -407,9 +413,15 @@ APIs:
407413
using the :c:type:`PyUnicodeWriter` API, or one of the ``PyUnicode_From*``
408414
functions below.
409415
416+
While :class:`str` objects are usually immutable in Python, this special C API
417+
returns a :class:`str` object that can be mutated, except if *size* is zero, in which
418+
case it returns the immutable empty string constant.
410419
411420
.. versionadded:: 3.3
412421
422+
.. soft-deprecated:: next
423+
Use the :c:type:`PyUnicodeWriter` API instead.
424+
413425
414426
.. c:function:: PyObject* PyUnicode_FromKindAndData(int kind, const void *buffer, \
415427
Py_ssize_t size)
@@ -754,11 +766,16 @@ APIs:
754766
possible. Returns ``-1`` and sets an exception on error, otherwise returns
755767
the number of copied characters.
756768
757-
The string must not have been “used” yet.
769+
While :class:`str` objects are usually immutable in Python, this special C API allows
770+
mutating a fresh :class:`str` object if the string has not been "used" yet.
771+
758772
See :c:func:`PyUnicode_New` for details.
759773
760774
.. versionadded:: 3.3
761775
776+
.. soft-deprecated:: next
777+
Use the :c:type:`PyUnicodeWriter` API instead.
778+
762779
763780
.. c:function:: int PyUnicode_Resize(PyObject **unicode, Py_ssize_t length);
764781
@@ -774,6 +791,14 @@ APIs:
774791
The function doesn't check string content, the result may not be a
775792
string in canonical representation.
776793
794+
While :class:`str` objects are usually immutable in Python, this special C API
795+
can resize a :class:`str` object in-place if the string has not been "used" yet.
796+
It returns a :class:`str` object which can be mutated, except if *size* is zero, in
797+
which case it returns the immutable empty string constant.
798+
799+
.. soft-deprecated:: next
800+
Use the :c:type:`PyUnicodeWriter` API instead.
801+
777802
778803
.. c:function:: Py_ssize_t PyUnicode_Fill(PyObject *unicode, Py_ssize_t start, \
779804
Py_ssize_t length, Py_UCS4 fill_char)
@@ -784,14 +809,19 @@ APIs:
784809
Fail if *fill_char* is bigger than the string maximum character, or if the
785810
string has more than 1 reference.
786811
787-
The string must not have been “used” yet.
788-
See :c:func:`PyUnicode_New` for details.
789-
790812
Return the number of written characters, or return ``-1`` and raise an
791813
exception on error.
792814
815+
While :class:`str` objects are usually immutable in Python, this special C API allows
816+
mutating a fresh :class:`str` object if the string has not been "used" yet.
817+
818+
See :c:func:`PyUnicode_New` for details.
819+
793820
.. versionadded:: 3.3
794821
822+
.. soft-deprecated:: next
823+
Use the :c:type:`PyUnicodeWriter` API instead.
824+
795825
796826
.. c:function:: int PyUnicode_WriteChar(PyObject *unicode, Py_ssize_t index, \
797827
Py_UCS4 character)
@@ -804,11 +834,16 @@ APIs:
804834
See :c:func:`PyUnicode_WRITE` for a version that skips these checks,
805835
making them your responsibility.
806836
807-
The string must not have been “used” yet.
837+
While :class:`str` objects are usually immutable in Python, this special C API allows
838+
mutating a fresh :class:`str` object if the string has not been "used" yet.
839+
808840
See :c:func:`PyUnicode_New` for details.
809841
810842
.. versionadded:: 3.3
811843
844+
.. soft-deprecated:: next
845+
Use the :c:type:`PyUnicodeWriter` API instead.
846+
812847
813848
.. c:function:: Py_UCS4 PyUnicode_ReadChar(PyObject *unicode, Py_ssize_t index)
814849

‎Doc/library/ctypes.rst‎

Lines changed: 3 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -1755,8 +1755,9 @@ These prefabricated library loaders are available:
17551755
:c:expr:`int`, which is of course not always the truth, so you have to assign
17561756
the correct :attr:`!restype` attribute to use these functions.
17571757

1758-
Note that if the Python interpreter is statically linked, this will be
1759-
``None``, as ``dlopen`` is not possible in this case.
1758+
.. note::
1759+
1760+
If the Python interpreter is statically linked, this may be ``None``.
17601761

17611762
.. audit-event:: ctypes.dlopen name ctypes.LibraryLoader
17621763

‎Doc/using/configure.rst‎

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1026,6 +1026,10 @@ Debug options
10261026

10271027
Enable MemorySanitizer allocation error detector, ``msan`` (default is no).
10281028

1029+
MSan reports false positives for memory initialized by libraries that are
1030+
not built with MSan, so either build all dependencies with MSan or disable
1031+
the extension modules that use them in :file:`Modules/Setup.local`.
1032+
10291033
.. versionadded:: 3.6
10301034

10311035
.. option:: --with-undefined-behavior-sanitizer

‎Doc/whatsnew/3.16.rst‎

Lines changed: 7 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1224,6 +1224,13 @@ Deprecated C APIs
12241224
:c:func:`PyModule_GetFilenameObject` instead is still recommended.
12251225
(Contributed by Victor Stinner in :gh:`154757`.)
12261226

1227+
* Soft deprecate functions modifying Unicode strings:
1228+
:c:func:`PyUnicode_New`, :c:func:`PyUnicode_CopyCharacters`,
1229+
:c:func:`PyUnicode_Fill`, :c:func:`PyUnicode_Resize`,
1230+
:c:func:`PyUnicode_WRITE` and :c:func:`PyUnicode_WriteChar`.
1231+
Use the safer :c:type:`PyUnicodeWriter` API instead.
1232+
(Contributed by Victor Stinner in :gh:`157710`.)
1233+
12271234
.. Add C API deprecations above alphabetically, not here at the end.
12281235
12291236
.. include:: ../deprecations/c-api-pending-removal-in-3.18.rst

‎Include/internal/pycore_fileutils.h‎

Lines changed: 11 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -24,27 +24,22 @@ extern "C" {
2424
PyAPI_FUNC(_Py_error_handler) _Py_GetErrorHandler(const char *errors);
2525

2626
// Export for '_testinternalcapi' shared extension
27-
PyAPI_FUNC(int) _Py_DecodeLocaleEx(
27+
PyAPI_FUNC(int) _Py_DecodeLocale(
2828
const char *arg,
2929
wchar_t **wstr,
3030
size_t *wlen,
31-
const char **reason,
3231
int current_locale,
3332
_Py_error_handler errors);
3433

3534
// Export for '_testinternalcapi' shared extension
36-
PyAPI_FUNC(int) _Py_EncodeLocaleEx(
35+
PyAPI_FUNC(int) _Py_EncodeLocale(
3736
const wchar_t *text,
3837
char **str,
38+
size_t *output_length,
3939
size_t *error_pos,
40-
const char **reason,
4140
int current_locale,
4241
_Py_error_handler errors);
4342

44-
extern char* _Py_EncodeLocaleRaw(
45-
const wchar_t *text,
46-
size_t *error_pos);
47-
4843
extern PyObject* _Py_device_encoding(int);
4944

5045
#if defined(MS_WINDOWS) || defined(__APPLE__)
@@ -190,19 +185,23 @@ extern int _Py_open_osfhandle(void *handle, int flags);
190185
? _PyStatus_ERR("cannot decode " NAME) \
191186
: _PyStatus_NO_MEMORY()
192187

193-
extern int _Py_DecodeUTF8Ex(
188+
#define _Py_CODEC_MEMORY_ERROR -1
189+
#define _Py_CODEC_DECODE_ERROR -2
190+
#define _Py_CODEC_ENCODE_ERROR -2
191+
#define _Py_CODEC_UNSUPPORTED_ERROR_HANDLER -3
192+
193+
extern int _Py_DecodeUTF8(
194194
const char *arg,
195195
Py_ssize_t arglen,
196196
wchar_t **wstr,
197197
size_t *wlen,
198-
const char **reason,
199198
_Py_error_handler errors);
200199

201-
extern int _Py_EncodeUTF8Ex(
200+
extern int _Py_EncodeUTF8(
202201
const wchar_t *text,
203202
char **str,
203+
size_t *output_length,
204204
size_t *error_pos,
205-
const char **reason,
206205
int raw_malloc,
207206
_Py_error_handler errors);
208207

‎Include/internal/pycore_global_objects_fini_generated.h‎

Lines changed: 3 additions & 0 deletions
Some generated files are not rendered by default. Learn more about customizing how changed files appear on GitHub.

‎Include/internal/pycore_global_strings.h‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -54,6 +54,9 @@ struct _Py_global_strings {
5454
STRUCT_FOR_STR(native, "<native>")
5555
STRUCT_FOR_STR(str_replace_inf, "1e309")
5656
STRUCT_FOR_STR(type_params, ".type_params")
57+
STRUCT_FOR_STR(unknown_file, "<unknown file>")
58+
STRUCT_FOR_STR(unknown_function, "<unknown function>")
59+
STRUCT_FOR_STR(unreadable_frame, "<unreadable frame>")
5760
STRUCT_FOR_STR(utf_8, "utf-8")
5861
} literals;
5962

0 commit comments

Comments
 (0)