Bug report
Extensions built for the stable ABI can't use PyList_SET_ITEM, so they fill new lists with PyList_SetItem. On the free-threaded build each call costs about 2.7x what it does on the GIL-enabled build, while PyTuple_SetItem costs about the same on both builds. Converting native sequences (pointer arrays, vectors) to lists is a hot path for bindings, so this becomes a visible slowdown when a project switches from abi3 to abi3t wheels. For example, in tokenizers a getter that returns three lists runs about 2x slower (see huggingface/tokenizers#2488 (comment)).
Reproducer
fill.c:
#include <Python.h>
static PyObject *fill_list(PyObject *m, PyObject *arg) {
Py_ssize_t n = PyLong_AsSsize_t(arg);
PyObject *list = PyList_New(n);
for (Py_ssize_t i = 0; i < n; i++)
PyList_SetItem(list, i, Py_NewRef(Py_None));
return list;
}
static PyObject *fill_tuple(PyObject *m, PyObject *arg) {
Py_ssize_t n = PyLong_AsSsize_t(arg);
PyObject *tuple = PyTuple_New(n);
for (Py_ssize_t i = 0; i < n; i++)
PyTuple_SetItem(tuple, i, Py_NewRef(Py_None));
return tuple;
}
static PyMethodDef methods[] = {
{"fill_list", fill_list, METH_O, NULL},
{"fill_tuple", fill_tuple, METH_O, NULL},
{NULL},
};
PyABIInfo_VAR(abi_info);
static PySlot slots[] = {
PySlot_DATA(Py_mod_abi, &abi_info),
PySlot_DATA(Py_mod_name, "fill"),
PySlot_STATIC_DATA(Py_mod_methods, methods),
PySlot_DATA(Py_mod_gil, Py_MOD_GIL_NOT_USED),
PySlot_END,
};
PyMODEXPORT_FUNC PyModExport_fill(void) { return slots; }
bench.py:
import sys, timeit, fill
n = 1000
print(sys.version.split()[0], "free-threaded" if not sys._is_gil_enabled() else "GIL", fill.__file__.rsplit("/", 1)[-1])
for name in ("fill_list", "fill_tuple"):
f = getattr(fill, name)
t = min(timeit.repeat(lambda: f(n), number=2000, repeat=9)) / 2000
print(f" {name}: {t / n * 1e9:.1f} ns/item")
Build one copy for abi3 and one for abi3t, then run them:
mkdir abi3 abi3t
gcc -O2 -shared -fPIC -DPy_LIMITED_API=0x030f0000 -I"$(python3.15 -c 'import sysconfig; print(sysconfig.get_path("include"))')" fill.c -o abi3/fill.abi3.so
gcc -O2 -shared -fPIC -DPy_TARGET_ABI3T=0x030f0000 -I"$(python3.15t -c 'import sysconfig; print(sysconfig.get_path("include"))')" fill.c -o abi3t/fill.abi3t.so
PYTHONPATH=abi3 python3.15 bench.py
PYTHONPATH=abi3t python3.15 bench.py
PYTHONPATH=abi3t python3.15t bench.py
Results
3.15.0rc2, x86-64 Linux, gcc -O2:
3.15.0rc2 GIL fill.abi3.so
fill_list: 7.2 ns/item
fill_tuple: 7.6 ns/item
3.15.0rc2 GIL fill.abi3t.so
fill_list: 6.6 ns/item
fill_tuple: 7.5 ns/item
3.15.0rc2 free-threaded fill.abi3t.so
fill_list: 19.8 ns/item
fill_tuple: 9.6 ns/item
CPython versions tested on:
3.15
Operating systems tested on:
Linux
Bug report
Extensions built for the stable ABI can't use
PyList_SET_ITEM, so they fill new lists withPyList_SetItem. On the free-threaded build each call costs about 2.7x what it does on the GIL-enabled build, whilePyTuple_SetItemcosts about the same on both builds. Converting native sequences (pointer arrays, vectors) to lists is a hot path for bindings, so this becomes a visible slowdown when a project switches from abi3 to abi3t wheels. For example, in tokenizers a getter that returns three lists runs about 2x slower (see huggingface/tokenizers#2488 (comment)).Reproducer
fill.c:bench.py:Build one copy for abi3 and one for abi3t, then run them:
Results
3.15.0rc2, x86-64 Linux, gcc -O2:
CPython versions tested on:
3.15
Operating systems tested on:
Linux