From f4a13b1daae06e8a1a77cf19680c7189ff829293 Mon Sep 17 00:00:00 2001 From: LucasZhou Date: Sun, 4 Oct 2026 01:40:51 -0500 Subject: [PATCH 1/2] gh-158790: Improve performance for list.insert() and del list[i] in free-threaded builds `ins1()`(used by `list.insert()`) and `list_ass_item_lock_held()` shift the list with one atomic release store per element in free-threaded builds. Those atomic stores prevent the compiler from vectorizing the loops, making them much slower than `memmove()` for large shifts. I propose using the existing `ptr_wise_atomic_memmove()` helper here. It uses `memmove()` as a fast path when the list is owned by the current thread and is not shared, and keeps the atomic element-wise stores for lists that other threads can observe. The default (GIL) build keeps the current loops because the compiler can optimize them directly in the calling function, and they perform better in benchmarks. pyperf, free-threaded release build (--disable-gil, -O3): del l[0] 2.26x faster del l[len(l)//2] 2.14x faster l.insert(0, x) 1.95x faster l.insert(len(l)//2, x) 1.86x faster insert(0)+del[0], n=1000 2.86x faster sliding window, n=1000 2.46x faster Compiling both versions in GIL mode produces the same code for `ins1()` and `list_ass_item_lock_held()`, so the default build is not affected. This continues gh-129069, which introduced `ptr_wise_atomic_memmove()`. --- ...-10-04-01-40-31.gh-issue-158790.F3kQ7z.rst | 1 + Objects/listobject.c | 23 +++++++++++++++++-- 2 files changed, 22 insertions(+), 2 deletions(-) create mode 100644 Misc/NEWS.d/next/Core_and_Builtins/2026-10-04-01-40-31.gh-issue-158790.F3kQ7z.rst diff --git a/Misc/NEWS.d/next/Core_and_Builtins/2026-10-04-01-40-31.gh-issue-158790.F3kQ7z.rst b/Misc/NEWS.d/next/Core_and_Builtins/2026-10-04-01-40-31.gh-issue-158790.F3kQ7z.rst new file mode 100644 index 000000000000000..6b9f71abf537ac7 --- /dev/null +++ b/Misc/NEWS.d/next/Core_and_Builtins/2026-10-04-01-40-31.gh-issue-158790.F3kQ7z.rst @@ -0,0 +1 @@ +Speed up ``list.insert()`` and ``del list[i]`` in free-threaded builds by using ``memmove()`` to shift elements when the list is not shared. diff --git a/Objects/listobject.c b/Objects/listobject.c index 81eb3e1102159b5..e61fb9b0bdf74fb 100644 --- a/Objects/listobject.c +++ b/Objects/listobject.c @@ -478,10 +478,13 @@ end:; return ret; } +static void ptr_wise_atomic_memmove(PyListObject *a, PyObject **dest, + PyObject **src, Py_ssize_t n); + static int ins1(PyListObject *self, Py_ssize_t where, PyObject *v) { - Py_ssize_t i, n = Py_SIZE(self); + Py_ssize_t n = Py_SIZE(self); PyObject **items; if (v == NULL) { PyErr_BadInternalCall(); @@ -500,8 +503,17 @@ ins1(PyListObject *self, Py_ssize_t where, PyObject *v) if (where > n) where = n; items = self->ob_item; - for (i = n; --i >= where; ) +#ifdef Py_GIL_DISABLED + if (where < n) { + ptr_wise_atomic_memmove(self, &items[where + 1], &items[where], + n - where); + } +#else + /* The compiler expands this loop inline. This is faster than a + memmove() call for short lists. */ + for (Py_ssize_t i = n; --i >= where; ) FT_ATOMIC_STORE_PTR_RELEASE(items[i+1], items[i]); +#endif FT_ATOMIC_STORE_PTR_RELEASE(items[where], Py_NewRef(v)); return 0; } @@ -1145,9 +1157,16 @@ list_ass_item_lock_held(PyListObject *a, Py_ssize_t i, PyObject *v) PyObject *tmp = a->ob_item[i]; if (v == NULL) { Py_ssize_t size = Py_SIZE(a); +#ifdef Py_GIL_DISABLED + if (i < size - 1) { + ptr_wise_atomic_memmove(a, &a->ob_item[i], &a->ob_item[i + 1], + size - 1 - i); + } +#else for (Py_ssize_t idx = i; idx < size - 1; idx++) { FT_ATOMIC_STORE_PTR_RELEASE(a->ob_item[idx], a->ob_item[idx + 1]); } +#endif Py_SET_SIZE(a, size - 1); } else { From 58c39395be123c4872144d1a19190aa319fb0c8a Mon Sep 17 00:00:00 2001 From: LucasZhou Date: Sun, 4 Oct 2026 08:41:25 -0500 Subject: [PATCH 2/2] gh-158790: Address review comments Use Sphinx roles in the NEWS entry (:meth:`list.insert`, :keyword:`del`, :manpage:`memmove(3)`) and drop a comment in ins1() that repeated the rationale already given in the commit message. --- .../2026-10-04-01-40-31.gh-issue-158790.F3kQ7z.rst | 2 +- Objects/listobject.c | 2 -- 2 files changed, 1 insertion(+), 3 deletions(-) diff --git a/Misc/NEWS.d/next/Core_and_Builtins/2026-10-04-01-40-31.gh-issue-158790.F3kQ7z.rst b/Misc/NEWS.d/next/Core_and_Builtins/2026-10-04-01-40-31.gh-issue-158790.F3kQ7z.rst index 6b9f71abf537ac7..b509071e16eea20 100644 --- a/Misc/NEWS.d/next/Core_and_Builtins/2026-10-04-01-40-31.gh-issue-158790.F3kQ7z.rst +++ b/Misc/NEWS.d/next/Core_and_Builtins/2026-10-04-01-40-31.gh-issue-158790.F3kQ7z.rst @@ -1 +1 @@ -Speed up ``list.insert()`` and ``del list[i]`` in free-threaded builds by using ``memmove()`` to shift elements when the list is not shared. +Speed up :meth:`list.insert` and item deletion via :keyword:`del` in free-threaded builds by using :manpage:`memmove(3)`. diff --git a/Objects/listobject.c b/Objects/listobject.c index e61fb9b0bdf74fb..4eb3036d862cb70 100644 --- a/Objects/listobject.c +++ b/Objects/listobject.c @@ -509,8 +509,6 @@ ins1(PyListObject *self, Py_ssize_t where, PyObject *v) n - where); } #else - /* The compiler expands this loop inline. This is faster than a - memmove() call for short lists. */ for (Py_ssize_t i = n; --i >= where; ) FT_ATOMIC_STORE_PTR_RELEASE(items[i+1], items[i]); #endif