Conversation
Benchmark
Scriptimport pyperf
INNER = 100_000
RANGE = range(INNER)
def _dealloc_only(loops, make):
total = 0.0
for _ in range(loops):
objs = [make(i) for i in RANGE]
t0 = pyperf.perf_counter()
objs.clear()
total += pyperf.perf_counter() - t0
return total
def dealloc_only_float(loops):
return _dealloc_only(loops, lambda i: i * 1.5)
def dealloc_only_bigint(loops):
big = 2 ** 70
return _dealloc_only(loops, lambda i: i + big)
def dealloc_only_str(loops):
return _dealloc_only(loops, str)
def dealloc_only_bytes(loops):
return _dealloc_only(loops, lambda i: bytes(3))
def alloc_dealloc_float(loops):
t0 = pyperf.perf_counter()
for _ in range(loops):
for i in RANGE:
x = i * 1.5
return pyperf.perf_counter() - t0
def alloc_dealloc_bigint(loops):
big = 2 ** 70
t0 = pyperf.perf_counter()
for _ in range(loops):
for i in RANGE:
x = i + big
return pyperf.perf_counter() - t0
def alloc_dealloc_str(loops):
t0 = pyperf.perf_counter()
for _ in range(loops):
for i in RANGE:
x = str(i)
return pyperf.perf_counter() - t0
def alloc_dealloc_bytes(loops):
t0 = pyperf.perf_counter()
for _ in range(loops):
for i in RANGE:
x = bytes(3)
return pyperf.perf_counter() - t0
runner = pyperf.Runner()
for func in (dealloc_only_float, dealloc_only_bigint, dealloc_only_str,
dealloc_only_bytes, alloc_dealloc_float, alloc_dealloc_bigint,
alloc_dealloc_str, alloc_dealloc_bytes):
runner.bench_time_func(func.__name__, func, inner_loops=INNER) |
|
cc @markshannon @vstinner who recently touched this function. |
vstinner
left a comment
There was a problem hiding this comment.
Is Py_NO_INLINE really useful here? Why not just a fast-path at the _Py_Dealloc() entry point? Usually, I prefer to let the compiler decides how to inline or not.
Something like that:
diff --git a/Objects/object.c b/Objects/object.c
index c7aeba0cee2..b30332f5945 100644
--- a/Objects/object.c
+++ b/Objects/object.c
@@ -3312,6 +3312,13 @@ _Py_Dealloc(PyObject *op)
PyTypeObject *type = Py_TYPE(op);
unsigned long gc_flag = type->tp_flags & Py_TPFLAGS_HAVE_GC;
destructor dealloc = type->tp_dealloc;
+#if !defined(Py_DEBUG) && !defined(Py_TRACE_REFS)
+ if (!gc_flag && _PyRuntime.ref_tracer.tracer_func == NULL) {
+ dealloc(op);
+ return;
+ }
+#endif
+
PyThreadState *tstate = NULL;
intptr_t margin = 0;
if (gc_flag) {
Let me check with the benchmark. |
|
There is also
|
The suggested version is also faster, but with the GC path in the same function, Clang still saves the callee-saved registers even on the non-GC path, which never uses them. The current version avoids this by moving the GC path into a non-inlined function, so _Py_Dealloc doesn't need to save those registers (compiler magic...). x86-64, suggested version: x86-64, this PR: AArch64, suggested version: AArch64, this PR:
|
I will handle this in a separate PR :) |
Okay, I 've verified this at the PGO + LTO build, and the result is the same. I would like to keep the current version. |
| stack is shallower */ | ||
| void | ||
| _Py_Dealloc(PyObject *op) | ||
| static Py_NO_INLINE void |
There was a problem hiding this comment.
Please explain in the comment the rationale for Py_NO_INLINE.
| } | ||
| } | ||
|
|
||
| void |
There was a problem hiding this comment.
Or you may explain here the dealloc_general() split with Py_NO_INLINE.
| void | ||
| _Py_Dealloc(PyObject *op) | ||
| static Py_NO_INLINE void | ||
| dealloc_general(PyObject *op) |
There was a problem hiding this comment.
I would prefer to rename the function to "py_dealloc()".
_Py_Deallocto non-inlined function call #130706