Description
When a value is yielded from an async generator and consumed via async for (or await agen.__anext__()), _PyAsyncGenASend_Send() (used by SEND_ASYNC_GEN and as am_send of the asend object since GH-148963) does the following for every value:
async_gen_asend_send() calls gen_send(), gets the wrapped value and passes it to async_gen_unwrap_value().
async_gen_unwrap_value() calls _PyGen_SetStopIterationValue(value): it allocates a StopIteration instance (including its args tuple) and sets it as the current exception.
_PyAsyncGenASend_Send() immediately calls _PyGen_FetchStopIterationValue(): it checks the exception type, takes the exception, reads .value and deallocates the exception.
The exception is never visible to Python code. It exists only to pass the value from one C function to its caller. With nested async generators this cost is paid at every nesting level for every value.
Profile
Benchmark: async_generators from pyperformance (recursive async generators over a binary tree of 100,000 nodes, nesting depth ~17).
Build: main at 7731da1, GCC 16.2.1, --with-tail-call-interp --enable-optimizations --with-lto, Linux x86-64.
Profiler: Perforator, 11,200 samples (cycles).
Inclusive time, recursion counted once:
| Function |
Share of cycles |
_PyGen_SetStopIterationValue |
12.2% |
_PyGen_FetchStopIterationValue (incl. .cold) |
12.6% |
StopIteration_dealloc |
4.3% |
In total, about 29% of the benchmark time is spent creating, raising, catching and destroying these StopIteration objects.
Proposed fix
- Split
async_gen_unwrap_value() into a new helper async_gen_unwrap_send() that returns a PySendResult. An async yield is reported as PYGEN_RETURN with the value itself, without creating an exception.
- Implement
_PyAsyncGenASend_Send() directly on top of this helper. async_gen_asend_send() (the Python-visible send() / __next__) becomes a thin wrapper that still raises StopIteration(value), so its behaviour is unchanged.
athrow() paths are unchanged.
- Additionally,
async_gen_asend_dealloc() now calls PyObject_CallFinalizerFromDealloc() only when ags_state == AWAITABLE_STATE_INIT, because the finalizer does nothing in any other state (it only warns about an asend that was never awaited). The effect of this part was not measured separately.
Patch
diff --git a/Objects/genobject.c b/Objects/genobject.c
index c313002c723..b575f860e7a 100644
--- a/Objects/genobject.c
+++ b/Objects/genobject.c
@@ -1963,8 +1963,11 @@ PyAsyncGen_New(PyFrameObject *f, PyObject *name, PyObject *qualname)
return (PyObject*)ag;
}
-static PyObject *
-async_gen_unwrap_value(PyAsyncGenObject *gen, PyObject *result)
+// Like async_gen_unwrap_value(), but an async yield is reported as
+// PYGEN_RETURN with the yielded value instead of raising StopIteration.
+static PySendResult
+async_gen_unwrap_send(PyAsyncGenObject *gen, PyObject *result,
+ PyObject **presult)
{
if (result == NULL) {
if (!PyErr_Occurred()) {
@@ -1977,17 +1980,31 @@ async_gen_unwrap_value(PyAsyncGenObject *gen, PyObject *result)
FT_ATOMIC_STORE_INT8_RELAXED(gen->ag_closed, 1);
}
- return NULL;
+ *presult = NULL;
+ return PYGEN_ERROR;
}
if (_PyAsyncGenWrappedValue_CheckExact(result)) {
/* async yield */
- _PyGen_SetStopIterationValue(((_PyAsyncGenWrappedValue*)result)->agw_val);
+ *presult = Py_NewRef(((_PyAsyncGenWrappedValue*)result)->agw_val);
Py_DECREF(result);
- return NULL;
+ return PYGEN_RETURN;
}
- return result;
+ *presult = result;
+ return PYGEN_NEXT;
+}
+
+static PyObject *
+async_gen_unwrap_value(PyAsyncGenObject *gen, PyObject *result)
+{
+ PyObject *value;
+ if (async_gen_unwrap_send(gen, result, &value) == PYGEN_RETURN) {
+ _PyGen_SetStopIterationValue(value);
+ Py_DECREF(value);
+ return NULL;
+ }
+ return value;
}
@@ -2000,7 +2017,10 @@ async_gen_asend_dealloc(PyObject *self)
assert(PyAsyncGenASend_CheckExact(self));
PyAsyncGenASend *ags = _PyAsyncGenASend_CAST(self);
- if (PyObject_CallFinalizerFromDealloc(self)) {
+ // The finalizer only warns about an asend() that was never awaited.
+ if (ags->ags_state == AWAITABLE_STATE_INIT
+ && PyObject_CallFinalizerFromDealloc(self))
+ {
return;
}
@@ -2023,18 +2043,19 @@ async_gen_asend_traverse(PyObject *self, visitproc visit, void *arg)
}
-static PyObject *
-async_gen_asend_send(PyObject *self, PyObject *arg)
+PySendResult
+_PyAsyncGenASend_Send(PyObject *self, PyObject *arg, PyObject **presult)
{
PyAsyncGenASend *o = _PyAsyncGenASend_CAST(self);
+ *presult = NULL;
int8_t state = FT_ATOMIC_LOAD_INT8_RELAXED(o->ags_state);
do {
if (state == AWAITABLE_STATE_CLOSED) {
PyErr_SetString(
PyExc_RuntimeError,
"cannot reuse already awaited __anext__()/asend()");
- return NULL;
+ return PYGEN_ERROR;
}
if (state == AWAITABLE_STATE_ITER) {
goto do_send;
@@ -2051,7 +2072,7 @@ async_gen_asend_send(PyObject *self, PyObject *arg)
PyErr_SetString(
PyExc_RuntimeError,
"anext(): asynchronous generator is already running");
- return NULL;
+ return PYGEN_ERROR;
}
if (arg == NULL || arg == Py_None) {
@@ -2059,29 +2080,29 @@ async_gen_asend_send(PyObject *self, PyObject *arg)
}
PyObject *result;
+ PySendResult res;
do_send:
result = gen_send((PyObject*)o->ags_gen, arg);
- result = async_gen_unwrap_value(o->ags_gen, result);
+ res = async_gen_unwrap_send(o->ags_gen, result, presult);
- if (result == NULL) {
+ if (res != PYGEN_NEXT) {
FT_ATOMIC_STORE_INT8_RELAXED(o->ags_state, AWAITABLE_STATE_CLOSED);
FT_ATOMIC_STORE_INT8_RELEASE(o->ags_gen->ag_running_async, 0);
}
- return result;
+ return res;
}
-PySendResult
-_PyAsyncGenASend_Send(PyObject *iter, PyObject *arg, PyObject **result)
+static PyObject *
+async_gen_asend_send(PyObject *self, PyObject *arg)
{
- *result = async_gen_asend_send(iter, arg);
- if (*result != NULL) {
- return PYGEN_NEXT;
- }
- if (_PyGen_FetchStopIterationValue(result) == 0) {
- return PYGEN_RETURN;
+ PyObject *result;
+ if (_PyAsyncGenASend_Send(self, arg, &result) == PYGEN_RETURN) {
+ _PyGen_SetStopIterationValue(result);
+ Py_DECREF(result);
+ return NULL;
}
- return PYGEN_ERROR;
+ return result;
}
Results
async_generators, same commit, only difference is the patch:
| Setup |
Before |
After |
Speedup |
| GCC tail-call, PGO+LTO, full pyperformance run |
585 ms |
440 ms |
1.33x |
GCC tail-call, PGO+LTO, async benchmarks, taskset to 4 cores |
625 ms |
444 ms |
1.41x |
| GCC tail-call, no PGO/LTO, pyperf |
552 ms |
384 ms |
1.44x |
2-vCPU VM, GCC 16.2.0 tail-call, no PGO/LTO, --fast |
807 ms |
527 ms |
1.53x |
Other benchmarks: differences are within the noise of the machines used. The same benchmark changed sign between runs (e.g. asyncio_tcp was faster with the patch in one run and slower in another), so they are not attributed to the patch.
Tests: the full test suite (./python -m test -j8) passes on the patched PGO+LTO build: 491 test files, same result as the unpatched build.
Linked PRs
Description
When a value is yielded from an async generator and consumed via
async for(orawait agen.__anext__()),_PyAsyncGenASend_Send()(used bySEND_ASYNC_GENand asam_sendof the asend object since GH-148963) does the following for every value:async_gen_asend_send()callsgen_send(), gets the wrapped value and passes it toasync_gen_unwrap_value().async_gen_unwrap_value()calls_PyGen_SetStopIterationValue(value): it allocates aStopIterationinstance (including itsargstuple) and sets it as the current exception._PyAsyncGenASend_Send()immediately calls_PyGen_FetchStopIterationValue(): it checks the exception type, takes the exception, reads.valueand deallocates the exception.The exception is never visible to Python code. It exists only to pass the value from one C function to its caller. With nested async generators this cost is paid at every nesting level for every value.
Profile
Benchmark:
async_generatorsfrom pyperformance (recursive async generators over a binary tree of 100,000 nodes, nesting depth ~17).Build:
mainat 7731da1, GCC 16.2.1,--with-tail-call-interp --enable-optimizations --with-lto, Linux x86-64.Profiler: Perforator, 11,200 samples (cycles).
Inclusive time, recursion counted once:
_PyGen_SetStopIterationValue_PyGen_FetchStopIterationValue(incl..cold)StopIteration_deallocIn total, about 29% of the benchmark time is spent creating, raising, catching and destroying these
StopIterationobjects.Proposed fix
async_gen_unwrap_value()into a new helperasync_gen_unwrap_send()that returns aPySendResult. An async yield is reported asPYGEN_RETURNwith the value itself, without creating an exception._PyAsyncGenASend_Send()directly on top of this helper.async_gen_asend_send()(the Python-visiblesend()/__next__) becomes a thin wrapper that still raisesStopIteration(value), so its behaviour is unchanged.athrow()paths are unchanged.async_gen_asend_dealloc()now callsPyObject_CallFinalizerFromDealloc()only whenags_state == AWAITABLE_STATE_INIT, because the finalizer does nothing in any other state (it only warns about an asend that was never awaited). The effect of this part was not measured separately.Patch
Results
async_generators, same commit, only difference is the patch:tasksetto 4 cores--fastOther benchmarks: differences are within the noise of the machines used. The same benchmark changed sign between runs (e.g.
asyncio_tcpwas faster with the patch in one run and slower in another), so they are not attributed to the patch.Tests: the full test suite (
./python -m test -j8) passes on the patched PGO+LTO build: 491 test files, same result as the unpatched build.Linked PRs