Skip to content

Commit a451080

Browse files
committed
fix(traces): keep backing off after a refused batch is dropped
Dropping the head batch after eight windows reset the failure count, so the next batch went straight to an endpoint that had just failed eight times and the depth trigger came back on. The backoff ceiling is now the events lane's single constant, and FakeTimer refuses to fire a timer the code cancelled.
1 parent eff1be7 commit a451080

3 files changed

Lines changed: 23 additions & 4 deletions

File tree

‎posthog/test/tracing/helpers.py‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -41,6 +41,7 @@ def cancel(self):
4141
self.cancelled = True
4242

4343
def fire(self):
44+
assert not self.cancelled, "fired a timer the code cancelled"
4445
self.fn()
4546

4647

‎posthog/test/tracing/test_export.py‎

Lines changed: 15 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -415,6 +415,21 @@ def test_drops_a_batch_the_endpoint_keeps_refusing_and_moves_to_the_next_one(
415415
assert [[s["name"] for s in b] for b in sender.batches()][-1] == ["next"]
416416
assert queued(pipeline) == []
417417

418+
def test_a_dropped_batch_keeps_the_backoff_for_the_next_one(self, clock):
419+
sender = FakeSender(*([SendOutcome("retry-later")] * MAX_RETRIES_PER_BATCH))
420+
pipeline, _, _ = make_traces(sender=sender, max_export_batch_size=1)
421+
pipeline.start_span("stuck").end()
422+
pipeline.start_span("next").end()
423+
for _ in range(MAX_RETRIES_PER_BATCH):
424+
pipeline.flush()
425+
clock["now"] += 100
426+
exporter = pipeline._exporter
427+
assert [s.name for s in queued(pipeline)] == ["next"]
428+
assert exporter._consecutive_failures >= MAX_RETRIES_PER_BATCH
429+
assert exporter._head_batch_failures == 1
430+
pipeline.start_span("c").end()
431+
assert FakeTimer.instances[-1].delay > 0
432+
418433
def test_charges_the_budget_once_per_backoff_window_not_per_attempt(self, clock):
419434
pipeline, _, _ = make_traces(sender=FakeSender(SendOutcome("retry-later")))
420435
pipeline.start_span("a").end()

‎posthog/tracing/_export.py‎

Lines changed: 7 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -12,6 +12,7 @@
1212
import time
1313
from typing import Any, Callable, List, Optional, Tuple
1414

15+
from ..capture_v1 import _MAX_BACKOFF_SECONDS
1516
from ._config import ResolvedTracesConfig
1617
from ._drops import DropLog
1718
from ._otlp import (
@@ -27,11 +28,12 @@
2728
MAX_RETRIES_PER_BATCH = 8
2829

2930
MAX_FLUSH_BACKOFF_EXPONENT = 6
30-
MAX_FLUSH_BACKOFF_SECONDS = 30.0
31+
# The events lane's ceiling, so both backoffs and the Retry-After clamp share one.
32+
MAX_FLUSH_BACKOFF_SECONDS = float(_MAX_BACKOFF_SECONDS)
3133

3234
# Nothing upstream bounds the header, and an unbounded value would strand the
33-
# queue. The same ceiling as the SDK's own backoff.
34-
MAX_RETRY_AFTER_SECONDS = 30.0
35+
# queue.
36+
MAX_RETRY_AFTER_SECONDS = MAX_FLUSH_BACKOFF_SECONDS
3537

3638
# Spread each backoff by up to a quarter, so clients refused together do not
3739
# return together.
@@ -348,7 +350,8 @@ def _apply_outcome_locked(
348350
if self._head_batch_failures < MAX_RETRIES_PER_BATCH:
349351
return 0, True, "Span export failed; retrying on the next flush"
350352
del self._queue[:size]
351-
self._end_failure_sequence_locked()
353+
# The endpoint is still failing: the next batch keeps backing off.
354+
self._reset_head_batch_budget_locked()
352355
self._drops.record(
353356
size,
354357
"the ingestion endpoint failed {} times in a row".format(

0 commit comments

Comments
 (0)