Skip to content

Commit bd5cff4

Browse files
fix(gemini): declare cache reporting as inclusive on generations (#860)
* fix(gemini): declare cache reporting as inclusive on generations Gemini counts `cached_content_token_count` inside `prompt_token_count`, but the SDK never said so, leaving ingestion to infer the accounting model from the token counts alone. That inference is unreliable here. Under explicit context caching the two counts come from separate measurements, the cache object at creation time and the prompt per request, so they can disagree by a few percent and the cache pool can land just above the input total. Set `cache_reporting_exclusive` to False on generations that report cache reads, and carry it through the streaming merge and both capture paths so ingestion prices cached tokens from the declared value. Generated-By: PostHog Code Task-Id: 06160e48-feb9-4d39-8b7b-3dcfd1d9ca24 * chore(ai): regenerate public API snapshot for TokenUsage field Generated-By: PostHog Code Task-Id: 06160e48-feb9-4d39-8b7b-3dcfd1d9ca24
1 parent 204522c commit bd5cff4

6 files changed

Lines changed: 41 additions & 0 deletions

File tree

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,5 @@
1+
---
2+
pypi/posthog: patch
3+
---
4+
5+
fix: declare Gemini's cache accounting model on generations with cache reads, so ingestion prices cached tokens from `$ai_cache_reporting_exclusive` instead of inferring it from the token counts.

‎posthog/ai/gemini/gemini_converter.py‎

Lines changed: 6 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -513,6 +513,12 @@ def _extract_usage_from_metadata(metadata: Any) -> TokenUsage:
513513
cache_tokens = metadata.cached_content_token_count
514514
if cache_tokens and cache_tokens > 0:
515515
usage["cache_read_input_tokens"] = cache_tokens
516+
# Gemini counts cached_content_token_count inside prompt_token_count, so
517+
# declare the accounting model rather than leaving ingestion to infer it.
518+
# Under explicit context caching the two counts come from separate
519+
# measurements and can disagree by a few percent, which makes inference
520+
# from the token counts alone unreliable.
521+
usage["cache_reporting_exclusive"] = False
516522

517523
# Add reasoning tokens if present (don't add if 0)
518524
if hasattr(metadata, "thoughts_token_count"):

‎posthog/ai/types.py‎

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -73,6 +73,11 @@ class TokenUsage(TypedDict, total=False):
7373
cache_creation_input_tokens: Optional[int]
7474
reasoning_tokens: Optional[int]
7575
web_search_count: Optional[int]
76+
# Whether cache tokens are counted separately from input_tokens. Providers that
77+
# report them as a subset of input_tokens set this False. Left unset when the
78+
# provider's accounting model is not known, in which case ingestion infers it
79+
# from the token counts.
80+
cache_reporting_exclusive: Optional[bool]
7681
raw_usage: Optional[Any] # Raw provider usage metadata for backend processing
7782

7883

‎posthog/ai/utils.py‎

Lines changed: 22 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -172,6 +172,12 @@ def merge_usage_stats(
172172
current = target.get("web_search_count") or 0
173173
target["web_search_count"] = max(current, source_web_search)
174174

175+
# Carried across, never accumulated: this describes the provider's accounting
176+
# model, so it is the same on every chunk.
177+
source_exclusive = source.get("cache_reporting_exclusive")
178+
if source_exclusive is not None:
179+
target["cache_reporting_exclusive"] = source_exclusive
180+
175181
# Merge raw_usage to avoid losing data from earlier events
176182
# For Anthropic streaming: message_start has input tokens, message_delta has output
177183
# Note: raw_usage is already serialized by converters, so it's a dict
@@ -199,6 +205,8 @@ def merge_usage_stats(
199205
target["reasoning_tokens"] = source["reasoning_tokens"]
200206
if source.get("web_search_count") is not None:
201207
target["web_search_count"] = source["web_search_count"]
208+
if source.get("cache_reporting_exclusive") is not None:
209+
target["cache_reporting_exclusive"] = source["cache_reporting_exclusive"]
202210
# Note: raw_usage is already serialized by converters, so it's a dict
203211
if source.get("raw_usage") is not None:
204212
target["raw_usage"] = source["raw_usage"]
@@ -482,6 +490,10 @@ def call_llm_and_track_usage(
482490
if cache_creation is not None and cache_creation > 0:
483491
tag("$ai_cache_creation_input_tokens", cache_creation)
484492

493+
cache_reporting_exclusive = usage.get("cache_reporting_exclusive")
494+
if cache_reporting_exclusive is not None:
495+
tag("$ai_cache_reporting_exclusive", cache_reporting_exclusive)
496+
485497
reasoning = usage.get("reasoning_tokens")
486498
if reasoning is not None and reasoning > 0:
487499
tag("$ai_reasoning_tokens", reasoning)
@@ -635,6 +647,10 @@ async def call_llm_and_track_usage_async(
635647
if cache_creation is not None and cache_creation > 0:
636648
tag("$ai_cache_creation_input_tokens", cache_creation)
637649

650+
cache_reporting_exclusive = usage.get("cache_reporting_exclusive")
651+
if cache_reporting_exclusive is not None:
652+
tag("$ai_cache_reporting_exclusive", cache_reporting_exclusive)
653+
638654
reasoning = usage.get("reasoning_tokens")
639655
if reasoning is not None and reasoning > 0:
640656
tag("$ai_reasoning_tokens", reasoning)
@@ -794,6 +810,12 @@ def capture_streaming_event(
794810
if value is not None and isinstance(value, int) and value > 0:
795811
event_properties[f"$ai_{field}"] = value
796812

813+
cache_reporting_exclusive = event_data["usage_stats"].get(
814+
"cache_reporting_exclusive"
815+
)
816+
if cache_reporting_exclusive is not None:
817+
event_properties["$ai_cache_reporting_exclusive"] = cache_reporting_exclusive
818+
797819
# Add web search count if present (all providers)
798820
web_search_count = event_data["usage_stats"].get("web_search_count")
799821
if (

‎posthog/test/ai/gemini/test_gemini.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -836,6 +836,7 @@ def test_cache_and_reasoning_tokens(mock_client, mock_google_genai_client):
836836
assert props["$ai_output_tokens"] == 50
837837
assert props["$ai_cache_read_input_tokens"] == 30
838838
assert props["$ai_reasoning_tokens"] == 10
839+
assert props["$ai_cache_reporting_exclusive"] is False
839840

840841

841842
def test_streaming_cache_and_reasoning_tokens(mock_client, mock_google_genai_client):
@@ -899,6 +900,7 @@ def test_streaming_cache_and_reasoning_tokens(mock_client, mock_google_genai_cli
899900
assert props["$ai_output_tokens"] == 10
900901
assert props["$ai_cache_read_input_tokens"] == 30
901902
assert props["$ai_reasoning_tokens"] == 5
903+
assert props["$ai_cache_reporting_exclusive"] is False
902904

903905
# Verify raw usage is captured in streaming mode (merged from chunks)
904906
assert "$ai_usage" in props

‎references/public_api_snapshot.txt‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -487,6 +487,7 @@ attribute posthog.ai.types.StreamingEventData.trace_id: Optional[str]
487487
attribute posthog.ai.types.StreamingEventData.usage_stats: TokenUsage
488488
attribute posthog.ai.types.TokenUsage.cache_creation_input_tokens: Optional[int]
489489
attribute posthog.ai.types.TokenUsage.cache_read_input_tokens: Optional[int]
490+
attribute posthog.ai.types.TokenUsage.cache_reporting_exclusive: Optional[bool]
490491
attribute posthog.ai.types.TokenUsage.input_tokens: int
491492
attribute posthog.ai.types.TokenUsage.output_tokens: int
492493
attribute posthog.ai.types.TokenUsage.raw_usage: Optional[Any]

0 commit comments

Comments
 (0)