@@ -128,6 +128,134 @@ def _strip_cache_control(obj: Any) -> Any:
128128 return obj
129129
130130
131+ # Keys that carry NO semantic payload for the model — transport / caching-directive
132+ # / telemetry / client-routing annotations that clients attach and vary turn-to-turn.
133+ # Grounded in provider API docs (Anthropic Messages, OpenAI Chat+Responses, Bedrock
134+ # Converse) + client-library field inventories (litellm, Vercel AI SDK, opencode,
135+ # Claude Code, Cline). Dropped from the cross-turn prefix-equality key ONLY.
136+ #
137+ # NOTE ON SAFETY: this projection is a COMPARISON KEY, never a source to rebuild
138+ # forwarded bytes — the cache-stable-delta path always forwards the previously
139+ # forwarded bytes + the raw appended delta. So dropping these can't deprive the
140+ # model. What we must NOT do is drop a *semantic* field (that would mask a real
141+ # divergence and replay a stale prefix), which is why: (1) reasoning SIGNATURES are
142+ # NOT in this set (Anthropic 400s if a thinking block is altered/missing, and a
143+ # present/absent flip is a real divergence we want to detect); (2) tool inputs /
144+ # arguments / json payloads are treated as OPAQUE and compared verbatim (see
145+ # _OPAQUE_PAYLOAD_KEYS) so a user key that happens to be named "index"/"state" is
146+ # never stripped from inside a tool call.
147+ _NON_SEMANTIC_KEYS = frozenset (
148+ {
149+ # cache-breakpoint markers (moved to the newest block every turn)
150+ "cache_control" , # Anthropic (per-block)
151+ "cachePoint" , # Bedrock (per-block content block)
152+ # litellm unified-message / tool annotations
153+ "caller" , # litellm programmatic-tool tag on tool_use
154+ "provider_specific_fields" ,
155+ "reasoning_content" , # litellm display echo (the paired signature is separate)
156+ "reasoning_items" ,
157+ "annotations" , # citation/display metadata
158+ # OpenAI response echoes that can ride on assistant messages
159+ "system_fingerprint" ,
160+ "service_tier" ,
161+ # Vercel AI SDK / opencode part transport
162+ "providerMetadata" ,
163+ "providerOptions" ,
164+ "callProviderMetadata" ,
165+ "state" ,
166+ "providerExecuted" ,
167+ "synthetic" ,
168+ "ignored" ,
169+ # streaming-assembly artifact
170+ "index" ,
171+ }
172+ )
173+
174+ # Values under these keys are opaque semantic payloads (tool-call input, OpenAI
175+ # stringified arguments, Bedrock tool_result json). They are compared VERBATIM — we
176+ # never recurse into them to strip "noise" keys, because arbitrary user data there
177+ # may legitimately contain keys that collide with _NON_SEMANTIC_KEYS (e.g. an
178+ # `input` of {"state": "CA", "index": 3}). Recursing would corrupt the comparison.
179+ _OPAQUE_PAYLOAD_KEYS = frozenset ({"input" , "arguments" , "json" })
180+
181+
182+ def _canonicalize_for_prefix_compare (obj : Any ) -> Any :
183+ """Representation-agnostic canonical form for cross-turn prefix equality.
184+
185+ Providers accept several *equivalent* encodings for the same message, and real
186+ clients vary them turn-to-turn; a raw-dict prefix compare then fails spuriously
187+ and drops cache mode to raw (uncompressed) forwarding. This normalizes ONLY
188+ representation:
189+ * drops non-semantic annotation / cache-directive / telemetry keys
190+ (_NON_SEMANTIC_KEYS) at any message/block level;
191+ * wraps a bare string ``content`` into ``[{"type": "text", "text": ...}]``
192+ (Anthropic's string sugar, which litellm flips per turn);
193+ * leaves tool ``input`` / ``arguments`` / ``json`` payloads verbatim
194+ (_OPAQUE_PAYLOAD_KEYS) so user data is never corrupted;
195+ * KEEPS all real content (text, tool name/input, tool_result content, reasoning
196+ signatures, ids) so two messages canonicalize-equal iff they are semantically
197+ identical.
198+
199+ Used ONLY as a comparison key for the cache-stable delta path; the original,
200+ unmodified messages are always what gets forwarded.
201+ """
202+ if isinstance (obj , dict ):
203+ out : dict [str , Any ] = {}
204+ for key , value in obj .items ():
205+ if key in _NON_SEMANTIC_KEYS :
206+ continue
207+ if key in _OPAQUE_PAYLOAD_KEYS :
208+ out [key ] = value # verbatim — do not recurse into user payloads
209+ elif key == "content" and isinstance (value , str ):
210+ out [key ] = [{"type" : "text" , "text" : value }]
211+ else :
212+ out [key ] = _canonicalize_for_prefix_compare (value )
213+ return out
214+ if isinstance (obj , list ):
215+ canon = [_canonicalize_for_prefix_compare (value ) for value in obj ]
216+ # Drop blocks that projected to {} — a pure cache-directive content block
217+ # (e.g. Bedrock {"cachePoint": {...}}) whose only key was non-semantic. Left
218+ # in place it would be an empty-dict entry, so a directive block moving
219+ # position across turns would spuriously fail the length/order compare.
220+ return [value for value in canon if value != {}]
221+ return obj
222+
223+
224+ def extract_cache_stable_delta (
225+ current_messages : list [dict [str , Any ]],
226+ previous_original_messages : list [dict [str , Any ]] | None ,
227+ previous_forwarded_messages : list [dict [str , Any ]] | None ,
228+ ) -> tuple [list [dict [str , Any ]], list [dict [str , Any ]]] | None :
229+ """Return ``(stable_forwarded_prefix, appended_delta_messages)`` when the current
230+ request append-only-extends the previous one, else ``None``.
231+
232+ Provider-agnostic delta engine for cache mode. "Append-only" is decided by comparing
233+ the *canonicalized* prefix (:func:`_canonicalize_for_prefix_compare`, which ignores
234+ per-turn transport / cache-directive / client-annotation noise across
235+ Anthropic / OpenAI / Bedrock and the common clients), so a moved cache marker or
236+ shape churn does not spuriously collapse cache mode to raw forwarding. On a match the
237+ caller replays the byte-identical previously-forwarded prefix and compresses ONLY the
238+ appended delta.
239+
240+ This is a COMPARISON + slice only: the returned prefix is the previously-forwarded
241+ bytes verbatim and the delta is the raw appended messages — never a rebuild from the
242+ canonical projection — so the projection dropping non-semantic fields is safe.
243+ """
244+ if not previous_original_messages or previous_forwarded_messages is None :
245+ return None
246+ prefix_len = len (previous_original_messages )
247+ if len (current_messages ) < prefix_len :
248+ return None
249+ if _canonicalize_for_prefix_compare (
250+ current_messages [:prefix_len ]
251+ ) != _canonicalize_for_prefix_compare (previous_original_messages ):
252+ return None
253+ return (
254+ copy .deepcopy (previous_forwarded_messages ),
255+ copy .deepcopy (current_messages [prefix_len :]),
256+ )
257+
258+
131259def overlay_cached_prefix (
132260 optimized_messages : list [dict [str , Any ]],
133261 current_original_messages : list [dict [str , Any ]],
@@ -168,12 +296,17 @@ def overlay_cached_prefix(
168296 if len (current_original_messages ) < n or len (optimized_messages ) < n :
169297 return optimized_messages
170298 # Append-only guard on CONTENT ONLY: the frozen region must be the same
171- # messages we cached. Compare with cache_control stripped — clients move that
172- # breakpoint to the newest message each turn, so a raw dict compare would
173- # spuriously fail whenever a marker lands in the frozen prefix, skip the
174- # replay, and bust the cache (the residual busts observed after the first
175- # fix). Content stability is what the provider's prefix cache actually keys on.
176- if _strip_cache_control (current_original_messages [:n ]) != _strip_cache_control (prev_orig ):
299+ # messages we cached. Compare with the shared canonicalizer (not just
300+ # cache_control-stripping) so the guard is robust to ALL per-turn transport /
301+ # annotation churn — cache_control movement (Anthropic), litellm `caller`,
302+ # provider_specific_fields, streaming `index`, string<->block content shape,
303+ # etc. — across providers/clients. Content stability is what the provider's
304+ # prefix cache actually keys on; a coarser cache_control-only strip let other
305+ # clients' noise spuriously fail the guard, skip the replay, and bust. This
306+ # helps every handler that shares overlay_cached_prefix (Anthropic + OpenAI).
307+ if _canonicalize_for_prefix_compare (
308+ current_original_messages [:n ]
309+ ) != _canonicalize_for_prefix_compare (prev_orig ):
177310 return optimized_messages
178311 # Replay the cached (compressed) prefix byte-identical; keep this turn's tail.
179312 return list (prev_fwd ) + list (optimized_messages [n :])
@@ -563,6 +696,22 @@ def _estimate_message_tokens(messages: list[dict[str, Any]]) -> list[int]:
563696 chars += len (text )
564697 else :
565698 chars = 0
699+ # OpenAI function-calling: the assistant's command lives in the
700+ # top-level `tool_calls` (or legacy `function_call`) field, NOT in
701+ # `content` (which is empty/None on a tool-call turn). Anthropic puts
702+ # the equivalent in a `tool_use` content BLOCK (counted above), but
703+ # the OpenAI shape was never counted here. That under-counted every
704+ # tool-based assistant turn to ~0, so the frozen-prefix estimate
705+ # overshot the real cache boundary and froze the NEWEST delta — which
706+ # is why OpenAI/Kimi (fireworks) tool harnesses got ~zero compression
707+ # while text/back-tick harnesses (command in `content`) compressed.
708+ for tc in msg .get ("tool_calls" ) or []:
709+ if isinstance (tc , dict ):
710+ fn = tc .get ("function" ) or {}
711+ chars += len (str (fn .get ("name" , "" ))) + len (str (fn .get ("arguments" , "" )))
712+ fc = msg .get ("function_call" )
713+ if isinstance (fc , dict ):
714+ chars += len (str (fc .get ("name" , "" ))) + len (str (fc .get ("arguments" , "" )))
566715 # Add overhead for role, block structure, etc.
567716 chars += 20
568717 counts .append (max (1 , int (chars / 3.5 )))
0 commit comments