Skip to content

Commit 0ba5065

Browse files
chopratejasclaude
andauthored
Tejas/tool search deferral (headroomlabs-ai#1885)
## Description <!-- Briefly explain the change and why it is needed. --> Closes # ## Type of Change - [ ] Bug fix (non-breaking change that fixes an issue) - [ ] New feature (non-breaking change that adds functionality) - [ ] Breaking change (fix or feature that would cause existing functionality to change) - [ ] Documentation update - [ ] Performance improvement - [ ] Code refactoring (no functional changes) ## Changes Made - ## Testing <!-- Check what you actually ran, then paste the real command output below. --> - [ ] Unit tests pass (`pytest`) - [ ] Linting passes (`ruff check .`) - [ ] Type checking passes (`mypy headroom`) - [ ] New tests added for new functionality - [ ] Manual testing performed ### Test Output ```text # Paste relevant command output or artifact links here ``` ## Real Behavior Proof - Environment: - Exact command / steps: - Observed result: - Not tested: ## Review Readiness - [ ] I have performed a self-review - [ ] This PR is ready for human review ## Checklist - [ ] My code follows the project's style guidelines - [ ] I have performed a self-review of my code - [ ] I have commented my code, particularly in hard-to-understand areas - [ ] I have made corresponding changes to the documentation - [ ] My changes generate no new warnings - [ ] I have added tests that prove my fix is effective or that my feature works - [ ] New and existing unit tests pass locally with my changes - [ ] I have updated the CHANGELOG.md if applicable ## Screenshots (if applicable) Add screenshots to help explain your changes. ## Additional Notes <!-- Mention any N/A checklist items, tradeoffs, follow-ups, or maintainer context. --> --------- Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
1 parent 7c2f0ea commit 0ba5065

7 files changed

Lines changed: 668 additions & 118 deletions

File tree

‎headroom/proxy/handlers/anthropic.py‎

Lines changed: 90 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -38,6 +38,33 @@
3838
logger = logging.getLogger("headroom.proxy")
3939

4040

41+
def _strip_streaming_only_content_fields(messages: Any) -> None:
42+
"""Remove streaming-only ``index`` keys from request content blocks, in place.
43+
44+
``index`` is a field Anthropic emits on streaming RESPONSE content-block deltas
45+
(see proxy/handlers/streaming.py). It is not part of the request-message schema, so
46+
forwarding it upstream triggers a 400 ("...content.N.text.index: Extra inputs are
47+
not permitted") that aborts multi-turn sessions once a client echoes a reconstructed
48+
assistant turn back. Strip it (including nested tool_result content) so requests are
49+
always schema-valid.
50+
"""
51+
if not isinstance(messages, list):
52+
return
53+
for message in messages:
54+
if isinstance(message, dict):
55+
_strip_index_from_content_blocks(message.get("content"))
56+
57+
58+
def _strip_index_from_content_blocks(content: Any) -> None:
59+
if not isinstance(content, list):
60+
return
61+
for block in content:
62+
if isinstance(block, dict):
63+
block.pop("index", None)
64+
# tool_result blocks nest their own content list of blocks.
65+
_strip_index_from_content_blocks(block.get("content"))
66+
67+
4168
class AnthropicHandlerMixin:
4269
"""Mixin providing Anthropic API handler methods for HeadroomProxy."""
4370

@@ -680,6 +707,16 @@ async def _finalize_pre_upstream() -> None:
680707
body["model"] = model
681708
body_mutation_tracker.mark_mutated("sanitize_model_id")
682709
messages = body.get("messages", [])
710+
# Strip streaming-only "index" keys from request content blocks BEFORE any
711+
# prefix-cache tracking or compression. The proxy's streaming reconstruction
712+
# tags assistant blocks with an "index" for SSE re-emission; clients (e.g.
713+
# opencode) persist that assistant message and echo it back next turn, but
714+
# "index" is a response-delta field that Anthropic REJECTS in a request
715+
# ("messages.N.content.0.text.index: Extra inputs are not permitted", 400),
716+
# aborting multi-turn sessions. Canonicalizing here (in place, so body,
717+
# original, forwarded, and the recorded/replayed prefix are all identical)
718+
# keeps it cache-safe: overlay_cached_prefix replays the same stripped bytes.
719+
_strip_streaming_only_content_fields(messages)
683720
pipeline_provider = provider_name
684721
pipeline_path = request.url.path if upstream_base_url else "/v1/messages"
685722
pipeline_stream = bool(body.get("stream", False) or force_stream)
@@ -2044,6 +2081,14 @@ class _DeferredCompressionResult:
20442081
if remembered_event.headers is not None:
20452082
headers = remembered_event.headers
20462083

2084+
# Final sanitization of the FORWARDED body: strip streaming-only "index"
2085+
# keys from content blocks. In cache mode the forwarded prefix is replayed
2086+
# from Headroom's own recorded/reconstructed messages (streaming.py tags
2087+
# blocks with "index"), so the inbound-side strip above doesn't cover it —
2088+
# this catches both the client-echoed and the cache-replayed forms. Applied
2089+
# deterministically to the exact bytes forwarded (and thus recorded as the
2090+
# next prefix), so Anthropic always caches/matches the same stripped prefix.
2091+
_strip_streaming_only_content_fields(optimized_messages)
20472092
# Update body
20482093
body["messages"] = optimized_messages
20492094
if tools or _original_tools is not None:
@@ -2085,6 +2130,51 @@ class _DeferredCompressionResult:
20852130
optimized_tokens = tokenizer.count_messages(body["messages"])
20862131
tokens_saved = max(0, original_tokens - optimized_tokens)
20872132

2133+
# Server-side Tool Search (opt-in HEADROOM_TOOL_SEARCH): defer the
2134+
# non-core tool schemas behind a tool_search tool so Anthropic excludes
2135+
# them from the context window — they stop counting as input tokens until
2136+
# the model searches for one — while every tool stays callable.
2137+
# Deterministic → the tools prefix still prompt-caches. Safe for opencode:
2138+
# its @ai-sdk/anthropic parses the server_tool_use / tool_search_tool_result
2139+
# round-trip natively, and its cache breakpoints sit on messages (never
2140+
# tools), so defer_loading cannot collide with cache_control. We COUNT the
2141+
# deferred tool-schema tokens as the tool-search input-token saving (those
2142+
# bytes are excluded from context this turn); the response usage confirms it.
2143+
#
2144+
# FIRST-PARTY ANTHROPIC ONLY: the tool_search_tool_* type + defer_loading
2145+
# here use the first-party Claude API shape (GA, no beta header). Bedrock
2146+
# (``anthropic_backend``) and Vertex/gateway providers gate tool search
2147+
# differently, so scope the injection to provider "anthropic" over the
2148+
# direct API and leave those paths untouched.
2149+
if (
2150+
provider_name == "anthropic"
2151+
and getattr(self, "anthropic_backend", None) is None
2152+
and os.environ.get("HEADROOM_TOOL_SEARCH", "").strip().lower()
2153+
in ("1", "true", "yes", "on", "auto")
2154+
):
2155+
from headroom.proxy.helpers import inject_tool_search_deferral
2156+
2157+
_ts_before = body.get("tools")
2158+
_ts_after = inject_tool_search_deferral(_ts_before)
2159+
if _ts_after is not _ts_before:
2160+
_ts_deferred = [
2161+
t for t in _ts_after if isinstance(t, dict) and t.get("defer_loading")
2162+
]
2163+
try:
2164+
_ts_saved_tokens = tokenizer.count_text(
2165+
json.dumps(_ts_deferred, default=str)
2166+
)
2167+
except Exception:
2168+
_ts_saved_tokens = 0
2169+
body["tools"] = _ts_after
2170+
tools = _ts_after
2171+
tags["tool_search_deferred_tools"] = len(_ts_deferred)
2172+
tags["tool_search_deferred_tokens"] = _ts_saved_tokens
2173+
transforms_applied.append(
2174+
f"router:tool_search_deferral:{len(_ts_deferred)}tools:"
2175+
f"{_ts_saved_tokens}tok"
2176+
)
2177+
20882178
# Output shaping (opt-in via HEADROOM_OUTPUT_SHAPER): verbosity
20892179
# steering appended to the system-prompt tail + effort routing on
20902180
# mechanical tool_result continuations. Runs after every other

‎headroom/proxy/handlers/openai.py‎

Lines changed: 18 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1691,6 +1691,24 @@ def _add_timing(name: str, started_at: float) -> None:
16911691
tools_bytes_saved=tools_before_bytes - tools_after_bytes,
16921692
)
16931693

1694+
# Server-side Tool Search deferral (OpenAI Responses, gpt-5.4+): mark
1695+
# non-core function/MCP tools defer_loading + inject {"type": "tool_search"}
1696+
# so OpenAI keeps their heavy parameter schemas out of the model's context
1697+
# until searched (every tool stays callable, cache preserved). No-op for
1698+
# older models / small tool sets / clients already using tool search. The
1699+
# deferred defs still ride in the request body (OpenAI needs them to load
1700+
# on demand), so this is a provider-side context saving, not a request-byte
1701+
# one — hence a transform tag but no tokens_saved claim.
1702+
from headroom.proxy.helpers import inject_tool_search_deferral_openai
1703+
1704+
_deferred_tools = inject_tool_search_deferral_openai(working.get("tools"), model)
1705+
if _deferred_tools is not working.get("tools"):
1706+
if working is payload:
1707+
working = copy.deepcopy(payload)
1708+
working["tools"] = _deferred_tools
1709+
modified = True
1710+
transforms.append("openai:responses:tool_search_deferral")
1711+
16941712
live_units_started = time.perf_counter()
16951713
(
16961714
router_payload,

‎headroom/proxy/helpers.py‎

Lines changed: 214 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -14,6 +14,7 @@
1414
import logging
1515
import os
1616
import random
17+
import re
1718
import threading
1819
import time
1920
from collections import OrderedDict
@@ -3145,3 +3146,216 @@ def reset_tool_search_hint_state() -> None:
31453146
global _tool_search_hint_emitted
31463147
with _tool_search_hint_lock:
31473148
_tool_search_hint_emitted = False
3149+
3150+
3151+
# ---------------------------------------------------------------------------
3152+
# Server-side Tool Search injection (opencode / non-Claude-Code clients).
3153+
#
3154+
# Clients that eagerly materialize every tool schema (opencode ships ~135 tool
3155+
# defs ≈ 28k tokens on EVERY request) never opt into Anthropic's Tool Search
3156+
# Tool themselves. Unlike the Claude Code case above — where the schemas are
3157+
# already in the client's own context and the proxy can't reverse it — a plain
3158+
# API client's tools live only in the request body, so the proxy CAN defer them:
3159+
# mark the non-core tools ``defer_loading: true`` and inject a tool_search tool.
3160+
# Anthropic then excludes deferred tools from the context window (they stop
3161+
# counting as input tokens until the model searches for one), while every tool
3162+
# stays callable. Deterministic output → the tools prefix still prompt-caches.
3163+
# ---------------------------------------------------------------------------
3164+
3165+
# Core coding tools kept non-deferred so routine edit/read/run loops never pay a
3166+
# search round-trip. Everything else (Slack/Linear/Sentry/Notion/Snowflake/…) is
3167+
# deferred and loaded on demand. Anthropic recommends keeping the 3–5 (here a few
3168+
# more) most frequent tools resident.
3169+
_TOOL_SEARCH_CORE_TOOLS = frozenset(
3170+
{
3171+
"bash",
3172+
"bash_background",
3173+
"bash_background_output",
3174+
"bash_background_wait",
3175+
"bash_background_kill",
3176+
"read",
3177+
"write",
3178+
"edit",
3179+
"multiedit",
3180+
"apply_patch",
3181+
"glob",
3182+
"grep",
3183+
"task",
3184+
"todowrite",
3185+
"todoread",
3186+
"webfetch",
3187+
"question",
3188+
"skill",
3189+
}
3190+
)
3191+
_TOOL_SEARCH_DEFAULT_TYPE = "tool_search_tool_regex_20251119"
3192+
_TOOL_SEARCH_DEFAULT_NAME = "tool_search_tool_regex"
3193+
# Below this many tools the ~search round-trip isn't worth it (Anthropic's own
3194+
# guidance: standard calling is better under ~10 tools).
3195+
_TOOL_SEARCH_MIN_TOOLS = 12
3196+
3197+
3198+
def inject_tool_search_deferral(
3199+
tools: Any,
3200+
*,
3201+
core_tools: frozenset[str] = _TOOL_SEARCH_CORE_TOOLS,
3202+
search_type: str = _TOOL_SEARCH_DEFAULT_TYPE,
3203+
search_name: str = _TOOL_SEARCH_DEFAULT_NAME,
3204+
) -> Any:
3205+
"""Return a new ``tools`` list with non-core tools deferred + a search tool
3206+
injected, or the original list unchanged when injection doesn't apply.
3207+
3208+
No-op when: not a list, fewer than ``_TOOL_SEARCH_MIN_TOOLS``, a tool_search
3209+
tool is already present (client already defers), or nothing would be deferred.
3210+
3211+
Invariants enforced (else Anthropic 400s): the search tool is never deferred;
3212+
at least one tool stays non-deferred; a deferred tool never carries
3213+
``cache_control`` — if the client's tools cache breakpoint sat on a now-deferred
3214+
tool, it is moved to the last non-deferred real tool so the (smaller) tools
3215+
prefix still caches.
3216+
"""
3217+
if not isinstance(tools, list) or len(tools) < _TOOL_SEARCH_MIN_TOOLS:
3218+
return tools
3219+
for tool in tools:
3220+
if isinstance(tool, dict) and str(tool.get("type", "")).startswith(
3221+
_TOOL_SEARCH_TOOL_TYPE_PREFIX
3222+
):
3223+
return tools # client already uses tool search — leave it alone
3224+
3225+
search_tool = {"type": search_type, "name": search_name}
3226+
out: list[Any] = [search_tool]
3227+
deferred = 0
3228+
dropped_cache_control = False
3229+
last_resident_real: dict[str, Any] | None = None
3230+
resident_has_cache_control = False
3231+
3232+
for tool in tools:
3233+
if not isinstance(tool, dict) or tool.get("type") or tool.get("name") in core_tools:
3234+
# Non-dict, server/typed tools (web_search, computer, …), and core
3235+
# tools stay resident and unchanged.
3236+
out.append(tool)
3237+
if isinstance(tool, dict) and not tool.get("type"):
3238+
last_resident_real = tool
3239+
resident_has_cache_control = resident_has_cache_control or bool(
3240+
tool.get("cache_control")
3241+
)
3242+
continue
3243+
new_tool = dict(tool)
3244+
new_tool["defer_loading"] = True
3245+
if new_tool.pop("cache_control", None) is not None:
3246+
dropped_cache_control = True
3247+
out.append(new_tool)
3248+
deferred += 1
3249+
3250+
if deferred == 0:
3251+
return tools # nothing to defer → don't perturb the cache prefix
3252+
# Preserve a tools cache breakpoint: if we stripped cache_control off a
3253+
# deferred tool and no resident tool carries one, move it to the last
3254+
# resident real tool (never the search tool, to keep its shape canonical).
3255+
if dropped_cache_control and not resident_has_cache_control and last_resident_real is not None:
3256+
last_resident_real["cache_control"] = {"type": "ephemeral"}
3257+
return out
3258+
3259+
3260+
# ---------------------------------------------------------------------------
3261+
# Server-side Tool Search injection — OpenAI Responses API (gpt-5.4+).
3262+
#
3263+
# The OpenAI-side analogue of inject_tool_search_deferral above. OpenAI shipped
3264+
# the same idea for the Responses API on gpt-5.4+: mark a function/MCP tool
3265+
# ``defer_loading: true`` and add a ``{"type": "tool_search"}`` tool, and OpenAI
3266+
# keeps the deferred tools' heavy parameter schemas OUT of the model's context
3267+
# (only name+description remain) until the model searches for one — while every
3268+
# tool stays callable and the prompt cache is preserved. Same win as Anthropic
3269+
# (~15-25k tool-schema tokens -> ~200) for clients that ship a big tool surface
3270+
# and never opt into tool search themselves (opencode, plain API clients).
3271+
#
3272+
# Differences from the Anthropic path that require a separate function:
3273+
# * Responses function tools carry ``type: "function"`` (Anthropic real tools
3274+
# have no ``type``), so the resident/defer test is inverted — we defer
3275+
# ``function`` (non-core) and ``mcp`` tools and keep OTHER typed/hosted tools
3276+
# (web_search, file_search, code_interpreter, computer, image_generation, and
3277+
# the search tool itself) resident.
3278+
# * Model-gated: only gpt-5.4+ support it; older models 400 on the fields.
3279+
# * No ``cache_control`` (OpenAI caches automatically), so no breakpoint move.
3280+
# ---------------------------------------------------------------------------
3281+
3282+
_OPENAI_TOOL_SEARCH_TYPE = "tool_search"
3283+
_OPENAI_TOOL_SEARCH_MIN_TOOLS = 12
3284+
# gpt-5.4 is the first model with Responses tool_search (OpenAI docs). Version-
3285+
# gated by default; overridable per deployment via a regex in
3286+
# HEADROOM_OPENAI_TOOL_SEARCH_MODELS (matched against the model name) so new
3287+
# model families can be enabled without a code edit + release.
3288+
_OPENAI_TOOL_SEARCH_MIN_VERSION = (5, 4)
3289+
3290+
3291+
def _model_supports_openai_tool_search(model: str | None) -> bool:
3292+
"""True when an OpenAI model supports the Responses ``tool_search`` feature.
3293+
3294+
Default gate: ``gpt-<major>.<minor>`` >= 5.4. A regex in
3295+
``HEADROOM_OPENAI_TOOL_SEARCH_MODELS`` (matched against the model name) wins
3296+
when set; a malformed pattern falls back to the version gate rather than
3297+
crashing.
3298+
"""
3299+
if not model:
3300+
return False
3301+
override = os.environ.get("HEADROOM_OPENAI_TOOL_SEARCH_MODELS", "").strip()
3302+
if override:
3303+
try:
3304+
return re.search(override, model) is not None
3305+
except re.error:
3306+
pass # malformed override → fall back to the version gate
3307+
match = re.match(r"gpt-(\d+)(?:\.(\d+))?", model.strip().lower())
3308+
if not match:
3309+
return False
3310+
major, minor = int(match.group(1)), int(match.group(2) or 0)
3311+
return (major, minor) >= _OPENAI_TOOL_SEARCH_MIN_VERSION
3312+
3313+
3314+
def inject_tool_search_deferral_openai(
3315+
tools: Any,
3316+
model: str | None,
3317+
*,
3318+
core_tools: frozenset[str] = _TOOL_SEARCH_CORE_TOOLS,
3319+
) -> Any:
3320+
"""Return a new Responses ``tools`` list with non-core function/MCP tools
3321+
deferred + a ``{"type": "tool_search"}`` tool injected, or the original list
3322+
unchanged when injection doesn't apply.
3323+
3324+
No-op when: the model doesn't support tool search (gpt-5.4+ only), ``tools``
3325+
is not a list, there are fewer than ``_OPENAI_TOOL_SEARCH_MIN_TOOLS``, a
3326+
tool_search tool is already present (client already defers), or nothing would
3327+
be deferred. Core coding tools and hosted/typed tools (web_search,
3328+
file_search, code_interpreter, computer, …) stay resident and unchanged, so
3329+
routine edit/read/run loops never pay a search round-trip and the request
3330+
stays valid; the injected search tool is itself resident.
3331+
"""
3332+
if not _model_supports_openai_tool_search(model):
3333+
return tools
3334+
if not isinstance(tools, list) or len(tools) < _OPENAI_TOOL_SEARCH_MIN_TOOLS:
3335+
return tools
3336+
for tool in tools:
3337+
if isinstance(tool, dict) and tool.get("type") == _OPENAI_TOOL_SEARCH_TYPE:
3338+
return tools # client already uses tool search — leave it alone
3339+
3340+
out: list[Any] = [{"type": _OPENAI_TOOL_SEARCH_TYPE}]
3341+
deferred = 0
3342+
for tool in tools:
3343+
if not isinstance(tool, dict):
3344+
out.append(tool)
3345+
continue
3346+
ttype = tool.get("type")
3347+
# Deferrable: a non-core function, or an MCP server (OpenAI models are
3348+
# trained to search namespaces / MCP servers). Everything else — core
3349+
# coding tools and other hosted tools — stays resident.
3350+
deferrable = (ttype == "function" and tool.get("name") not in core_tools) or ttype == "mcp"
3351+
if deferrable and not tool.get("defer_loading"):
3352+
new_tool = dict(tool)
3353+
new_tool["defer_loading"] = True
3354+
out.append(new_tool)
3355+
deferred += 1
3356+
else:
3357+
out.append(tool)
3358+
3359+
if deferred == 0:
3360+
return tools # nothing to defer → don't perturb the request / cache prefix
3361+
return out

‎pyproject.toml‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -323,6 +323,8 @@ constraint-dependencies = [
323323
"gitpython>=3.1.50",
324324
# GHSA-f4xh-w4cj-qxq8 (High) — transitive via langchain-core; fix at 0.8.18
325325
"langsmith>=0.9.0",
326+
# CVE-2026-49825 (High, XSS) — transitive via lxml[html-clean]; fix at 0.4.5
327+
"lxml-html-clean>=0.4.5",
326328
]
327329

328330
# Pin the project's package index to public PyPI. Without this, `uv lock`

0 commit comments

Comments
 (0)