|
14 | 14 | import logging |
15 | 15 | import os |
16 | 16 | import random |
| 17 | +import re |
17 | 18 | import threading |
18 | 19 | import time |
19 | 20 | from collections import OrderedDict |
@@ -3145,3 +3146,216 @@ def reset_tool_search_hint_state() -> None: |
3145 | 3146 | global _tool_search_hint_emitted |
3146 | 3147 | with _tool_search_hint_lock: |
3147 | 3148 | _tool_search_hint_emitted = False |
| 3149 | + |
| 3150 | + |
| 3151 | +# --------------------------------------------------------------------------- |
| 3152 | +# Server-side Tool Search injection (opencode / non-Claude-Code clients). |
| 3153 | +# |
| 3154 | +# Clients that eagerly materialize every tool schema (opencode ships ~135 tool |
| 3155 | +# defs ≈ 28k tokens on EVERY request) never opt into Anthropic's Tool Search |
| 3156 | +# Tool themselves. Unlike the Claude Code case above — where the schemas are |
| 3157 | +# already in the client's own context and the proxy can't reverse it — a plain |
| 3158 | +# API client's tools live only in the request body, so the proxy CAN defer them: |
| 3159 | +# mark the non-core tools ``defer_loading: true`` and inject a tool_search tool. |
| 3160 | +# Anthropic then excludes deferred tools from the context window (they stop |
| 3161 | +# counting as input tokens until the model searches for one), while every tool |
| 3162 | +# stays callable. Deterministic output → the tools prefix still prompt-caches. |
| 3163 | +# --------------------------------------------------------------------------- |
| 3164 | + |
| 3165 | +# Core coding tools kept non-deferred so routine edit/read/run loops never pay a |
| 3166 | +# search round-trip. Everything else (Slack/Linear/Sentry/Notion/Snowflake/…) is |
| 3167 | +# deferred and loaded on demand. Anthropic recommends keeping the 3–5 (here a few |
| 3168 | +# more) most frequent tools resident. |
| 3169 | +_TOOL_SEARCH_CORE_TOOLS = frozenset( |
| 3170 | + { |
| 3171 | + "bash", |
| 3172 | + "bash_background", |
| 3173 | + "bash_background_output", |
| 3174 | + "bash_background_wait", |
| 3175 | + "bash_background_kill", |
| 3176 | + "read", |
| 3177 | + "write", |
| 3178 | + "edit", |
| 3179 | + "multiedit", |
| 3180 | + "apply_patch", |
| 3181 | + "glob", |
| 3182 | + "grep", |
| 3183 | + "task", |
| 3184 | + "todowrite", |
| 3185 | + "todoread", |
| 3186 | + "webfetch", |
| 3187 | + "question", |
| 3188 | + "skill", |
| 3189 | + } |
| 3190 | +) |
| 3191 | +_TOOL_SEARCH_DEFAULT_TYPE = "tool_search_tool_regex_20251119" |
| 3192 | +_TOOL_SEARCH_DEFAULT_NAME = "tool_search_tool_regex" |
| 3193 | +# Below this many tools the ~search round-trip isn't worth it (Anthropic's own |
| 3194 | +# guidance: standard calling is better under ~10 tools). |
| 3195 | +_TOOL_SEARCH_MIN_TOOLS = 12 |
| 3196 | + |
| 3197 | + |
| 3198 | +def inject_tool_search_deferral( |
| 3199 | + tools: Any, |
| 3200 | + *, |
| 3201 | + core_tools: frozenset[str] = _TOOL_SEARCH_CORE_TOOLS, |
| 3202 | + search_type: str = _TOOL_SEARCH_DEFAULT_TYPE, |
| 3203 | + search_name: str = _TOOL_SEARCH_DEFAULT_NAME, |
| 3204 | +) -> Any: |
| 3205 | + """Return a new ``tools`` list with non-core tools deferred + a search tool |
| 3206 | + injected, or the original list unchanged when injection doesn't apply. |
| 3207 | +
|
| 3208 | + No-op when: not a list, fewer than ``_TOOL_SEARCH_MIN_TOOLS``, a tool_search |
| 3209 | + tool is already present (client already defers), or nothing would be deferred. |
| 3210 | +
|
| 3211 | + Invariants enforced (else Anthropic 400s): the search tool is never deferred; |
| 3212 | + at least one tool stays non-deferred; a deferred tool never carries |
| 3213 | + ``cache_control`` — if the client's tools cache breakpoint sat on a now-deferred |
| 3214 | + tool, it is moved to the last non-deferred real tool so the (smaller) tools |
| 3215 | + prefix still caches. |
| 3216 | + """ |
| 3217 | + if not isinstance(tools, list) or len(tools) < _TOOL_SEARCH_MIN_TOOLS: |
| 3218 | + return tools |
| 3219 | + for tool in tools: |
| 3220 | + if isinstance(tool, dict) and str(tool.get("type", "")).startswith( |
| 3221 | + _TOOL_SEARCH_TOOL_TYPE_PREFIX |
| 3222 | + ): |
| 3223 | + return tools # client already uses tool search — leave it alone |
| 3224 | + |
| 3225 | + search_tool = {"type": search_type, "name": search_name} |
| 3226 | + out: list[Any] = [search_tool] |
| 3227 | + deferred = 0 |
| 3228 | + dropped_cache_control = False |
| 3229 | + last_resident_real: dict[str, Any] | None = None |
| 3230 | + resident_has_cache_control = False |
| 3231 | + |
| 3232 | + for tool in tools: |
| 3233 | + if not isinstance(tool, dict) or tool.get("type") or tool.get("name") in core_tools: |
| 3234 | + # Non-dict, server/typed tools (web_search, computer, …), and core |
| 3235 | + # tools stay resident and unchanged. |
| 3236 | + out.append(tool) |
| 3237 | + if isinstance(tool, dict) and not tool.get("type"): |
| 3238 | + last_resident_real = tool |
| 3239 | + resident_has_cache_control = resident_has_cache_control or bool( |
| 3240 | + tool.get("cache_control") |
| 3241 | + ) |
| 3242 | + continue |
| 3243 | + new_tool = dict(tool) |
| 3244 | + new_tool["defer_loading"] = True |
| 3245 | + if new_tool.pop("cache_control", None) is not None: |
| 3246 | + dropped_cache_control = True |
| 3247 | + out.append(new_tool) |
| 3248 | + deferred += 1 |
| 3249 | + |
| 3250 | + if deferred == 0: |
| 3251 | + return tools # nothing to defer → don't perturb the cache prefix |
| 3252 | + # Preserve a tools cache breakpoint: if we stripped cache_control off a |
| 3253 | + # deferred tool and no resident tool carries one, move it to the last |
| 3254 | + # resident real tool (never the search tool, to keep its shape canonical). |
| 3255 | + if dropped_cache_control and not resident_has_cache_control and last_resident_real is not None: |
| 3256 | + last_resident_real["cache_control"] = {"type": "ephemeral"} |
| 3257 | + return out |
| 3258 | + |
| 3259 | + |
| 3260 | +# --------------------------------------------------------------------------- |
| 3261 | +# Server-side Tool Search injection — OpenAI Responses API (gpt-5.4+). |
| 3262 | +# |
| 3263 | +# The OpenAI-side analogue of inject_tool_search_deferral above. OpenAI shipped |
| 3264 | +# the same idea for the Responses API on gpt-5.4+: mark a function/MCP tool |
| 3265 | +# ``defer_loading: true`` and add a ``{"type": "tool_search"}`` tool, and OpenAI |
| 3266 | +# keeps the deferred tools' heavy parameter schemas OUT of the model's context |
| 3267 | +# (only name+description remain) until the model searches for one — while every |
| 3268 | +# tool stays callable and the prompt cache is preserved. Same win as Anthropic |
| 3269 | +# (~15-25k tool-schema tokens -> ~200) for clients that ship a big tool surface |
| 3270 | +# and never opt into tool search themselves (opencode, plain API clients). |
| 3271 | +# |
| 3272 | +# Differences from the Anthropic path that require a separate function: |
| 3273 | +# * Responses function tools carry ``type: "function"`` (Anthropic real tools |
| 3274 | +# have no ``type``), so the resident/defer test is inverted — we defer |
| 3275 | +# ``function`` (non-core) and ``mcp`` tools and keep OTHER typed/hosted tools |
| 3276 | +# (web_search, file_search, code_interpreter, computer, image_generation, and |
| 3277 | +# the search tool itself) resident. |
| 3278 | +# * Model-gated: only gpt-5.4+ support it; older models 400 on the fields. |
| 3279 | +# * No ``cache_control`` (OpenAI caches automatically), so no breakpoint move. |
| 3280 | +# --------------------------------------------------------------------------- |
| 3281 | + |
| 3282 | +_OPENAI_TOOL_SEARCH_TYPE = "tool_search" |
| 3283 | +_OPENAI_TOOL_SEARCH_MIN_TOOLS = 12 |
| 3284 | +# gpt-5.4 is the first model with Responses tool_search (OpenAI docs). Version- |
| 3285 | +# gated by default; overridable per deployment via a regex in |
| 3286 | +# HEADROOM_OPENAI_TOOL_SEARCH_MODELS (matched against the model name) so new |
| 3287 | +# model families can be enabled without a code edit + release. |
| 3288 | +_OPENAI_TOOL_SEARCH_MIN_VERSION = (5, 4) |
| 3289 | + |
| 3290 | + |
| 3291 | +def _model_supports_openai_tool_search(model: str | None) -> bool: |
| 3292 | + """True when an OpenAI model supports the Responses ``tool_search`` feature. |
| 3293 | +
|
| 3294 | + Default gate: ``gpt-<major>.<minor>`` >= 5.4. A regex in |
| 3295 | + ``HEADROOM_OPENAI_TOOL_SEARCH_MODELS`` (matched against the model name) wins |
| 3296 | + when set; a malformed pattern falls back to the version gate rather than |
| 3297 | + crashing. |
| 3298 | + """ |
| 3299 | + if not model: |
| 3300 | + return False |
| 3301 | + override = os.environ.get("HEADROOM_OPENAI_TOOL_SEARCH_MODELS", "").strip() |
| 3302 | + if override: |
| 3303 | + try: |
| 3304 | + return re.search(override, model) is not None |
| 3305 | + except re.error: |
| 3306 | + pass # malformed override → fall back to the version gate |
| 3307 | + match = re.match(r"gpt-(\d+)(?:\.(\d+))?", model.strip().lower()) |
| 3308 | + if not match: |
| 3309 | + return False |
| 3310 | + major, minor = int(match.group(1)), int(match.group(2) or 0) |
| 3311 | + return (major, minor) >= _OPENAI_TOOL_SEARCH_MIN_VERSION |
| 3312 | + |
| 3313 | + |
| 3314 | +def inject_tool_search_deferral_openai( |
| 3315 | + tools: Any, |
| 3316 | + model: str | None, |
| 3317 | + *, |
| 3318 | + core_tools: frozenset[str] = _TOOL_SEARCH_CORE_TOOLS, |
| 3319 | +) -> Any: |
| 3320 | + """Return a new Responses ``tools`` list with non-core function/MCP tools |
| 3321 | + deferred + a ``{"type": "tool_search"}`` tool injected, or the original list |
| 3322 | + unchanged when injection doesn't apply. |
| 3323 | +
|
| 3324 | + No-op when: the model doesn't support tool search (gpt-5.4+ only), ``tools`` |
| 3325 | + is not a list, there are fewer than ``_OPENAI_TOOL_SEARCH_MIN_TOOLS``, a |
| 3326 | + tool_search tool is already present (client already defers), or nothing would |
| 3327 | + be deferred. Core coding tools and hosted/typed tools (web_search, |
| 3328 | + file_search, code_interpreter, computer, …) stay resident and unchanged, so |
| 3329 | + routine edit/read/run loops never pay a search round-trip and the request |
| 3330 | + stays valid; the injected search tool is itself resident. |
| 3331 | + """ |
| 3332 | + if not _model_supports_openai_tool_search(model): |
| 3333 | + return tools |
| 3334 | + if not isinstance(tools, list) or len(tools) < _OPENAI_TOOL_SEARCH_MIN_TOOLS: |
| 3335 | + return tools |
| 3336 | + for tool in tools: |
| 3337 | + if isinstance(tool, dict) and tool.get("type") == _OPENAI_TOOL_SEARCH_TYPE: |
| 3338 | + return tools # client already uses tool search — leave it alone |
| 3339 | + |
| 3340 | + out: list[Any] = [{"type": _OPENAI_TOOL_SEARCH_TYPE}] |
| 3341 | + deferred = 0 |
| 3342 | + for tool in tools: |
| 3343 | + if not isinstance(tool, dict): |
| 3344 | + out.append(tool) |
| 3345 | + continue |
| 3346 | + ttype = tool.get("type") |
| 3347 | + # Deferrable: a non-core function, or an MCP server (OpenAI models are |
| 3348 | + # trained to search namespaces / MCP servers). Everything else — core |
| 3349 | + # coding tools and other hosted tools — stays resident. |
| 3350 | + deferrable = (ttype == "function" and tool.get("name") not in core_tools) or ttype == "mcp" |
| 3351 | + if deferrable and not tool.get("defer_loading"): |
| 3352 | + new_tool = dict(tool) |
| 3353 | + new_tool["defer_loading"] = True |
| 3354 | + out.append(new_tool) |
| 3355 | + deferred += 1 |
| 3356 | + else: |
| 3357 | + out.append(tool) |
| 3358 | + |
| 3359 | + if deferred == 0: |
| 3360 | + return tools # nothing to defer → don't perturb the request / cache prefix |
| 3361 | + return out |
0 commit comments