From 64d76a241e9573aa3aa5d5a5a9e76cec1efcce0c Mon Sep 17 00:00:00 2001 From: Lee Jackson <130007945+Imagineer99@users.noreply.github.com> Date: Tue, 28 Jul 2026 11:14:13 +0100 Subject: [PATCH] Handle llama.cpp tool schema limits (#7512) * Handle llama.cpp tool schema limits * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --------- Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com> --- studio/backend/routes/inference.py | 109 +++++++++++++++++- .../tests/test_openai_tool_passthrough.py | 35 ++++++ 2 files changed, 143 insertions(+), 1 deletion(-) diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 4e89522434..9843dc6378 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -14737,6 +14737,113 @@ async def _anthropic_plain_non_streaming(run_gen, message_id, model_name): # ===================================================================== +_JSON_SCHEMA_MAP_KEYWORDS = frozenset( + { + "$defs", + "definitions", + "dependentSchemas", + "patternProperties", + "properties", + } +) +_JSON_SCHEMA_SINGLE_KEYWORDS = frozenset( + { + "additionalProperties", + "contains", + "contentSchema", + "else", + "if", + "items", + "not", + "propertyNames", + "then", + "unevaluatedItems", + "unevaluatedProperties", + } +) +_JSON_SCHEMA_LIST_KEYWORDS = frozenset({"allOf", "anyOf", "oneOf", "prefixItems"}) +_LLAMA_GRAMMAR_MAX_REPETITION = 2000 +_JSON_SCHEMA_REPETITION_KEYWORDS = frozenset({"maxItems", "maxLength", "minItems", "minLength"}) + + +def _llama_compatible_tool_schema(schema): + """Return a llama.cpp-compatible copy of one JSON Schema node. + + JSON Schema ``pattern`` expressions match anywhere in a string, so an + unanchored pattern is valid and cannot be made compatible by merely adding + ``^`` and ``$`` without changing its meaning. llama.cpp's grammar converter + currently rejects those patterns outright. Its grammar parser likewise + rejects repetition bounds above 2000. Omit only those unsupported + constraints from the local-backend copy; the agent retains and validates + its original schema, while every compatible constraint still reaches + llama.cpp. + """ + if not isinstance(schema, dict): + return schema + + compatible = dict(schema) + pattern = compatible.get("pattern") + if isinstance(pattern, str) and not (pattern.startswith("^") and pattern.endswith("$")): + compatible.pop("pattern") + # llama-grammar.cpp refuses repetition bounds above its sane-default + # threshold. Dropping the local-backend constraint preserves every value + # the client schema accepts; capping it would incorrectly reject otherwise + # valid tool arguments. + for keyword in _JSON_SCHEMA_REPETITION_KEYWORDS: + bound = compatible.get(keyword) + if ( + isinstance(bound, int) + and not isinstance(bound, bool) + and bound > _LLAMA_GRAMMAR_MAX_REPETITION + ): + compatible.pop(keyword) + + for keyword in _JSON_SCHEMA_MAP_KEYWORDS: + children = compatible.get(keyword) + if isinstance(children, dict): + compatible[keyword] = { + key: _llama_compatible_tool_schema(value) for key, value in children.items() + } + + for keyword in _JSON_SCHEMA_SINGLE_KEYWORDS: + child = compatible.get(keyword) + if isinstance(child, dict): + compatible[keyword] = _llama_compatible_tool_schema(child) + + for keyword in _JSON_SCHEMA_LIST_KEYWORDS: + children = compatible.get(keyword) + if isinstance(children, list): + compatible[keyword] = [_llama_compatible_tool_schema(value) for value in children] + + return compatible + + +def _llama_compatible_tools(openai_tools): + if not isinstance(openai_tools, list): + return openai_tools + + compatible_tools = [] + for tool in openai_tools: + if not isinstance(tool, dict): + compatible_tools.append(tool) + continue + function = tool.get("function") + parameters = function.get("parameters") if isinstance(function, dict) else None + if not isinstance(parameters, dict): + compatible_tools.append(tool) + continue + compatible_tools.append( + { + **tool, + "function": { + **function, + "parameters": _llama_compatible_tool_schema(parameters), + }, + } + ) + return compatible_tools + + def _build_passthrough_payload( openai_messages, openai_tools, @@ -14764,7 +14871,7 @@ def _build_passthrough_payload( "stream": stream, } if openai_tools: - body["tools"] = openai_tools + body["tools"] = _llama_compatible_tools(openai_tools) if tool_choice is not None: body["tool_choice"] = tool_choice if seed is not None: diff --git a/studio/backend/tests/test_openai_tool_passthrough.py b/studio/backend/tests/test_openai_tool_passthrough.py index d98e08db93..7758339070 100644 --- a/studio/backend/tests/test_openai_tool_passthrough.py +++ b/studio/backend/tests/test_openai_tool_passthrough.py @@ -1191,6 +1191,41 @@ class TestBuildPassthroughPayloadToolChoice: body = _build_passthrough_payload(**self._args(), tool_choice = tc) assert body["tool_choice"] == tc + def test_llama_incompatible_tool_constraints_are_omitted(self): + args = self._args() + schema = args["openai_tools"][0]["function"]["parameters"] + schema["properties"] = { + "declarationKey": {"type": "string", "pattern": r"\S"}, + "exactKey": {"type": "string", "pattern": r"^[A-Z]+$"}, + "nested": { + "type": "array", + "items": { + "anyOf": [ + {"type": "string", "pattern": "token"}, + {"type": "string", "pattern": "^fixed$"}, + ], + "default": {"pattern": "annotation data"}, + }, + }, + "largeScript": {"type": "string", "minLength": 1, "maxLength": 65536}, + "boundedScript": {"type": "string", "maxLength": 2000}, + } + + body = _build_passthrough_payload(**args) + forwarded = body["tools"][0]["function"]["parameters"]["properties"] + + assert forwarded["declarationKey"] == {"type": "string"} + assert forwarded["exactKey"]["pattern"] == r"^[A-Z]+$" + nested = forwarded["nested"]["items"] + assert nested["anyOf"][0] == {"type": "string"} + assert nested["anyOf"][1]["pattern"] == "^fixed$" + assert nested["default"] == {"pattern": "annotation data"} + assert forwarded["largeScript"] == {"type": "string", "minLength": 1} + assert forwarded["boundedScript"]["maxLength"] == 2000 + assert schema["properties"]["declarationKey"]["pattern"] == r"\S" + assert schema["properties"]["nested"]["items"]["anyOf"][0]["pattern"] == "token" + assert schema["properties"]["largeScript"]["maxLength"] == 65536 + def test_stream_omits_usage_options_when_client_did_not_request_them(self): args = self._args() args["stream"] = True