fix(ai): recover schema-echo leak in tool-call arguments
Weak local models (seen with qwen2.5-coder:3b at temp 0) sometimes emit a
parameter's JSON *schema* fragment as its *value*, e.g.
run_shell(command={'type':'string','description':'bash ./add.py'}). _exec_tool
does str(args["command"]), so the stringified dict was run as a command →
exit 127 and a hollow "task done" claim.
Every native tool arg is a plain string, so a dict-valued arg is always this
leak. Add _unleak_str/_clean_args to OllamaProvider: pull the intended string
from a value-ish key (or `description`), ignore JSON-schema scaffolding keys
like `type`, else drop to "" so the tool reports a clean error instead of
running garbage. Applied on both the structured tool_calls path and the
text-recovery path (_coerce_call). New tests/test_agent_providers.py pins it.
Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -180,7 +180,7 @@ class OllamaProvider:
|
||||
args = json.loads(args)
|
||||
except ValueError:
|
||||
args = {}
|
||||
calls.append({"name": fn.get("name", ""), "arguments": args or {}})
|
||||
calls.append({"name": fn.get("name", ""), "arguments": self._clean_args(args or {})})
|
||||
# Small/quantized models (notably qwen2.5 on CPU) intermittently emit a valid
|
||||
# tool call as literal text in `content` instead of the structured `tool_calls`
|
||||
# field — qwen's `<tool_call>{…}</tool_call>`, but also bare/fenced JSON and
|
||||
@@ -215,6 +215,37 @@ class OllamaProvider:
|
||||
re.I,
|
||||
)
|
||||
|
||||
# Per-argument schema-echo leak: a weak model sometimes emits a parameter's
|
||||
# JSON *schema* fragment as its *value*, e.g.
|
||||
# run_shell(command={'type': 'string', 'description': 'bash ./add.py'})
|
||||
# Every native tool arg is a plain string, so a dict-valued arg is always this
|
||||
# leak — recover the intended string (the model stows it in a value-ish key, or
|
||||
# in `description`), else drop to "" so the tool reports a clean error instead
|
||||
# of running the stringified dict.
|
||||
_LEAK_VALUE_KEYS = ("value", "default", "command", "content", "path",
|
||||
"text", "input", "arg", "description")
|
||||
# JSON-schema scaffolding keys — never a recovered value (e.g. `type: string`
|
||||
# must not be mistaken for the single string value of a bare `{'type':'string'}`).
|
||||
_SCHEMA_META = frozenset({"type", "format", "enum", "items", "properties",
|
||||
"required", "title", "additionalproperties"})
|
||||
|
||||
@classmethod
|
||||
def _unleak_str(cls, v):
|
||||
if not isinstance(v, dict):
|
||||
return v
|
||||
for k in cls._LEAK_VALUE_KEYS:
|
||||
if isinstance(v.get(k), str) and v[k].strip():
|
||||
return v[k]
|
||||
strs = [x for k, x in v.items()
|
||||
if k not in cls._SCHEMA_META and isinstance(x, str) and x.strip()]
|
||||
return strs[0] if len(strs) == 1 else ""
|
||||
|
||||
@classmethod
|
||||
def _clean_args(cls, args):
|
||||
if not isinstance(args, dict):
|
||||
return args
|
||||
return {k: cls._unleak_str(v) for k, v in args.items()}
|
||||
|
||||
@classmethod
|
||||
def _coerce_call(cls, obj, valid_names) -> dict | None:
|
||||
"""Turn a decoded JSON object into a `{"name","arguments"}` call IF it
|
||||
@@ -243,7 +274,7 @@ class OllamaProvider:
|
||||
args = {}
|
||||
if not isinstance(args, dict):
|
||||
args = {}
|
||||
return {"name": name, "arguments": args}
|
||||
return {"name": name, "arguments": cls._clean_args(args)}
|
||||
|
||||
@classmethod
|
||||
def _extract_text_tool_calls(
|
||||
|
||||
@@ -0,0 +1,48 @@
|
||||
"""Unit tests for OllamaProvider tool-call argument recovery.
|
||||
|
||||
Weak/quantized local models intermittently emit a parameter's JSON *schema*
|
||||
fragment as its *value* (e.g. run_shell(command={'type':'string',
|
||||
'description':'bash ./add.py'})). Every native tool arg is a plain string, so a
|
||||
dict-valued arg is always this leak. These tests pin the recovery so the harness
|
||||
never runs a stringified schema dict as a command again.
|
||||
"""
|
||||
from cmd_chat.agent.providers import OllamaProvider as P
|
||||
|
||||
|
||||
def test_unleak_recovers_command_from_schema_fragment():
|
||||
assert P._unleak_str({"type": "string", "description": "bash ./add.py"}) == "bash ./add.py"
|
||||
|
||||
|
||||
def test_unleak_prefers_value_keys_over_description():
|
||||
assert P._unleak_str({"description": "help text", "value": "ls -la"}) == "ls -la"
|
||||
assert P._unleak_str({"description": "help text", "default": "echo hi"}) == "echo hi"
|
||||
|
||||
|
||||
def test_unleak_passes_plain_strings_through():
|
||||
assert P._unleak_str("mkdir data") == "mkdir data"
|
||||
|
||||
|
||||
def test_unleak_single_string_value_fallback():
|
||||
assert P._unleak_str({"foo": "only-one"}) == "only-one"
|
||||
|
||||
|
||||
def test_unleak_ignores_bare_schema_meta():
|
||||
# {'type':'string'} has no real value — must NOT recover the type name.
|
||||
assert P._unleak_str({"type": "string"}) == ""
|
||||
assert P._unleak_str({}) == ""
|
||||
|
||||
|
||||
def test_clean_args_flattens_leaked_arg_keeps_real_ones():
|
||||
out = P._clean_args({"command": {"type": "string", "description": "bash ./add.py"}, "x": "keep"})
|
||||
assert out == {"command": "bash ./add.py", "x": "keep"}
|
||||
|
||||
|
||||
def test_clean_args_passes_non_dict_through():
|
||||
assert P._clean_args("passthrough") == "passthrough"
|
||||
|
||||
|
||||
def test_coerce_call_cleans_recovered_arguments():
|
||||
# The text-recovery path must also unleak args.
|
||||
obj = {"name": "run_shell", "arguments": {"command": {"type": "string", "description": "ls"}}}
|
||||
call = P._coerce_call(obj, valid_names={"run_shell"})
|
||||
assert call == {"name": "run_shell", "arguments": {"command": "ls"}}
|
||||
Reference in New Issue
Block a user