Merge branch 'main' of github.com:NousResearch/hermes-agent into bb/tui-audit-followup
# Conflicts: # ui-tui/src/components/markdown.tsx # ui-tui/src/types/hermes-ink.d.ts
This commit is contained in:
@@ -697,7 +697,12 @@ class TestIsConnectionError:
|
||||
|
||||
|
||||
class TestKimiForCodingTemperature:
|
||||
"""kimi-for-coding now requires temperature=0.6 exactly."""
|
||||
"""Moonshot kimi-for-coding models require fixed temperatures.
|
||||
|
||||
k2.5 / k2-turbo-preview / k2-0905-preview → 0.6 (non-thinking lock).
|
||||
k2-thinking / k2-thinking-turbo → 1.0 (thinking lock).
|
||||
kimi-k2-instruct* and every other model preserve the caller's temperature.
|
||||
"""
|
||||
|
||||
def test_build_call_kwargs_forces_fixed_temperature(self):
|
||||
from agent.auxiliary_client import _build_call_kwargs
|
||||
@@ -772,12 +777,55 @@ class TestKimiForCodingTemperature:
|
||||
assert kwargs["model"] == "kimi-for-coding"
|
||||
assert kwargs["temperature"] == 0.6
|
||||
|
||||
def test_non_kimi_model_still_preserves_temperature(self):
|
||||
@pytest.mark.parametrize(
|
||||
"model,expected",
|
||||
[
|
||||
("kimi-k2.5", 0.6),
|
||||
("kimi-k2-turbo-preview", 0.6),
|
||||
("kimi-k2-0905-preview", 0.6),
|
||||
("kimi-k2-thinking", 1.0),
|
||||
("kimi-k2-thinking-turbo", 1.0),
|
||||
("moonshotai/kimi-k2.5", 0.6),
|
||||
("moonshotai/Kimi-K2-Thinking", 1.0),
|
||||
],
|
||||
)
|
||||
def test_kimi_k2_family_temperature_override(self, model, expected):
|
||||
"""Moonshot kimi-k2.* models only accept fixed temperatures.
|
||||
|
||||
Non-thinking models → 0.6, thinking-mode models → 1.0.
|
||||
"""
|
||||
from agent.auxiliary_client import _build_call_kwargs
|
||||
|
||||
kwargs = _build_call_kwargs(
|
||||
provider="kimi-coding",
|
||||
model="kimi-k2.5",
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
temperature=0.3,
|
||||
)
|
||||
|
||||
assert kwargs["temperature"] == expected
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
[
|
||||
"anthropic/claude-sonnet-4-6",
|
||||
"gpt-5.4",
|
||||
# kimi-k2-instruct is the non-coding K2 family — temperature is
|
||||
# variable (recommended 0.6 but not enforced). Must not clamp.
|
||||
"kimi-k2-instruct",
|
||||
"moonshotai/Kimi-K2-Instruct",
|
||||
"moonshotai/Kimi-K2-Instruct-0905",
|
||||
"kimi-k2-instruct-0905",
|
||||
# Hypothetical future kimi name not in the whitelist.
|
||||
"kimi-k2-experimental",
|
||||
],
|
||||
)
|
||||
def test_non_restricted_model_preserves_temperature(self, model):
|
||||
from agent.auxiliary_client import _build_call_kwargs
|
||||
|
||||
kwargs = _build_call_kwargs(
|
||||
provider="openrouter",
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
temperature=0.3,
|
||||
)
|
||||
|
||||
@@ -781,3 +781,127 @@ class TestTokenBudgetTailProtection:
|
||||
# Tool at index 2 is outside the protected tail (last 3 = indices 2,3,4)
|
||||
# so it might or might not be pruned depending on boundary
|
||||
assert isinstance(pruned, int)
|
||||
|
||||
|
||||
class TestTruncateToolCallArgsJson:
|
||||
"""Regression tests for #11762.
|
||||
|
||||
The previous implementation produced invalid JSON by slicing
|
||||
``function.arguments`` mid-string, which caused non-retryable 400s from
|
||||
strict providers (observed on MiniMax) and stuck long sessions in a
|
||||
re-send loop. The helper here must always emit parseable JSON whose
|
||||
shape matches the original — shrunken, not corrupted.
|
||||
"""
|
||||
|
||||
def _helper(self):
|
||||
from agent.context_compressor import _truncate_tool_call_args_json
|
||||
return _truncate_tool_call_args_json
|
||||
|
||||
def test_shrunken_args_remain_valid_json(self):
|
||||
import json as _json
|
||||
shrink = self._helper()
|
||||
original = _json.dumps({
|
||||
"path": "~/.hermes/skills/shopping/browser-setup-notes.md",
|
||||
"content": "# Shopping Browser Setup Notes\n\n" + "abc " * 400,
|
||||
})
|
||||
assert len(original) > 500
|
||||
shrunk = shrink(original)
|
||||
parsed = _json.loads(shrunk) # must not raise
|
||||
assert parsed["path"] == "~/.hermes/skills/shopping/browser-setup-notes.md"
|
||||
assert parsed["content"].endswith("...[truncated]")
|
||||
assert len(shrunk) < len(original)
|
||||
|
||||
def test_non_json_arguments_pass_through(self):
|
||||
shrink = self._helper()
|
||||
not_json = "this is not json at all, " * 50
|
||||
assert shrink(not_json) == not_json
|
||||
|
||||
def test_short_string_leaves_unchanged(self):
|
||||
import json as _json
|
||||
shrink = self._helper()
|
||||
payload = _json.dumps({"command": "ls -la", "cwd": "/tmp"})
|
||||
assert _json.loads(shrink(payload)) == {"command": "ls -la", "cwd": "/tmp"}
|
||||
|
||||
def test_nested_structures_are_walked(self):
|
||||
import json as _json
|
||||
shrink = self._helper()
|
||||
payload = _json.dumps({
|
||||
"messages": [
|
||||
{"role": "user", "content": "x" * 500},
|
||||
{"role": "assistant", "content": "ok"},
|
||||
],
|
||||
"meta": {"note": "y" * 500},
|
||||
})
|
||||
parsed = _json.loads(shrink(payload))
|
||||
assert parsed["messages"][0]["content"].endswith("...[truncated]")
|
||||
assert parsed["messages"][1]["content"] == "ok"
|
||||
assert parsed["meta"]["note"].endswith("...[truncated]")
|
||||
|
||||
def test_non_string_leaves_preserved(self):
|
||||
import json as _json
|
||||
shrink = self._helper()
|
||||
payload = _json.dumps({
|
||||
"retries": 3,
|
||||
"enabled": True,
|
||||
"timeout": None,
|
||||
"items": [1, 2, 3],
|
||||
"note": "z" * 500,
|
||||
})
|
||||
parsed = _json.loads(shrink(payload))
|
||||
assert parsed["retries"] == 3
|
||||
assert parsed["enabled"] is True
|
||||
assert parsed["timeout"] is None
|
||||
assert parsed["items"] == [1, 2, 3]
|
||||
assert parsed["note"].endswith("...[truncated]")
|
||||
|
||||
def test_scalar_json_string_gets_shrunk(self):
|
||||
import json as _json
|
||||
shrink = self._helper()
|
||||
payload = _json.dumps("q" * 500)
|
||||
parsed = _json.loads(shrink(payload))
|
||||
assert isinstance(parsed, str)
|
||||
assert parsed.endswith("...[truncated]")
|
||||
|
||||
def test_unicode_preserved(self):
|
||||
import json as _json
|
||||
shrink = self._helper()
|
||||
payload = _json.dumps({"content": "非德满" + ("a" * 500)})
|
||||
out = shrink(payload)
|
||||
# ensure_ascii=False keeps CJK intact rather than emitting \uXXXX
|
||||
assert "非德满" in out
|
||||
|
||||
def test_pass3_emits_valid_json_for_downstream_provider(self):
|
||||
"""End-to-end: Pass 3 must never produce the exact failure payload
|
||||
that caused the 400 loop (unterminated string, missing brace)."""
|
||||
import json as _json
|
||||
with patch("agent.context_compressor.get_model_context_length", return_value=100000):
|
||||
c = ContextCompressor(
|
||||
model="test/model",
|
||||
threshold_percent=0.85,
|
||||
protect_first_n=1,
|
||||
protect_last_n=1,
|
||||
quiet_mode=True,
|
||||
)
|
||||
huge_content = "# Shopping Browser Setup Notes\n\n## Overview\n" + "x " * 400
|
||||
args_payload = _json.dumps({
|
||||
"path": "~/.hermes/skills/shopping/browser-setup-notes.md",
|
||||
"content": huge_content,
|
||||
})
|
||||
assert len(args_payload) > 500 # triggers the Pass-3 shrink
|
||||
messages = [
|
||||
{"role": "user", "content": "please write two files"},
|
||||
{"role": "assistant", "content": None, "tool_calls": [
|
||||
{"id": "call_1", "type": "function",
|
||||
"function": {"name": "write_file", "arguments": args_payload}},
|
||||
]},
|
||||
{"role": "tool", "tool_call_id": "call_1",
|
||||
"content": '{"bytes_written": 727}'},
|
||||
{"role": "user", "content": "ok"},
|
||||
{"role": "assistant", "content": "done"},
|
||||
]
|
||||
result, _ = c._prune_old_tool_results(messages, protect_tail_count=2)
|
||||
shrunk = result[1]["tool_calls"][0]["function"]["arguments"]
|
||||
# Must parse — otherwise downstream provider returns 400
|
||||
parsed = _json.loads(shrunk)
|
||||
assert parsed["path"] == "~/.hermes/skills/shopping/browser-setup-notes.md"
|
||||
assert parsed["content"].endswith("...[truncated]")
|
||||
|
||||
@@ -231,3 +231,201 @@ def test_cli_exec_blocked(server, argv):
|
||||
])
|
||||
def test_cli_exec_allowed(server, argv):
|
||||
assert server._cli_exec_blocked(argv) is None
|
||||
|
||||
|
||||
# ── slash.exec skill command interception ────────────────────────────
|
||||
|
||||
|
||||
def test_slash_exec_rejects_skill_commands(server):
|
||||
"""slash.exec must reject skill commands so the TUI falls through to command.dispatch."""
|
||||
# Register a mock session
|
||||
sid = "test-session"
|
||||
server._sessions[sid] = {"session_key": sid, "agent": None}
|
||||
|
||||
# Mock scan_skill_commands to return a known skill
|
||||
fake_skills = {"/hermes-agent-dev": {"name": "hermes-agent-dev", "description": "Dev workflow"}}
|
||||
|
||||
with patch("agent.skill_commands.get_skill_commands", return_value=fake_skills):
|
||||
resp = server.handle_request({
|
||||
"id": "r1",
|
||||
"method": "slash.exec",
|
||||
"params": {"command": "hermes-agent-dev", "session_id": sid},
|
||||
})
|
||||
|
||||
# Should return an error so the TUI's .catch() fires command.dispatch
|
||||
assert "error" in resp
|
||||
assert resp["error"]["code"] == 4018
|
||||
assert "skill command" in resp["error"]["message"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("cmd", ["retry", "queue hello", "q hello", "steer fix the test", "plan"])
|
||||
def test_slash_exec_rejects_pending_input_commands(server, cmd):
|
||||
"""slash.exec must reject commands that use _pending_input in the CLI."""
|
||||
sid = "test-session"
|
||||
server._sessions[sid] = {"session_key": sid, "agent": None}
|
||||
|
||||
resp = server.handle_request({
|
||||
"id": "r1",
|
||||
"method": "slash.exec",
|
||||
"params": {"command": cmd, "session_id": sid},
|
||||
})
|
||||
|
||||
assert "error" in resp
|
||||
assert resp["error"]["code"] == 4018
|
||||
assert "pending-input command" in resp["error"]["message"]
|
||||
|
||||
|
||||
def test_command_dispatch_queue_sends_message(server):
|
||||
"""command.dispatch /queue returns {type: 'send', message: ...} for the TUI."""
|
||||
sid = "test-session"
|
||||
server._sessions[sid] = {"session_key": sid}
|
||||
|
||||
resp = server.handle_request({
|
||||
"id": "r1",
|
||||
"method": "command.dispatch",
|
||||
"params": {"name": "queue", "arg": "tell me about quantum computing", "session_id": sid},
|
||||
})
|
||||
|
||||
assert "error" not in resp
|
||||
result = resp["result"]
|
||||
assert result["type"] == "send"
|
||||
assert result["message"] == "tell me about quantum computing"
|
||||
|
||||
|
||||
def test_command_dispatch_queue_requires_arg(server):
|
||||
"""command.dispatch /queue without an argument returns an error."""
|
||||
sid = "test-session"
|
||||
server._sessions[sid] = {"session_key": sid}
|
||||
|
||||
resp = server.handle_request({
|
||||
"id": "r2",
|
||||
"method": "command.dispatch",
|
||||
"params": {"name": "queue", "arg": "", "session_id": sid},
|
||||
})
|
||||
|
||||
assert "error" in resp
|
||||
assert resp["error"]["code"] == 4004
|
||||
|
||||
|
||||
def test_command_dispatch_steer_fallback_sends_message(server):
|
||||
"""command.dispatch /steer with no active agent falls back to send."""
|
||||
sid = "test-session"
|
||||
server._sessions[sid] = {"session_key": sid, "agent": None}
|
||||
|
||||
resp = server.handle_request({
|
||||
"id": "r3",
|
||||
"method": "command.dispatch",
|
||||
"params": {"name": "steer", "arg": "focus on testing", "session_id": sid},
|
||||
})
|
||||
|
||||
assert "error" not in resp
|
||||
result = resp["result"]
|
||||
assert result["type"] == "send"
|
||||
assert result["message"] == "focus on testing"
|
||||
|
||||
|
||||
def test_command_dispatch_retry_finds_last_user_message(server):
|
||||
"""command.dispatch /retry walks session['history'] to find the last user message."""
|
||||
sid = "test-session"
|
||||
history = [
|
||||
{"role": "user", "content": "first question"},
|
||||
{"role": "assistant", "content": "first answer"},
|
||||
{"role": "user", "content": "second question"},
|
||||
{"role": "assistant", "content": "second answer"},
|
||||
]
|
||||
server._sessions[sid] = {
|
||||
"session_key": sid,
|
||||
"agent": None,
|
||||
"history": history,
|
||||
"history_lock": threading.Lock(),
|
||||
"history_version": 0,
|
||||
}
|
||||
|
||||
resp = server.handle_request({
|
||||
"id": "r4",
|
||||
"method": "command.dispatch",
|
||||
"params": {"name": "retry", "session_id": sid},
|
||||
})
|
||||
|
||||
assert "error" not in resp
|
||||
result = resp["result"]
|
||||
assert result["type"] == "send"
|
||||
assert result["message"] == "second question"
|
||||
# Verify history was truncated: everything from last user message onward removed
|
||||
assert len(server._sessions[sid]["history"]) == 2
|
||||
assert server._sessions[sid]["history"][-1]["role"] == "assistant"
|
||||
assert server._sessions[sid]["history_version"] == 1
|
||||
|
||||
|
||||
def test_command_dispatch_retry_empty_history(server):
|
||||
"""command.dispatch /retry with empty history returns error."""
|
||||
sid = "test-session"
|
||||
server._sessions[sid] = {
|
||||
"session_key": sid,
|
||||
"agent": None,
|
||||
"history": [],
|
||||
"history_lock": threading.Lock(),
|
||||
"history_version": 0,
|
||||
}
|
||||
|
||||
resp = server.handle_request({
|
||||
"id": "r5",
|
||||
"method": "command.dispatch",
|
||||
"params": {"name": "retry", "session_id": sid},
|
||||
})
|
||||
|
||||
assert "error" in resp
|
||||
assert resp["error"]["code"] == 4018
|
||||
|
||||
|
||||
def test_command_dispatch_retry_handles_multipart_content(server):
|
||||
"""command.dispatch /retry extracts text from multipart content lists."""
|
||||
sid = "test-session"
|
||||
history = [
|
||||
{"role": "user", "content": [
|
||||
{"type": "text", "text": "analyze this"},
|
||||
{"type": "image_url", "image_url": {"url": "data:image/png;base64,..."}}
|
||||
]},
|
||||
{"role": "assistant", "content": "I see the image."},
|
||||
]
|
||||
server._sessions[sid] = {
|
||||
"session_key": sid,
|
||||
"agent": None,
|
||||
"history": history,
|
||||
"history_lock": threading.Lock(),
|
||||
"history_version": 0,
|
||||
}
|
||||
|
||||
resp = server.handle_request({
|
||||
"id": "r6",
|
||||
"method": "command.dispatch",
|
||||
"params": {"name": "retry", "session_id": sid},
|
||||
})
|
||||
|
||||
assert "error" not in resp
|
||||
result = resp["result"]
|
||||
assert result["type"] == "send"
|
||||
assert result["message"] == "analyze this"
|
||||
|
||||
|
||||
def test_command_dispatch_returns_skill_payload(server):
|
||||
"""command.dispatch returns structured skill payload for the TUI to send()."""
|
||||
sid = "test-session"
|
||||
server._sessions[sid] = {"session_key": sid}
|
||||
|
||||
fake_skills = {"/hermes-agent-dev": {"name": "hermes-agent-dev", "description": "Dev workflow"}}
|
||||
fake_msg = "Loaded skill content here"
|
||||
|
||||
with patch("agent.skill_commands.scan_skill_commands", return_value=fake_skills), \
|
||||
patch("agent.skill_commands.build_skill_invocation_message", return_value=fake_msg):
|
||||
resp = server.handle_request({
|
||||
"id": "r2",
|
||||
"method": "command.dispatch",
|
||||
"params": {"name": "hermes-agent-dev", "session_id": sid},
|
||||
})
|
||||
|
||||
assert "error" not in resp
|
||||
result = resp["result"]
|
||||
assert result["type"] == "skill"
|
||||
assert result["message"] == fake_msg
|
||||
assert result["name"] == "hermes-agent-dev"
|
||||
|
||||
Reference in New Issue
Block a user