Files
nousresearch--hermes-agent/tests/run_agent/test_empty_response_recovery_persistence.py
wehub-resource-sync b4fbd6fe9f
Deploy Site / deploy-vercel (push) Has been skipped
Deploy Site / deploy-docs (push) Has been skipped
Build Skills Index / build-index (push) Has been skipped
CI / Deny unrelated histories (push) Has been skipped
CI / Detect affected areas (push) Successful in 27m35s
CI / OSV scan (push) Failing after 4s
CI / Build&Test Docker image (push) Successful in 9s
CI / Supply-chain scan (push) Has been skipped
CI / Lint Docker scripts (push) Failing after 5m13s
CI / Check contributors (push) Failing after 12m8s
CI / Docs Site (push) Failing after 12m8s
CI / TypeScript (push) Failing after 12m8s
CI / Python lints (push) Failing after 12m9s
CI / Python tests (push) Failing after 12m9s
CI / Check uv.lock (push) Failing after 23m22s
CI / CI timing report (push) Has been cancelled
Build Skills Index / trigger-deploy (push) Has been cancelled
CI / All required checks pass (push) Has been cancelled
chore: import upstream snapshot with attribution
2026-07-13 11:56:03 +08:00

177 lines
6.5 KiB
Python

"""Regression tests for empty-response recovery transcript persistence."""
from run_agent import AIAgent
class _CapturingSessionDB:
"""Minimal SessionDB stand-in that records every appended message."""
def __init__(self):
self.rows = []
def append_message(self, session_id, role, content=None, **kwargs):
self.rows.append({"role": role, "content": content})
return len(self.rows)
def _agent_with_capturing_db():
agent = AIAgent.__new__(AIAgent)
agent._persist_user_message_idx = None
agent._persist_user_message_override = None
agent._session_db = _CapturingSessionDB()
agent._session_db_created = True
agent._last_flushed_db_idx = 0
agent.session_id = "sess-test"
return agent
def _agent_with_stubbed_persistence():
agent = AIAgent.__new__(AIAgent)
agent._persist_user_message_idx = None
agent._persist_user_message_override = None
agent._session_db = None
agent._session_messages = []
agent.flushed_session_db_messages = []
agent._flush_messages_to_session_db = lambda messages, conversation_history=None: (
agent.flushed_session_db_messages.append([m.copy() for m in messages])
)
return agent
def test_persist_session_strips_trailing_empty_recovery_scaffolding():
"""After stripping scaffolding, also rewind past orphan trailing tool-result
messages that the failed iteration left behind. Otherwise the next user
message lands after a bare ``tool`` and produces a protocol-invalid
sequence that most providers silently fail on, retriggering the empty-
retry loop indefinitely.
"""
agent = _agent_with_stubbed_persistence()
messages = [
{"role": "user", "content": "run the task"},
{
"role": "assistant",
"content": "",
"tool_calls": [{"id": "call_1", "type": "function",
"function": {"name": "x", "arguments": "{}"}}],
},
{"role": "tool", "content": "{}", "tool_call_id": "call_1"},
{
"role": "assistant",
"content": "(empty)",
"_empty_recovery_synthetic": True,
},
{
"role": "user",
"content": (
"You just executed tool calls but returned an empty response. "
"Please process the tool results above and continue with the task."
),
"_empty_recovery_synthetic": True,
},
]
AIAgent._persist_session(agent, messages, conversation_history=[])
# After strip + rewind, only the original user message remains. The
# assistant(tool_calls) + tool pair is dropped because its iteration
# never produced a real response.
assert messages == [
{"role": "user", "content": "run the task"},
]
assert agent.flushed_session_db_messages[-1] == messages
assert all(not msg.get("_empty_recovery_synthetic") for msg in messages)
def test_persist_session_keeps_unmarked_terminal_empty_response():
agent = _agent_with_stubbed_persistence()
messages = [
{"role": "user", "content": "run the task"},
{"role": "assistant", "content": "(empty)"},
]
AIAgent._persist_session(agent, messages, conversation_history=[])
assert messages == [
{"role": "user", "content": "run the task"},
{"role": "assistant", "content": "(empty)"},
]
assert agent.flushed_session_db_messages[-1] == messages
def test_persist_session_strips_marked_terminal_empty_sentinel():
agent = _agent_with_stubbed_persistence()
messages = [
{"role": "user", "content": "continue"},
{
"role": "assistant",
"content": "(empty)",
"_empty_terminal_sentinel": True,
},
]
AIAgent._persist_session(agent, messages, conversation_history=[])
assert messages == [{"role": "user", "content": "continue"}]
assert agent.flushed_session_db_messages[-1] == messages
assert all(not msg.get("_empty_terminal_sentinel") for msg in messages)
def test_flush_never_writes_buried_empty_recovery_scaffolding():
"""When an empty-after-tools nudge is followed by a tool-calling response,
the synthetic ``(empty)`` + nudge pair stays buried in the live message
list (only the trailing copies are ever dropped). The append-only flush
must skip it regardless of position, otherwise the synthetic turns land in
the session store and pollute every resumed transcript.
"""
agent = _agent_with_capturing_db()
messages = [
{"role": "user", "content": "run the task"},
{
"role": "assistant",
"content": "",
"tool_calls": [{"id": "call_1", "type": "function",
"function": {"name": "x", "arguments": "{}"}}],
},
{"role": "tool", "content": "{}", "tool_call_id": "call_1"},
# Synthetic recovery scaffolding, now buried because the model answered
# the nudge with another tool call rather than terminating.
{"role": "assistant", "content": "(empty)", "_empty_recovery_synthetic": True},
{
"role": "user",
"content": "You just executed tool calls but returned an empty response.",
"_empty_recovery_synthetic": True,
},
{
"role": "assistant",
"content": "",
"tool_calls": [{"id": "call_2", "type": "function",
"function": {"name": "x", "arguments": "{}"}}],
},
{"role": "tool", "content": "{}", "tool_call_id": "call_2"},
{"role": "assistant", "content": "All done."},
]
agent._flush_messages_to_session_db(messages, conversation_history=[])
persisted = agent._session_db.rows
assert all(row["content"] != "(empty)" for row in persisted)
assert all("empty response" not in (row["content"] or "") for row in persisted)
# Only the genuine turns reach the store, in order.
assert [r["role"] for r in persisted] == [
"user", "assistant", "tool", "assistant", "tool", "assistant",
]
assert persisted[-1]["content"] == "All done."
def test_flush_skips_thinking_prefill_scaffolding():
agent = _agent_with_capturing_db()
messages = [
{"role": "user", "content": "hi"},
{"role": "assistant", "content": "", "_thinking_prefill": True},
{"role": "assistant", "content": "Hello!"},
]
agent._flush_messages_to_session_db(messages, conversation_history=[])
assert [r["content"] for r in agent._session_db.rows] == ["hi", "Hello!"]