Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
47 changes: 46 additions & 1 deletion raven/agent/loop/main.py
Original file line number Diff line number Diff line change
Expand Up @@ -51,6 +51,12 @@
from raven.tracing import semconv, trace
from raven.utils.helpers import estimate_prompt_tokens

_ABORTED_ACTION_REPLY = (
"The operation was not completed, and no alternative method will be attempted. "
"Would you like me to continue with the remaining parts of the task that do not "
"require this operation?"
)

# NOTE: ``raven.context_engine`` is intentionally imported lazily (inside
# ``__init__`` and ``_assemble_context_messages``) to break a runtime
# import cycle: ``raven.agent.__init__`` eagerly loads AgentLoop,
Expand Down Expand Up @@ -1584,6 +1590,7 @@ async def _run_agent_loop(
continue

if response.has_tool_calls:
abort_action = False
if on_progress:
thought = self._strip_think(response.content)
if thought:
Expand All @@ -1599,7 +1606,7 @@ async def _run_agent_loop(
thinking_blocks=response.thinking_blocks,
)

for tool_call in response.tool_calls:
for tool_call_index, tool_call in enumerate(response.tool_calls):
tools_used.append(tool_call.name)
args_str = json.dumps(tool_call.arguments, ensure_ascii=False)
logger.info("Tool call: {}({})", tool_call.name, args_str[:200])
Expand All @@ -1618,6 +1625,10 @@ async def _run_agent_loop(
"display": _tool.display_call(tool_call.arguments) if _tool else None,
},
)
if tool_call.name == "exec":
exec_tool = self.tools.get("exec")
if isinstance(exec_tool, ExecTool):
exec_tool.set_tool_call_id(tool_call.id)
tool_t0 = time.monotonic()
result = await self.tools.execute(tool_call.name, tool_call.arguments)
duration_ms = int((time.monotonic() - tool_t0) * 1000)
Expand Down Expand Up @@ -1647,6 +1658,26 @@ async def _run_agent_loop(
},
)
messages = self.context.add_tool_result(messages, tool_call.id, tool_call.name, model_text)
if getattr(result, "abort_action", False):
abort_action = True
# A single assistant message may contain several parallel
# tool calls (for example ``rm`` followed by a Python
# fallback). Once policy terminates the action, none of
# the siblings may execute. We must nevertheless append
# one result for every advertised call id: OpenAI-style
# providers reject conversation history containing an
# assistant tool call without its matching tool result.
for skipped_call in response.tool_calls[tool_call_index + 1 :]:
messages = self.context.add_tool_result(
messages,
skipped_call.id,
skipped_call.name,
(
"Error: Tool call was not executed because a prior safety "
"decision terminated this action."
),
)
break
# #1b Track consecutive same-tool deterministic failures
# (transient errors excluded — a retry would clear those).
if _is_hard_tool_failure(model_text):
Expand All @@ -1657,6 +1688,20 @@ async def _run_agent_loop(
else:
loop_fail_tool, loop_fail_streak = None, 0

if abort_action:
# A normal tool result starts another model iteration. That
# is specifically unsafe here: the next plan can translate
# the rejected operation into an equivalent interpreter,
# script, or tool call. Finish the turn in runtime code and
# expose only the non-destructive continuation question.
# Streaming callers need the explicit callback because no
# final model response exists to generate token deltas.
messages = self.context.add_assistant_message(messages, _ABORTED_ACTION_REPLY)
final_content = _ABORTED_ACTION_REPLY
if on_token_delta is not None:
await on_token_delta(_ABORTED_ACTION_REPLY)
break

# #1b Failure-loop break: the same tool failed deterministically
# `threshold` times running → append a change-approach nudge to
# the last tool result so the model stops repeating a dead call.
Expand Down
20 changes: 18 additions & 2 deletions raven/agent/subagent/manager.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,10 @@
# One hour: a runaway re-injection loop fires fast and trips the limit quickly,
# while legitimate spawns spread over time and age out before it bites.
_SPAWN_WINDOW_SECONDS = 3600
_ABORTED_ACTION_RESULT = (
"The subtask stopped because a safety decision terminated the requested operation. "
"No alternative method was attempted."
)


class SubagentManager:
Expand Down Expand Up @@ -182,6 +186,7 @@ async def _run_subagent_inner(
max_iterations = 15
iteration = 0
final_result: str | None = None
final_status = "ok"

while iteration < max_iterations:
iteration += 1
Expand Down Expand Up @@ -220,15 +225,26 @@ async def _run_subagent_inner(
"content": wrap_untrusted(result, source=tool_call.name),
}
)
if getattr(result, "abort_action", False):
# Subagents must enforce the same terminal safety
# signal as the main loop. Returning to the model
# would let it translate a rejected operation into
# another command or interpreter, while continuing
# this batch would execute already-proposed siblings.
final_result = _ABORTED_ACTION_RESULT
final_status = "error"
break
if final_result is not None:
break
else:
final_result = response.content
break

if final_result is None:
final_result = "Task completed but no final response was generated."

logger.info("Subagent [{}] completed successfully", task_id)
await self._announce_result(task_id, label, task, final_result, origin, "ok")
logger.info("Subagent [{}] finished with status {}", task_id, final_status)
await self._announce_result(task_id, label, task, final_result, origin, final_status)

except Exception as e:
error_msg = f"Error: {str(e)}"
Expand Down
22 changes: 19 additions & 3 deletions raven/agent/tools/base.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,10 +14,15 @@ class ToolResult:
to a generic preview of it) or this, when the tool wants a cleaner
transcript rendering than what it feeds the model. ``display_text`` must be
built from the tool's own execution data, not by re-parsing ``model_text``.
``retryable=False`` suppresses the registry's generic change-approach hint.
``abort_action=True`` tells the agent loop not to execute sibling calls or
ask the model for another approach.
"""

model_text: str
display_text: str | None = None
retryable: bool = True
abort_action: bool = False


class ToolOutput(str):
Expand All @@ -30,14 +35,25 @@ class ToolOutput(str):
artifact, so the boundary has to return something that *is* a str; handing
them a :class:`ToolResult` would format its repr into model context and
user-facing replies. The agent loop reads ``display_text`` off it to render
the transcript row.
the transcript row and the control flags to enforce terminal tool decisions.
"""

display_text: str | None

def __new__(cls, model_text: str, display_text: str | None = None) -> "ToolOutput":
retryable: bool
abort_action: bool

def __new__(
cls,
model_text: str,
display_text: str | None = None,
*,
retryable: bool = True,
abort_action: bool = False,
) -> "ToolOutput":
out = super().__new__(cls, model_text)
out.display_text = display_text
out.retryable = retryable
out.abort_action = abort_action
return out


Expand Down
21 changes: 19 additions & 2 deletions raven/agent/tools/registry.py
Original file line number Diff line number Diff line change
Expand Up @@ -72,12 +72,29 @@ async def execute(self, name: str, params: dict[str, Any]) -> str:
# string (which rides along on ToolOutput).
if isinstance(result, ToolResult):
model_text, display_text = result.model_text, result.display_text
retryable, abort_action = result.retryable, result.abort_action
else:
model_text, display_text = str(result), None
retryable, abort_action = True, False

if model_text.startswith("Error"):
return ToolOutput(model_text + _hint, display_text)
return ToolOutput(model_text, display_text)
# ``Error:`` describes presentation, not retry semantics.
# Policy-aware tools return explicit control metadata so the
# registry does not accidentally turn a security decision into
# the generic invitation to find an equivalent implementation.
suffix = _hint if retryable else ""
return ToolOutput(
model_text + suffix,
display_text,
retryable=retryable,
abort_action=abort_action,
)
return ToolOutput(
model_text,
display_text,
retryable=retryable,
abort_action=abort_action,
)
except asyncio.TimeoutError:
return f"Error: Tool '{name}' timed out after {ceiling:.0f}s." + _hint
except Exception as e:
Expand Down
Loading
Loading