Source code for toolscore.metrics.efficiency

"""Tool usage efficiency metrics."""

import json
from itertools import pairwise

from toolscore.adapters.base import ToolCall


def _call_signature(call: ToolCall) -> tuple[str, str]:
    """Tool name plus canonical arguments; ``None`` and ``{}`` both mean no arguments."""
    return call.tool, json.dumps(call.args or {}, sort_keys=True, default=str)


[docs] def calculate_redundant_call_rate( gold_calls: list[ToolCall], trace_calls: list[ToolCall], ) -> dict[str, float]: """Calculate redundant call rate. Measures inefficiency in tool use by quantifying how many tool calls were unnecessary or redundant. Args: gold_calls: Expected tool calls from gold standard. trace_calls: Actual tool calls from agent trace. Returns: Dictionary containing: - redundant_count: Number of redundant calls - total_calls: Total number of calls made - redundant_rate: Proportion of calls that were redundant (0-1) - identical_count: Calls that exactly repeat an earlier call (same tool and same arguments). Unlike ``redundant_count``, which counts calls beyond the gold's per-tool expectation, this isolates loop-like repetition from productive repeated use of a tool. - identical_rate: ``identical_count`` / ``total_calls`` (0-1) - error_count: Calls that failed (see :attr:`ToolCall.is_error`). Zero when the trace carries no error information. - error_rate: ``error_count`` / ``total_calls`` (0-1) - retry_after_error_count: Calls that repeat the immediately preceding call exactly (same tool and arguments) after that call failed, i.e. retrying a failure without changing anything. """ if not trace_calls: return { "redundant_count": 0, "total_calls": 0, "redundant_rate": 0.0, "identical_count": 0, "identical_rate": 0.0, "error_count": 0, "error_rate": 0.0, "retry_after_error_count": 0, } gold_tool_names = [call.tool for call in gold_calls] trace_tool_names = [call.tool for call in trace_calls] # Count expected occurrences of each tool expected_counts: dict[str, int] = {} for tool in gold_tool_names: expected_counts[tool] = expected_counts.get(tool, 0) + 1 # Count actual occurrences actual_counts: dict[str, int] = {} for tool in trace_tool_names: actual_counts[tool] = actual_counts.get(tool, 0) + 1 # Calculate redundant calls redundant_count = 0 for tool, actual_count in actual_counts.items(): expected_count = expected_counts.get(tool, 0) if actual_count > expected_count: redundant_count += actual_count - expected_count total_calls = len(trace_calls) redundant_rate = redundant_count / total_calls if total_calls > 0 else 0.0 identical_count = total_calls - len({_call_signature(call) for call in trace_calls}) error_count = sum(1 for call in trace_calls if call.is_error) retry_after_error_count = sum( 1 for previous, current in pairwise(trace_calls) if previous.is_error and _call_signature(previous) == _call_signature(current) ) return { "redundant_count": redundant_count, "total_calls": total_calls, "redundant_rate": redundant_rate, "identical_count": identical_count, "identical_rate": identical_count / total_calls, "error_count": error_count, "error_rate": error_count / total_calls, "retry_after_error_count": retry_after_error_count, }