Source code for toolscore.mcp.scorecard

"""Scoring and reporting for the MCP Scorecard.

This module aggregates the raw outputs of :mod:`toolscore.mcp.harness` -- the
server's tools, the executed :class:`~toolscore.mcp.harness.ScenarioResult`
list, and the :class:`~toolscore.mcp.harness.LintIssue` list -- into a single
:class:`MCPScorecard` with an A--F letter grade, and renders it as a rich
console panel, as PR-comment-friendly Markdown, or as a JSON dict.

Scoring
-------
The overall score is a weighted blend in ``[0, 1]``::

    score = 0.6 * happy_pass_rate
          + 0.2 * edge_resilience_rate
          + 0.2 * lint_score

where

* ``happy_pass_rate`` is the fraction of *happy* scenarios that passed (a tool
  that does what it advertises). With no happy scenarios this term is ``1.0``
  (full credit -- there is nothing to fail).
* ``edge_resilience_rate`` is the fraction of *edge* scenarios where the server
  responded without crashing or timing out. With no edge scenarios this term is
  ``1.0``.
* ``lint_score`` rewards clean schemas::

      lint_score = max(0, 1 - (errors * 0.25 + warnings * 0.1) / max(num_tools, 1))

Letter grades follow the usual bands: ``>= 0.9`` → A, ``>= 0.8`` → B,
``>= 0.7`` → C, ``>= 0.6`` → D, otherwise F. All rates guard against division by
zero, so a server with zero scenarios still scores cleanly.
"""

from __future__ import annotations

from dataclasses import dataclass
from typing import TYPE_CHECKING, Any

from rich.console import Console
from rich.panel import Panel
from rich.table import Table
from rich.text import Text

from toolscore.mcp.harness import estimate_tokens, tool_definition_tokens
from toolscore.verdict import GRADE_ORDER, FixSuggestion, letter_grade

if TYPE_CHECKING:
    from toolscore.mcp.client import MCPToolDef
    from toolscore.mcp.harness import LintIssue, ScenarioResult

#: Weight of the happy-path pass rate in the overall score.
_WEIGHT_HAPPY = 0.6
#: Weight of the edge-case resilience rate in the overall score.
_WEIGHT_EDGE = 0.2
#: Weight of the lint score in the overall score.
_WEIGHT_LINT = 0.2

#: Per-error penalty applied to the lint score.
_LINT_ERROR_PENALTY = 0.25
#: Per-warning penalty applied to the lint score.
_LINT_WARNING_PENALTY = 0.1

#: Rich color per grade for console rendering.
_GRADE_COLORS: dict[str, str] = {
    "A": "bright_green",
    "B": "green",
    "C": "yellow",
    "D": "orange3",
    "F": "red",
}


[docs] @dataclass class MCPScorecard: """An aggregated quality report for a single MCP server. Attributes: server_info: The ``serverInfo`` dict reported during the handshake. tools: The tools advertised by the server. results: The executed scenario results. lint: The lint issues found in the tool schemas and the text a model reads. instructions: The server ``instructions`` from the handshake, if any. MCP clients send them to the model with every session, so they count toward the context cost. """ server_info: dict[str, Any] tools: list[MCPToolDef] results: list[ScenarioResult] lint: list[LintIssue] instructions: str | None = None # -- component rates --------------------------------------------------- @property def happy_pass_rate(self) -> float: """Fraction of happy-path scenarios that passed (``1.0`` if none).""" happy = [r for r in self.results if r.scenario.kind == "happy"] if not happy: return 1.0 return sum(1 for r in happy if r.ok) / len(happy) @property def edge_resilience_rate(self) -> float: """Fraction of edge scenarios the server survived (``1.0`` if none).""" edge = [r for r in self.results if r.scenario.kind == "edge"] if not edge: return 1.0 return sum(1 for r in edge if r.ok) / len(edge) @property def lint_error_count(self) -> int: """Number of ``error``-severity lint issues.""" return sum(1 for i in self.lint if i.severity == "error") @property def lint_warning_count(self) -> int: """Number of ``warning``-severity lint issues.""" return sum(1 for i in self.lint if i.severity == "warning") @property def lint_score(self) -> float: """Schema cleanliness score in ``[0, 1]`` (1.0 is spotless).""" denominator = max(len(self.tools), 1) penalty = ( self.lint_error_count * _LINT_ERROR_PENALTY + self.lint_warning_count * _LINT_WARNING_PENALTY ) / denominator return max(0.0, 1.0 - penalty) # -- aggregate --------------------------------------------------------- @property def score(self) -> float: """The overall blended score in ``[0, 1]``.""" return ( _WEIGHT_HAPPY * self.happy_pass_rate + _WEIGHT_EDGE * self.edge_resilience_rate + _WEIGHT_LINT * self.lint_score ) @property def grade(self) -> str: """The letter grade (``"A"`` … ``"F"``) for :attr:`score`.""" return letter_grade(self.score) @property def total_tool_tokens(self) -> int: """Estimated total context tokens consumed by all tool definitions. Tool definitions are sent to the model on every request, so this is a proxy for how much of the context window the server's tools occupy. """ return sum(tool_definition_tokens(t) for t in self.tools) @property def instructions_tokens(self) -> int: """Estimated context tokens of the server ``instructions`` (``0`` if none).""" return estimate_tokens(self.instructions) if self.instructions else 0 @property def context_tokens(self) -> int: """Estimated context the server costs on every request: tools plus instructions.""" return self.total_tool_tokens + self.instructions_tokens
def _tool_summaries(card: MCPScorecard) -> list[dict[str, Any]]: """Summarize per-tool scenario outcomes. Args: card: The scorecard to summarize. Returns: One dict per tool with scenario pass counts and average latency, in the order the tools were advertised. """ summaries: list[dict[str, Any]] = [] for tool in card.tools: results = [r for r in card.results if r.scenario.tool == tool.name] total = len(results) passed = sum(1 for r in results if r.ok) durations = [r.duration for r in results] avg_latency = sum(durations) / len(durations) if durations else 0.0 summaries.append( { "name": tool.name, "passed": passed, "total": total, "avg_latency_ms": round(avg_latency * 1000.0, 1), "tokens": tool_definition_tokens(tool), } ) return summaries
[docs] def grade_meets(grade: str, threshold: str) -> bool: """Report whether ``grade`` is at least as good as ``threshold``. Args: grade: The achieved grade (case-insensitive). threshold: The minimum acceptable grade (case-insensitive). Returns: ``True`` if ``grade`` ranks the same as or better than ``threshold``. Unknown grades are treated as the worst possible. """ order = {letter: rank for rank, letter in enumerate(GRADE_ORDER)} worst = len(GRADE_ORDER) return order.get(grade.upper(), worst) <= order.get(threshold.upper(), worst)
#: Priority bands for fix suggestions (lower numbers are more urgent). _PRIORITY_HAPPY_FAIL = 0 _PRIORITY_EDGE_CRASH = 1 _PRIORITY_LINT_ERROR = 2 _PRIORITY_LINT_WARNING = 3 #: Rich color per fix priority, worst-first. _PRIORITY_COLORS: dict[int, str] = { _PRIORITY_HAPPY_FAIL: "bold red", _PRIORITY_EDGE_CRASH: "red", _PRIORITY_LINT_ERROR: "yellow", _PRIORITY_LINT_WARNING: "dim", } #: Max number of fix items rendered in the console verdict before truncating. _MAX_FIXES_SHOWN = 8 def build_fix_list(card: MCPScorecard, limit: int | None = None) -> list[FixSuggestion]: """Merge scenario failures and lint issues into one ranked, actionable list. The ranking, worst first: 1. tools that **fail on valid input** (a happy-path scenario errored), 2. tools that **crash on bad input** (an edge scenario timed out / crashed), 3. schema **errors**, then 4. schema **warnings**. Scenario failures are de-duplicated per tool (one entry per tool per kind), since a tool that fails one happy path usually fails them all. Args: card: The scorecard to analyze. limit: If given, return at most this many suggestions. Returns: The suggestions, sorted by priority then tool name. """ suggestions: list[FixSuggestion] = [] happy_seen: set[str] = set() edge_seen: set[str] = set() for result in card.results: if result.ok: continue tool = result.scenario.tool detail = result.detail or "no detail" if result.scenario.kind == "happy" and tool not in happy_seen: happy_seen.add(tool) suggestions.append( FixSuggestion( tool=tool, problem=f"fails on valid input ({detail})", fix=( "The tool errors on well-formed arguments - check the handler and that " "the input schema matches what the tool actually accepts." ), priority=_PRIORITY_HAPPY_FAIL, ) ) elif result.scenario.kind == "edge" and tool not in edge_seen: edge_seen.add(tool) suggestions.append( FixSuggestion( tool=tool, problem=f"crashes on bad input ({detail})", fix=( "Validate and guard arguments so the tool returns a clean error result " "instead of crashing or timing out." ), priority=_PRIORITY_EDGE_CRASH, ) ) for issue in card.lint: priority = _PRIORITY_LINT_ERROR if issue.severity == "error" else _PRIORITY_LINT_WARNING suggestions.append( FixSuggestion( tool=issue.tool, problem=issue.message, fix=issue.fix or "Review the tool's schema and description.", priority=priority, ) ) suggestions.sort(key=lambda s: (s.priority, s.tool)) if limit is not None: suggestions = suggestions[:limit] return suggestions
[docs] def scorecard_to_markdown(card: MCPScorecard) -> str: """Render the scorecard as Markdown for READMEs or PR comments. Args: card: The scorecard to render. Returns: A Markdown document with a grade line, component scores (including the tool-definition token cost), a per-tool summary table, and a ranked "Top issues to fix" list. """ name = str(card.server_info.get("name", "unknown server")) version = str(card.server_info.get("version", "")) title = f"{name} {version}".strip() lines: list[str] = [] lines.append(f"# MCP Scorecard: {title}") lines.append("") lines.append(f"**Grade: {card.grade}** &middot; Score {card.score:.0%}") lines.append("") lines.append( f"- Happy-path pass rate: {card.happy_pass_rate:.0%}\n" f"- Edge-case resilience: {card.edge_resilience_rate:.0%}\n" f"- Lint score: {card.lint_score:.0%} " f"({card.lint_error_count} errors, {card.lint_warning_count} warnings)\n" f"- Tool-definition tokens (estimated): ~{card.total_tool_tokens} " f"across {len(card.tools)} tool(s)" ) if card.instructions: lines.append( f"- Server instructions (estimated): ~{card.instructions_tokens} tokens " f"(~{card.context_tokens} in total, sent with every request)" ) lines.append("") lines.append("## Tools") lines.append("") lines.append("| Tool | Scenarios | Avg latency | Def. tokens |") lines.append("| --- | --- | --- | --- |") for summary in _tool_summaries(card): lines.append( f"| `{summary['name']}` | {summary['passed']}/{summary['total']} | " f"{summary['avg_latency_ms']:.1f} ms | {summary['tokens']} |" ) lines.append("") lines.append("## Top issues to fix") lines.append("") fixes = build_fix_list(card) if fixes: for suggestion in fixes: lines.append( f"- **`{suggestion.tool}`** — {suggestion.problem}. _Fix:_ {suggestion.fix}" ) else: lines.append("- No issues found — an LLM should be able to use this server cleanly.") lines.append("") return "\n".join(lines)
[docs] def scorecard_to_json(card: MCPScorecard) -> dict[str, Any]: """Render the scorecard as a JSON-serializable dict. Args: card: The scorecard to render. Returns: A plain dict (lists/dicts/strings/numbers only) capturing the grade, component scores, per-tool summaries, every scenario result, and every lint issue. """ return { "server": { "name": card.server_info.get("name", ""), "version": card.server_info.get("version", ""), }, "grade": card.grade, "score": card.score, "total_tool_tokens": card.total_tool_tokens, "instructions_tokens": card.instructions_tokens, "context_tokens": card.context_tokens, "scores": { "happy_pass_rate": card.happy_pass_rate, "edge_resilience_rate": card.edge_resilience_rate, "lint_score": card.lint_score, }, "tools": _tool_summaries(card), "results": [ { "tool": r.scenario.tool, "kind": r.scenario.kind, "description": r.scenario.description, "arguments": r.scenario.arguments, "ok": r.ok, "is_error": r.is_error, "duration": r.duration, "detail": r.detail, } for r in card.results ], "lint": [ {"tool": i.tool, "severity": i.severity, "message": i.message, "fix": i.fix} for i in card.lint ], "fixes": [ {"tool": s.tool, "problem": s.problem, "fix": s.fix, "priority": s.priority} for s in build_fix_list(card) ], }