Skip to content

Aggregate Comparison

Compare metric results across multiple models side-by-side — as a table or a grouped bar chart.

Install the metrics extra before using these APIs:

pip install "supervision[metrics]"

Compare Metric Results

Compute the same metric for each model, then aggregate the resulting scores in a table or a grouped bar chart.

import numpy as np
import supervision as sv
from supervision.metrics import (
    F1Score,
    aggregate_metric_results,
    plot_aggregate_metric_results,
)

targets = sv.Detections(
    xyxy=np.array([[0, 0, 10, 10]]),
    class_id=np.array([0]),
)
model_a_predictions = sv.Detections(
    xyxy=np.array([[0, 0, 10, 10]]),
    class_id=np.array([0]),
    confidence=np.array([0.9]),
)
model_b_predictions = sv.Detections(
    xyxy=np.array([[3, 3, 10, 10]]),
    class_id=np.array([0]),
    confidence=np.array([0.9]),
)

metric_results = [
    F1Score().update(model_a_predictions, targets).compute(),
    F1Score().update(model_b_predictions, targets).compute(),
]
model_names = ["Model A", "Model B"]

comparison = aggregate_metric_results(metric_results, model_names=model_names)
print(comparison[["F1@50", "F1@75"]])

plot_aggregate_metric_results(
    metric_results,
    model_names=model_names,
    show=True,
)

Functions

supervision.metrics.utils.aggregate.aggregate_metric_results(metric_results: list[MetricResult], *, model_names: list[str] | None = None, include_object_sizes: bool = False) -> pd.DataFrame

Combine several :class:MetricResult objects into a single DataFrame.

Each row corresponds to one result (one model). All results must be of the same concrete type (e.g. all :class:F1ScoreResult).

Parameters:

Name Type Description Default

metric_results

list[MetricResult]

A list of metric results to aggregate.

required

model_names

list[str] | None

Optional display names for each result. When provided, the DataFrame index is set to these names. Must have the same length as metric_results.

None

include_object_sizes

bool

When True, include columns for small / medium / large object-size categories.

False

Returns:

Name Type Description
A DataFrame

class:~pandas.DataFrame with one row per result and columns for

DataFrame

each metric value.

Raises:

Type Description
ValueError

If the list is empty or model_names length does not match metric_results.

TypeError

If the list contains mixed result types.

Source code in src/supervision/metrics/utils/aggregate.py
def aggregate_metric_results(
    metric_results: list[MetricResult],
    *,
    model_names: list[str] | None = None,
    include_object_sizes: bool = False,
) -> pd.DataFrame:
    """Combine several :class:`MetricResult` objects into a single DataFrame.

    Each row corresponds to one result (one model). All results must be of the
    same concrete type (e.g. all :class:`F1ScoreResult`).

    Args:
        metric_results: A list of metric results to aggregate.
        model_names: Optional display names for each result. When provided,
            the DataFrame index is set to these names. Must have the same
            length as *metric_results*.
        include_object_sizes: When ``True``, include columns for
            small / medium / large object-size categories.

    Returns:
        A :class:`~pandas.DataFrame` with one row per result and columns for
        each metric value.

    Raises:
        ValueError: If the list is empty or *model_names* length does not
            match *metric_results*.
        TypeError: If the list contains mixed result types.
    """
    if not metric_results:
        raise ValueError("metric_results must not be empty.")

    ensure_pandas_installed()
    import pandas as pd

    result_type = type(metric_results[0])
    for result in metric_results[1:]:
        if type(result) is not result_type:
            raise TypeError(
                f"All metric results must be the same type. "
                f"Expected {result_type.__name__}, "
                f"got {type(result).__name__}."
            )

    if model_names is not None and len(model_names) != len(metric_results):
        raise ValueError(
            f"model_names length ({len(model_names)}) must match "
            f"metric_results length ({len(metric_results)})."
        )

    frames = [result.to_pandas() for result in metric_results]
    df = pd.concat(frames, ignore_index=True)

    if not include_object_sizes:
        size_prefixes = ("small_objects_", "medium_objects_", "large_objects_")
        cols_to_drop = [col for col in df.columns if col.startswith(size_prefixes)]
        df = df.drop(columns=cols_to_drop)

    if model_names is not None:
        df.index = pd.Index(model_names)

    return df

supervision.metrics.utils.aggregate.plot_aggregate_metric_results(metric_results: list[MetricResult], *, model_names: list[str] | None = None, include_object_sizes: bool = False, show: bool = False) -> None

Plot multiple :class:MetricResult objects on a single grouped bar chart.

Each group of bars corresponds to a metric label (e.g. "F1@50"), and each bar within the group corresponds to one model.

Parameters:

Name Type Description Default

metric_results

list[MetricResult]

A list of metric results to plot.

required

model_names

list[str] | None

Optional display names for each result (used in the legend). When None, results are labelled "Model 1", "Model 2", etc.

None

include_object_sizes

bool

When True, include bars for small / medium / large object-size categories.

False

show

bool

When True, display the completed plot with plt.show().

False

Raises:

Type Description
ValueError

If the list is empty, model_names length does not match metric_results, or results have mismatched plot details.

TypeError

If the list contains mixed result types.

Source code in src/supervision/metrics/utils/aggregate.py
def plot_aggregate_metric_results(
    metric_results: list[MetricResult],
    *,
    model_names: list[str] | None = None,
    include_object_sizes: bool = False,
    show: bool = False,
) -> None:
    """Plot multiple :class:`MetricResult` objects on a single grouped bar chart.

    Each group of bars corresponds to a metric label (e.g. ``"F1@50"``), and
    each bar within the group corresponds to one model.

    Args:
        metric_results: A list of metric results to plot.
        model_names: Optional display names for each result (used in the
            legend). When ``None``, results are labelled ``"Model 1"``,
            ``"Model 2"``, etc.
        include_object_sizes: When ``True``, include bars for
            small / medium / large object-size categories.
        show: When ``True``, display the completed plot with ``plt.show()``.

    Raises:
        ValueError: If the list is empty, *model_names* length does not
            match *metric_results*, or results have mismatched plot details.
        TypeError: If the list contains mixed result types.
    """
    from matplotlib import pyplot as plt

    if not metric_results:
        raise ValueError("metric_results must not be empty.")

    result_type = type(metric_results[0])
    for result in metric_results[1:]:
        if type(result) is not result_type:
            raise TypeError(
                f"All metric results must be the same type. "
                f"Expected {result_type.__name__}, "
                f"got {type(result).__name__}."
            )

    if model_names is not None and len(model_names) != len(metric_results):
        raise ValueError(
            f"model_names length ({len(model_names)}) must match "
            f"metric_results length ({len(metric_results)})."
        )

    if model_names is None:
        model_names = [f"Model {i + 1}" for i in range(len(metric_results))]

    all_details = [
        result._get_plot_details(include_object_sizes=include_object_sizes)
        for result in metric_results
    ]

    labels = all_details[0].labels
    title = all_details[0].title
    for i, details in enumerate(all_details[1:], 1):
        if details.labels != labels:
            raise ValueError(
                f"Label mismatch: result 0 has {labels}, "
                f"result {i} has {details.labels}."
            )
        if details.title != title:
            raise ValueError(
                f"Plot configuration mismatch: result 0 has title {title!r}, "
                f"result {i} has title {details.title!r}."
            )
    num_models = len(metric_results)
    num_labels = len(labels)

    x = np.arange(num_labels)
    bar_width = 0.8 / num_models

    plt.rcParams["font.family"] = "monospace"
    _, ax = plt.subplots(figsize=(max(10, num_labels * 1.5), 6))
    ax.set_ylim(0, 1)
    ax.set_ylabel("Value", fontweight="bold")
    ax.set_title(title, fontweight="bold")

    for model_idx, (name, details) in enumerate(zip(model_names, all_details)):
        offset = (model_idx - num_models / 2 + 0.5) * bar_width
        color = LEGACY_COLOR_PALETTE[model_idx % len(LEGACY_COLOR_PALETTE)]
        bars = ax.bar(
            x + offset,
            details.values,
            bar_width,
            label=name,
            color=color,
        )
        for bar in bars:
            y_value = bar.get_height()
            ax.text(
                bar.get_x() + bar.get_width() / 2,
                y_value + 0.02,
                f"{y_value:.2f}",
                ha="center",
                va="bottom",
                fontsize=max(6, 8 - num_models),
            )

    ax.set_xticks(x)
    ax.set_xticklabels(labels, rotation=45, ha="right")
    ax.legend()

    plt.rcParams["font.family"] = "sans-serif"

    plt.tight_layout()
    if show:
        plt.show()

Supporting Types

supervision.metrics.core.MetricResult

Bases: ABC

Abstract base class shared by all metric result dataclasses.

Source code in src/supervision/metrics/core.py
class MetricResult(ABC):
    """Abstract base class shared by all metric result dataclasses."""

    @abstractmethod
    def to_pandas(self) -> pd.DataFrame:
        """Convert the result to a :class:`~pandas.DataFrame`."""
        raise NotImplementedError

    @abstractmethod
    def plot(self) -> None:
        """Render a bar-chart of the result."""
        raise NotImplementedError

    @abstractmethod
    def _get_plot_details(self, include_object_sizes: bool = True) -> PlotDetails:
        """Return labels, values, colors, and title for a bar chart.

        Args:
            include_object_sizes: When ``True`` (default), include bars for
                small / medium / large object-size categories.
        """
        raise NotImplementedError

Methods:

plot() -> None abstractmethod

Render a bar-chart of the result.

Source code in src/supervision/metrics/core.py
@abstractmethod
def plot(self) -> None:
    """Render a bar-chart of the result."""
    raise NotImplementedError

to_pandas() -> pd.DataFrame abstractmethod

Convert the result to a :class:~pandas.DataFrame.

Source code in src/supervision/metrics/core.py
@abstractmethod
def to_pandas(self) -> pd.DataFrame:
    """Convert the result to a :class:`~pandas.DataFrame`."""
    raise NotImplementedError

supervision.metrics.core.PlotDetails dataclass

Container for bar-chart data returned by MetricResult._get_plot_details.

Attributes:

Name Type Description
labels list[str]

Bar labels (x-axis tick labels).

values list[float]

Bar heights (metric values).

colors list[str]

One hex color string per bar (e.g. "#A351FB").

title str

Chart title.

Source code in src/supervision/metrics/core.py
@dataclass
class PlotDetails:
    """Container for bar-chart data returned by ``MetricResult._get_plot_details``.

    Attributes:
        labels: Bar labels (x-axis tick labels).
        values: Bar heights (metric values).
        colors: One hex color string per bar (e.g. ``"#A351FB"``).
        title: Chart title.
    """

    labels: list[str] = field(default_factory=list)
    values: list[float] = field(default_factory=list)
    colors: list[str] = field(default_factory=list)
    title: str = ""

Comments