cockpit-ui-grounding/tests/test_benchmark_statistics.py
2026-08-24 16:29:35 +08:00

90 lines
3.1 KiB
Python

import unittest
from pathlib import Path
from tempfile import TemporaryDirectory
from cockpit_grounding.benchmark.config import BenchmarkSample
from cockpit_grounding.benchmark.runner import _benchmark_sample
from cockpit_grounding.benchmark.statistics import (
percentile,
summarize_model,
summarize_timings,
)
from cockpit_grounding.models.base import InferenceTiming
def _timing(total: float) -> InferenceTiming:
return InferenceTiming(
preprocess_ms=1.0,
generate_ms=total - 2.0,
decode_ms=1.0,
total_ms=total,
)
class BenchmarkStatisticsTest(unittest.TestCase):
def test_percentile_uses_linear_interpolation(self) -> None:
self.assertEqual(percentile([10.0, 20.0, 30.0], 0.5), 20.0)
self.assertAlmostEqual(percentile([10.0, 20.0], 0.95), 19.5)
def test_summarize_timings_handles_empty_input(self) -> None:
self.assertIsNone(summarize_timings([])["total_mean_ms"])
def test_summarize_model_without_gt_uses_null_accuracy(self) -> None:
predictions = [
{
"parse_success": True,
"point_in_box": None,
"bbox_iou": None,
"normalized_center_error": None,
"peak_cuda_memory_mb": 123.0,
}
]
summary = summarize_model(
model_name="model",
model_path="/model",
model_load_seconds=2.0,
memory_metrics={
"model_cuda_allocated_mb": 100.0,
"model_cuda_reserved_mb": 120.0,
},
predictions=predictions,
all_timings=[_timing(10.0), _timing(20.0)],
)
self.assertEqual(summary["parse_success_rate"], 1.0)
self.assertEqual(summary["mean_total_ms"], 15.0)
self.assertAlmostEqual(summary["throughput_samples_per_sec"], 1000 / 15)
self.assertEqual(summary["peak_cuda_memory_mb"], 123.0)
self.assertIsNone(summary["accuracy_point_in_box"])
def test_parse_failure_is_returned_instead_of_raised(self) -> None:
class InvalidOutputGrounder:
def generate_with_metrics(self, **_: object):
return (
"not a bbox",
_timing(10.0),
{"peak_cuda_memory_mb": 100.0},
)
with TemporaryDirectory() as directory:
tmp_path = Path(directory)
image = tmp_path / "image.jpg"
image.touch()
prediction, timings = _benchmark_sample(
grounder=InvalidOutputGrounder(), # type: ignore[arg-type]
model_name="model",
sample=BenchmarkSample(
sample_id="sample",
image=image,
target="button",
),
repeats=3,
max_new_tokens=128,
visualization_path=tmp_path / "visualization.jpg",
)
self.assertFalse(prediction["parse_success"])
self.assertIn("ValueError", prediction["parse_error"])
self.assertEqual(len(timings), 3)
self.assertEqual(prediction["total_mean_ms"], 10.0)