import unittest from pathlib import Path from tempfile import TemporaryDirectory from cockpit_grounding.benchmark.config import BenchmarkSample from cockpit_grounding.benchmark.runner import _benchmark_sample from cockpit_grounding.benchmark.statistics import ( percentile, summarize_model, summarize_timings, ) from cockpit_grounding.models.base import InferenceTiming def _timing(total: float) -> InferenceTiming: return InferenceTiming( preprocess_ms=1.0, generate_ms=total - 2.0, decode_ms=1.0, total_ms=total, ) class BenchmarkStatisticsTest(unittest.TestCase): def test_percentile_uses_linear_interpolation(self) -> None: self.assertEqual(percentile([10.0, 20.0, 30.0], 0.5), 20.0) self.assertAlmostEqual(percentile([10.0, 20.0], 0.95), 19.5) def test_summarize_timings_handles_empty_input(self) -> None: self.assertIsNone(summarize_timings([])["total_mean_ms"]) def test_summarize_model_without_gt_uses_null_accuracy(self) -> None: predictions = [ { "parse_success": True, "point_in_box": None, "bbox_iou": None, "normalized_center_error": None, "peak_cuda_memory_mb": 123.0, } ] summary = summarize_model( model_name="model", model_path="/model", model_load_seconds=2.0, memory_metrics={ "model_cuda_allocated_mb": 100.0, "model_cuda_reserved_mb": 120.0, }, predictions=predictions, all_timings=[_timing(10.0), _timing(20.0)], ) self.assertEqual(summary["parse_success_rate"], 1.0) self.assertEqual(summary["mean_total_ms"], 15.0) self.assertAlmostEqual(summary["throughput_samples_per_sec"], 1000 / 15) self.assertEqual(summary["peak_cuda_memory_mb"], 123.0) self.assertIsNone(summary["accuracy_point_in_box"]) def test_parse_failure_is_returned_instead_of_raised(self) -> None: class InvalidOutputGrounder: def generate_with_metrics(self, **_: object): return ( "not a bbox", _timing(10.0), {"peak_cuda_memory_mb": 100.0}, ) with TemporaryDirectory() as directory: tmp_path = Path(directory) image = tmp_path / "image.jpg" image.touch() prediction, timings = _benchmark_sample( grounder=InvalidOutputGrounder(), # type: ignore[arg-type] model_name="model", sample=BenchmarkSample( sample_id="sample", image=image, target="button", ), repeats=3, max_new_tokens=128, visualization_path=tmp_path / "visualization.jpg", ) self.assertFalse(prediction["parse_success"]) self.assertIn("ValueError", prediction["parse_error"]) self.assertEqual(len(timings), 3) self.assertEqual(prediction["total_mean_ms"], 10.0)