commit c9bcfaa6e8e12f6ae81cf3683d128bb5d516c82d Author: lgv Date: Mon Aug 24 16:29:35 2026 +0800 first commit diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..e69de29 diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..43ae0e2 --- /dev/null +++ b/.gitignore @@ -0,0 +1,2 @@ +__pycache__/ +*.py[cod] diff --git a/.idea/.gitignore b/.idea/.gitignore new file mode 100644 index 0000000..30cf57e --- /dev/null +++ b/.idea/.gitignore @@ -0,0 +1,10 @@ +# Default ignored files +/shelf/ +/workspace.xml +# Editor-based HTTP Client requests +/httpRequests/ +# Ignored default folder with query files +/queries/ +# Datasource local storage ignored files +/dataSources/ +/dataSources.local.xml diff --git a/.idea/cockpit-ui-grounding.iml b/.idea/cockpit-ui-grounding.iml new file mode 100644 index 0000000..60665ef --- /dev/null +++ b/.idea/cockpit-ui-grounding.iml @@ -0,0 +1,14 @@ + + + + + + + + + + + + + + \ No newline at end of file diff --git a/.idea/inspectionProfiles/profiles_settings.xml b/.idea/inspectionProfiles/profiles_settings.xml new file mode 100644 index 0000000..105ce2d --- /dev/null +++ b/.idea/inspectionProfiles/profiles_settings.xml @@ -0,0 +1,6 @@ + + + + \ No newline at end of file diff --git a/.idea/modules.xml b/.idea/modules.xml new file mode 100644 index 0000000..1716bf5 --- /dev/null +++ b/.idea/modules.xml @@ -0,0 +1,8 @@ + + + + + + + + \ No newline at end of file diff --git a/.idea/vcs.xml b/.idea/vcs.xml new file mode 100644 index 0000000..35eb1dd --- /dev/null +++ b/.idea/vcs.xml @@ -0,0 +1,6 @@ + + + + + + \ No newline at end of file diff --git a/README.md b/README.md new file mode 100644 index 0000000..e69de29 diff --git a/configs/benchmark/qwen35_08b_vs_2b.toml b/configs/benchmark/qwen35_08b_vs_2b.toml new file mode 100644 index 0000000..894ec9c --- /dev/null +++ b/configs/benchmark/qwen35_08b_vs_2b.toml @@ -0,0 +1,13 @@ +[benchmark] +warmup = 3 +repeats = 5 +max_new_tokens = 128 +output_root = "/data/lgv/runs/cockpit-ui-grounding" + +[models.qwen35_08b] +backend = "qwen35" +path = "/data/lgv/models/pretrained/Qwen3.5-0.8B" + +[models.qwen35_2b] +backend = "qwen35" +path = "/data/lgv/models/pretrained/Qwen3.5-2B" diff --git a/configs/benchmark/qwen3vl2b_vs_qwen35_2b.toml b/configs/benchmark/qwen3vl2b_vs_qwen35_2b.toml new file mode 100644 index 0000000..aea0cd2 --- /dev/null +++ b/configs/benchmark/qwen3vl2b_vs_qwen35_2b.toml @@ -0,0 +1,14 @@ +[benchmark] +warmup = 3 +repeats = 5 +max_new_tokens = 128 +output_root = "/data/lgv/runs/cockpit-ui-grounding" +timestamp_run_directory = false + +[models.qwen3vl_2b] +backend = "qwen3vl" +path = "/data/lgv/models/pretrained/Qwen3-VL-2B-Instruct" + +[models.qwen35_2b] +backend = "qwen35" +path = "/data/lgv/models/pretrained/Qwen3.5-2B" diff --git a/configs/benchmark/qwen3vl_2b_vs_4b.toml b/configs/benchmark/qwen3vl_2b_vs_4b.toml new file mode 100644 index 0000000..119a981 --- /dev/null +++ b/configs/benchmark/qwen3vl_2b_vs_4b.toml @@ -0,0 +1,11 @@ +[benchmark] +warmup = 1 +repeats = 3 +max_new_tokens = 128 +output_root = "/data/lgv/runs/cockpit-ui-grounding" + +[models.qwen3vl_2b] +path = "/data/lgv/models/pretrained/Qwen3-VL-2B-Instruct" + +[models.qwen3vl_4b] +path = "/data/lgv/models/pretrained/Qwen3-VL-4B-Instruct" diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..811b6d3 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,12 @@ +[build-system] +requires = ["setuptools>=68"] +build-backend = "setuptools.build_meta" + +[project] +name = "cockpit-ui-grounding" +version = "0.1.0" +description = "Camera-view automotive cockpit UI grounding" +requires-python = ">=3.11" + +[tool.setuptools.packages.find] +where = ["src"] diff --git a/scripts/benchmark_models.py b/scripts/benchmark_models.py new file mode 100644 index 0000000..490cdb3 --- /dev/null +++ b/scripts/benchmark_models.py @@ -0,0 +1,75 @@ +import argparse +import logging +from pathlib import Path + +from cockpit_grounding.benchmark.config import ( + load_benchmark_config, + load_manifest, +) +from cockpit_grounding.benchmark.reporting import ( + print_comparison, + print_model_selection, +) +from cockpit_grounding.benchmark.runner import run_benchmark + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser( + description="Benchmark local multimodal models sequentially on one GPU", + ) + parser.add_argument( + "--config", + type=Path, + required=True, + help="Benchmark TOML configuration", + ) + parser.add_argument( + "--manifest", + type=Path, + required=True, + help="JSONL benchmark manifest", + ) + parser.add_argument( + "--run-name", + help="Optional run directory suffix", + ) + return parser.parse_args() + + +def main() -> None: + args = parse_args() + logging.basicConfig( + level=logging.INFO, + format="%(asctime)s | %(levelname)s | %(message)s", + ) + + config = load_benchmark_config(args.config) + samples = load_manifest(args.manifest) + run_directory, summaries = run_benchmark( + config=config, + samples=samples, + config_path=args.config, + manifest_path=args.manifest, + run_name=args.run_name, + ) + comparison_title = ( + "QWEN3.5 MODEL COMPARISON" + if all(model.backend == "qwen35" for model in config.models) + else "MODEL COMPARISON" + ) + models_by_name = {model.name: model for model in config.models} + display_summaries = { + ( + Path(models_by_name[name].path).name + if models_by_name[name].backend == "qwen35" + else name + ): summary + for name, summary in summaries.items() + } + print_comparison(display_summaries, title=comparison_title) + print_model_selection(summaries) + print(f"Results: {run_directory}") + + +if __name__ == "__main__": + main() diff --git a/scripts/run_grounding.py b/scripts/run_grounding.py new file mode 100644 index 0000000..99abe52 --- /dev/null +++ b/scripts/run_grounding.py @@ -0,0 +1,128 @@ +import argparse +import json + +from cockpit_grounding.models.factory import create_grounder + +from cockpit_grounding.grounding.predictor import ( + build_grounding_prompt, + parse_grounding_output, +) + +from cockpit_grounding.vision.visualize import ( + visualize_grounding, +) + + +def main(): + parser = argparse.ArgumentParser() + + parser.add_argument( + "--image", + required=True, + ) + + parser.add_argument( + "--target", + required=True, + ) + + parser.add_argument( + "--output", + required=True, + ) + + parser.add_argument( + "--model", + required=True, + help="Local model path", + ) + + parser.add_argument( + "--backend", + choices=("qwen3vl", "qwen35"), + default="qwen3vl", + help="Model backend (default: qwen3vl)", + ) + + args = parser.parse_args() + + # ---------------------------- + # Load model + # ---------------------------- + + grounder = create_grounder( + backend=args.backend, + model_path=args.model, + ) + + # ---------------------------- + # Prompt + # ---------------------------- + + prompt = build_grounding_prompt( + args.target + ) + + # ---------------------------- + # Inference + # ---------------------------- + + raw_output = grounder.generate( + args.image, + prompt, + ) + + print() + print("========== RAW MODEL OUTPUT ==========") + print(raw_output) + + # ---------------------------- + # Parse bbox + # ---------------------------- + + result = parse_grounding_output( + raw_output + ) + + # ---------------------------- + # Convert + draw + # ---------------------------- + + pixel_result = visualize_grounding( + args.image, + result, + args.output, + ) + + final_result = { + "target": args.target, + + "bbox_relative": [ + result.x1, + result.y1, + result.x2, + result.y2, + ], + + **pixel_result, + } + + print() + print("========== FINAL RESULT ==========") + + print( + json.dumps( + final_result, + indent=2, + ensure_ascii=False, + ) + ) + + print() + print( + f"Result image: {args.output}" + ) + + +if __name__ == "__main__": + main() diff --git a/src/cockpit_grounding/__init__.py b/src/cockpit_grounding/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/cockpit_grounding/api/__init__.py b/src/cockpit_grounding/api/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/cockpit_grounding/benchmark/__init__.py b/src/cockpit_grounding/benchmark/__init__.py new file mode 100644 index 0000000..f2f997e --- /dev/null +++ b/src/cockpit_grounding/benchmark/__init__.py @@ -0,0 +1,11 @@ +from cockpit_grounding.benchmark.metrics import ( + bbox_iou, + normalized_center_error, + point_in_box, +) + +__all__ = [ + "bbox_iou", + "normalized_center_error", + "point_in_box", +] diff --git a/src/cockpit_grounding/benchmark/config.py b/src/cockpit_grounding/benchmark/config.py new file mode 100644 index 0000000..75e8ee8 --- /dev/null +++ b/src/cockpit_grounding/benchmark/config.py @@ -0,0 +1,190 @@ +from __future__ import annotations + +import json +import tomllib +from dataclasses import dataclass +from pathlib import Path +from typing import Any + + +@dataclass(frozen=True) +class BenchmarkSettings: + warmup: int + repeats: int + max_new_tokens: int + output_root: Path + timestamp_run_directory: bool = True + + +@dataclass(frozen=True) +class ModelConfig: + name: str + backend: str + path: Path + + +@dataclass(frozen=True) +class BenchmarkConfig: + settings: BenchmarkSettings + models: tuple[ModelConfig, ...] + + +@dataclass(frozen=True) +class BenchmarkSample: + sample_id: str + image: Path + target: str + gt_bbox_pixel: tuple[float, float, float, float] | None = None + + +def load_benchmark_config(path: str | Path) -> BenchmarkConfig: + config_path = Path(path).expanduser().resolve() + with config_path.open("rb") as config_file: + data = tomllib.load(config_file) + + benchmark = _require_mapping(data, "benchmark") + models_data = _require_mapping(data, "models") + + settings = BenchmarkSettings( + warmup=_non_negative_int(benchmark.get("warmup"), "benchmark.warmup"), + repeats=_positive_int(benchmark.get("repeats"), "benchmark.repeats"), + max_new_tokens=_positive_int( + benchmark.get("max_new_tokens"), + "benchmark.max_new_tokens", + ), + output_root=Path( + _non_empty_string( + benchmark.get("output_root"), + "benchmark.output_root", + ) + ).expanduser(), + timestamp_run_directory=_optional_bool( + benchmark.get("timestamp_run_directory"), + "benchmark.timestamp_run_directory", + default=True, + ), + ) + + models: list[ModelConfig] = [] + for name, model_data in models_data.items(): + if not isinstance(model_data, dict): + raise ValueError(f"models.{name} must be a TOML table") + model_path = Path( + _non_empty_string(model_data.get("path"), f"models.{name}.path") + ).expanduser().resolve() + if not model_path.is_dir(): + raise FileNotFoundError(f"Local model directory not found: {model_path}") + backend = _non_empty_string( + model_data.get("backend", "qwen3vl"), + f"models.{name}.backend", + ) + models.append(ModelConfig(name=name, backend=backend, path=model_path)) + + if not models: + raise ValueError("At least one model must be configured") + + return BenchmarkConfig(settings=settings, models=tuple(models)) + + +def load_manifest(path: str | Path) -> tuple[BenchmarkSample, ...]: + manifest_path = Path(path).expanduser().resolve() + samples: list[BenchmarkSample] = [] + seen_ids: set[str] = set() + + with manifest_path.open("r", encoding="utf-8") as manifest_file: + for line_number, line in enumerate(manifest_file, start=1): + if not line.strip(): + continue + try: + item = json.loads(line) + except json.JSONDecodeError as exc: + raise ValueError( + f"Invalid JSON at {manifest_path}:{line_number}: {exc.msg}" + ) from exc + if not isinstance(item, dict): + raise ValueError( + f"Manifest entry at line {line_number} must be an object" + ) + + sample = _parse_sample(item, manifest_path, line_number) + if sample.sample_id in seen_ids: + raise ValueError(f"Duplicate sample id: {sample.sample_id}") + seen_ids.add(sample.sample_id) + samples.append(sample) + + if not samples: + raise ValueError(f"Manifest contains no samples: {manifest_path}") + return tuple(samples) + + +def _parse_sample( + item: dict[str, Any], + manifest_path: Path, + line_number: int, +) -> BenchmarkSample: + prefix = f"{manifest_path}:{line_number}" + sample_id = _non_empty_string(item.get("id"), f"{prefix} id") + if Path(sample_id).name != sample_id or sample_id in {".", ".."}: + raise ValueError(f"{prefix} id must be a filename-safe identifier") + + image = Path( + _non_empty_string(item.get("image"), f"{prefix} image") + ).expanduser().resolve() + if not image.is_file(): + raise FileNotFoundError(f"Image not found at {prefix}: {image}") + + target = _non_empty_string(item.get("target"), f"{prefix} target") + gt_bbox = item.get("gt_bbox_pixel") + parsed_gt: tuple[float, float, float, float] | None = None + if gt_bbox is not None: + if not isinstance(gt_bbox, list) or len(gt_bbox) != 4: + raise ValueError(f"{prefix} gt_bbox_pixel must contain four numbers") + try: + parsed_gt = tuple(float(value) for value in gt_bbox) # type: ignore[assignment] + except (TypeError, ValueError) as exc: + raise ValueError( + f"{prefix} gt_bbox_pixel must contain four numbers" + ) from exc + x1, y1, x2, y2 = parsed_gt + if x1 > x2 or y1 > y2: + raise ValueError(f"{prefix} gt_bbox_pixel has invalid corner order") + + return BenchmarkSample( + sample_id=sample_id, + image=image, + target=target, + gt_bbox_pixel=parsed_gt, + ) + + +def _require_mapping(data: dict[str, Any], key: str) -> dict[str, Any]: + value = data.get(key) + if not isinstance(value, dict): + raise ValueError(f"Missing or invalid [{key}] table") + return value + + +def _non_empty_string(value: Any, name: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ValueError(f"{name} must be a non-empty string") + return value + + +def _positive_int(value: Any, name: str) -> int: + if not isinstance(value, int) or isinstance(value, bool) or value <= 0: + raise ValueError(f"{name} must be a positive integer") + return value + + +def _non_negative_int(value: Any, name: str) -> int: + if not isinstance(value, int) or isinstance(value, bool) or value < 0: + raise ValueError(f"{name} must be a non-negative integer") + return value + + +def _optional_bool(value: Any, name: str, *, default: bool) -> bool: + if value is None: + return default + if not isinstance(value, bool): + raise ValueError(f"{name} must be a boolean") + return value diff --git a/src/cockpit_grounding/benchmark/metrics.py b/src/cockpit_grounding/benchmark/metrics.py new file mode 100644 index 0000000..133c3dd --- /dev/null +++ b/src/cockpit_grounding/benchmark/metrics.py @@ -0,0 +1,60 @@ +from math import hypot +from typing import Sequence + +Point = Sequence[float] +BBox = Sequence[float] + + +def _validate_point(point: Point, name: str) -> None: + if len(point) != 2: + raise ValueError(f"{name} must contain two coordinates") + + +def _validate_bbox(bbox: BBox, name: str) -> None: + if len(bbox) != 4: + raise ValueError(f"{name} must contain four coordinates") + x1, y1, x2, y2 = bbox + if x1 > x2 or y1 > y2: + raise ValueError(f"{name} must satisfy x1 <= x2 and y1 <= y2") + + +def point_in_box(pred_center: Point, gt_bbox: BBox) -> bool: + _validate_point(pred_center, "pred_center") + _validate_bbox(gt_bbox, "gt_bbox") + x, y = pred_center + x1, y1, x2, y2 = gt_bbox + return x1 <= x <= x2 and y1 <= y <= y2 + + +def bbox_iou(pred_bbox: BBox, gt_bbox: BBox) -> float: + _validate_bbox(pred_bbox, "pred_bbox") + _validate_bbox(gt_bbox, "gt_bbox") + pred_x1, pred_y1, pred_x2, pred_y2 = pred_bbox + gt_x1, gt_y1, gt_x2, gt_y2 = gt_bbox + + intersection_width = max(0.0, min(pred_x2, gt_x2) - max(pred_x1, gt_x1)) + intersection_height = max(0.0, min(pred_y2, gt_y2) - max(pred_y1, gt_y1)) + intersection = intersection_width * intersection_height + pred_area = (pred_x2 - pred_x1) * (pred_y2 - pred_y1) + gt_area = (gt_x2 - gt_x1) * (gt_y2 - gt_y1) + union = pred_area + gt_area - intersection + return intersection / union if union > 0 else 0.0 + + +def normalized_center_error( + pred_center: Point, + gt_bbox: BBox, + image_width: int, + image_height: int, +) -> float: + """Return center distance normalized by the image diagonal.""" + _validate_point(pred_center, "pred_center") + _validate_bbox(gt_bbox, "gt_bbox") + if image_width <= 0 or image_height <= 0: + raise ValueError("image dimensions must be positive") + + gt_x1, gt_y1, gt_x2, gt_y2 = gt_bbox + gt_center_x = (gt_x1 + gt_x2) / 2.0 + gt_center_y = (gt_y1 + gt_y2) / 2.0 + distance = hypot(pred_center[0] - gt_center_x, pred_center[1] - gt_center_y) + return distance / hypot(image_width, image_height) diff --git a/src/cockpit_grounding/benchmark/reporting.py b/src/cockpit_grounding/benchmark/reporting.py new file mode 100644 index 0000000..d818bab --- /dev/null +++ b/src/cockpit_grounding/benchmark/reporting.py @@ -0,0 +1,317 @@ +from __future__ import annotations + +import csv +import json +from datetime import datetime +from pathlib import Path +from typing import Any + + +RESULT_COLUMNS = ( + "model", + "id", + "target", + "parse_success", + "point_in_box", + "bbox_iou", + "normalized_center_error", + "preprocess_mean_ms", + "generate_mean_ms", + "decode_mean_ms", + "total_mean_ms", + "total_p50_ms", + "total_p95_ms", + "peak_cuda_memory_mb", +) + + +def create_run_directory( + output_root: Path, + run_name: str, + *, + timestamped: bool = True, +) -> Path: + safe_name = _safe_run_name(run_name) + directory_name = ( + f"{datetime.now().strftime('%Y%m%d_%H%M%S')}_{safe_name}" + if timestamped + else safe_name + ) + run_directory = output_root / directory_name + run_directory.mkdir(parents=True, exist_ok=False) + return run_directory + + +def write_model_outputs( + model_directory: Path, + summary: dict[str, Any], + predictions: list[dict[str, Any]], +) -> None: + model_directory.mkdir(parents=True, exist_ok=True) + write_json(model_directory / "summary.json", summary) + with (model_directory / "predictions.jsonl").open( + "w", + encoding="utf-8", + ) as predictions_file: + for prediction in predictions: + predictions_file.write( + json.dumps(prediction, ensure_ascii=False, allow_nan=False) + "\n" + ) + + +def write_root_outputs( + run_directory: Path, + metadata: dict[str, Any], + summaries: dict[str, dict[str, Any]], + predictions: list[dict[str, Any]], +) -> None: + write_json(run_directory / "benchmark_meta.json", metadata) + write_json( + run_directory / "summary.json", + {"models": summaries}, + ) + + with (run_directory / "results.csv").open( + "w", + encoding="utf-8", + newline="", + ) as csv_file: + writer = csv.DictWriter(csv_file, fieldnames=RESULT_COLUMNS) + writer.writeheader() + for prediction in predictions: + writer.writerow({column: prediction.get(column) for column in RESULT_COLUMNS}) + + _write_case_results( + run_directory / "case_results.csv", + tuple(summaries), + predictions, + ) + _write_review( + run_directory / "review.md", + tuple(summaries), + predictions, + ) + + +def _write_case_results( + path: Path, + model_names: tuple[str, ...], + predictions: list[dict[str, Any]], +) -> None: + columns = ["id", "target"] + for model_name in model_names: + columns.extend( + ( + f"{model_name}_center_x", + f"{model_name}_center_y", + f"{model_name}_latency_ms", + f"{model_name}_parse_success", + f"{model_name}_bbox_x1", + f"{model_name}_bbox_y1", + f"{model_name}_bbox_x2", + f"{model_name}_bbox_y2", + f"manual_{model_name}", + ) + ) + + cases: dict[str, dict[str, Any]] = {} + for prediction in predictions: + sample_id = prediction["id"] + target = prediction["target"] + case = cases.setdefault(sample_id, {"id": sample_id, "target": target}) + if case["target"] != target: + raise ValueError(f"Mismatched targets for benchmark case: {sample_id}") + + model_name = prediction["model"] + if model_name not in model_names: + raise ValueError(f"Unexpected model in predictions: {model_name}") + center = prediction.get("pred_center_pixel") or (None, None) + bbox = prediction.get("pred_bbox_pixel") or (None, None, None, None) + case[f"{model_name}_center_x"] = center[0] + case[f"{model_name}_center_y"] = center[1] + case[f"{model_name}_latency_ms"] = prediction.get("total_mean_ms") + case[f"{model_name}_parse_success"] = prediction.get("parse_success") + case[f"{model_name}_bbox_x1"] = bbox[0] + case[f"{model_name}_bbox_y1"] = bbox[1] + case[f"{model_name}_bbox_x2"] = bbox[2] + case[f"{model_name}_bbox_y2"] = bbox[3] + case[f"manual_{model_name}"] = "" + + with path.open("w", encoding="utf-8", newline="") as csv_file: + writer = csv.DictWriter(csv_file, fieldnames=columns) + writer.writeheader() + for case in cases.values(): + writer.writerow({column: case.get(column) for column in columns}) + + +def _write_review( + path: Path, + model_names: tuple[str, ...], + predictions: list[dict[str, Any]], +) -> None: + cases: dict[str, dict[str, Any]] = {} + for prediction in predictions: + sample_id = prediction["id"] + case = cases.setdefault( + sample_id, + { + "target": prediction["target"], + "image": prediction["image"], + "models": {}, + }, + ) + case["models"][prediction["model"]] = prediction + + lines = [ + "# Manual Grounding Review", + "", + "Use `correct`, `wrong`, or `partial` in the corresponding ", + "`manual_` columns of `case_results.csv` after visual review.", + "", + ] + for sample_id, case in cases.items(): + lines.extend( + ( + f"## {sample_id}", + "", + f"Target: {case['target']}", + "", + f"Source: `{case['image']}`", + "", + ) + ) + for model_name in model_names: + prediction = case["models"].get(model_name) + if prediction is None: + lines.extend((f"### {model_name}", "", "Missing prediction.", "")) + continue + visualization = Path(model_name) / "visualizations" / f"{sample_id}.jpg" + lines.extend( + ( + f"### {model_name}", + "", + f"Parse success: `{prediction.get('parse_success')}`", + "", + f"Predicted bbox: `{prediction.get('pred_bbox_pixel')}`", + "", + f"![{model_name} {sample_id}]({visualization.as_posix()})", + "", + ) + ) + path.write_text("\n".join(lines), encoding="utf-8") + + +def write_json(path: Path, value: dict[str, Any]) -> None: + with path.open("w", encoding="utf-8") as output_file: + json.dump( + value, + output_file, + indent=2, + ensure_ascii=False, + allow_nan=False, + ) + output_file.write("\n") + + +def print_comparison( + summaries: dict[str, dict[str, Any]], + title: str = "MODEL COMPARISON", +) -> None: + print() + print("=" * 60) + print(title) + print("=" * 60) + for model_name, summary in summaries.items(): + print() + print(model_name) + print() + print(f" model load : {_format(summary['model_load_seconds'], '.2f', 's')}") + print( + " CUDA memory : " + f"{_format(summary['model_cuda_allocated_mb'], '.0f', 'MB')} allocated, " + f"{_format(summary['model_cuda_reserved_mb'], '.0f', 'MB')} reserved" + ) + print( + " peak memory : " + f"{_format(summary['peak_cuda_memory_mb'], '.0f', 'MB')}" + ) + print(f" mean latency : {_format(summary['mean_total_ms'], '.1f', 'ms')}") + print(f" p50 latency : {_format(summary['p50_total_ms'], '.1f', 'ms')}") + print(f" p95 latency : {_format(summary['p95_total_ms'], '.1f', 'ms')}") + print( + " generation mean : " + f"{_format(summary['mean_generate_ms'], '.1f', 'ms')}" + ) + print( + " throughput : " + f"{_format(summary['throughput_samples_per_sec'], '.2f', 'samples/s')}" + ) + print(f" parse success : {summary['parse_success_rate'] * 100:.1f} %") + if summary["accuracy_point_in_box"] is not None: + print( + " grounding acc : " + f"{summary['accuracy_point_in_box'] * 100:.1f} %" + ) + print() + print("=" * 60) + + +def print_model_selection(summaries: dict[str, dict[str, Any]]) -> None: + model_names = tuple(summaries) + print() + print("=" * 60) + print("MODEL SELECTION") + print("=" * 60) + print() + header = f"{'Metric':<32}" + "".join(f"{name:>20}" for name in model_names) + print(header) + rows = ( + ("Parse Success", "parse_success_rate", ".1%"), + ("Mean Latency", "mean_total_ms", ".1f"), + ("P50", "p50_total_ms", ".1f"), + ("P95", "p95_total_ms", ".1f"), + ("Peak GPU Memory", "peak_cuda_memory_mb", ".0f"), + ("Throughput", "throughput_samples_per_sec", ".3f"), + ) + for label, key, number_format in rows: + values = "".join( + f"{_format_table_value(summaries[name].get(key), number_format):>20}" + for name in model_names + ) + print(f"{label:<32}{values}") + print() + manual = "N/A - manual review required" + for label in ( + "Text UI Accuracy", + "Generic Icon Accuracy", + "Automotive Icon Accuracy", + "Function Region Accuracy", + "Small Sub-Control Accuracy", + "Slider Geometry Accuracy", + ): + values = "".join(f"{'N/A':>20}" for _ in model_names) + print(f"{label:<32}{values}") + print() + print(manual) + print() + print("=" * 60) + + +def _format_table_value(value: Any, number_format: str) -> str: + if not isinstance(value, (int, float)): + return "n/a" + return format(value, number_format) + + +def _format(value: float | None, number_format: str, unit: str) -> str: + if value is None: + return "n/a" + return f"{value:{number_format}} {unit}" + + +def _safe_run_name(run_name: str) -> str: + if not run_name or any(character in run_name for character in "/\\"): + raise ValueError("run name must be non-empty and cannot contain path separators") + if run_name in {".", ".."}: + raise ValueError("invalid run name") + return run_name diff --git a/src/cockpit_grounding/benchmark/runner.py b/src/cockpit_grounding/benchmark/runner.py new file mode 100644 index 0000000..6d8dcd0 --- /dev/null +++ b/src/cockpit_grounding/benchmark/runner.py @@ -0,0 +1,319 @@ +from __future__ import annotations + +import gc +import logging +import os +import platform +import hashlib +from datetime import datetime +from pathlib import Path +from time import perf_counter +from typing import Any + +import torch +import transformers + +from cockpit_grounding.benchmark.config import ( + BenchmarkConfig, + BenchmarkSample, +) +from cockpit_grounding.benchmark.metrics import ( + bbox_iou, + normalized_center_error, + point_in_box, +) +from cockpit_grounding.benchmark.reporting import ( + create_run_directory, + write_json, + write_model_outputs, + write_root_outputs, +) +from cockpit_grounding.benchmark.statistics import ( + summarize_model, + summarize_timings, +) +from cockpit_grounding.grounding.predictor import ( + build_grounding_prompt, + parse_grounding_output, +) +from cockpit_grounding.models.base import Grounder, InferenceTiming +from cockpit_grounding.models.factory import create_grounder +from cockpit_grounding.vision.visualize import visualize_grounding + +LOGGER = logging.getLogger(__name__) +DEFAULT_RUN_NAME = "qwen3vl_2b_vs_4b" + + +def run_benchmark( + *, + config: BenchmarkConfig, + samples: tuple[BenchmarkSample, ...], + config_path: Path, + manifest_path: Path, + run_name: str | None = None, +) -> tuple[Path, dict[str, dict[str, Any]]]: + _validate_cuda_environment() + settings = config.settings + run_directory = create_run_directory( + settings.output_root, + run_name or DEFAULT_RUN_NAME, + timestamped=settings.timestamp_run_directory, + ) + started_at = datetime.now().astimezone() + metadata: dict[str, Any] = { + "started_at": started_at.isoformat(), + "completed_at": None, + "config": str(config_path.resolve()), + "manifest": str(manifest_path.resolve()), + "run_directory": str(run_directory.resolve()), + "cuda_visible_devices": os.environ.get("CUDA_VISIBLE_DEVICES"), + "cuda_device": torch.cuda.get_device_name(0), + "python_version": platform.python_version(), + "torch_version": torch.__version__, + "transformers_version": transformers.__version__, + "grounding_prompt_sha256": hashlib.sha256( + build_grounding_prompt("__SEMANTIC_TARGET__").encode("utf-8") + ).hexdigest(), + "preprocessing_policy": ( + "source image at native resolution; no benchmark-side resize" + ), + "benchmark": { + "warmup": settings.warmup, + "repeats": settings.repeats, + "max_new_tokens": settings.max_new_tokens, + }, + "models": {model.name: str(model.path) for model in config.models}, + "model_backends": {model.name: model.backend for model in config.models}, + "sample_count": len(samples), + } + write_json(run_directory / "benchmark_meta.json", metadata) + + summaries: dict[str, dict[str, Any]] = {} + all_predictions: list[dict[str, Any]] = [] + + for model in config.models: + summary, predictions = _benchmark_model( + model_name=model.name, + backend=model.backend, + model_path=model.path, + samples=samples, + warmup=settings.warmup, + repeats=settings.repeats, + max_new_tokens=settings.max_new_tokens, + run_directory=run_directory, + ) + summaries[model.name] = summary + all_predictions.extend(predictions) + + completed_at = datetime.now().astimezone() + metadata["completed_at"] = completed_at.isoformat() + metadata["duration_seconds"] = (completed_at - started_at).total_seconds() + write_root_outputs( + run_directory, + metadata, + summaries, + all_predictions, + ) + return run_directory, summaries + + +def _benchmark_model( + *, + model_name: str, + backend: str, + model_path: Path, + samples: tuple[BenchmarkSample, ...], + warmup: int, + repeats: int, + max_new_tokens: int, + run_directory: Path, +) -> tuple[dict[str, Any], list[dict[str, Any]]]: + LOGGER.info("Loading %s from %s", model_name, model_path) + grounder: Grounder | None = None + try: + torch.cuda.synchronize() + load_start = perf_counter() + grounder = create_grounder(backend=backend, model_path=model_path) + torch.cuda.synchronize() + model_load_seconds = perf_counter() - load_start + memory_metrics = grounder.cuda_memory_metrics() + + if warmup: + LOGGER.info("Running %d warmup inference(s) for %s", warmup, model_name) + prompt = build_grounding_prompt(samples[0].target) + for _ in range(warmup): + grounder.generate( + image_path=str(samples[0].image), + prompt=prompt, + max_new_tokens=max_new_tokens, + ) + + model_directory = run_directory / model_name + visualization_directory = model_directory / "visualizations" + visualization_directory.mkdir(parents=True, exist_ok=True) + + predictions: list[dict[str, Any]] = [] + all_timings: list[InferenceTiming] = [] + for index, sample in enumerate(samples, start=1): + LOGGER.info( + "[%s] sample %d/%d: %s", + model_name, + index, + len(samples), + sample.sample_id, + ) + prediction, timings = _benchmark_sample( + grounder=grounder, + model_name=model_name, + sample=sample, + repeats=repeats, + max_new_tokens=max_new_tokens, + visualization_path=( + visualization_directory / f"{sample.sample_id}.jpg" + ), + ) + predictions.append(prediction) + all_timings.extend(timings) + + summary = summarize_model( + model_name=model_name, + model_path=str(model_path), + model_load_seconds=model_load_seconds, + memory_metrics=memory_metrics, + predictions=predictions, + all_timings=all_timings, + ) + summary.update(grounder.implementation_metadata()) + write_model_outputs(model_directory, summary, predictions) + LOGGER.info("Saved %s results to %s", model_name, model_directory) + return summary, predictions + finally: + if grounder is not None: + del grounder + gc.collect() + torch.cuda.empty_cache() + torch.cuda.synchronize() + LOGGER.info("Released model %s", model_name) + + +def _benchmark_sample( + *, + grounder: Grounder, + model_name: str, + sample: BenchmarkSample, + repeats: int, + max_new_tokens: int, + visualization_path: Path, +) -> tuple[dict[str, Any], list[InferenceTiming]]: + prompt = build_grounding_prompt(sample.target) + raw_output: str | None = None + timings: list[InferenceTiming] = [] + peak_memory_values: list[float] = [] + inference_errors: list[str] = [] + + for repeat_index in range(1, repeats + 1): + try: + output, timing, extra_metrics = grounder.generate_with_metrics( + image_path=str(sample.image), + prompt=prompt, + max_new_tokens=max_new_tokens, + ) + if raw_output is None: + raw_output = output + timings.append(timing) + peak_memory_values.append(extra_metrics["peak_cuda_memory_mb"]) + except Exception as exc: # Keep the batch running after a sample failure. + error = f"repeat {repeat_index}: {type(exc).__name__}: {exc}" + inference_errors.append(error) + LOGGER.exception( + "[%s] inference failed for %s (%s)", + model_name, + sample.sample_id, + error, + ) + + timing_summary = summarize_timings(timings) + prediction: dict[str, Any] = { + "model": model_name, + "id": sample.sample_id, + "image": str(sample.image), + "target": sample.target, + "raw_output": raw_output, + "parse_success": False, + "parse_error": None, + "inference_errors": inference_errors, + "pred_bbox_pixel": None, + "pred_center_pixel": None, + "gt_bbox_pixel": ( + list(sample.gt_bbox_pixel) if sample.gt_bbox_pixel is not None else None + ), + "point_in_box": None, + "bbox_iou": None, + "normalized_center_error": None, + **timing_summary, + "peak_cuda_memory_mb": max(peak_memory_values) if peak_memory_values else None, + } + + if raw_output is None: + prediction["parse_error"] = "; ".join(inference_errors) or "No model output" + return prediction, timings + + try: + parsed = parse_grounding_output(raw_output) + prediction["parse_success"] = True + except Exception as exc: + prediction["parse_error"] = f"{type(exc).__name__}: {exc}" + LOGGER.warning( + "[%s] parse failed for %s: %s", + model_name, + sample.sample_id, + exc, + ) + return prediction, timings + + try: + pixel_result = visualize_grounding( + str(sample.image), + parsed, + str(visualization_path), + ) + pred_bbox = pixel_result["bbox_pixel"] + pred_center = pixel_result["center_pixel"] + prediction["pred_bbox_pixel"] = pred_bbox + prediction["pred_center_pixel"] = pred_center + + if sample.gt_bbox_pixel is not None: + prediction["point_in_box"] = point_in_box( + pred_center, + sample.gt_bbox_pixel, + ) + prediction["bbox_iou"] = bbox_iou( + pred_bbox, + sample.gt_bbox_pixel, + ) + prediction["normalized_center_error"] = normalized_center_error( + pred_center, + sample.gt_bbox_pixel, + pixel_result["image_width"], + pixel_result["image_height"], + ) + except Exception as exc: + prediction["visualization_error"] = f"{type(exc).__name__}: {exc}" + LOGGER.exception( + "[%s] visualization failed for %s", + model_name, + sample.sample_id, + ) + + return prediction, timings + + +def _validate_cuda_environment() -> None: + if not torch.cuda.is_available(): + raise RuntimeError("CUDA is required for the multimodal benchmark") + visible_device_count = torch.cuda.device_count() + if visible_device_count != 1: + raise RuntimeError( + "Benchmark requires exactly one visible CUDA device; " + f"found {visible_device_count}. Set CUDA_VISIBLE_DEVICES=0." + ) diff --git a/src/cockpit_grounding/benchmark/statistics.py b/src/cockpit_grounding/benchmark/statistics.py new file mode 100644 index 0000000..a2f835b --- /dev/null +++ b/src/cockpit_grounding/benchmark/statistics.py @@ -0,0 +1,96 @@ +from __future__ import annotations + +from collections.abc import Iterable, Sequence +from statistics import fmean +from typing import Any + +from cockpit_grounding.models.base import InferenceTiming + + +def percentile(values: Sequence[float], quantile: float) -> float: + if not values: + raise ValueError("percentile requires at least one value") + if not 0.0 <= quantile <= 1.0: + raise ValueError("quantile must be between 0 and 1") + + ordered = sorted(values) + position = (len(ordered) - 1) * quantile + lower = int(position) + upper = min(lower + 1, len(ordered) - 1) + fraction = position - lower + return ordered[lower] + (ordered[upper] - ordered[lower]) * fraction + + +def summarize_timings(timings: Sequence[InferenceTiming]) -> dict[str, float | None]: + if not timings: + return { + "preprocess_mean_ms": None, + "generate_mean_ms": None, + "decode_mean_ms": None, + "total_mean_ms": None, + "total_p50_ms": None, + "total_p95_ms": None, + } + + totals = [timing.total_ms for timing in timings] + return { + "preprocess_mean_ms": fmean(t.preprocess_ms for t in timings), + "generate_mean_ms": fmean(t.generate_ms for t in timings), + "decode_mean_ms": fmean(t.decode_ms for t in timings), + "total_mean_ms": fmean(totals), + "total_p50_ms": percentile(totals, 0.50), + "total_p95_ms": percentile(totals, 0.95), + } + + +def summarize_model( + *, + model_name: str, + model_path: str, + model_load_seconds: float, + memory_metrics: dict[str, float], + predictions: Sequence[dict[str, Any]], + all_timings: Sequence[InferenceTiming], +) -> dict[str, Any]: + timing = summarize_timings(all_timings) + sample_count = len(predictions) + parse_successes = sum(bool(item["parse_success"]) for item in predictions) + + point_values = _present_values(predictions, "point_in_box") + iou_values = _present_values(predictions, "bbox_iou") + center_error_values = _present_values(predictions, "normalized_center_error") + peak_memory_values = _present_values(predictions, "peak_cuda_memory_mb") + mean_total = timing["total_mean_ms"] + + return { + "model": model_name, + "model_path": model_path, + "model_load_seconds": model_load_seconds, + **memory_metrics, + "samples": sample_count, + "parse_success_rate": parse_successes / sample_count if sample_count else 0.0, + "mean_preprocess_ms": timing["preprocess_mean_ms"], + "mean_generate_ms": timing["generate_mean_ms"], + "mean_decode_ms": timing["decode_mean_ms"], + "mean_total_ms": mean_total, + "p50_total_ms": timing["total_p50_ms"], + "p95_total_ms": timing["total_p95_ms"], + "throughput_samples_per_sec": ( + 1000.0 / mean_total if isinstance(mean_total, float) and mean_total > 0 else None + ), + "peak_cuda_memory_mb": max(peak_memory_values) if peak_memory_values else None, + "accuracy_point_in_box": ( + fmean(point_values) if point_values else None + ), + "mean_bbox_iou": fmean(iou_values) if iou_values else None, + "mean_normalized_center_error": ( + fmean(center_error_values) if center_error_values else None + ), + } + + +def _present_values( + predictions: Iterable[dict[str, Any]], + key: str, +) -> list[float]: + return [float(item[key]) for item in predictions if item.get(key) is not None] diff --git a/src/cockpit_grounding/grounding/__init__.py b/src/cockpit_grounding/grounding/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/cockpit_grounding/grounding/predictor.py b/src/cockpit_grounding/grounding/predictor.py new file mode 100644 index 0000000..22ea1f2 --- /dev/null +++ b/src/cockpit_grounding/grounding/predictor.py @@ -0,0 +1,94 @@ +import json +import re +from dataclasses import dataclass + + +@dataclass +class GroundingResult: + x1: float + y1: float + x2: float + y2: float + + @property + def center_relative(self): + return ( + (self.x1 + self.x2) / 2.0, + (self.y1 + self.y2) / 2.0, + ) + + +def build_grounding_prompt(target: str) -> str: + return f""" +你是一个汽车座舱可操作UI元素定位器。 + +输入图像来自真实相机拍摄,而不是系统截图。 + +请找到以下目标: + +{target} + +要求: + +1. 找到真正可以被用户点击的控件。 +2. 不要选择标题、说明文字或者附近的无关区域。 +3. 返回整个可点击控件的边界框。 +4. 使用相对坐标: + 左上角 = (0, 0) + 右下角 = (1000, 1000) +5. x1 < x2,y1 < y2。 +6. 只返回 JSON,不要解释。 + +严格输出: + +{{"bbox_2d": [x1, y1, x2, y2]}} +""".strip() + + +def parse_grounding_output( + text: str, +) -> GroundingResult: + + # 防止模型偶尔包 markdown code block + match = re.search( + r'"bbox_2d"\s*:\s*\[\s*' + r'([0-9.]+)\s*,\s*' + r'([0-9.]+)\s*,\s*' + r'([0-9.]+)\s*,\s*' + r'([0-9.]+)\s*\]', + text, + ) + + if match is None: + raise ValueError( + "无法从模型输出解析 bbox_2d:\n" + + text + ) + + x1, y1, x2, y2 = ( + float(match.group(i)) + for i in range(1, 5) + ) + + # 基础合法性检查 + values = [x1, y1, x2, y2] + + if not all( + 0 <= value <= 1000 + for value in values + ): + raise ValueError( + f"坐标超出 0~1000: {values}" + ) + + if x1 >= x2 or y1 >= y2: + raise ValueError( + f"非法 bbox: {values}" + ) + + return GroundingResult( + x1=x1, + y1=y1, + x2=x2, + y2=y2, + ) diff --git a/src/cockpit_grounding/models/__init__.py b/src/cockpit_grounding/models/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/cockpit_grounding/models/base.py b/src/cockpit_grounding/models/base.py new file mode 100644 index 0000000..6b6d3a3 --- /dev/null +++ b/src/cockpit_grounding/models/base.py @@ -0,0 +1,42 @@ +from __future__ import annotations + +from dataclasses import dataclass +from typing import Protocol + + +@dataclass(frozen=True) +class InferenceTiming: + preprocess_ms: float + generate_ms: float + decode_ms: float + total_ms: float + + +class Grounder(Protocol): + model_path: str + + def generate( + self, + image_path: str, + prompt: str, + max_new_tokens: int = 128, + ) -> str: ... + + def generate_with_metrics( + self, + image_path: str, + prompt: str, + max_new_tokens: int = 128, + ) -> tuple[str, InferenceTiming, dict[str, float]]: ... + + def cuda_memory_metrics(self) -> dict[str, float]: ... + + def implementation_metadata(self) -> dict[str, str]: ... + + +class TextGenerator(Protocol): + def generate_text( + self, + prompt: str, + max_new_tokens: int = 256, + ) -> str: ... diff --git a/src/cockpit_grounding/models/factory.py b/src/cockpit_grounding/models/factory.py new file mode 100644 index 0000000..4d8cb9f --- /dev/null +++ b/src/cockpit_grounding/models/factory.py @@ -0,0 +1,20 @@ +from pathlib import Path + +from cockpit_grounding.models.base import Grounder + +SUPPORTED_BACKENDS = ("qwen3vl", "qwen35") + + +def create_grounder(backend: str, model_path: str | Path) -> Grounder: + if backend == "qwen3vl": + from cockpit_grounding.models.qwen3vl import Qwen3VLGrounder + + return Qwen3VLGrounder(model_path=str(model_path)) + if backend == "qwen35": + from cockpit_grounding.models.qwen35 import Qwen35Grounder + + return Qwen35Grounder(model_path=str(model_path)) + raise ValueError( + f"Unsupported model backend {backend!r}; " + f"expected one of {', '.join(SUPPORTED_BACKENDS)}" + ) diff --git a/src/cockpit_grounding/models/qwen35.py b/src/cockpit_grounding/models/qwen35.py new file mode 100644 index 0000000..6c265a9 --- /dev/null +++ b/src/cockpit_grounding/models/qwen35.py @@ -0,0 +1,177 @@ +from pathlib import Path +from time import perf_counter + +import torch +from transformers import AutoModelForMultimodalLM, AutoProcessor + +from cockpit_grounding.models.base import InferenceTiming + + +class Qwen35Grounder: + def __init__(self, model_path: str) -> None: + local_model_path = Path(model_path).expanduser().resolve() + if not local_model_path.is_dir(): + raise FileNotFoundError( + f"Local model directory not found: {local_model_path}" + ) + if not torch.cuda.is_available(): + raise RuntimeError("Qwen35Grounder requires a CUDA device") + + self.model_path = str(local_model_path) + print("[Model] Loading local Qwen3.5 model:") + print(self.model_path) + + self.model = AutoModelForMultimodalLM.from_pretrained( + self.model_path, + dtype=torch.bfloat16, + device_map={"": 0}, + local_files_only=True, + ) + self.processor = AutoProcessor.from_pretrained( + self.model_path, + local_files_only=True, + ) + self.model.eval() + + print(f"[Model] Class: {type(self.model).__name__}") + print("[Model] Loaded successfully") + print("[Model] GPU:", torch.cuda.get_device_name(self.model.device)) + + @torch.inference_mode() + def generate( + self, + image_path: str, + prompt: str, + max_new_tokens: int = 128, + ) -> str: + raw_output, _, _ = self.generate_with_metrics( + image_path=image_path, + prompt=prompt, + max_new_tokens=max_new_tokens, + ) + return raw_output + + @torch.inference_mode() + def generate_text( + self, + prompt: str, + max_new_tokens: int = 256, + ) -> str: + messages = [ + { + "role": "user", + "content": [{"type": "text", "text": prompt}], + } + ] + inputs = self.processor.apply_chat_template( + messages, + add_generation_prompt=True, + tokenize=True, + return_dict=True, + return_tensors="pt", + ) + inputs = inputs.to(self.model.device) + output_ids = self.model.generate( + **inputs, + max_new_tokens=max_new_tokens, + do_sample=False, + ) + generated_ids = output_ids[:, inputs["input_ids"].shape[-1] :] + return self.processor.batch_decode( + generated_ids, + skip_special_tokens=True, + clean_up_tokenization_spaces=False, + )[0] + + @torch.inference_mode() + def generate_with_metrics( + self, + image_path: str, + prompt: str, + max_new_tokens: int = 128, + ) -> tuple[str, InferenceTiming, dict[str, float]]: + image = Path(image_path).expanduser().resolve() + if not image.is_file(): + raise FileNotFoundError(image) + + self._synchronize_cuda() + torch.cuda.reset_peak_memory_stats(self.model.device) + total_start = perf_counter() + preprocess_start = total_start + + messages = [ + { + "role": "user", + "content": [ + {"type": "image", "path": str(image)}, + {"type": "text", "text": prompt}, + ], + } + ] + inputs = self.processor.apply_chat_template( + messages, + add_generation_prompt=True, + tokenize=True, + return_dict=True, + return_tensors="pt", + ) + inputs = inputs.to(self.model.device) + + self._synchronize_cuda() + preprocess_ms = (perf_counter() - preprocess_start) * 1000.0 + + generate_start = perf_counter() + output_ids = self.model.generate( + **inputs, + max_new_tokens=max_new_tokens, + do_sample=False, + ) + self._synchronize_cuda() + generate_ms = (perf_counter() - generate_start) * 1000.0 + + decode_start = perf_counter() + generated_ids = output_ids[:, inputs["input_ids"].shape[-1] :] + output_text = self.processor.batch_decode( + generated_ids, + skip_special_tokens=True, + clean_up_tokenization_spaces=False, + )[0] + self._synchronize_cuda() + decode_ms = (perf_counter() - decode_start) * 1000.0 + total_ms = (perf_counter() - total_start) * 1000.0 + + timing = InferenceTiming( + preprocess_ms=preprocess_ms, + generate_ms=generate_ms, + decode_ms=decode_ms, + total_ms=total_ms, + ) + metrics = { + "peak_cuda_memory_mb": self._bytes_to_mb( + torch.cuda.max_memory_allocated(self.model.device) + ) + } + return output_text, timing, metrics + + def cuda_memory_metrics(self) -> dict[str, float]: + return { + "model_cuda_allocated_mb": self._bytes_to_mb( + torch.cuda.memory_allocated(self.model.device) + ), + "model_cuda_reserved_mb": self._bytes_to_mb( + torch.cuda.memory_reserved(self.model.device) + ), + } + + def implementation_metadata(self) -> dict[str, str]: + return { + "transformers_model_class": type(self.model).__name__, + "transformers_processor_class": type(self.processor).__name__, + } + + @staticmethod + def _bytes_to_mb(value: int) -> float: + return value / (1024.0 * 1024.0) + + def _synchronize_cuda(self) -> None: + torch.cuda.synchronize(self.model.device) diff --git a/src/cockpit_grounding/models/qwen3vl.py b/src/cockpit_grounding/models/qwen3vl.py new file mode 100644 index 0000000..05d2a21 --- /dev/null +++ b/src/cockpit_grounding/models/qwen3vl.py @@ -0,0 +1,204 @@ +from pathlib import Path +from time import perf_counter + +import torch +from transformers import ( + AutoProcessor, + Qwen3VLForConditionalGeneration, +) +from qwen_vl_utils import process_vision_info + +from cockpit_grounding.models.base import InferenceTiming + + +class Qwen3VLGrounder: + + def __init__( + self, + model_path: str, + ) -> None: + local_model_path = Path(model_path).expanduser().resolve() + if not local_model_path.is_dir(): + raise FileNotFoundError( + f"Local model directory not found: {local_model_path}" + ) + if not torch.cuda.is_available(): + raise RuntimeError("Qwen3VLGrounder requires a CUDA device") + + self.model_path = str(local_model_path) + + print("[Model] Loading local model:") + print(self.model_path) + + self.model = ( + Qwen3VLForConditionalGeneration + .from_pretrained( + self.model_path, + dtype=torch.bfloat16, + device_map={"": 0}, + local_files_only=True, + ) + ) + + self.processor = AutoProcessor.from_pretrained( + self.model_path, + local_files_only=True, + ) + + self.model.eval() + + print("[Model] Loaded successfully") + print( + "[Model] GPU:", + torch.cuda.get_device_name(self.model.device) + ) + + @torch.inference_mode() + def generate( + self, + image_path: str, + prompt: str, + max_new_tokens: int = 128, + ) -> str: + raw_output, _, _ = self.generate_with_metrics( + image_path=image_path, + prompt=prompt, + max_new_tokens=max_new_tokens, + ) + return raw_output + + @torch.inference_mode() + def generate_with_metrics( + self, + image_path: str, + prompt: str, + max_new_tokens: int = 128, + ) -> tuple[str, InferenceTiming, dict[str, float]]: + """Generate a response and report synchronized stage timings.""" + + image_path = Path(image_path).resolve() + + if not image_path.exists(): + raise FileNotFoundError(image_path) + + self._synchronize_cuda() + torch.cuda.reset_peak_memory_stats(self.model.device) + total_start = perf_counter() + preprocess_start = total_start + + messages = [ + { + "role": "user", + "content": [ + { + "type": "image", + "image": image_path.as_uri(), + }, + { + "type": "text", + "text": prompt, + }, + ], + } + ] + + text = self.processor.apply_chat_template( + messages, + tokenize=False, + add_generation_prompt=True, + ) + + images, videos, video_kwargs = process_vision_info( + messages, + image_patch_size=16, + return_video_kwargs=True, + return_video_metadata=True, + ) + + if videos is not None: + videos, video_metadatas = zip(*videos) + videos = list(videos) + video_metadatas = list(video_metadatas) + else: + video_metadatas = None + + inputs = self.processor( + text=text, + images=images, + videos=videos, + video_metadata=video_metadatas, + return_tensors="pt", + do_resize=False, + **video_kwargs, + ) + + inputs = inputs.to(self.model.device) + + self._synchronize_cuda() + preprocess_ms = (perf_counter() - preprocess_start) * 1000.0 + + generate_start = perf_counter() + generated_ids = self.model.generate( + **inputs, + max_new_tokens=max_new_tokens, + do_sample=False, + ) + self._synchronize_cuda() + generate_ms = (perf_counter() - generate_start) * 1000.0 + + decode_start = perf_counter() + generated_ids_trimmed = [ + output_ids[len(input_ids):] + for input_ids, output_ids + in zip( + inputs.input_ids, + generated_ids, + ) + ] + + output_text = self.processor.batch_decode( + generated_ids_trimmed, + skip_special_tokens=True, + clean_up_tokenization_spaces=False, + )[0] + + self._synchronize_cuda() + decode_ms = (perf_counter() - decode_start) * 1000.0 + total_ms = (perf_counter() - total_start) * 1000.0 + peak_cuda_memory_mb = self._bytes_to_mb( + torch.cuda.max_memory_allocated(self.model.device) + ) + + timing = InferenceTiming( + preprocess_ms=preprocess_ms, + generate_ms=generate_ms, + decode_ms=decode_ms, + total_ms=total_ms, + ) + extra_metrics = { + "peak_cuda_memory_mb": peak_cuda_memory_mb, + } + return output_text, timing, extra_metrics + + @staticmethod + def _bytes_to_mb(value: int) -> float: + return value / (1024.0 * 1024.0) + + def cuda_memory_metrics(self) -> dict[str, float]: + return { + "model_cuda_allocated_mb": self._bytes_to_mb( + torch.cuda.memory_allocated(self.model.device) + ), + "model_cuda_reserved_mb": self._bytes_to_mb( + torch.cuda.memory_reserved(self.model.device) + ), + } + + def implementation_metadata(self) -> dict[str, str]: + return { + "transformers_model_class": type(self.model).__name__, + "transformers_processor_class": type(self.processor).__name__, + } + + def _synchronize_cuda(self) -> None: + torch.cuda.synchronize(self.model.device) diff --git a/src/cockpit_grounding/utils/__init__.py b/src/cockpit_grounding/utils/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/cockpit_grounding/vision/__init__.py b/src/cockpit_grounding/vision/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/cockpit_grounding/vision/visualize.py b/src/cockpit_grounding/vision/visualize.py new file mode 100644 index 0000000..3dc1a82 --- /dev/null +++ b/src/cockpit_grounding/vision/visualize.py @@ -0,0 +1,109 @@ +from pathlib import Path + +import cv2 + +from cockpit_grounding.grounding.predictor import ( + GroundingResult, +) + + +def relative_bbox_to_pixels( + result: GroundingResult, + width: int, + height: int, +): + x1 = round(result.x1 / 1000.0 * width) + y1 = round(result.y1 / 1000.0 * height) + x2 = round(result.x2 / 1000.0 * width) + y2 = round(result.y2 / 1000.0 * height) + + return x1, y1, x2, y2 + + +def visualize_grounding( + image_path: str, + result: GroundingResult, + output_path: str, +): + image = cv2.imread(image_path) + + if image is None: + raise RuntimeError( + f"无法读取图片: {image_path}" + ) + + height, width = image.shape[:2] + + x1, y1, x2, y2 = relative_bbox_to_pixels( + result, + width, + height, + ) + + # 后面给机器人使用的点 + u = round((x1 + x2) / 2) + v = round((y1 + y2) / 2) + + # bbox + cv2.rectangle( + image, + (x1, y1), + (x2, y2), + (0, 0, 255), + 3, + ) + + # 中心 + cv2.circle( + image, + (u, v), + 12, + (0, 0, 255), + -1, + ) + + # 十字线 + cv2.line( + image, + (u - 25, v), + (u + 25, v), + (0, 0, 255), + 3, + ) + + cv2.line( + image, + (u, v - 25), + (u, v + 25), + (0, 0, 255), + 3, + ) + + output_path = Path(output_path) + + output_path.parent.mkdir( + parents=True, + exist_ok=True, + ) + + cv2.imwrite( + str(output_path), + image, + ) + + return { + "image_width": width, + "image_height": height, + + "bbox_pixel": [ + x1, + y1, + x2, + y2, + ], + + "center_pixel": [ + u, + v, + ], + } diff --git a/src/cockpit_ui_grounding.egg-info/PKG-INFO b/src/cockpit_ui_grounding.egg-info/PKG-INFO new file mode 100644 index 0000000..e2624b7 --- /dev/null +++ b/src/cockpit_ui_grounding.egg-info/PKG-INFO @@ -0,0 +1,5 @@ +Metadata-Version: 2.4 +Name: cockpit-ui-grounding +Version: 0.1.0 +Summary: Camera-view automotive cockpit UI grounding +Requires-Python: >=3.11 diff --git a/src/cockpit_ui_grounding.egg-info/SOURCES.txt b/src/cockpit_ui_grounding.egg-info/SOURCES.txt new file mode 100644 index 0000000..3f29a42 --- /dev/null +++ b/src/cockpit_ui_grounding.egg-info/SOURCES.txt @@ -0,0 +1,12 @@ +README.md +pyproject.toml +src/cockpit_grounding/__init__.py +src/cockpit_grounding/api/__init__.py +src/cockpit_grounding/grounding/__init__.py +src/cockpit_grounding/models/__init__.py +src/cockpit_grounding/utils/__init__.py +src/cockpit_grounding/vision/__init__.py +src/cockpit_ui_grounding.egg-info/PKG-INFO +src/cockpit_ui_grounding.egg-info/SOURCES.txt +src/cockpit_ui_grounding.egg-info/dependency_links.txt +src/cockpit_ui_grounding.egg-info/top_level.txt \ No newline at end of file diff --git a/src/cockpit_ui_grounding.egg-info/dependency_links.txt b/src/cockpit_ui_grounding.egg-info/dependency_links.txt new file mode 100644 index 0000000..8b13789 --- /dev/null +++ b/src/cockpit_ui_grounding.egg-info/dependency_links.txt @@ -0,0 +1 @@ + diff --git a/src/cockpit_ui_grounding.egg-info/top_level.txt b/src/cockpit_ui_grounding.egg-info/top_level.txt new file mode 100644 index 0000000..7bc1034 --- /dev/null +++ b/src/cockpit_ui_grounding.egg-info/top_level.txt @@ -0,0 +1 @@ +cockpit_grounding diff --git a/tests/test_benchmark_config.py b/tests/test_benchmark_config.py new file mode 100644 index 0000000..dea7a05 --- /dev/null +++ b/tests/test_benchmark_config.py @@ -0,0 +1,141 @@ +import json +import unittest +from pathlib import Path +from tempfile import TemporaryDirectory + +from cockpit_grounding.benchmark.config import ( + load_benchmark_config, + load_manifest, +) + + +class BenchmarkConfigTest(unittest.TestCase): + def test_load_benchmark_config_preserves_model_order(self) -> None: + with TemporaryDirectory() as directory: + tmp_path = Path(directory) + model_a = tmp_path / "model-a" + model_b = tmp_path / "model-b" + model_a.mkdir() + model_b.mkdir() + config = tmp_path / "benchmark.toml" + config.write_text( + "\n".join( + ( + "[benchmark]", + "warmup = 1", + "repeats = 3", + "max_new_tokens = 128", + f'output_root = "{tmp_path}"', + "[models.first]", + f'path = "{model_a}"', + "[models.second]", + f'path = "{model_b}"', + ) + ), + encoding="utf-8", + ) + + loaded = load_benchmark_config(config) + + self.assertEqual( + [model.name for model in loaded.models], + ["first", "second"], + ) + self.assertEqual( + [model.backend for model in loaded.models], + ["qwen3vl", "qwen3vl"], + ) + self.assertEqual(loaded.settings.repeats, 3) + self.assertTrue(loaded.settings.timestamp_run_directory) + + def test_load_benchmark_config_supports_explicit_backend(self) -> None: + with TemporaryDirectory() as directory: + tmp_path = Path(directory) + model_path = tmp_path / "model" + model_path.mkdir() + config = tmp_path / "benchmark.toml" + config.write_text( + "\n".join( + ( + "[benchmark]", + "warmup = 0", + "repeats = 1", + "max_new_tokens = 16", + f'output_root = "{tmp_path}"', + "[models.qwen35]", + 'backend = "qwen35"', + f'path = "{model_path}"', + ) + ), + encoding="utf-8", + ) + + loaded = load_benchmark_config(config) + + self.assertEqual(loaded.models[0].backend, "qwen35") + + def test_load_benchmark_config_supports_exact_run_directory(self) -> None: + with TemporaryDirectory() as directory: + tmp_path = Path(directory) + model_path = tmp_path / "model" + model_path.mkdir() + config = tmp_path / "benchmark.toml" + config.write_text( + "\n".join( + ( + "[benchmark]", + "warmup = 3", + "repeats = 5", + "max_new_tokens = 128", + f'output_root = "{tmp_path}"', + "timestamp_run_directory = false", + "[models.model]", + f'path = "{model_path}"', + ) + ), + encoding="utf-8", + ) + + loaded = load_benchmark_config(config) + + self.assertFalse(loaded.settings.timestamp_run_directory) + + def test_manifest_supports_optional_gt(self) -> None: + with TemporaryDirectory() as directory: + tmp_path = Path(directory) + image = tmp_path / "image.jpg" + image.touch() + manifest = tmp_path / "samples.jsonl" + entries = ( + {"id": "without-gt", "image": str(image), "target": "button"}, + { + "id": "with-gt", + "image": str(image), + "target": "button", + "gt_bbox_pixel": [1, 2, 3, 4], + }, + ) + manifest.write_text( + "\n".join(json.dumps(item) for item in entries) + "\n", + encoding="utf-8", + ) + + samples = load_manifest(manifest) + + self.assertIsNone(samples[0].gt_bbox_pixel) + self.assertEqual(samples[1].gt_bbox_pixel, (1.0, 2.0, 3.0, 4.0)) + + def test_manifest_rejects_duplicate_ids(self) -> None: + with TemporaryDirectory() as directory: + tmp_path = Path(directory) + image = tmp_path / "image.jpg" + image.touch() + manifest = tmp_path / "samples.jsonl" + item = {"id": "same", "image": str(image), "target": "button"} + manifest.write_text( + json.dumps(item) + "\n" + json.dumps(item) + "\n", + encoding="utf-8", + ) + + with self.assertRaisesRegex(ValueError, "Duplicate"): + load_manifest(manifest) diff --git a/tests/test_benchmark_reporting.py b/tests/test_benchmark_reporting.py new file mode 100644 index 0000000..403460b --- /dev/null +++ b/tests/test_benchmark_reporting.py @@ -0,0 +1,90 @@ +import csv +import io +import unittest +from contextlib import redirect_stdout +from pathlib import Path +from tempfile import TemporaryDirectory + +from cockpit_grounding.benchmark.reporting import ( + create_run_directory, + print_model_selection, + write_root_outputs, +) + + +class BenchmarkReportingTest(unittest.TestCase): + def test_case_results_joins_models_by_case(self) -> None: + predictions = [ + { + "model": "qwen3vl_2b", + "id": "button", + "target": "按钮", + "pred_center_pixel": [10, 20], + "total_mean_ms": 100.0, + "parse_success": True, + "pred_bbox_pixel": [1, 2, 19, 38], + "image": "/dataset/image.jpg", + }, + { + "model": "qwen3vl_4b", + "id": "button", + "target": "按钮", + "pred_center_pixel": [11, 21], + "total_mean_ms": 120.0, + "parse_success": True, + "pred_bbox_pixel": [2, 3, 20, 39], + "image": "/dataset/image.jpg", + }, + ] + summaries = {"qwen3vl_2b": {}, "qwen3vl_4b": {}} + + with TemporaryDirectory() as directory: + output = Path(directory) + write_root_outputs(output, {}, summaries, predictions) + with (output / "case_results.csv").open( + encoding="utf-8", + newline="", + ) as csv_file: + rows = list(csv.DictReader(csv_file)) + review = (output / "review.md").read_text(encoding="utf-8") + + self.assertEqual(len(rows), 1) + self.assertEqual(rows[0]["id"], "button") + self.assertEqual(rows[0]["qwen3vl_2b_center_x"], "10") + self.assertEqual(rows[0]["qwen3vl_4b_center_y"], "21") + self.assertEqual(rows[0]["qwen3vl_2b_latency_ms"], "100.0") + self.assertEqual(rows[0]["manual_qwen3vl_2b"], "") + self.assertEqual(rows[0]["qwen3vl_4b_bbox_x2"], "20") + self.assertIn("qwen3vl_2b/visualizations/button.jpg", review) + self.assertIn("qwen3vl_4b/visualizations/button.jpg", review) + + def test_exact_run_directory_uses_run_name_without_timestamp(self) -> None: + with TemporaryDirectory() as directory: + path = create_run_directory( + Path(directory), + "selection_run", + timestamped=False, + ) + + self.assertEqual(path.name, "selection_run") + + def test_model_selection_marks_accuracy_as_manual_without_gt(self) -> None: + summary = { + "parse_success_rate": 1.0, + "mean_total_ms": 100.0, + "p50_total_ms": 99.0, + "p95_total_ms": 110.0, + "peak_cuda_memory_mb": 1000.0, + "throughput_samples_per_sec": 10.0, + } + output = io.StringIO() + with redirect_stdout(output): + print_model_selection({"model_a": summary, "model_b": summary}) + + printed = output.getvalue() + self.assertIn("MODEL SELECTION", printed) + self.assertIn("N/A - manual review required", printed) + accuracy_row = next( + line for line in printed.splitlines() if line.startswith("Text UI Accuracy") + ) + self.assertEqual(accuracy_row.count("N/A"), 2) diff --git a/tests/test_benchmark_statistics.py b/tests/test_benchmark_statistics.py new file mode 100644 index 0000000..65be1e2 --- /dev/null +++ b/tests/test_benchmark_statistics.py @@ -0,0 +1,89 @@ +import unittest +from pathlib import Path +from tempfile import TemporaryDirectory + +from cockpit_grounding.benchmark.config import BenchmarkSample +from cockpit_grounding.benchmark.runner import _benchmark_sample +from cockpit_grounding.benchmark.statistics import ( + percentile, + summarize_model, + summarize_timings, +) +from cockpit_grounding.models.base import InferenceTiming + + +def _timing(total: float) -> InferenceTiming: + return InferenceTiming( + preprocess_ms=1.0, + generate_ms=total - 2.0, + decode_ms=1.0, + total_ms=total, + ) + + +class BenchmarkStatisticsTest(unittest.TestCase): + def test_percentile_uses_linear_interpolation(self) -> None: + self.assertEqual(percentile([10.0, 20.0, 30.0], 0.5), 20.0) + self.assertAlmostEqual(percentile([10.0, 20.0], 0.95), 19.5) + + def test_summarize_timings_handles_empty_input(self) -> None: + self.assertIsNone(summarize_timings([])["total_mean_ms"]) + + def test_summarize_model_without_gt_uses_null_accuracy(self) -> None: + predictions = [ + { + "parse_success": True, + "point_in_box": None, + "bbox_iou": None, + "normalized_center_error": None, + "peak_cuda_memory_mb": 123.0, + } + ] + summary = summarize_model( + model_name="model", + model_path="/model", + model_load_seconds=2.0, + memory_metrics={ + "model_cuda_allocated_mb": 100.0, + "model_cuda_reserved_mb": 120.0, + }, + predictions=predictions, + all_timings=[_timing(10.0), _timing(20.0)], + ) + + self.assertEqual(summary["parse_success_rate"], 1.0) + self.assertEqual(summary["mean_total_ms"], 15.0) + self.assertAlmostEqual(summary["throughput_samples_per_sec"], 1000 / 15) + self.assertEqual(summary["peak_cuda_memory_mb"], 123.0) + self.assertIsNone(summary["accuracy_point_in_box"]) + + def test_parse_failure_is_returned_instead_of_raised(self) -> None: + class InvalidOutputGrounder: + def generate_with_metrics(self, **_: object): + return ( + "not a bbox", + _timing(10.0), + {"peak_cuda_memory_mb": 100.0}, + ) + + with TemporaryDirectory() as directory: + tmp_path = Path(directory) + image = tmp_path / "image.jpg" + image.touch() + prediction, timings = _benchmark_sample( + grounder=InvalidOutputGrounder(), # type: ignore[arg-type] + model_name="model", + sample=BenchmarkSample( + sample_id="sample", + image=image, + target="button", + ), + repeats=3, + max_new_tokens=128, + visualization_path=tmp_path / "visualization.jpg", + ) + + self.assertFalse(prediction["parse_success"]) + self.assertIn("ValueError", prediction["parse_error"]) + self.assertEqual(len(timings), 3) + self.assertEqual(prediction["total_mean_ms"], 10.0) diff --git a/tests/test_metrics.py b/tests/test_metrics.py new file mode 100644 index 0000000..f05ef38 --- /dev/null +++ b/tests/test_metrics.py @@ -0,0 +1,33 @@ +import unittest + +from cockpit_grounding.benchmark.metrics import ( + bbox_iou, + normalized_center_error, + point_in_box, +) + + +class MetricsTest(unittest.TestCase): + def test_point_in_box_includes_boundary(self) -> None: + self.assertTrue(point_in_box((10, 20), (10, 20, 30, 40))) + self.assertFalse(point_in_box((9, 20), (10, 20, 30, 40))) + + def test_bbox_iou(self) -> None: + self.assertAlmostEqual( + bbox_iou((0, 0, 10, 10), (5, 5, 15, 15)), + 25 / 175, + ) + self.assertEqual(bbox_iou((0, 0, 2, 2), (3, 3, 4, 4)), 0.0) + + def test_normalized_center_error_uses_image_diagonal(self) -> None: + value = normalized_center_error( + pred_center=(5, 5), + gt_bbox=(0, 0, 0, 0), + image_width=10, + image_height=10, + ) + self.assertAlmostEqual(value, 0.5) + + def test_metrics_reject_invalid_bbox(self) -> None: + with self.assertRaisesRegex(ValueError, "satisfy"): + bbox_iou((10, 0, 0, 10), (0, 0, 10, 10)) diff --git a/tests/test_model_factory.py b/tests/test_model_factory.py new file mode 100644 index 0000000..e59075a --- /dev/null +++ b/tests/test_model_factory.py @@ -0,0 +1,9 @@ +import unittest + +from cockpit_grounding.models.factory import create_grounder + + +class ModelFactoryTest(unittest.TestCase): + def test_rejects_unknown_backend(self) -> None: + with self.assertRaisesRegex(ValueError, "Unsupported model backend"): + create_grounder("unknown", "/model")