first commit

This commit is contained in:
lgv 2026-08-24 16:29:35 +08:00
commit c9bcfaa6e8
41 changed files with 2319 additions and 0 deletions

0
.env.example Normal file
View File

2
.gitignore vendored Normal file
View File

@ -0,0 +1,2 @@
__pycache__/
*.py[cod]

10
.idea/.gitignore generated vendored Normal file
View File

@ -0,0 +1,10 @@
# Default ignored files
/shelf/
/workspace.xml
# Editor-based HTTP Client requests
/httpRequests/
# Ignored default folder with query files
/queries/
# Datasource local storage ignored files
/dataSources/
/dataSources.local.xml

14
.idea/cockpit-ui-grounding.iml generated Normal file
View File

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<module external.system.id="pyproject.toml" type="PYTHON_MODULE" version="4">
<component name="NewModuleRootManager">
<content url="file://$MODULE_DIR$">
<sourceFolder url="file://$MODULE_DIR$/src" />
</content>
<orderEntry type="jdk" jdkName="~/miniconda3/envs/qwen3vl" jdkType="Python SDK" />
<orderEntry type="sourceFolder" forTests="false" />
</component>
<component name="PackageRequirementsSettings" />
<component name="PyDocumentationSettings" />
<component name="ReSTService" />
<component name="TestRunnerService" />
</module>

View File

@ -0,0 +1,6 @@
<component name="InspectionProjectProfileManager">
<settings>
<option name="USE_PROJECT_PROFILE" value="false" />
<version value="1.0" />
</settings>
</component>

8
.idea/modules.xml generated Normal file
View File

@ -0,0 +1,8 @@
<?xml version="1.0" encoding="UTF-8"?>
<project version="4">
<component name="ProjectModuleManager">
<modules>
<module fileurl="file://$PROJECT_DIR$/.idea/cockpit-ui-grounding.iml" filepath="$PROJECT_DIR$/.idea/cockpit-ui-grounding.iml" />
</modules>
</component>
</project>

6
.idea/vcs.xml generated Normal file
View File

@ -0,0 +1,6 @@
<?xml version="1.0" encoding="UTF-8"?>
<project version="4">
<component name="VcsDirectoryMappings">
<mapping directory="" vcs="Git" />
</component>
</project>

0
README.md Normal file
View File

View File

@ -0,0 +1,13 @@
[benchmark]
warmup = 3
repeats = 5
max_new_tokens = 128
output_root = "/data/lgv/runs/cockpit-ui-grounding"
[models.qwen35_08b]
backend = "qwen35"
path = "/data/lgv/models/pretrained/Qwen3.5-0.8B"
[models.qwen35_2b]
backend = "qwen35"
path = "/data/lgv/models/pretrained/Qwen3.5-2B"

View File

@ -0,0 +1,14 @@
[benchmark]
warmup = 3
repeats = 5
max_new_tokens = 128
output_root = "/data/lgv/runs/cockpit-ui-grounding"
timestamp_run_directory = false
[models.qwen3vl_2b]
backend = "qwen3vl"
path = "/data/lgv/models/pretrained/Qwen3-VL-2B-Instruct"
[models.qwen35_2b]
backend = "qwen35"
path = "/data/lgv/models/pretrained/Qwen3.5-2B"

View File

@ -0,0 +1,11 @@
[benchmark]
warmup = 1
repeats = 3
max_new_tokens = 128
output_root = "/data/lgv/runs/cockpit-ui-grounding"
[models.qwen3vl_2b]
path = "/data/lgv/models/pretrained/Qwen3-VL-2B-Instruct"
[models.qwen3vl_4b]
path = "/data/lgv/models/pretrained/Qwen3-VL-4B-Instruct"

12
pyproject.toml Normal file
View File

@ -0,0 +1,12 @@
[build-system]
requires = ["setuptools>=68"]
build-backend = "setuptools.build_meta"
[project]
name = "cockpit-ui-grounding"
version = "0.1.0"
description = "Camera-view automotive cockpit UI grounding"
requires-python = ">=3.11"
[tool.setuptools.packages.find]
where = ["src"]

View File

@ -0,0 +1,75 @@
import argparse
import logging
from pathlib import Path
from cockpit_grounding.benchmark.config import (
load_benchmark_config,
load_manifest,
)
from cockpit_grounding.benchmark.reporting import (
print_comparison,
print_model_selection,
)
from cockpit_grounding.benchmark.runner import run_benchmark
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Benchmark local multimodal models sequentially on one GPU",
)
parser.add_argument(
"--config",
type=Path,
required=True,
help="Benchmark TOML configuration",
)
parser.add_argument(
"--manifest",
type=Path,
required=True,
help="JSONL benchmark manifest",
)
parser.add_argument(
"--run-name",
help="Optional run directory suffix",
)
return parser.parse_args()
def main() -> None:
args = parse_args()
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s | %(levelname)s | %(message)s",
)
config = load_benchmark_config(args.config)
samples = load_manifest(args.manifest)
run_directory, summaries = run_benchmark(
config=config,
samples=samples,
config_path=args.config,
manifest_path=args.manifest,
run_name=args.run_name,
)
comparison_title = (
"QWEN3.5 MODEL COMPARISON"
if all(model.backend == "qwen35" for model in config.models)
else "MODEL COMPARISON"
)
models_by_name = {model.name: model for model in config.models}
display_summaries = {
(
Path(models_by_name[name].path).name
if models_by_name[name].backend == "qwen35"
else name
): summary
for name, summary in summaries.items()
}
print_comparison(display_summaries, title=comparison_title)
print_model_selection(summaries)
print(f"Results: {run_directory}")
if __name__ == "__main__":
main()

128
scripts/run_grounding.py Normal file
View File

@ -0,0 +1,128 @@
import argparse
import json
from cockpit_grounding.models.factory import create_grounder
from cockpit_grounding.grounding.predictor import (
build_grounding_prompt,
parse_grounding_output,
)
from cockpit_grounding.vision.visualize import (
visualize_grounding,
)
def main():
parser = argparse.ArgumentParser()
parser.add_argument(
"--image",
required=True,
)
parser.add_argument(
"--target",
required=True,
)
parser.add_argument(
"--output",
required=True,
)
parser.add_argument(
"--model",
required=True,
help="Local model path",
)
parser.add_argument(
"--backend",
choices=("qwen3vl", "qwen35"),
default="qwen3vl",
help="Model backend (default: qwen3vl)",
)
args = parser.parse_args()
# ----------------------------
# Load model
# ----------------------------
grounder = create_grounder(
backend=args.backend,
model_path=args.model,
)
# ----------------------------
# Prompt
# ----------------------------
prompt = build_grounding_prompt(
args.target
)
# ----------------------------
# Inference
# ----------------------------
raw_output = grounder.generate(
args.image,
prompt,
)
print()
print("========== RAW MODEL OUTPUT ==========")
print(raw_output)
# ----------------------------
# Parse bbox
# ----------------------------
result = parse_grounding_output(
raw_output
)
# ----------------------------
# Convert + draw
# ----------------------------
pixel_result = visualize_grounding(
args.image,
result,
args.output,
)
final_result = {
"target": args.target,
"bbox_relative": [
result.x1,
result.y1,
result.x2,
result.y2,
],
**pixel_result,
}
print()
print("========== FINAL RESULT ==========")
print(
json.dumps(
final_result,
indent=2,
ensure_ascii=False,
)
)
print()
print(
f"Result image: {args.output}"
)
if __name__ == "__main__":
main()

View File

View File

View File

@ -0,0 +1,11 @@
from cockpit_grounding.benchmark.metrics import (
bbox_iou,
normalized_center_error,
point_in_box,
)
__all__ = [
"bbox_iou",
"normalized_center_error",
"point_in_box",
]

View File

@ -0,0 +1,190 @@
from __future__ import annotations
import json
import tomllib
from dataclasses import dataclass
from pathlib import Path
from typing import Any
@dataclass(frozen=True)
class BenchmarkSettings:
warmup: int
repeats: int
max_new_tokens: int
output_root: Path
timestamp_run_directory: bool = True
@dataclass(frozen=True)
class ModelConfig:
name: str
backend: str
path: Path
@dataclass(frozen=True)
class BenchmarkConfig:
settings: BenchmarkSettings
models: tuple[ModelConfig, ...]
@dataclass(frozen=True)
class BenchmarkSample:
sample_id: str
image: Path
target: str
gt_bbox_pixel: tuple[float, float, float, float] | None = None
def load_benchmark_config(path: str | Path) -> BenchmarkConfig:
config_path = Path(path).expanduser().resolve()
with config_path.open("rb") as config_file:
data = tomllib.load(config_file)
benchmark = _require_mapping(data, "benchmark")
models_data = _require_mapping(data, "models")
settings = BenchmarkSettings(
warmup=_non_negative_int(benchmark.get("warmup"), "benchmark.warmup"),
repeats=_positive_int(benchmark.get("repeats"), "benchmark.repeats"),
max_new_tokens=_positive_int(
benchmark.get("max_new_tokens"),
"benchmark.max_new_tokens",
),
output_root=Path(
_non_empty_string(
benchmark.get("output_root"),
"benchmark.output_root",
)
).expanduser(),
timestamp_run_directory=_optional_bool(
benchmark.get("timestamp_run_directory"),
"benchmark.timestamp_run_directory",
default=True,
),
)
models: list[ModelConfig] = []
for name, model_data in models_data.items():
if not isinstance(model_data, dict):
raise ValueError(f"models.{name} must be a TOML table")
model_path = Path(
_non_empty_string(model_data.get("path"), f"models.{name}.path")
).expanduser().resolve()
if not model_path.is_dir():
raise FileNotFoundError(f"Local model directory not found: {model_path}")
backend = _non_empty_string(
model_data.get("backend", "qwen3vl"),
f"models.{name}.backend",
)
models.append(ModelConfig(name=name, backend=backend, path=model_path))
if not models:
raise ValueError("At least one model must be configured")
return BenchmarkConfig(settings=settings, models=tuple(models))
def load_manifest(path: str | Path) -> tuple[BenchmarkSample, ...]:
manifest_path = Path(path).expanduser().resolve()
samples: list[BenchmarkSample] = []
seen_ids: set[str] = set()
with manifest_path.open("r", encoding="utf-8") as manifest_file:
for line_number, line in enumerate(manifest_file, start=1):
if not line.strip():
continue
try:
item = json.loads(line)
except json.JSONDecodeError as exc:
raise ValueError(
f"Invalid JSON at {manifest_path}:{line_number}: {exc.msg}"
) from exc
if not isinstance(item, dict):
raise ValueError(
f"Manifest entry at line {line_number} must be an object"
)
sample = _parse_sample(item, manifest_path, line_number)
if sample.sample_id in seen_ids:
raise ValueError(f"Duplicate sample id: {sample.sample_id}")
seen_ids.add(sample.sample_id)
samples.append(sample)
if not samples:
raise ValueError(f"Manifest contains no samples: {manifest_path}")
return tuple(samples)
def _parse_sample(
item: dict[str, Any],
manifest_path: Path,
line_number: int,
) -> BenchmarkSample:
prefix = f"{manifest_path}:{line_number}"
sample_id = _non_empty_string(item.get("id"), f"{prefix} id")
if Path(sample_id).name != sample_id or sample_id in {".", ".."}:
raise ValueError(f"{prefix} id must be a filename-safe identifier")
image = Path(
_non_empty_string(item.get("image"), f"{prefix} image")
).expanduser().resolve()
if not image.is_file():
raise FileNotFoundError(f"Image not found at {prefix}: {image}")
target = _non_empty_string(item.get("target"), f"{prefix} target")
gt_bbox = item.get("gt_bbox_pixel")
parsed_gt: tuple[float, float, float, float] | None = None
if gt_bbox is not None:
if not isinstance(gt_bbox, list) or len(gt_bbox) != 4:
raise ValueError(f"{prefix} gt_bbox_pixel must contain four numbers")
try:
parsed_gt = tuple(float(value) for value in gt_bbox) # type: ignore[assignment]
except (TypeError, ValueError) as exc:
raise ValueError(
f"{prefix} gt_bbox_pixel must contain four numbers"
) from exc
x1, y1, x2, y2 = parsed_gt
if x1 > x2 or y1 > y2:
raise ValueError(f"{prefix} gt_bbox_pixel has invalid corner order")
return BenchmarkSample(
sample_id=sample_id,
image=image,
target=target,
gt_bbox_pixel=parsed_gt,
)
def _require_mapping(data: dict[str, Any], key: str) -> dict[str, Any]:
value = data.get(key)
if not isinstance(value, dict):
raise ValueError(f"Missing or invalid [{key}] table")
return value
def _non_empty_string(value: Any, name: str) -> str:
if not isinstance(value, str) or not value.strip():
raise ValueError(f"{name} must be a non-empty string")
return value
def _positive_int(value: Any, name: str) -> int:
if not isinstance(value, int) or isinstance(value, bool) or value <= 0:
raise ValueError(f"{name} must be a positive integer")
return value
def _non_negative_int(value: Any, name: str) -> int:
if not isinstance(value, int) or isinstance(value, bool) or value < 0:
raise ValueError(f"{name} must be a non-negative integer")
return value
def _optional_bool(value: Any, name: str, *, default: bool) -> bool:
if value is None:
return default
if not isinstance(value, bool):
raise ValueError(f"{name} must be a boolean")
return value

View File

@ -0,0 +1,60 @@
from math import hypot
from typing import Sequence
Point = Sequence[float]
BBox = Sequence[float]
def _validate_point(point: Point, name: str) -> None:
if len(point) != 2:
raise ValueError(f"{name} must contain two coordinates")
def _validate_bbox(bbox: BBox, name: str) -> None:
if len(bbox) != 4:
raise ValueError(f"{name} must contain four coordinates")
x1, y1, x2, y2 = bbox
if x1 > x2 or y1 > y2:
raise ValueError(f"{name} must satisfy x1 <= x2 and y1 <= y2")
def point_in_box(pred_center: Point, gt_bbox: BBox) -> bool:
_validate_point(pred_center, "pred_center")
_validate_bbox(gt_bbox, "gt_bbox")
x, y = pred_center
x1, y1, x2, y2 = gt_bbox
return x1 <= x <= x2 and y1 <= y <= y2
def bbox_iou(pred_bbox: BBox, gt_bbox: BBox) -> float:
_validate_bbox(pred_bbox, "pred_bbox")
_validate_bbox(gt_bbox, "gt_bbox")
pred_x1, pred_y1, pred_x2, pred_y2 = pred_bbox
gt_x1, gt_y1, gt_x2, gt_y2 = gt_bbox
intersection_width = max(0.0, min(pred_x2, gt_x2) - max(pred_x1, gt_x1))
intersection_height = max(0.0, min(pred_y2, gt_y2) - max(pred_y1, gt_y1))
intersection = intersection_width * intersection_height
pred_area = (pred_x2 - pred_x1) * (pred_y2 - pred_y1)
gt_area = (gt_x2 - gt_x1) * (gt_y2 - gt_y1)
union = pred_area + gt_area - intersection
return intersection / union if union > 0 else 0.0
def normalized_center_error(
pred_center: Point,
gt_bbox: BBox,
image_width: int,
image_height: int,
) -> float:
"""Return center distance normalized by the image diagonal."""
_validate_point(pred_center, "pred_center")
_validate_bbox(gt_bbox, "gt_bbox")
if image_width <= 0 or image_height <= 0:
raise ValueError("image dimensions must be positive")
gt_x1, gt_y1, gt_x2, gt_y2 = gt_bbox
gt_center_x = (gt_x1 + gt_x2) / 2.0
gt_center_y = (gt_y1 + gt_y2) / 2.0
distance = hypot(pred_center[0] - gt_center_x, pred_center[1] - gt_center_y)
return distance / hypot(image_width, image_height)

View File

@ -0,0 +1,317 @@
from __future__ import annotations
import csv
import json
from datetime import datetime
from pathlib import Path
from typing import Any
RESULT_COLUMNS = (
"model",
"id",
"target",
"parse_success",
"point_in_box",
"bbox_iou",
"normalized_center_error",
"preprocess_mean_ms",
"generate_mean_ms",
"decode_mean_ms",
"total_mean_ms",
"total_p50_ms",
"total_p95_ms",
"peak_cuda_memory_mb",
)
def create_run_directory(
output_root: Path,
run_name: str,
*,
timestamped: bool = True,
) -> Path:
safe_name = _safe_run_name(run_name)
directory_name = (
f"{datetime.now().strftime('%Y%m%d_%H%M%S')}_{safe_name}"
if timestamped
else safe_name
)
run_directory = output_root / directory_name
run_directory.mkdir(parents=True, exist_ok=False)
return run_directory
def write_model_outputs(
model_directory: Path,
summary: dict[str, Any],
predictions: list[dict[str, Any]],
) -> None:
model_directory.mkdir(parents=True, exist_ok=True)
write_json(model_directory / "summary.json", summary)
with (model_directory / "predictions.jsonl").open(
"w",
encoding="utf-8",
) as predictions_file:
for prediction in predictions:
predictions_file.write(
json.dumps(prediction, ensure_ascii=False, allow_nan=False) + "\n"
)
def write_root_outputs(
run_directory: Path,
metadata: dict[str, Any],
summaries: dict[str, dict[str, Any]],
predictions: list[dict[str, Any]],
) -> None:
write_json(run_directory / "benchmark_meta.json", metadata)
write_json(
run_directory / "summary.json",
{"models": summaries},
)
with (run_directory / "results.csv").open(
"w",
encoding="utf-8",
newline="",
) as csv_file:
writer = csv.DictWriter(csv_file, fieldnames=RESULT_COLUMNS)
writer.writeheader()
for prediction in predictions:
writer.writerow({column: prediction.get(column) for column in RESULT_COLUMNS})
_write_case_results(
run_directory / "case_results.csv",
tuple(summaries),
predictions,
)
_write_review(
run_directory / "review.md",
tuple(summaries),
predictions,
)
def _write_case_results(
path: Path,
model_names: tuple[str, ...],
predictions: list[dict[str, Any]],
) -> None:
columns = ["id", "target"]
for model_name in model_names:
columns.extend(
(
f"{model_name}_center_x",
f"{model_name}_center_y",
f"{model_name}_latency_ms",
f"{model_name}_parse_success",
f"{model_name}_bbox_x1",
f"{model_name}_bbox_y1",
f"{model_name}_bbox_x2",
f"{model_name}_bbox_y2",
f"manual_{model_name}",
)
)
cases: dict[str, dict[str, Any]] = {}
for prediction in predictions:
sample_id = prediction["id"]
target = prediction["target"]
case = cases.setdefault(sample_id, {"id": sample_id, "target": target})
if case["target"] != target:
raise ValueError(f"Mismatched targets for benchmark case: {sample_id}")
model_name = prediction["model"]
if model_name not in model_names:
raise ValueError(f"Unexpected model in predictions: {model_name}")
center = prediction.get("pred_center_pixel") or (None, None)
bbox = prediction.get("pred_bbox_pixel") or (None, None, None, None)
case[f"{model_name}_center_x"] = center[0]
case[f"{model_name}_center_y"] = center[1]
case[f"{model_name}_latency_ms"] = prediction.get("total_mean_ms")
case[f"{model_name}_parse_success"] = prediction.get("parse_success")
case[f"{model_name}_bbox_x1"] = bbox[0]
case[f"{model_name}_bbox_y1"] = bbox[1]
case[f"{model_name}_bbox_x2"] = bbox[2]
case[f"{model_name}_bbox_y2"] = bbox[3]
case[f"manual_{model_name}"] = ""
with path.open("w", encoding="utf-8", newline="") as csv_file:
writer = csv.DictWriter(csv_file, fieldnames=columns)
writer.writeheader()
for case in cases.values():
writer.writerow({column: case.get(column) for column in columns})
def _write_review(
path: Path,
model_names: tuple[str, ...],
predictions: list[dict[str, Any]],
) -> None:
cases: dict[str, dict[str, Any]] = {}
for prediction in predictions:
sample_id = prediction["id"]
case = cases.setdefault(
sample_id,
{
"target": prediction["target"],
"image": prediction["image"],
"models": {},
},
)
case["models"][prediction["model"]] = prediction
lines = [
"# Manual Grounding Review",
"",
"Use `correct`, `wrong`, or `partial` in the corresponding ",
"`manual_<model>` columns of `case_results.csv` after visual review.",
"",
]
for sample_id, case in cases.items():
lines.extend(
(
f"## {sample_id}",
"",
f"Target: {case['target']}",
"",
f"Source: `{case['image']}`",
"",
)
)
for model_name in model_names:
prediction = case["models"].get(model_name)
if prediction is None:
lines.extend((f"### {model_name}", "", "Missing prediction.", ""))
continue
visualization = Path(model_name) / "visualizations" / f"{sample_id}.jpg"
lines.extend(
(
f"### {model_name}",
"",
f"Parse success: `{prediction.get('parse_success')}`",
"",
f"Predicted bbox: `{prediction.get('pred_bbox_pixel')}`",
"",
f"![{model_name} {sample_id}]({visualization.as_posix()})",
"",
)
)
path.write_text("\n".join(lines), encoding="utf-8")
def write_json(path: Path, value: dict[str, Any]) -> None:
with path.open("w", encoding="utf-8") as output_file:
json.dump(
value,
output_file,
indent=2,
ensure_ascii=False,
allow_nan=False,
)
output_file.write("\n")
def print_comparison(
summaries: dict[str, dict[str, Any]],
title: str = "MODEL COMPARISON",
) -> None:
print()
print("=" * 60)
print(title)
print("=" * 60)
for model_name, summary in summaries.items():
print()
print(model_name)
print()
print(f" model load : {_format(summary['model_load_seconds'], '.2f', 's')}")
print(
" CUDA memory : "
f"{_format(summary['model_cuda_allocated_mb'], '.0f', 'MB')} allocated, "
f"{_format(summary['model_cuda_reserved_mb'], '.0f', 'MB')} reserved"
)
print(
" peak memory : "
f"{_format(summary['peak_cuda_memory_mb'], '.0f', 'MB')}"
)
print(f" mean latency : {_format(summary['mean_total_ms'], '.1f', 'ms')}")
print(f" p50 latency : {_format(summary['p50_total_ms'], '.1f', 'ms')}")
print(f" p95 latency : {_format(summary['p95_total_ms'], '.1f', 'ms')}")
print(
" generation mean : "
f"{_format(summary['mean_generate_ms'], '.1f', 'ms')}"
)
print(
" throughput : "
f"{_format(summary['throughput_samples_per_sec'], '.2f', 'samples/s')}"
)
print(f" parse success : {summary['parse_success_rate'] * 100:.1f} %")
if summary["accuracy_point_in_box"] is not None:
print(
" grounding acc : "
f"{summary['accuracy_point_in_box'] * 100:.1f} %"
)
print()
print("=" * 60)
def print_model_selection(summaries: dict[str, dict[str, Any]]) -> None:
model_names = tuple(summaries)
print()
print("=" * 60)
print("MODEL SELECTION")
print("=" * 60)
print()
header = f"{'Metric':<32}" + "".join(f"{name:>20}" for name in model_names)
print(header)
rows = (
("Parse Success", "parse_success_rate", ".1%"),
("Mean Latency", "mean_total_ms", ".1f"),
("P50", "p50_total_ms", ".1f"),
("P95", "p95_total_ms", ".1f"),
("Peak GPU Memory", "peak_cuda_memory_mb", ".0f"),
("Throughput", "throughput_samples_per_sec", ".3f"),
)
for label, key, number_format in rows:
values = "".join(
f"{_format_table_value(summaries[name].get(key), number_format):>20}"
for name in model_names
)
print(f"{label:<32}{values}")
print()
manual = "N/A - manual review required"
for label in (
"Text UI Accuracy",
"Generic Icon Accuracy",
"Automotive Icon Accuracy",
"Function Region Accuracy",
"Small Sub-Control Accuracy",
"Slider Geometry Accuracy",
):
values = "".join(f"{'N/A':>20}" for _ in model_names)
print(f"{label:<32}{values}")
print()
print(manual)
print()
print("=" * 60)
def _format_table_value(value: Any, number_format: str) -> str:
if not isinstance(value, (int, float)):
return "n/a"
return format(value, number_format)
def _format(value: float | None, number_format: str, unit: str) -> str:
if value is None:
return "n/a"
return f"{value:{number_format}} {unit}"
def _safe_run_name(run_name: str) -> str:
if not run_name or any(character in run_name for character in "/\\"):
raise ValueError("run name must be non-empty and cannot contain path separators")
if run_name in {".", ".."}:
raise ValueError("invalid run name")
return run_name

View File

@ -0,0 +1,319 @@
from __future__ import annotations
import gc
import logging
import os
import platform
import hashlib
from datetime import datetime
from pathlib import Path
from time import perf_counter
from typing import Any
import torch
import transformers
from cockpit_grounding.benchmark.config import (
BenchmarkConfig,
BenchmarkSample,
)
from cockpit_grounding.benchmark.metrics import (
bbox_iou,
normalized_center_error,
point_in_box,
)
from cockpit_grounding.benchmark.reporting import (
create_run_directory,
write_json,
write_model_outputs,
write_root_outputs,
)
from cockpit_grounding.benchmark.statistics import (
summarize_model,
summarize_timings,
)
from cockpit_grounding.grounding.predictor import (
build_grounding_prompt,
parse_grounding_output,
)
from cockpit_grounding.models.base import Grounder, InferenceTiming
from cockpit_grounding.models.factory import create_grounder
from cockpit_grounding.vision.visualize import visualize_grounding
LOGGER = logging.getLogger(__name__)
DEFAULT_RUN_NAME = "qwen3vl_2b_vs_4b"
def run_benchmark(
*,
config: BenchmarkConfig,
samples: tuple[BenchmarkSample, ...],
config_path: Path,
manifest_path: Path,
run_name: str | None = None,
) -> tuple[Path, dict[str, dict[str, Any]]]:
_validate_cuda_environment()
settings = config.settings
run_directory = create_run_directory(
settings.output_root,
run_name or DEFAULT_RUN_NAME,
timestamped=settings.timestamp_run_directory,
)
started_at = datetime.now().astimezone()
metadata: dict[str, Any] = {
"started_at": started_at.isoformat(),
"completed_at": None,
"config": str(config_path.resolve()),
"manifest": str(manifest_path.resolve()),
"run_directory": str(run_directory.resolve()),
"cuda_visible_devices": os.environ.get("CUDA_VISIBLE_DEVICES"),
"cuda_device": torch.cuda.get_device_name(0),
"python_version": platform.python_version(),
"torch_version": torch.__version__,
"transformers_version": transformers.__version__,
"grounding_prompt_sha256": hashlib.sha256(
build_grounding_prompt("__SEMANTIC_TARGET__").encode("utf-8")
).hexdigest(),
"preprocessing_policy": (
"source image at native resolution; no benchmark-side resize"
),
"benchmark": {
"warmup": settings.warmup,
"repeats": settings.repeats,
"max_new_tokens": settings.max_new_tokens,
},
"models": {model.name: str(model.path) for model in config.models},
"model_backends": {model.name: model.backend for model in config.models},
"sample_count": len(samples),
}
write_json(run_directory / "benchmark_meta.json", metadata)
summaries: dict[str, dict[str, Any]] = {}
all_predictions: list[dict[str, Any]] = []
for model in config.models:
summary, predictions = _benchmark_model(
model_name=model.name,
backend=model.backend,
model_path=model.path,
samples=samples,
warmup=settings.warmup,
repeats=settings.repeats,
max_new_tokens=settings.max_new_tokens,
run_directory=run_directory,
)
summaries[model.name] = summary
all_predictions.extend(predictions)
completed_at = datetime.now().astimezone()
metadata["completed_at"] = completed_at.isoformat()
metadata["duration_seconds"] = (completed_at - started_at).total_seconds()
write_root_outputs(
run_directory,
metadata,
summaries,
all_predictions,
)
return run_directory, summaries
def _benchmark_model(
*,
model_name: str,
backend: str,
model_path: Path,
samples: tuple[BenchmarkSample, ...],
warmup: int,
repeats: int,
max_new_tokens: int,
run_directory: Path,
) -> tuple[dict[str, Any], list[dict[str, Any]]]:
LOGGER.info("Loading %s from %s", model_name, model_path)
grounder: Grounder | None = None
try:
torch.cuda.synchronize()
load_start = perf_counter()
grounder = create_grounder(backend=backend, model_path=model_path)
torch.cuda.synchronize()
model_load_seconds = perf_counter() - load_start
memory_metrics = grounder.cuda_memory_metrics()
if warmup:
LOGGER.info("Running %d warmup inference(s) for %s", warmup, model_name)
prompt = build_grounding_prompt(samples[0].target)
for _ in range(warmup):
grounder.generate(
image_path=str(samples[0].image),
prompt=prompt,
max_new_tokens=max_new_tokens,
)
model_directory = run_directory / model_name
visualization_directory = model_directory / "visualizations"
visualization_directory.mkdir(parents=True, exist_ok=True)
predictions: list[dict[str, Any]] = []
all_timings: list[InferenceTiming] = []
for index, sample in enumerate(samples, start=1):
LOGGER.info(
"[%s] sample %d/%d: %s",
model_name,
index,
len(samples),
sample.sample_id,
)
prediction, timings = _benchmark_sample(
grounder=grounder,
model_name=model_name,
sample=sample,
repeats=repeats,
max_new_tokens=max_new_tokens,
visualization_path=(
visualization_directory / f"{sample.sample_id}.jpg"
),
)
predictions.append(prediction)
all_timings.extend(timings)
summary = summarize_model(
model_name=model_name,
model_path=str(model_path),
model_load_seconds=model_load_seconds,
memory_metrics=memory_metrics,
predictions=predictions,
all_timings=all_timings,
)
summary.update(grounder.implementation_metadata())
write_model_outputs(model_directory, summary, predictions)
LOGGER.info("Saved %s results to %s", model_name, model_directory)
return summary, predictions
finally:
if grounder is not None:
del grounder
gc.collect()
torch.cuda.empty_cache()
torch.cuda.synchronize()
LOGGER.info("Released model %s", model_name)
def _benchmark_sample(
*,
grounder: Grounder,
model_name: str,
sample: BenchmarkSample,
repeats: int,
max_new_tokens: int,
visualization_path: Path,
) -> tuple[dict[str, Any], list[InferenceTiming]]:
prompt = build_grounding_prompt(sample.target)
raw_output: str | None = None
timings: list[InferenceTiming] = []
peak_memory_values: list[float] = []
inference_errors: list[str] = []
for repeat_index in range(1, repeats + 1):
try:
output, timing, extra_metrics = grounder.generate_with_metrics(
image_path=str(sample.image),
prompt=prompt,
max_new_tokens=max_new_tokens,
)
if raw_output is None:
raw_output = output
timings.append(timing)
peak_memory_values.append(extra_metrics["peak_cuda_memory_mb"])
except Exception as exc: # Keep the batch running after a sample failure.
error = f"repeat {repeat_index}: {type(exc).__name__}: {exc}"
inference_errors.append(error)
LOGGER.exception(
"[%s] inference failed for %s (%s)",
model_name,
sample.sample_id,
error,
)
timing_summary = summarize_timings(timings)
prediction: dict[str, Any] = {
"model": model_name,
"id": sample.sample_id,
"image": str(sample.image),
"target": sample.target,
"raw_output": raw_output,
"parse_success": False,
"parse_error": None,
"inference_errors": inference_errors,
"pred_bbox_pixel": None,
"pred_center_pixel": None,
"gt_bbox_pixel": (
list(sample.gt_bbox_pixel) if sample.gt_bbox_pixel is not None else None
),
"point_in_box": None,
"bbox_iou": None,
"normalized_center_error": None,
**timing_summary,
"peak_cuda_memory_mb": max(peak_memory_values) if peak_memory_values else None,
}
if raw_output is None:
prediction["parse_error"] = "; ".join(inference_errors) or "No model output"
return prediction, timings
try:
parsed = parse_grounding_output(raw_output)
prediction["parse_success"] = True
except Exception as exc:
prediction["parse_error"] = f"{type(exc).__name__}: {exc}"
LOGGER.warning(
"[%s] parse failed for %s: %s",
model_name,
sample.sample_id,
exc,
)
return prediction, timings
try:
pixel_result = visualize_grounding(
str(sample.image),
parsed,
str(visualization_path),
)
pred_bbox = pixel_result["bbox_pixel"]
pred_center = pixel_result["center_pixel"]
prediction["pred_bbox_pixel"] = pred_bbox
prediction["pred_center_pixel"] = pred_center
if sample.gt_bbox_pixel is not None:
prediction["point_in_box"] = point_in_box(
pred_center,
sample.gt_bbox_pixel,
)
prediction["bbox_iou"] = bbox_iou(
pred_bbox,
sample.gt_bbox_pixel,
)
prediction["normalized_center_error"] = normalized_center_error(
pred_center,
sample.gt_bbox_pixel,
pixel_result["image_width"],
pixel_result["image_height"],
)
except Exception as exc:
prediction["visualization_error"] = f"{type(exc).__name__}: {exc}"
LOGGER.exception(
"[%s] visualization failed for %s",
model_name,
sample.sample_id,
)
return prediction, timings
def _validate_cuda_environment() -> None:
if not torch.cuda.is_available():
raise RuntimeError("CUDA is required for the multimodal benchmark")
visible_device_count = torch.cuda.device_count()
if visible_device_count != 1:
raise RuntimeError(
"Benchmark requires exactly one visible CUDA device; "
f"found {visible_device_count}. Set CUDA_VISIBLE_DEVICES=0."
)

View File

@ -0,0 +1,96 @@
from __future__ import annotations
from collections.abc import Iterable, Sequence
from statistics import fmean
from typing import Any
from cockpit_grounding.models.base import InferenceTiming
def percentile(values: Sequence[float], quantile: float) -> float:
if not values:
raise ValueError("percentile requires at least one value")
if not 0.0 <= quantile <= 1.0:
raise ValueError("quantile must be between 0 and 1")
ordered = sorted(values)
position = (len(ordered) - 1) * quantile
lower = int(position)
upper = min(lower + 1, len(ordered) - 1)
fraction = position - lower
return ordered[lower] + (ordered[upper] - ordered[lower]) * fraction
def summarize_timings(timings: Sequence[InferenceTiming]) -> dict[str, float | None]:
if not timings:
return {
"preprocess_mean_ms": None,
"generate_mean_ms": None,
"decode_mean_ms": None,
"total_mean_ms": None,
"total_p50_ms": None,
"total_p95_ms": None,
}
totals = [timing.total_ms for timing in timings]
return {
"preprocess_mean_ms": fmean(t.preprocess_ms for t in timings),
"generate_mean_ms": fmean(t.generate_ms for t in timings),
"decode_mean_ms": fmean(t.decode_ms for t in timings),
"total_mean_ms": fmean(totals),
"total_p50_ms": percentile(totals, 0.50),
"total_p95_ms": percentile(totals, 0.95),
}
def summarize_model(
*,
model_name: str,
model_path: str,
model_load_seconds: float,
memory_metrics: dict[str, float],
predictions: Sequence[dict[str, Any]],
all_timings: Sequence[InferenceTiming],
) -> dict[str, Any]:
timing = summarize_timings(all_timings)
sample_count = len(predictions)
parse_successes = sum(bool(item["parse_success"]) for item in predictions)
point_values = _present_values(predictions, "point_in_box")
iou_values = _present_values(predictions, "bbox_iou")
center_error_values = _present_values(predictions, "normalized_center_error")
peak_memory_values = _present_values(predictions, "peak_cuda_memory_mb")
mean_total = timing["total_mean_ms"]
return {
"model": model_name,
"model_path": model_path,
"model_load_seconds": model_load_seconds,
**memory_metrics,
"samples": sample_count,
"parse_success_rate": parse_successes / sample_count if sample_count else 0.0,
"mean_preprocess_ms": timing["preprocess_mean_ms"],
"mean_generate_ms": timing["generate_mean_ms"],
"mean_decode_ms": timing["decode_mean_ms"],
"mean_total_ms": mean_total,
"p50_total_ms": timing["total_p50_ms"],
"p95_total_ms": timing["total_p95_ms"],
"throughput_samples_per_sec": (
1000.0 / mean_total if isinstance(mean_total, float) and mean_total > 0 else None
),
"peak_cuda_memory_mb": max(peak_memory_values) if peak_memory_values else None,
"accuracy_point_in_box": (
fmean(point_values) if point_values else None
),
"mean_bbox_iou": fmean(iou_values) if iou_values else None,
"mean_normalized_center_error": (
fmean(center_error_values) if center_error_values else None
),
}
def _present_values(
predictions: Iterable[dict[str, Any]],
key: str,
) -> list[float]:
return [float(item[key]) for item in predictions if item.get(key) is not None]

View File

@ -0,0 +1,94 @@
import json
import re
from dataclasses import dataclass
@dataclass
class GroundingResult:
x1: float
y1: float
x2: float
y2: float
@property
def center_relative(self):
return (
(self.x1 + self.x2) / 2.0,
(self.y1 + self.y2) / 2.0,
)
def build_grounding_prompt(target: str) -> str:
return f"""
你是一个汽车座舱可操作UI元素定位器。
输入图像来自真实相机拍摄,而不是系统截图。
请找到以下目标:
{target}
要求:
1. 找到真正可以被用户点击的控件。
2. 不要选择标题、说明文字或者附近的无关区域。
3. 返回整个可点击控件的边界框。
4. 使用相对坐标:
左上角 = (0, 0)
右下角 = (1000, 1000)
5. x1 < x2,y1 < y2。
6. 只返回 JSON,不要解释。
严格输出:
{{"bbox_2d": [x1, y1, x2, y2]}}
""".strip()
def parse_grounding_output(
text: str,
) -> GroundingResult:
# 防止模型偶尔包 markdown code block
match = re.search(
r'"bbox_2d"\s*:\s*\[\s*'
r'([0-9.]+)\s*,\s*'
r'([0-9.]+)\s*,\s*'
r'([0-9.]+)\s*,\s*'
r'([0-9.]+)\s*\]',
text,
)
if match is None:
raise ValueError(
"无法从模型输出解析 bbox_2d:\n"
+ text
)
x1, y1, x2, y2 = (
float(match.group(i))
for i in range(1, 5)
)
# 基础合法性检查
values = [x1, y1, x2, y2]
if not all(
0 <= value <= 1000
for value in values
):
raise ValueError(
f"坐标超出 0~1000: {values}"
)
if x1 >= x2 or y1 >= y2:
raise ValueError(
f"非法 bbox: {values}"
)
return GroundingResult(
x1=x1,
y1=y1,
x2=x2,
y2=y2,
)

View File

View File

@ -0,0 +1,42 @@
from __future__ import annotations
from dataclasses import dataclass
from typing import Protocol
@dataclass(frozen=True)
class InferenceTiming:
preprocess_ms: float
generate_ms: float
decode_ms: float
total_ms: float
class Grounder(Protocol):
model_path: str
def generate(
self,
image_path: str,
prompt: str,
max_new_tokens: int = 128,
) -> str: ...
def generate_with_metrics(
self,
image_path: str,
prompt: str,
max_new_tokens: int = 128,
) -> tuple[str, InferenceTiming, dict[str, float]]: ...
def cuda_memory_metrics(self) -> dict[str, float]: ...
def implementation_metadata(self) -> dict[str, str]: ...
class TextGenerator(Protocol):
def generate_text(
self,
prompt: str,
max_new_tokens: int = 256,
) -> str: ...

View File

@ -0,0 +1,20 @@
from pathlib import Path
from cockpit_grounding.models.base import Grounder
SUPPORTED_BACKENDS = ("qwen3vl", "qwen35")
def create_grounder(backend: str, model_path: str | Path) -> Grounder:
if backend == "qwen3vl":
from cockpit_grounding.models.qwen3vl import Qwen3VLGrounder
return Qwen3VLGrounder(model_path=str(model_path))
if backend == "qwen35":
from cockpit_grounding.models.qwen35 import Qwen35Grounder
return Qwen35Grounder(model_path=str(model_path))
raise ValueError(
f"Unsupported model backend {backend!r}; "
f"expected one of {', '.join(SUPPORTED_BACKENDS)}"
)

View File

@ -0,0 +1,177 @@
from pathlib import Path
from time import perf_counter
import torch
from transformers import AutoModelForMultimodalLM, AutoProcessor
from cockpit_grounding.models.base import InferenceTiming
class Qwen35Grounder:
def __init__(self, model_path: str) -> None:
local_model_path = Path(model_path).expanduser().resolve()
if not local_model_path.is_dir():
raise FileNotFoundError(
f"Local model directory not found: {local_model_path}"
)
if not torch.cuda.is_available():
raise RuntimeError("Qwen35Grounder requires a CUDA device")
self.model_path = str(local_model_path)
print("[Model] Loading local Qwen3.5 model:")
print(self.model_path)
self.model = AutoModelForMultimodalLM.from_pretrained(
self.model_path,
dtype=torch.bfloat16,
device_map={"": 0},
local_files_only=True,
)
self.processor = AutoProcessor.from_pretrained(
self.model_path,
local_files_only=True,
)
self.model.eval()
print(f"[Model] Class: {type(self.model).__name__}")
print("[Model] Loaded successfully")
print("[Model] GPU:", torch.cuda.get_device_name(self.model.device))
@torch.inference_mode()
def generate(
self,
image_path: str,
prompt: str,
max_new_tokens: int = 128,
) -> str:
raw_output, _, _ = self.generate_with_metrics(
image_path=image_path,
prompt=prompt,
max_new_tokens=max_new_tokens,
)
return raw_output
@torch.inference_mode()
def generate_text(
self,
prompt: str,
max_new_tokens: int = 256,
) -> str:
messages = [
{
"role": "user",
"content": [{"type": "text", "text": prompt}],
}
]
inputs = self.processor.apply_chat_template(
messages,
add_generation_prompt=True,
tokenize=True,
return_dict=True,
return_tensors="pt",
)
inputs = inputs.to(self.model.device)
output_ids = self.model.generate(
**inputs,
max_new_tokens=max_new_tokens,
do_sample=False,
)
generated_ids = output_ids[:, inputs["input_ids"].shape[-1] :]
return self.processor.batch_decode(
generated_ids,
skip_special_tokens=True,
clean_up_tokenization_spaces=False,
)[0]
@torch.inference_mode()
def generate_with_metrics(
self,
image_path: str,
prompt: str,
max_new_tokens: int = 128,
) -> tuple[str, InferenceTiming, dict[str, float]]:
image = Path(image_path).expanduser().resolve()
if not image.is_file():
raise FileNotFoundError(image)
self._synchronize_cuda()
torch.cuda.reset_peak_memory_stats(self.model.device)
total_start = perf_counter()
preprocess_start = total_start
messages = [
{
"role": "user",
"content": [
{"type": "image", "path": str(image)},
{"type": "text", "text": prompt},
],
}
]
inputs = self.processor.apply_chat_template(
messages,
add_generation_prompt=True,
tokenize=True,
return_dict=True,
return_tensors="pt",
)
inputs = inputs.to(self.model.device)
self._synchronize_cuda()
preprocess_ms = (perf_counter() - preprocess_start) * 1000.0
generate_start = perf_counter()
output_ids = self.model.generate(
**inputs,
max_new_tokens=max_new_tokens,
do_sample=False,
)
self._synchronize_cuda()
generate_ms = (perf_counter() - generate_start) * 1000.0
decode_start = perf_counter()
generated_ids = output_ids[:, inputs["input_ids"].shape[-1] :]
output_text = self.processor.batch_decode(
generated_ids,
skip_special_tokens=True,
clean_up_tokenization_spaces=False,
)[0]
self._synchronize_cuda()
decode_ms = (perf_counter() - decode_start) * 1000.0
total_ms = (perf_counter() - total_start) * 1000.0
timing = InferenceTiming(
preprocess_ms=preprocess_ms,
generate_ms=generate_ms,
decode_ms=decode_ms,
total_ms=total_ms,
)
metrics = {
"peak_cuda_memory_mb": self._bytes_to_mb(
torch.cuda.max_memory_allocated(self.model.device)
)
}
return output_text, timing, metrics
def cuda_memory_metrics(self) -> dict[str, float]:
return {
"model_cuda_allocated_mb": self._bytes_to_mb(
torch.cuda.memory_allocated(self.model.device)
),
"model_cuda_reserved_mb": self._bytes_to_mb(
torch.cuda.memory_reserved(self.model.device)
),
}
def implementation_metadata(self) -> dict[str, str]:
return {
"transformers_model_class": type(self.model).__name__,
"transformers_processor_class": type(self.processor).__name__,
}
@staticmethod
def _bytes_to_mb(value: int) -> float:
return value / (1024.0 * 1024.0)
def _synchronize_cuda(self) -> None:
torch.cuda.synchronize(self.model.device)

View File

@ -0,0 +1,204 @@
from pathlib import Path
from time import perf_counter
import torch
from transformers import (
AutoProcessor,
Qwen3VLForConditionalGeneration,
)
from qwen_vl_utils import process_vision_info
from cockpit_grounding.models.base import InferenceTiming
class Qwen3VLGrounder:
def __init__(
self,
model_path: str,
) -> None:
local_model_path = Path(model_path).expanduser().resolve()
if not local_model_path.is_dir():
raise FileNotFoundError(
f"Local model directory not found: {local_model_path}"
)
if not torch.cuda.is_available():
raise RuntimeError("Qwen3VLGrounder requires a CUDA device")
self.model_path = str(local_model_path)
print("[Model] Loading local model:")
print(self.model_path)
self.model = (
Qwen3VLForConditionalGeneration
.from_pretrained(
self.model_path,
dtype=torch.bfloat16,
device_map={"": 0},
local_files_only=True,
)
)
self.processor = AutoProcessor.from_pretrained(
self.model_path,
local_files_only=True,
)
self.model.eval()
print("[Model] Loaded successfully")
print(
"[Model] GPU:",
torch.cuda.get_device_name(self.model.device)
)
@torch.inference_mode()
def generate(
self,
image_path: str,
prompt: str,
max_new_tokens: int = 128,
) -> str:
raw_output, _, _ = self.generate_with_metrics(
image_path=image_path,
prompt=prompt,
max_new_tokens=max_new_tokens,
)
return raw_output
@torch.inference_mode()
def generate_with_metrics(
self,
image_path: str,
prompt: str,
max_new_tokens: int = 128,
) -> tuple[str, InferenceTiming, dict[str, float]]:
"""Generate a response and report synchronized stage timings."""
image_path = Path(image_path).resolve()
if not image_path.exists():
raise FileNotFoundError(image_path)
self._synchronize_cuda()
torch.cuda.reset_peak_memory_stats(self.model.device)
total_start = perf_counter()
preprocess_start = total_start
messages = [
{
"role": "user",
"content": [
{
"type": "image",
"image": image_path.as_uri(),
},
{
"type": "text",
"text": prompt,
},
],
}
]
text = self.processor.apply_chat_template(
messages,
tokenize=False,
add_generation_prompt=True,
)
images, videos, video_kwargs = process_vision_info(
messages,
image_patch_size=16,
return_video_kwargs=True,
return_video_metadata=True,
)
if videos is not None:
videos, video_metadatas = zip(*videos)
videos = list(videos)
video_metadatas = list(video_metadatas)
else:
video_metadatas = None
inputs = self.processor(
text=text,
images=images,
videos=videos,
video_metadata=video_metadatas,
return_tensors="pt",
do_resize=False,
**video_kwargs,
)
inputs = inputs.to(self.model.device)
self._synchronize_cuda()
preprocess_ms = (perf_counter() - preprocess_start) * 1000.0
generate_start = perf_counter()
generated_ids = self.model.generate(
**inputs,
max_new_tokens=max_new_tokens,
do_sample=False,
)
self._synchronize_cuda()
generate_ms = (perf_counter() - generate_start) * 1000.0
decode_start = perf_counter()
generated_ids_trimmed = [
output_ids[len(input_ids):]
for input_ids, output_ids
in zip(
inputs.input_ids,
generated_ids,
)
]
output_text = self.processor.batch_decode(
generated_ids_trimmed,
skip_special_tokens=True,
clean_up_tokenization_spaces=False,
)[0]
self._synchronize_cuda()
decode_ms = (perf_counter() - decode_start) * 1000.0
total_ms = (perf_counter() - total_start) * 1000.0
peak_cuda_memory_mb = self._bytes_to_mb(
torch.cuda.max_memory_allocated(self.model.device)
)
timing = InferenceTiming(
preprocess_ms=preprocess_ms,
generate_ms=generate_ms,
decode_ms=decode_ms,
total_ms=total_ms,
)
extra_metrics = {
"peak_cuda_memory_mb": peak_cuda_memory_mb,
}
return output_text, timing, extra_metrics
@staticmethod
def _bytes_to_mb(value: int) -> float:
return value / (1024.0 * 1024.0)
def cuda_memory_metrics(self) -> dict[str, float]:
return {
"model_cuda_allocated_mb": self._bytes_to_mb(
torch.cuda.memory_allocated(self.model.device)
),
"model_cuda_reserved_mb": self._bytes_to_mb(
torch.cuda.memory_reserved(self.model.device)
),
}
def implementation_metadata(self) -> dict[str, str]:
return {
"transformers_model_class": type(self.model).__name__,
"transformers_processor_class": type(self.processor).__name__,
}
def _synchronize_cuda(self) -> None:
torch.cuda.synchronize(self.model.device)

View File

View File

View File

@ -0,0 +1,109 @@
from pathlib import Path
import cv2
from cockpit_grounding.grounding.predictor import (
GroundingResult,
)
def relative_bbox_to_pixels(
result: GroundingResult,
width: int,
height: int,
):
x1 = round(result.x1 / 1000.0 * width)
y1 = round(result.y1 / 1000.0 * height)
x2 = round(result.x2 / 1000.0 * width)
y2 = round(result.y2 / 1000.0 * height)
return x1, y1, x2, y2
def visualize_grounding(
image_path: str,
result: GroundingResult,
output_path: str,
):
image = cv2.imread(image_path)
if image is None:
raise RuntimeError(
f"无法读取图片: {image_path}"
)
height, width = image.shape[:2]
x1, y1, x2, y2 = relative_bbox_to_pixels(
result,
width,
height,
)
# 后面给机器人使用的点
u = round((x1 + x2) / 2)
v = round((y1 + y2) / 2)
# bbox
cv2.rectangle(
image,
(x1, y1),
(x2, y2),
(0, 0, 255),
3,
)
# 中心
cv2.circle(
image,
(u, v),
12,
(0, 0, 255),
-1,
)
# 十字线
cv2.line(
image,
(u - 25, v),
(u + 25, v),
(0, 0, 255),
3,
)
cv2.line(
image,
(u, v - 25),
(u, v + 25),
(0, 0, 255),
3,
)
output_path = Path(output_path)
output_path.parent.mkdir(
parents=True,
exist_ok=True,
)
cv2.imwrite(
str(output_path),
image,
)
return {
"image_width": width,
"image_height": height,
"bbox_pixel": [
x1,
y1,
x2,
y2,
],
"center_pixel": [
u,
v,
],
}

View File

@ -0,0 +1,5 @@
Metadata-Version: 2.4
Name: cockpit-ui-grounding
Version: 0.1.0
Summary: Camera-view automotive cockpit UI grounding
Requires-Python: >=3.11

View File

@ -0,0 +1,12 @@
README.md
pyproject.toml
src/cockpit_grounding/__init__.py
src/cockpit_grounding/api/__init__.py
src/cockpit_grounding/grounding/__init__.py
src/cockpit_grounding/models/__init__.py
src/cockpit_grounding/utils/__init__.py
src/cockpit_grounding/vision/__init__.py
src/cockpit_ui_grounding.egg-info/PKG-INFO
src/cockpit_ui_grounding.egg-info/SOURCES.txt
src/cockpit_ui_grounding.egg-info/dependency_links.txt
src/cockpit_ui_grounding.egg-info/top_level.txt

View File

@ -0,0 +1 @@

View File

@ -0,0 +1 @@
cockpit_grounding

View File

@ -0,0 +1,141 @@
import json
import unittest
from pathlib import Path
from tempfile import TemporaryDirectory
from cockpit_grounding.benchmark.config import (
load_benchmark_config,
load_manifest,
)
class BenchmarkConfigTest(unittest.TestCase):
def test_load_benchmark_config_preserves_model_order(self) -> None:
with TemporaryDirectory() as directory:
tmp_path = Path(directory)
model_a = tmp_path / "model-a"
model_b = tmp_path / "model-b"
model_a.mkdir()
model_b.mkdir()
config = tmp_path / "benchmark.toml"
config.write_text(
"\n".join(
(
"[benchmark]",
"warmup = 1",
"repeats = 3",
"max_new_tokens = 128",
f'output_root = "{tmp_path}"',
"[models.first]",
f'path = "{model_a}"',
"[models.second]",
f'path = "{model_b}"',
)
),
encoding="utf-8",
)
loaded = load_benchmark_config(config)
self.assertEqual(
[model.name for model in loaded.models],
["first", "second"],
)
self.assertEqual(
[model.backend for model in loaded.models],
["qwen3vl", "qwen3vl"],
)
self.assertEqual(loaded.settings.repeats, 3)
self.assertTrue(loaded.settings.timestamp_run_directory)
def test_load_benchmark_config_supports_explicit_backend(self) -> None:
with TemporaryDirectory() as directory:
tmp_path = Path(directory)
model_path = tmp_path / "model"
model_path.mkdir()
config = tmp_path / "benchmark.toml"
config.write_text(
"\n".join(
(
"[benchmark]",
"warmup = 0",
"repeats = 1",
"max_new_tokens = 16",
f'output_root = "{tmp_path}"',
"[models.qwen35]",
'backend = "qwen35"',
f'path = "{model_path}"',
)
),
encoding="utf-8",
)
loaded = load_benchmark_config(config)
self.assertEqual(loaded.models[0].backend, "qwen35")
def test_load_benchmark_config_supports_exact_run_directory(self) -> None:
with TemporaryDirectory() as directory:
tmp_path = Path(directory)
model_path = tmp_path / "model"
model_path.mkdir()
config = tmp_path / "benchmark.toml"
config.write_text(
"\n".join(
(
"[benchmark]",
"warmup = 3",
"repeats = 5",
"max_new_tokens = 128",
f'output_root = "{tmp_path}"',
"timestamp_run_directory = false",
"[models.model]",
f'path = "{model_path}"',
)
),
encoding="utf-8",
)
loaded = load_benchmark_config(config)
self.assertFalse(loaded.settings.timestamp_run_directory)
def test_manifest_supports_optional_gt(self) -> None:
with TemporaryDirectory() as directory:
tmp_path = Path(directory)
image = tmp_path / "image.jpg"
image.touch()
manifest = tmp_path / "samples.jsonl"
entries = (
{"id": "without-gt", "image": str(image), "target": "button"},
{
"id": "with-gt",
"image": str(image),
"target": "button",
"gt_bbox_pixel": [1, 2, 3, 4],
},
)
manifest.write_text(
"\n".join(json.dumps(item) for item in entries) + "\n",
encoding="utf-8",
)
samples = load_manifest(manifest)
self.assertIsNone(samples[0].gt_bbox_pixel)
self.assertEqual(samples[1].gt_bbox_pixel, (1.0, 2.0, 3.0, 4.0))
def test_manifest_rejects_duplicate_ids(self) -> None:
with TemporaryDirectory() as directory:
tmp_path = Path(directory)
image = tmp_path / "image.jpg"
image.touch()
manifest = tmp_path / "samples.jsonl"
item = {"id": "same", "image": str(image), "target": "button"}
manifest.write_text(
json.dumps(item) + "\n" + json.dumps(item) + "\n",
encoding="utf-8",
)
with self.assertRaisesRegex(ValueError, "Duplicate"):
load_manifest(manifest)

View File

@ -0,0 +1,90 @@
import csv
import io
import unittest
from contextlib import redirect_stdout
from pathlib import Path
from tempfile import TemporaryDirectory
from cockpit_grounding.benchmark.reporting import (
create_run_directory,
print_model_selection,
write_root_outputs,
)
class BenchmarkReportingTest(unittest.TestCase):
def test_case_results_joins_models_by_case(self) -> None:
predictions = [
{
"model": "qwen3vl_2b",
"id": "button",
"target": "按钮",
"pred_center_pixel": [10, 20],
"total_mean_ms": 100.0,
"parse_success": True,
"pred_bbox_pixel": [1, 2, 19, 38],
"image": "/dataset/image.jpg",
},
{
"model": "qwen3vl_4b",
"id": "button",
"target": "按钮",
"pred_center_pixel": [11, 21],
"total_mean_ms": 120.0,
"parse_success": True,
"pred_bbox_pixel": [2, 3, 20, 39],
"image": "/dataset/image.jpg",
},
]
summaries = {"qwen3vl_2b": {}, "qwen3vl_4b": {}}
with TemporaryDirectory() as directory:
output = Path(directory)
write_root_outputs(output, {}, summaries, predictions)
with (output / "case_results.csv").open(
encoding="utf-8",
newline="",
) as csv_file:
rows = list(csv.DictReader(csv_file))
review = (output / "review.md").read_text(encoding="utf-8")
self.assertEqual(len(rows), 1)
self.assertEqual(rows[0]["id"], "button")
self.assertEqual(rows[0]["qwen3vl_2b_center_x"], "10")
self.assertEqual(rows[0]["qwen3vl_4b_center_y"], "21")
self.assertEqual(rows[0]["qwen3vl_2b_latency_ms"], "100.0")
self.assertEqual(rows[0]["manual_qwen3vl_2b"], "")
self.assertEqual(rows[0]["qwen3vl_4b_bbox_x2"], "20")
self.assertIn("qwen3vl_2b/visualizations/button.jpg", review)
self.assertIn("qwen3vl_4b/visualizations/button.jpg", review)
def test_exact_run_directory_uses_run_name_without_timestamp(self) -> None:
with TemporaryDirectory() as directory:
path = create_run_directory(
Path(directory),
"selection_run",
timestamped=False,
)
self.assertEqual(path.name, "selection_run")
def test_model_selection_marks_accuracy_as_manual_without_gt(self) -> None:
summary = {
"parse_success_rate": 1.0,
"mean_total_ms": 100.0,
"p50_total_ms": 99.0,
"p95_total_ms": 110.0,
"peak_cuda_memory_mb": 1000.0,
"throughput_samples_per_sec": 10.0,
}
output = io.StringIO()
with redirect_stdout(output):
print_model_selection({"model_a": summary, "model_b": summary})
printed = output.getvalue()
self.assertIn("MODEL SELECTION", printed)
self.assertIn("N/A - manual review required", printed)
accuracy_row = next(
line for line in printed.splitlines() if line.startswith("Text UI Accuracy")
)
self.assertEqual(accuracy_row.count("N/A"), 2)

View File

@ -0,0 +1,89 @@
import unittest
from pathlib import Path
from tempfile import TemporaryDirectory
from cockpit_grounding.benchmark.config import BenchmarkSample
from cockpit_grounding.benchmark.runner import _benchmark_sample
from cockpit_grounding.benchmark.statistics import (
percentile,
summarize_model,
summarize_timings,
)
from cockpit_grounding.models.base import InferenceTiming
def _timing(total: float) -> InferenceTiming:
return InferenceTiming(
preprocess_ms=1.0,
generate_ms=total - 2.0,
decode_ms=1.0,
total_ms=total,
)
class BenchmarkStatisticsTest(unittest.TestCase):
def test_percentile_uses_linear_interpolation(self) -> None:
self.assertEqual(percentile([10.0, 20.0, 30.0], 0.5), 20.0)
self.assertAlmostEqual(percentile([10.0, 20.0], 0.95), 19.5)
def test_summarize_timings_handles_empty_input(self) -> None:
self.assertIsNone(summarize_timings([])["total_mean_ms"])
def test_summarize_model_without_gt_uses_null_accuracy(self) -> None:
predictions = [
{
"parse_success": True,
"point_in_box": None,
"bbox_iou": None,
"normalized_center_error": None,
"peak_cuda_memory_mb": 123.0,
}
]
summary = summarize_model(
model_name="model",
model_path="/model",
model_load_seconds=2.0,
memory_metrics={
"model_cuda_allocated_mb": 100.0,
"model_cuda_reserved_mb": 120.0,
},
predictions=predictions,
all_timings=[_timing(10.0), _timing(20.0)],
)
self.assertEqual(summary["parse_success_rate"], 1.0)
self.assertEqual(summary["mean_total_ms"], 15.0)
self.assertAlmostEqual(summary["throughput_samples_per_sec"], 1000 / 15)
self.assertEqual(summary["peak_cuda_memory_mb"], 123.0)
self.assertIsNone(summary["accuracy_point_in_box"])
def test_parse_failure_is_returned_instead_of_raised(self) -> None:
class InvalidOutputGrounder:
def generate_with_metrics(self, **_: object):
return (
"not a bbox",
_timing(10.0),
{"peak_cuda_memory_mb": 100.0},
)
with TemporaryDirectory() as directory:
tmp_path = Path(directory)
image = tmp_path / "image.jpg"
image.touch()
prediction, timings = _benchmark_sample(
grounder=InvalidOutputGrounder(), # type: ignore[arg-type]
model_name="model",
sample=BenchmarkSample(
sample_id="sample",
image=image,
target="button",
),
repeats=3,
max_new_tokens=128,
visualization_path=tmp_path / "visualization.jpg",
)
self.assertFalse(prediction["parse_success"])
self.assertIn("ValueError", prediction["parse_error"])
self.assertEqual(len(timings), 3)
self.assertEqual(prediction["total_mean_ms"], 10.0)

33
tests/test_metrics.py Normal file
View File

@ -0,0 +1,33 @@
import unittest
from cockpit_grounding.benchmark.metrics import (
bbox_iou,
normalized_center_error,
point_in_box,
)
class MetricsTest(unittest.TestCase):
def test_point_in_box_includes_boundary(self) -> None:
self.assertTrue(point_in_box((10, 20), (10, 20, 30, 40)))
self.assertFalse(point_in_box((9, 20), (10, 20, 30, 40)))
def test_bbox_iou(self) -> None:
self.assertAlmostEqual(
bbox_iou((0, 0, 10, 10), (5, 5, 15, 15)),
25 / 175,
)
self.assertEqual(bbox_iou((0, 0, 2, 2), (3, 3, 4, 4)), 0.0)
def test_normalized_center_error_uses_image_diagonal(self) -> None:
value = normalized_center_error(
pred_center=(5, 5),
gt_bbox=(0, 0, 0, 0),
image_width=10,
image_height=10,
)
self.assertAlmostEqual(value, 0.5)
def test_metrics_reject_invalid_bbox(self) -> None:
with self.assertRaisesRegex(ValueError, "satisfy"):
bbox_iou((10, 0, 0, 10), (0, 0, 10, 10))

View File

@ -0,0 +1,9 @@
import unittest
from cockpit_grounding.models.factory import create_grounder
class ModelFactoryTest(unittest.TestCase):
def test_rejects_unknown_backend(self) -> None:
with self.assertRaisesRegex(ValueError, "Unsupported model backend"):
create_grounder("unknown", "/model")