first commit
This commit is contained in:
commit
c9bcfaa6e8
0
.env.example
Normal file
0
.env.example
Normal file
2
.gitignore
vendored
Normal file
2
.gitignore
vendored
Normal file
@ -0,0 +1,2 @@
|
|||||||
|
__pycache__/
|
||||||
|
*.py[cod]
|
||||||
10
.idea/.gitignore
generated
vendored
Normal file
10
.idea/.gitignore
generated
vendored
Normal file
@ -0,0 +1,10 @@
|
|||||||
|
# Default ignored files
|
||||||
|
/shelf/
|
||||||
|
/workspace.xml
|
||||||
|
# Editor-based HTTP Client requests
|
||||||
|
/httpRequests/
|
||||||
|
# Ignored default folder with query files
|
||||||
|
/queries/
|
||||||
|
# Datasource local storage ignored files
|
||||||
|
/dataSources/
|
||||||
|
/dataSources.local.xml
|
||||||
14
.idea/cockpit-ui-grounding.iml
generated
Normal file
14
.idea/cockpit-ui-grounding.iml
generated
Normal file
@ -0,0 +1,14 @@
|
|||||||
|
<?xml version="1.0" encoding="UTF-8"?>
|
||||||
|
<module external.system.id="pyproject.toml" type="PYTHON_MODULE" version="4">
|
||||||
|
<component name="NewModuleRootManager">
|
||||||
|
<content url="file://$MODULE_DIR$">
|
||||||
|
<sourceFolder url="file://$MODULE_DIR$/src" />
|
||||||
|
</content>
|
||||||
|
<orderEntry type="jdk" jdkName="~/miniconda3/envs/qwen3vl" jdkType="Python SDK" />
|
||||||
|
<orderEntry type="sourceFolder" forTests="false" />
|
||||||
|
</component>
|
||||||
|
<component name="PackageRequirementsSettings" />
|
||||||
|
<component name="PyDocumentationSettings" />
|
||||||
|
<component name="ReSTService" />
|
||||||
|
<component name="TestRunnerService" />
|
||||||
|
</module>
|
||||||
6
.idea/inspectionProfiles/profiles_settings.xml
generated
Normal file
6
.idea/inspectionProfiles/profiles_settings.xml
generated
Normal file
@ -0,0 +1,6 @@
|
|||||||
|
<component name="InspectionProjectProfileManager">
|
||||||
|
<settings>
|
||||||
|
<option name="USE_PROJECT_PROFILE" value="false" />
|
||||||
|
<version value="1.0" />
|
||||||
|
</settings>
|
||||||
|
</component>
|
||||||
8
.idea/modules.xml
generated
Normal file
8
.idea/modules.xml
generated
Normal file
@ -0,0 +1,8 @@
|
|||||||
|
<?xml version="1.0" encoding="UTF-8"?>
|
||||||
|
<project version="4">
|
||||||
|
<component name="ProjectModuleManager">
|
||||||
|
<modules>
|
||||||
|
<module fileurl="file://$PROJECT_DIR$/.idea/cockpit-ui-grounding.iml" filepath="$PROJECT_DIR$/.idea/cockpit-ui-grounding.iml" />
|
||||||
|
</modules>
|
||||||
|
</component>
|
||||||
|
</project>
|
||||||
6
.idea/vcs.xml
generated
Normal file
6
.idea/vcs.xml
generated
Normal file
@ -0,0 +1,6 @@
|
|||||||
|
<?xml version="1.0" encoding="UTF-8"?>
|
||||||
|
<project version="4">
|
||||||
|
<component name="VcsDirectoryMappings">
|
||||||
|
<mapping directory="" vcs="Git" />
|
||||||
|
</component>
|
||||||
|
</project>
|
||||||
13
configs/benchmark/qwen35_08b_vs_2b.toml
Normal file
13
configs/benchmark/qwen35_08b_vs_2b.toml
Normal file
@ -0,0 +1,13 @@
|
|||||||
|
[benchmark]
|
||||||
|
warmup = 3
|
||||||
|
repeats = 5
|
||||||
|
max_new_tokens = 128
|
||||||
|
output_root = "/data/lgv/runs/cockpit-ui-grounding"
|
||||||
|
|
||||||
|
[models.qwen35_08b]
|
||||||
|
backend = "qwen35"
|
||||||
|
path = "/data/lgv/models/pretrained/Qwen3.5-0.8B"
|
||||||
|
|
||||||
|
[models.qwen35_2b]
|
||||||
|
backend = "qwen35"
|
||||||
|
path = "/data/lgv/models/pretrained/Qwen3.5-2B"
|
||||||
14
configs/benchmark/qwen3vl2b_vs_qwen35_2b.toml
Normal file
14
configs/benchmark/qwen3vl2b_vs_qwen35_2b.toml
Normal file
@ -0,0 +1,14 @@
|
|||||||
|
[benchmark]
|
||||||
|
warmup = 3
|
||||||
|
repeats = 5
|
||||||
|
max_new_tokens = 128
|
||||||
|
output_root = "/data/lgv/runs/cockpit-ui-grounding"
|
||||||
|
timestamp_run_directory = false
|
||||||
|
|
||||||
|
[models.qwen3vl_2b]
|
||||||
|
backend = "qwen3vl"
|
||||||
|
path = "/data/lgv/models/pretrained/Qwen3-VL-2B-Instruct"
|
||||||
|
|
||||||
|
[models.qwen35_2b]
|
||||||
|
backend = "qwen35"
|
||||||
|
path = "/data/lgv/models/pretrained/Qwen3.5-2B"
|
||||||
11
configs/benchmark/qwen3vl_2b_vs_4b.toml
Normal file
11
configs/benchmark/qwen3vl_2b_vs_4b.toml
Normal file
@ -0,0 +1,11 @@
|
|||||||
|
[benchmark]
|
||||||
|
warmup = 1
|
||||||
|
repeats = 3
|
||||||
|
max_new_tokens = 128
|
||||||
|
output_root = "/data/lgv/runs/cockpit-ui-grounding"
|
||||||
|
|
||||||
|
[models.qwen3vl_2b]
|
||||||
|
path = "/data/lgv/models/pretrained/Qwen3-VL-2B-Instruct"
|
||||||
|
|
||||||
|
[models.qwen3vl_4b]
|
||||||
|
path = "/data/lgv/models/pretrained/Qwen3-VL-4B-Instruct"
|
||||||
12
pyproject.toml
Normal file
12
pyproject.toml
Normal file
@ -0,0 +1,12 @@
|
|||||||
|
[build-system]
|
||||||
|
requires = ["setuptools>=68"]
|
||||||
|
build-backend = "setuptools.build_meta"
|
||||||
|
|
||||||
|
[project]
|
||||||
|
name = "cockpit-ui-grounding"
|
||||||
|
version = "0.1.0"
|
||||||
|
description = "Camera-view automotive cockpit UI grounding"
|
||||||
|
requires-python = ">=3.11"
|
||||||
|
|
||||||
|
[tool.setuptools.packages.find]
|
||||||
|
where = ["src"]
|
||||||
75
scripts/benchmark_models.py
Normal file
75
scripts/benchmark_models.py
Normal file
@ -0,0 +1,75 @@
|
|||||||
|
import argparse
|
||||||
|
import logging
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from cockpit_grounding.benchmark.config import (
|
||||||
|
load_benchmark_config,
|
||||||
|
load_manifest,
|
||||||
|
)
|
||||||
|
from cockpit_grounding.benchmark.reporting import (
|
||||||
|
print_comparison,
|
||||||
|
print_model_selection,
|
||||||
|
)
|
||||||
|
from cockpit_grounding.benchmark.runner import run_benchmark
|
||||||
|
|
||||||
|
|
||||||
|
def parse_args() -> argparse.Namespace:
|
||||||
|
parser = argparse.ArgumentParser(
|
||||||
|
description="Benchmark local multimodal models sequentially on one GPU",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--config",
|
||||||
|
type=Path,
|
||||||
|
required=True,
|
||||||
|
help="Benchmark TOML configuration",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--manifest",
|
||||||
|
type=Path,
|
||||||
|
required=True,
|
||||||
|
help="JSONL benchmark manifest",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--run-name",
|
||||||
|
help="Optional run directory suffix",
|
||||||
|
)
|
||||||
|
return parser.parse_args()
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
args = parse_args()
|
||||||
|
logging.basicConfig(
|
||||||
|
level=logging.INFO,
|
||||||
|
format="%(asctime)s | %(levelname)s | %(message)s",
|
||||||
|
)
|
||||||
|
|
||||||
|
config = load_benchmark_config(args.config)
|
||||||
|
samples = load_manifest(args.manifest)
|
||||||
|
run_directory, summaries = run_benchmark(
|
||||||
|
config=config,
|
||||||
|
samples=samples,
|
||||||
|
config_path=args.config,
|
||||||
|
manifest_path=args.manifest,
|
||||||
|
run_name=args.run_name,
|
||||||
|
)
|
||||||
|
comparison_title = (
|
||||||
|
"QWEN3.5 MODEL COMPARISON"
|
||||||
|
if all(model.backend == "qwen35" for model in config.models)
|
||||||
|
else "MODEL COMPARISON"
|
||||||
|
)
|
||||||
|
models_by_name = {model.name: model for model in config.models}
|
||||||
|
display_summaries = {
|
||||||
|
(
|
||||||
|
Path(models_by_name[name].path).name
|
||||||
|
if models_by_name[name].backend == "qwen35"
|
||||||
|
else name
|
||||||
|
): summary
|
||||||
|
for name, summary in summaries.items()
|
||||||
|
}
|
||||||
|
print_comparison(display_summaries, title=comparison_title)
|
||||||
|
print_model_selection(summaries)
|
||||||
|
print(f"Results: {run_directory}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
128
scripts/run_grounding.py
Normal file
128
scripts/run_grounding.py
Normal file
@ -0,0 +1,128 @@
|
|||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
|
||||||
|
from cockpit_grounding.models.factory import create_grounder
|
||||||
|
|
||||||
|
from cockpit_grounding.grounding.predictor import (
|
||||||
|
build_grounding_prompt,
|
||||||
|
parse_grounding_output,
|
||||||
|
)
|
||||||
|
|
||||||
|
from cockpit_grounding.vision.visualize import (
|
||||||
|
visualize_grounding,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
|
||||||
|
parser.add_argument(
|
||||||
|
"--image",
|
||||||
|
required=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
parser.add_argument(
|
||||||
|
"--target",
|
||||||
|
required=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
parser.add_argument(
|
||||||
|
"--output",
|
||||||
|
required=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
parser.add_argument(
|
||||||
|
"--model",
|
||||||
|
required=True,
|
||||||
|
help="Local model path",
|
||||||
|
)
|
||||||
|
|
||||||
|
parser.add_argument(
|
||||||
|
"--backend",
|
||||||
|
choices=("qwen3vl", "qwen35"),
|
||||||
|
default="qwen3vl",
|
||||||
|
help="Model backend (default: qwen3vl)",
|
||||||
|
)
|
||||||
|
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
# ----------------------------
|
||||||
|
# Load model
|
||||||
|
# ----------------------------
|
||||||
|
|
||||||
|
grounder = create_grounder(
|
||||||
|
backend=args.backend,
|
||||||
|
model_path=args.model,
|
||||||
|
)
|
||||||
|
|
||||||
|
# ----------------------------
|
||||||
|
# Prompt
|
||||||
|
# ----------------------------
|
||||||
|
|
||||||
|
prompt = build_grounding_prompt(
|
||||||
|
args.target
|
||||||
|
)
|
||||||
|
|
||||||
|
# ----------------------------
|
||||||
|
# Inference
|
||||||
|
# ----------------------------
|
||||||
|
|
||||||
|
raw_output = grounder.generate(
|
||||||
|
args.image,
|
||||||
|
prompt,
|
||||||
|
)
|
||||||
|
|
||||||
|
print()
|
||||||
|
print("========== RAW MODEL OUTPUT ==========")
|
||||||
|
print(raw_output)
|
||||||
|
|
||||||
|
# ----------------------------
|
||||||
|
# Parse bbox
|
||||||
|
# ----------------------------
|
||||||
|
|
||||||
|
result = parse_grounding_output(
|
||||||
|
raw_output
|
||||||
|
)
|
||||||
|
|
||||||
|
# ----------------------------
|
||||||
|
# Convert + draw
|
||||||
|
# ----------------------------
|
||||||
|
|
||||||
|
pixel_result = visualize_grounding(
|
||||||
|
args.image,
|
||||||
|
result,
|
||||||
|
args.output,
|
||||||
|
)
|
||||||
|
|
||||||
|
final_result = {
|
||||||
|
"target": args.target,
|
||||||
|
|
||||||
|
"bbox_relative": [
|
||||||
|
result.x1,
|
||||||
|
result.y1,
|
||||||
|
result.x2,
|
||||||
|
result.y2,
|
||||||
|
],
|
||||||
|
|
||||||
|
**pixel_result,
|
||||||
|
}
|
||||||
|
|
||||||
|
print()
|
||||||
|
print("========== FINAL RESULT ==========")
|
||||||
|
|
||||||
|
print(
|
||||||
|
json.dumps(
|
||||||
|
final_result,
|
||||||
|
indent=2,
|
||||||
|
ensure_ascii=False,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
print()
|
||||||
|
print(
|
||||||
|
f"Result image: {args.output}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
0
src/cockpit_grounding/__init__.py
Normal file
0
src/cockpit_grounding/__init__.py
Normal file
0
src/cockpit_grounding/api/__init__.py
Normal file
0
src/cockpit_grounding/api/__init__.py
Normal file
11
src/cockpit_grounding/benchmark/__init__.py
Normal file
11
src/cockpit_grounding/benchmark/__init__.py
Normal file
@ -0,0 +1,11 @@
|
|||||||
|
from cockpit_grounding.benchmark.metrics import (
|
||||||
|
bbox_iou,
|
||||||
|
normalized_center_error,
|
||||||
|
point_in_box,
|
||||||
|
)
|
||||||
|
|
||||||
|
__all__ = [
|
||||||
|
"bbox_iou",
|
||||||
|
"normalized_center_error",
|
||||||
|
"point_in_box",
|
||||||
|
]
|
||||||
190
src/cockpit_grounding/benchmark/config.py
Normal file
190
src/cockpit_grounding/benchmark/config.py
Normal file
@ -0,0 +1,190 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import tomllib
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class BenchmarkSettings:
|
||||||
|
warmup: int
|
||||||
|
repeats: int
|
||||||
|
max_new_tokens: int
|
||||||
|
output_root: Path
|
||||||
|
timestamp_run_directory: bool = True
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class ModelConfig:
|
||||||
|
name: str
|
||||||
|
backend: str
|
||||||
|
path: Path
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class BenchmarkConfig:
|
||||||
|
settings: BenchmarkSettings
|
||||||
|
models: tuple[ModelConfig, ...]
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class BenchmarkSample:
|
||||||
|
sample_id: str
|
||||||
|
image: Path
|
||||||
|
target: str
|
||||||
|
gt_bbox_pixel: tuple[float, float, float, float] | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def load_benchmark_config(path: str | Path) -> BenchmarkConfig:
|
||||||
|
config_path = Path(path).expanduser().resolve()
|
||||||
|
with config_path.open("rb") as config_file:
|
||||||
|
data = tomllib.load(config_file)
|
||||||
|
|
||||||
|
benchmark = _require_mapping(data, "benchmark")
|
||||||
|
models_data = _require_mapping(data, "models")
|
||||||
|
|
||||||
|
settings = BenchmarkSettings(
|
||||||
|
warmup=_non_negative_int(benchmark.get("warmup"), "benchmark.warmup"),
|
||||||
|
repeats=_positive_int(benchmark.get("repeats"), "benchmark.repeats"),
|
||||||
|
max_new_tokens=_positive_int(
|
||||||
|
benchmark.get("max_new_tokens"),
|
||||||
|
"benchmark.max_new_tokens",
|
||||||
|
),
|
||||||
|
output_root=Path(
|
||||||
|
_non_empty_string(
|
||||||
|
benchmark.get("output_root"),
|
||||||
|
"benchmark.output_root",
|
||||||
|
)
|
||||||
|
).expanduser(),
|
||||||
|
timestamp_run_directory=_optional_bool(
|
||||||
|
benchmark.get("timestamp_run_directory"),
|
||||||
|
"benchmark.timestamp_run_directory",
|
||||||
|
default=True,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
models: list[ModelConfig] = []
|
||||||
|
for name, model_data in models_data.items():
|
||||||
|
if not isinstance(model_data, dict):
|
||||||
|
raise ValueError(f"models.{name} must be a TOML table")
|
||||||
|
model_path = Path(
|
||||||
|
_non_empty_string(model_data.get("path"), f"models.{name}.path")
|
||||||
|
).expanduser().resolve()
|
||||||
|
if not model_path.is_dir():
|
||||||
|
raise FileNotFoundError(f"Local model directory not found: {model_path}")
|
||||||
|
backend = _non_empty_string(
|
||||||
|
model_data.get("backend", "qwen3vl"),
|
||||||
|
f"models.{name}.backend",
|
||||||
|
)
|
||||||
|
models.append(ModelConfig(name=name, backend=backend, path=model_path))
|
||||||
|
|
||||||
|
if not models:
|
||||||
|
raise ValueError("At least one model must be configured")
|
||||||
|
|
||||||
|
return BenchmarkConfig(settings=settings, models=tuple(models))
|
||||||
|
|
||||||
|
|
||||||
|
def load_manifest(path: str | Path) -> tuple[BenchmarkSample, ...]:
|
||||||
|
manifest_path = Path(path).expanduser().resolve()
|
||||||
|
samples: list[BenchmarkSample] = []
|
||||||
|
seen_ids: set[str] = set()
|
||||||
|
|
||||||
|
with manifest_path.open("r", encoding="utf-8") as manifest_file:
|
||||||
|
for line_number, line in enumerate(manifest_file, start=1):
|
||||||
|
if not line.strip():
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
item = json.loads(line)
|
||||||
|
except json.JSONDecodeError as exc:
|
||||||
|
raise ValueError(
|
||||||
|
f"Invalid JSON at {manifest_path}:{line_number}: {exc.msg}"
|
||||||
|
) from exc
|
||||||
|
if not isinstance(item, dict):
|
||||||
|
raise ValueError(
|
||||||
|
f"Manifest entry at line {line_number} must be an object"
|
||||||
|
)
|
||||||
|
|
||||||
|
sample = _parse_sample(item, manifest_path, line_number)
|
||||||
|
if sample.sample_id in seen_ids:
|
||||||
|
raise ValueError(f"Duplicate sample id: {sample.sample_id}")
|
||||||
|
seen_ids.add(sample.sample_id)
|
||||||
|
samples.append(sample)
|
||||||
|
|
||||||
|
if not samples:
|
||||||
|
raise ValueError(f"Manifest contains no samples: {manifest_path}")
|
||||||
|
return tuple(samples)
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_sample(
|
||||||
|
item: dict[str, Any],
|
||||||
|
manifest_path: Path,
|
||||||
|
line_number: int,
|
||||||
|
) -> BenchmarkSample:
|
||||||
|
prefix = f"{manifest_path}:{line_number}"
|
||||||
|
sample_id = _non_empty_string(item.get("id"), f"{prefix} id")
|
||||||
|
if Path(sample_id).name != sample_id or sample_id in {".", ".."}:
|
||||||
|
raise ValueError(f"{prefix} id must be a filename-safe identifier")
|
||||||
|
|
||||||
|
image = Path(
|
||||||
|
_non_empty_string(item.get("image"), f"{prefix} image")
|
||||||
|
).expanduser().resolve()
|
||||||
|
if not image.is_file():
|
||||||
|
raise FileNotFoundError(f"Image not found at {prefix}: {image}")
|
||||||
|
|
||||||
|
target = _non_empty_string(item.get("target"), f"{prefix} target")
|
||||||
|
gt_bbox = item.get("gt_bbox_pixel")
|
||||||
|
parsed_gt: tuple[float, float, float, float] | None = None
|
||||||
|
if gt_bbox is not None:
|
||||||
|
if not isinstance(gt_bbox, list) or len(gt_bbox) != 4:
|
||||||
|
raise ValueError(f"{prefix} gt_bbox_pixel must contain four numbers")
|
||||||
|
try:
|
||||||
|
parsed_gt = tuple(float(value) for value in gt_bbox) # type: ignore[assignment]
|
||||||
|
except (TypeError, ValueError) as exc:
|
||||||
|
raise ValueError(
|
||||||
|
f"{prefix} gt_bbox_pixel must contain four numbers"
|
||||||
|
) from exc
|
||||||
|
x1, y1, x2, y2 = parsed_gt
|
||||||
|
if x1 > x2 or y1 > y2:
|
||||||
|
raise ValueError(f"{prefix} gt_bbox_pixel has invalid corner order")
|
||||||
|
|
||||||
|
return BenchmarkSample(
|
||||||
|
sample_id=sample_id,
|
||||||
|
image=image,
|
||||||
|
target=target,
|
||||||
|
gt_bbox_pixel=parsed_gt,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _require_mapping(data: dict[str, Any], key: str) -> dict[str, Any]:
|
||||||
|
value = data.get(key)
|
||||||
|
if not isinstance(value, dict):
|
||||||
|
raise ValueError(f"Missing or invalid [{key}] table")
|
||||||
|
return value
|
||||||
|
|
||||||
|
|
||||||
|
def _non_empty_string(value: Any, name: str) -> str:
|
||||||
|
if not isinstance(value, str) or not value.strip():
|
||||||
|
raise ValueError(f"{name} must be a non-empty string")
|
||||||
|
return value
|
||||||
|
|
||||||
|
|
||||||
|
def _positive_int(value: Any, name: str) -> int:
|
||||||
|
if not isinstance(value, int) or isinstance(value, bool) or value <= 0:
|
||||||
|
raise ValueError(f"{name} must be a positive integer")
|
||||||
|
return value
|
||||||
|
|
||||||
|
|
||||||
|
def _non_negative_int(value: Any, name: str) -> int:
|
||||||
|
if not isinstance(value, int) or isinstance(value, bool) or value < 0:
|
||||||
|
raise ValueError(f"{name} must be a non-negative integer")
|
||||||
|
return value
|
||||||
|
|
||||||
|
|
||||||
|
def _optional_bool(value: Any, name: str, *, default: bool) -> bool:
|
||||||
|
if value is None:
|
||||||
|
return default
|
||||||
|
if not isinstance(value, bool):
|
||||||
|
raise ValueError(f"{name} must be a boolean")
|
||||||
|
return value
|
||||||
60
src/cockpit_grounding/benchmark/metrics.py
Normal file
60
src/cockpit_grounding/benchmark/metrics.py
Normal file
@ -0,0 +1,60 @@
|
|||||||
|
from math import hypot
|
||||||
|
from typing import Sequence
|
||||||
|
|
||||||
|
Point = Sequence[float]
|
||||||
|
BBox = Sequence[float]
|
||||||
|
|
||||||
|
|
||||||
|
def _validate_point(point: Point, name: str) -> None:
|
||||||
|
if len(point) != 2:
|
||||||
|
raise ValueError(f"{name} must contain two coordinates")
|
||||||
|
|
||||||
|
|
||||||
|
def _validate_bbox(bbox: BBox, name: str) -> None:
|
||||||
|
if len(bbox) != 4:
|
||||||
|
raise ValueError(f"{name} must contain four coordinates")
|
||||||
|
x1, y1, x2, y2 = bbox
|
||||||
|
if x1 > x2 or y1 > y2:
|
||||||
|
raise ValueError(f"{name} must satisfy x1 <= x2 and y1 <= y2")
|
||||||
|
|
||||||
|
|
||||||
|
def point_in_box(pred_center: Point, gt_bbox: BBox) -> bool:
|
||||||
|
_validate_point(pred_center, "pred_center")
|
||||||
|
_validate_bbox(gt_bbox, "gt_bbox")
|
||||||
|
x, y = pred_center
|
||||||
|
x1, y1, x2, y2 = gt_bbox
|
||||||
|
return x1 <= x <= x2 and y1 <= y <= y2
|
||||||
|
|
||||||
|
|
||||||
|
def bbox_iou(pred_bbox: BBox, gt_bbox: BBox) -> float:
|
||||||
|
_validate_bbox(pred_bbox, "pred_bbox")
|
||||||
|
_validate_bbox(gt_bbox, "gt_bbox")
|
||||||
|
pred_x1, pred_y1, pred_x2, pred_y2 = pred_bbox
|
||||||
|
gt_x1, gt_y1, gt_x2, gt_y2 = gt_bbox
|
||||||
|
|
||||||
|
intersection_width = max(0.0, min(pred_x2, gt_x2) - max(pred_x1, gt_x1))
|
||||||
|
intersection_height = max(0.0, min(pred_y2, gt_y2) - max(pred_y1, gt_y1))
|
||||||
|
intersection = intersection_width * intersection_height
|
||||||
|
pred_area = (pred_x2 - pred_x1) * (pred_y2 - pred_y1)
|
||||||
|
gt_area = (gt_x2 - gt_x1) * (gt_y2 - gt_y1)
|
||||||
|
union = pred_area + gt_area - intersection
|
||||||
|
return intersection / union if union > 0 else 0.0
|
||||||
|
|
||||||
|
|
||||||
|
def normalized_center_error(
|
||||||
|
pred_center: Point,
|
||||||
|
gt_bbox: BBox,
|
||||||
|
image_width: int,
|
||||||
|
image_height: int,
|
||||||
|
) -> float:
|
||||||
|
"""Return center distance normalized by the image diagonal."""
|
||||||
|
_validate_point(pred_center, "pred_center")
|
||||||
|
_validate_bbox(gt_bbox, "gt_bbox")
|
||||||
|
if image_width <= 0 or image_height <= 0:
|
||||||
|
raise ValueError("image dimensions must be positive")
|
||||||
|
|
||||||
|
gt_x1, gt_y1, gt_x2, gt_y2 = gt_bbox
|
||||||
|
gt_center_x = (gt_x1 + gt_x2) / 2.0
|
||||||
|
gt_center_y = (gt_y1 + gt_y2) / 2.0
|
||||||
|
distance = hypot(pred_center[0] - gt_center_x, pred_center[1] - gt_center_y)
|
||||||
|
return distance / hypot(image_width, image_height)
|
||||||
317
src/cockpit_grounding/benchmark/reporting.py
Normal file
317
src/cockpit_grounding/benchmark/reporting.py
Normal file
@ -0,0 +1,317 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import csv
|
||||||
|
import json
|
||||||
|
from datetime import datetime
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
|
||||||
|
RESULT_COLUMNS = (
|
||||||
|
"model",
|
||||||
|
"id",
|
||||||
|
"target",
|
||||||
|
"parse_success",
|
||||||
|
"point_in_box",
|
||||||
|
"bbox_iou",
|
||||||
|
"normalized_center_error",
|
||||||
|
"preprocess_mean_ms",
|
||||||
|
"generate_mean_ms",
|
||||||
|
"decode_mean_ms",
|
||||||
|
"total_mean_ms",
|
||||||
|
"total_p50_ms",
|
||||||
|
"total_p95_ms",
|
||||||
|
"peak_cuda_memory_mb",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def create_run_directory(
|
||||||
|
output_root: Path,
|
||||||
|
run_name: str,
|
||||||
|
*,
|
||||||
|
timestamped: bool = True,
|
||||||
|
) -> Path:
|
||||||
|
safe_name = _safe_run_name(run_name)
|
||||||
|
directory_name = (
|
||||||
|
f"{datetime.now().strftime('%Y%m%d_%H%M%S')}_{safe_name}"
|
||||||
|
if timestamped
|
||||||
|
else safe_name
|
||||||
|
)
|
||||||
|
run_directory = output_root / directory_name
|
||||||
|
run_directory.mkdir(parents=True, exist_ok=False)
|
||||||
|
return run_directory
|
||||||
|
|
||||||
|
|
||||||
|
def write_model_outputs(
|
||||||
|
model_directory: Path,
|
||||||
|
summary: dict[str, Any],
|
||||||
|
predictions: list[dict[str, Any]],
|
||||||
|
) -> None:
|
||||||
|
model_directory.mkdir(parents=True, exist_ok=True)
|
||||||
|
write_json(model_directory / "summary.json", summary)
|
||||||
|
with (model_directory / "predictions.jsonl").open(
|
||||||
|
"w",
|
||||||
|
encoding="utf-8",
|
||||||
|
) as predictions_file:
|
||||||
|
for prediction in predictions:
|
||||||
|
predictions_file.write(
|
||||||
|
json.dumps(prediction, ensure_ascii=False, allow_nan=False) + "\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def write_root_outputs(
|
||||||
|
run_directory: Path,
|
||||||
|
metadata: dict[str, Any],
|
||||||
|
summaries: dict[str, dict[str, Any]],
|
||||||
|
predictions: list[dict[str, Any]],
|
||||||
|
) -> None:
|
||||||
|
write_json(run_directory / "benchmark_meta.json", metadata)
|
||||||
|
write_json(
|
||||||
|
run_directory / "summary.json",
|
||||||
|
{"models": summaries},
|
||||||
|
)
|
||||||
|
|
||||||
|
with (run_directory / "results.csv").open(
|
||||||
|
"w",
|
||||||
|
encoding="utf-8",
|
||||||
|
newline="",
|
||||||
|
) as csv_file:
|
||||||
|
writer = csv.DictWriter(csv_file, fieldnames=RESULT_COLUMNS)
|
||||||
|
writer.writeheader()
|
||||||
|
for prediction in predictions:
|
||||||
|
writer.writerow({column: prediction.get(column) for column in RESULT_COLUMNS})
|
||||||
|
|
||||||
|
_write_case_results(
|
||||||
|
run_directory / "case_results.csv",
|
||||||
|
tuple(summaries),
|
||||||
|
predictions,
|
||||||
|
)
|
||||||
|
_write_review(
|
||||||
|
run_directory / "review.md",
|
||||||
|
tuple(summaries),
|
||||||
|
predictions,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _write_case_results(
|
||||||
|
path: Path,
|
||||||
|
model_names: tuple[str, ...],
|
||||||
|
predictions: list[dict[str, Any]],
|
||||||
|
) -> None:
|
||||||
|
columns = ["id", "target"]
|
||||||
|
for model_name in model_names:
|
||||||
|
columns.extend(
|
||||||
|
(
|
||||||
|
f"{model_name}_center_x",
|
||||||
|
f"{model_name}_center_y",
|
||||||
|
f"{model_name}_latency_ms",
|
||||||
|
f"{model_name}_parse_success",
|
||||||
|
f"{model_name}_bbox_x1",
|
||||||
|
f"{model_name}_bbox_y1",
|
||||||
|
f"{model_name}_bbox_x2",
|
||||||
|
f"{model_name}_bbox_y2",
|
||||||
|
f"manual_{model_name}",
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
cases: dict[str, dict[str, Any]] = {}
|
||||||
|
for prediction in predictions:
|
||||||
|
sample_id = prediction["id"]
|
||||||
|
target = prediction["target"]
|
||||||
|
case = cases.setdefault(sample_id, {"id": sample_id, "target": target})
|
||||||
|
if case["target"] != target:
|
||||||
|
raise ValueError(f"Mismatched targets for benchmark case: {sample_id}")
|
||||||
|
|
||||||
|
model_name = prediction["model"]
|
||||||
|
if model_name not in model_names:
|
||||||
|
raise ValueError(f"Unexpected model in predictions: {model_name}")
|
||||||
|
center = prediction.get("pred_center_pixel") or (None, None)
|
||||||
|
bbox = prediction.get("pred_bbox_pixel") or (None, None, None, None)
|
||||||
|
case[f"{model_name}_center_x"] = center[0]
|
||||||
|
case[f"{model_name}_center_y"] = center[1]
|
||||||
|
case[f"{model_name}_latency_ms"] = prediction.get("total_mean_ms")
|
||||||
|
case[f"{model_name}_parse_success"] = prediction.get("parse_success")
|
||||||
|
case[f"{model_name}_bbox_x1"] = bbox[0]
|
||||||
|
case[f"{model_name}_bbox_y1"] = bbox[1]
|
||||||
|
case[f"{model_name}_bbox_x2"] = bbox[2]
|
||||||
|
case[f"{model_name}_bbox_y2"] = bbox[3]
|
||||||
|
case[f"manual_{model_name}"] = ""
|
||||||
|
|
||||||
|
with path.open("w", encoding="utf-8", newline="") as csv_file:
|
||||||
|
writer = csv.DictWriter(csv_file, fieldnames=columns)
|
||||||
|
writer.writeheader()
|
||||||
|
for case in cases.values():
|
||||||
|
writer.writerow({column: case.get(column) for column in columns})
|
||||||
|
|
||||||
|
|
||||||
|
def _write_review(
|
||||||
|
path: Path,
|
||||||
|
model_names: tuple[str, ...],
|
||||||
|
predictions: list[dict[str, Any]],
|
||||||
|
) -> None:
|
||||||
|
cases: dict[str, dict[str, Any]] = {}
|
||||||
|
for prediction in predictions:
|
||||||
|
sample_id = prediction["id"]
|
||||||
|
case = cases.setdefault(
|
||||||
|
sample_id,
|
||||||
|
{
|
||||||
|
"target": prediction["target"],
|
||||||
|
"image": prediction["image"],
|
||||||
|
"models": {},
|
||||||
|
},
|
||||||
|
)
|
||||||
|
case["models"][prediction["model"]] = prediction
|
||||||
|
|
||||||
|
lines = [
|
||||||
|
"# Manual Grounding Review",
|
||||||
|
"",
|
||||||
|
"Use `correct`, `wrong`, or `partial` in the corresponding ",
|
||||||
|
"`manual_<model>` columns of `case_results.csv` after visual review.",
|
||||||
|
"",
|
||||||
|
]
|
||||||
|
for sample_id, case in cases.items():
|
||||||
|
lines.extend(
|
||||||
|
(
|
||||||
|
f"## {sample_id}",
|
||||||
|
"",
|
||||||
|
f"Target: {case['target']}",
|
||||||
|
"",
|
||||||
|
f"Source: `{case['image']}`",
|
||||||
|
"",
|
||||||
|
)
|
||||||
|
)
|
||||||
|
for model_name in model_names:
|
||||||
|
prediction = case["models"].get(model_name)
|
||||||
|
if prediction is None:
|
||||||
|
lines.extend((f"### {model_name}", "", "Missing prediction.", ""))
|
||||||
|
continue
|
||||||
|
visualization = Path(model_name) / "visualizations" / f"{sample_id}.jpg"
|
||||||
|
lines.extend(
|
||||||
|
(
|
||||||
|
f"### {model_name}",
|
||||||
|
"",
|
||||||
|
f"Parse success: `{prediction.get('parse_success')}`",
|
||||||
|
"",
|
||||||
|
f"Predicted bbox: `{prediction.get('pred_bbox_pixel')}`",
|
||||||
|
"",
|
||||||
|
f"})",
|
||||||
|
"",
|
||||||
|
)
|
||||||
|
)
|
||||||
|
path.write_text("\n".join(lines), encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def write_json(path: Path, value: dict[str, Any]) -> None:
|
||||||
|
with path.open("w", encoding="utf-8") as output_file:
|
||||||
|
json.dump(
|
||||||
|
value,
|
||||||
|
output_file,
|
||||||
|
indent=2,
|
||||||
|
ensure_ascii=False,
|
||||||
|
allow_nan=False,
|
||||||
|
)
|
||||||
|
output_file.write("\n")
|
||||||
|
|
||||||
|
|
||||||
|
def print_comparison(
|
||||||
|
summaries: dict[str, dict[str, Any]],
|
||||||
|
title: str = "MODEL COMPARISON",
|
||||||
|
) -> None:
|
||||||
|
print()
|
||||||
|
print("=" * 60)
|
||||||
|
print(title)
|
||||||
|
print("=" * 60)
|
||||||
|
for model_name, summary in summaries.items():
|
||||||
|
print()
|
||||||
|
print(model_name)
|
||||||
|
print()
|
||||||
|
print(f" model load : {_format(summary['model_load_seconds'], '.2f', 's')}")
|
||||||
|
print(
|
||||||
|
" CUDA memory : "
|
||||||
|
f"{_format(summary['model_cuda_allocated_mb'], '.0f', 'MB')} allocated, "
|
||||||
|
f"{_format(summary['model_cuda_reserved_mb'], '.0f', 'MB')} reserved"
|
||||||
|
)
|
||||||
|
print(
|
||||||
|
" peak memory : "
|
||||||
|
f"{_format(summary['peak_cuda_memory_mb'], '.0f', 'MB')}"
|
||||||
|
)
|
||||||
|
print(f" mean latency : {_format(summary['mean_total_ms'], '.1f', 'ms')}")
|
||||||
|
print(f" p50 latency : {_format(summary['p50_total_ms'], '.1f', 'ms')}")
|
||||||
|
print(f" p95 latency : {_format(summary['p95_total_ms'], '.1f', 'ms')}")
|
||||||
|
print(
|
||||||
|
" generation mean : "
|
||||||
|
f"{_format(summary['mean_generate_ms'], '.1f', 'ms')}"
|
||||||
|
)
|
||||||
|
print(
|
||||||
|
" throughput : "
|
||||||
|
f"{_format(summary['throughput_samples_per_sec'], '.2f', 'samples/s')}"
|
||||||
|
)
|
||||||
|
print(f" parse success : {summary['parse_success_rate'] * 100:.1f} %")
|
||||||
|
if summary["accuracy_point_in_box"] is not None:
|
||||||
|
print(
|
||||||
|
" grounding acc : "
|
||||||
|
f"{summary['accuracy_point_in_box'] * 100:.1f} %"
|
||||||
|
)
|
||||||
|
print()
|
||||||
|
print("=" * 60)
|
||||||
|
|
||||||
|
|
||||||
|
def print_model_selection(summaries: dict[str, dict[str, Any]]) -> None:
|
||||||
|
model_names = tuple(summaries)
|
||||||
|
print()
|
||||||
|
print("=" * 60)
|
||||||
|
print("MODEL SELECTION")
|
||||||
|
print("=" * 60)
|
||||||
|
print()
|
||||||
|
header = f"{'Metric':<32}" + "".join(f"{name:>20}" for name in model_names)
|
||||||
|
print(header)
|
||||||
|
rows = (
|
||||||
|
("Parse Success", "parse_success_rate", ".1%"),
|
||||||
|
("Mean Latency", "mean_total_ms", ".1f"),
|
||||||
|
("P50", "p50_total_ms", ".1f"),
|
||||||
|
("P95", "p95_total_ms", ".1f"),
|
||||||
|
("Peak GPU Memory", "peak_cuda_memory_mb", ".0f"),
|
||||||
|
("Throughput", "throughput_samples_per_sec", ".3f"),
|
||||||
|
)
|
||||||
|
for label, key, number_format in rows:
|
||||||
|
values = "".join(
|
||||||
|
f"{_format_table_value(summaries[name].get(key), number_format):>20}"
|
||||||
|
for name in model_names
|
||||||
|
)
|
||||||
|
print(f"{label:<32}{values}")
|
||||||
|
print()
|
||||||
|
manual = "N/A - manual review required"
|
||||||
|
for label in (
|
||||||
|
"Text UI Accuracy",
|
||||||
|
"Generic Icon Accuracy",
|
||||||
|
"Automotive Icon Accuracy",
|
||||||
|
"Function Region Accuracy",
|
||||||
|
"Small Sub-Control Accuracy",
|
||||||
|
"Slider Geometry Accuracy",
|
||||||
|
):
|
||||||
|
values = "".join(f"{'N/A':>20}" for _ in model_names)
|
||||||
|
print(f"{label:<32}{values}")
|
||||||
|
print()
|
||||||
|
print(manual)
|
||||||
|
print()
|
||||||
|
print("=" * 60)
|
||||||
|
|
||||||
|
|
||||||
|
def _format_table_value(value: Any, number_format: str) -> str:
|
||||||
|
if not isinstance(value, (int, float)):
|
||||||
|
return "n/a"
|
||||||
|
return format(value, number_format)
|
||||||
|
|
||||||
|
|
||||||
|
def _format(value: float | None, number_format: str, unit: str) -> str:
|
||||||
|
if value is None:
|
||||||
|
return "n/a"
|
||||||
|
return f"{value:{number_format}} {unit}"
|
||||||
|
|
||||||
|
|
||||||
|
def _safe_run_name(run_name: str) -> str:
|
||||||
|
if not run_name or any(character in run_name for character in "/\\"):
|
||||||
|
raise ValueError("run name must be non-empty and cannot contain path separators")
|
||||||
|
if run_name in {".", ".."}:
|
||||||
|
raise ValueError("invalid run name")
|
||||||
|
return run_name
|
||||||
319
src/cockpit_grounding/benchmark/runner.py
Normal file
319
src/cockpit_grounding/benchmark/runner.py
Normal file
@ -0,0 +1,319 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import gc
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
import platform
|
||||||
|
import hashlib
|
||||||
|
from datetime import datetime
|
||||||
|
from pathlib import Path
|
||||||
|
from time import perf_counter
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import torch
|
||||||
|
import transformers
|
||||||
|
|
||||||
|
from cockpit_grounding.benchmark.config import (
|
||||||
|
BenchmarkConfig,
|
||||||
|
BenchmarkSample,
|
||||||
|
)
|
||||||
|
from cockpit_grounding.benchmark.metrics import (
|
||||||
|
bbox_iou,
|
||||||
|
normalized_center_error,
|
||||||
|
point_in_box,
|
||||||
|
)
|
||||||
|
from cockpit_grounding.benchmark.reporting import (
|
||||||
|
create_run_directory,
|
||||||
|
write_json,
|
||||||
|
write_model_outputs,
|
||||||
|
write_root_outputs,
|
||||||
|
)
|
||||||
|
from cockpit_grounding.benchmark.statistics import (
|
||||||
|
summarize_model,
|
||||||
|
summarize_timings,
|
||||||
|
)
|
||||||
|
from cockpit_grounding.grounding.predictor import (
|
||||||
|
build_grounding_prompt,
|
||||||
|
parse_grounding_output,
|
||||||
|
)
|
||||||
|
from cockpit_grounding.models.base import Grounder, InferenceTiming
|
||||||
|
from cockpit_grounding.models.factory import create_grounder
|
||||||
|
from cockpit_grounding.vision.visualize import visualize_grounding
|
||||||
|
|
||||||
|
LOGGER = logging.getLogger(__name__)
|
||||||
|
DEFAULT_RUN_NAME = "qwen3vl_2b_vs_4b"
|
||||||
|
|
||||||
|
|
||||||
|
def run_benchmark(
|
||||||
|
*,
|
||||||
|
config: BenchmarkConfig,
|
||||||
|
samples: tuple[BenchmarkSample, ...],
|
||||||
|
config_path: Path,
|
||||||
|
manifest_path: Path,
|
||||||
|
run_name: str | None = None,
|
||||||
|
) -> tuple[Path, dict[str, dict[str, Any]]]:
|
||||||
|
_validate_cuda_environment()
|
||||||
|
settings = config.settings
|
||||||
|
run_directory = create_run_directory(
|
||||||
|
settings.output_root,
|
||||||
|
run_name or DEFAULT_RUN_NAME,
|
||||||
|
timestamped=settings.timestamp_run_directory,
|
||||||
|
)
|
||||||
|
started_at = datetime.now().astimezone()
|
||||||
|
metadata: dict[str, Any] = {
|
||||||
|
"started_at": started_at.isoformat(),
|
||||||
|
"completed_at": None,
|
||||||
|
"config": str(config_path.resolve()),
|
||||||
|
"manifest": str(manifest_path.resolve()),
|
||||||
|
"run_directory": str(run_directory.resolve()),
|
||||||
|
"cuda_visible_devices": os.environ.get("CUDA_VISIBLE_DEVICES"),
|
||||||
|
"cuda_device": torch.cuda.get_device_name(0),
|
||||||
|
"python_version": platform.python_version(),
|
||||||
|
"torch_version": torch.__version__,
|
||||||
|
"transformers_version": transformers.__version__,
|
||||||
|
"grounding_prompt_sha256": hashlib.sha256(
|
||||||
|
build_grounding_prompt("__SEMANTIC_TARGET__").encode("utf-8")
|
||||||
|
).hexdigest(),
|
||||||
|
"preprocessing_policy": (
|
||||||
|
"source image at native resolution; no benchmark-side resize"
|
||||||
|
),
|
||||||
|
"benchmark": {
|
||||||
|
"warmup": settings.warmup,
|
||||||
|
"repeats": settings.repeats,
|
||||||
|
"max_new_tokens": settings.max_new_tokens,
|
||||||
|
},
|
||||||
|
"models": {model.name: str(model.path) for model in config.models},
|
||||||
|
"model_backends": {model.name: model.backend for model in config.models},
|
||||||
|
"sample_count": len(samples),
|
||||||
|
}
|
||||||
|
write_json(run_directory / "benchmark_meta.json", metadata)
|
||||||
|
|
||||||
|
summaries: dict[str, dict[str, Any]] = {}
|
||||||
|
all_predictions: list[dict[str, Any]] = []
|
||||||
|
|
||||||
|
for model in config.models:
|
||||||
|
summary, predictions = _benchmark_model(
|
||||||
|
model_name=model.name,
|
||||||
|
backend=model.backend,
|
||||||
|
model_path=model.path,
|
||||||
|
samples=samples,
|
||||||
|
warmup=settings.warmup,
|
||||||
|
repeats=settings.repeats,
|
||||||
|
max_new_tokens=settings.max_new_tokens,
|
||||||
|
run_directory=run_directory,
|
||||||
|
)
|
||||||
|
summaries[model.name] = summary
|
||||||
|
all_predictions.extend(predictions)
|
||||||
|
|
||||||
|
completed_at = datetime.now().astimezone()
|
||||||
|
metadata["completed_at"] = completed_at.isoformat()
|
||||||
|
metadata["duration_seconds"] = (completed_at - started_at).total_seconds()
|
||||||
|
write_root_outputs(
|
||||||
|
run_directory,
|
||||||
|
metadata,
|
||||||
|
summaries,
|
||||||
|
all_predictions,
|
||||||
|
)
|
||||||
|
return run_directory, summaries
|
||||||
|
|
||||||
|
|
||||||
|
def _benchmark_model(
|
||||||
|
*,
|
||||||
|
model_name: str,
|
||||||
|
backend: str,
|
||||||
|
model_path: Path,
|
||||||
|
samples: tuple[BenchmarkSample, ...],
|
||||||
|
warmup: int,
|
||||||
|
repeats: int,
|
||||||
|
max_new_tokens: int,
|
||||||
|
run_directory: Path,
|
||||||
|
) -> tuple[dict[str, Any], list[dict[str, Any]]]:
|
||||||
|
LOGGER.info("Loading %s from %s", model_name, model_path)
|
||||||
|
grounder: Grounder | None = None
|
||||||
|
try:
|
||||||
|
torch.cuda.synchronize()
|
||||||
|
load_start = perf_counter()
|
||||||
|
grounder = create_grounder(backend=backend, model_path=model_path)
|
||||||
|
torch.cuda.synchronize()
|
||||||
|
model_load_seconds = perf_counter() - load_start
|
||||||
|
memory_metrics = grounder.cuda_memory_metrics()
|
||||||
|
|
||||||
|
if warmup:
|
||||||
|
LOGGER.info("Running %d warmup inference(s) for %s", warmup, model_name)
|
||||||
|
prompt = build_grounding_prompt(samples[0].target)
|
||||||
|
for _ in range(warmup):
|
||||||
|
grounder.generate(
|
||||||
|
image_path=str(samples[0].image),
|
||||||
|
prompt=prompt,
|
||||||
|
max_new_tokens=max_new_tokens,
|
||||||
|
)
|
||||||
|
|
||||||
|
model_directory = run_directory / model_name
|
||||||
|
visualization_directory = model_directory / "visualizations"
|
||||||
|
visualization_directory.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
predictions: list[dict[str, Any]] = []
|
||||||
|
all_timings: list[InferenceTiming] = []
|
||||||
|
for index, sample in enumerate(samples, start=1):
|
||||||
|
LOGGER.info(
|
||||||
|
"[%s] sample %d/%d: %s",
|
||||||
|
model_name,
|
||||||
|
index,
|
||||||
|
len(samples),
|
||||||
|
sample.sample_id,
|
||||||
|
)
|
||||||
|
prediction, timings = _benchmark_sample(
|
||||||
|
grounder=grounder,
|
||||||
|
model_name=model_name,
|
||||||
|
sample=sample,
|
||||||
|
repeats=repeats,
|
||||||
|
max_new_tokens=max_new_tokens,
|
||||||
|
visualization_path=(
|
||||||
|
visualization_directory / f"{sample.sample_id}.jpg"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
predictions.append(prediction)
|
||||||
|
all_timings.extend(timings)
|
||||||
|
|
||||||
|
summary = summarize_model(
|
||||||
|
model_name=model_name,
|
||||||
|
model_path=str(model_path),
|
||||||
|
model_load_seconds=model_load_seconds,
|
||||||
|
memory_metrics=memory_metrics,
|
||||||
|
predictions=predictions,
|
||||||
|
all_timings=all_timings,
|
||||||
|
)
|
||||||
|
summary.update(grounder.implementation_metadata())
|
||||||
|
write_model_outputs(model_directory, summary, predictions)
|
||||||
|
LOGGER.info("Saved %s results to %s", model_name, model_directory)
|
||||||
|
return summary, predictions
|
||||||
|
finally:
|
||||||
|
if grounder is not None:
|
||||||
|
del grounder
|
||||||
|
gc.collect()
|
||||||
|
torch.cuda.empty_cache()
|
||||||
|
torch.cuda.synchronize()
|
||||||
|
LOGGER.info("Released model %s", model_name)
|
||||||
|
|
||||||
|
|
||||||
|
def _benchmark_sample(
|
||||||
|
*,
|
||||||
|
grounder: Grounder,
|
||||||
|
model_name: str,
|
||||||
|
sample: BenchmarkSample,
|
||||||
|
repeats: int,
|
||||||
|
max_new_tokens: int,
|
||||||
|
visualization_path: Path,
|
||||||
|
) -> tuple[dict[str, Any], list[InferenceTiming]]:
|
||||||
|
prompt = build_grounding_prompt(sample.target)
|
||||||
|
raw_output: str | None = None
|
||||||
|
timings: list[InferenceTiming] = []
|
||||||
|
peak_memory_values: list[float] = []
|
||||||
|
inference_errors: list[str] = []
|
||||||
|
|
||||||
|
for repeat_index in range(1, repeats + 1):
|
||||||
|
try:
|
||||||
|
output, timing, extra_metrics = grounder.generate_with_metrics(
|
||||||
|
image_path=str(sample.image),
|
||||||
|
prompt=prompt,
|
||||||
|
max_new_tokens=max_new_tokens,
|
||||||
|
)
|
||||||
|
if raw_output is None:
|
||||||
|
raw_output = output
|
||||||
|
timings.append(timing)
|
||||||
|
peak_memory_values.append(extra_metrics["peak_cuda_memory_mb"])
|
||||||
|
except Exception as exc: # Keep the batch running after a sample failure.
|
||||||
|
error = f"repeat {repeat_index}: {type(exc).__name__}: {exc}"
|
||||||
|
inference_errors.append(error)
|
||||||
|
LOGGER.exception(
|
||||||
|
"[%s] inference failed for %s (%s)",
|
||||||
|
model_name,
|
||||||
|
sample.sample_id,
|
||||||
|
error,
|
||||||
|
)
|
||||||
|
|
||||||
|
timing_summary = summarize_timings(timings)
|
||||||
|
prediction: dict[str, Any] = {
|
||||||
|
"model": model_name,
|
||||||
|
"id": sample.sample_id,
|
||||||
|
"image": str(sample.image),
|
||||||
|
"target": sample.target,
|
||||||
|
"raw_output": raw_output,
|
||||||
|
"parse_success": False,
|
||||||
|
"parse_error": None,
|
||||||
|
"inference_errors": inference_errors,
|
||||||
|
"pred_bbox_pixel": None,
|
||||||
|
"pred_center_pixel": None,
|
||||||
|
"gt_bbox_pixel": (
|
||||||
|
list(sample.gt_bbox_pixel) if sample.gt_bbox_pixel is not None else None
|
||||||
|
),
|
||||||
|
"point_in_box": None,
|
||||||
|
"bbox_iou": None,
|
||||||
|
"normalized_center_error": None,
|
||||||
|
**timing_summary,
|
||||||
|
"peak_cuda_memory_mb": max(peak_memory_values) if peak_memory_values else None,
|
||||||
|
}
|
||||||
|
|
||||||
|
if raw_output is None:
|
||||||
|
prediction["parse_error"] = "; ".join(inference_errors) or "No model output"
|
||||||
|
return prediction, timings
|
||||||
|
|
||||||
|
try:
|
||||||
|
parsed = parse_grounding_output(raw_output)
|
||||||
|
prediction["parse_success"] = True
|
||||||
|
except Exception as exc:
|
||||||
|
prediction["parse_error"] = f"{type(exc).__name__}: {exc}"
|
||||||
|
LOGGER.warning(
|
||||||
|
"[%s] parse failed for %s: %s",
|
||||||
|
model_name,
|
||||||
|
sample.sample_id,
|
||||||
|
exc,
|
||||||
|
)
|
||||||
|
return prediction, timings
|
||||||
|
|
||||||
|
try:
|
||||||
|
pixel_result = visualize_grounding(
|
||||||
|
str(sample.image),
|
||||||
|
parsed,
|
||||||
|
str(visualization_path),
|
||||||
|
)
|
||||||
|
pred_bbox = pixel_result["bbox_pixel"]
|
||||||
|
pred_center = pixel_result["center_pixel"]
|
||||||
|
prediction["pred_bbox_pixel"] = pred_bbox
|
||||||
|
prediction["pred_center_pixel"] = pred_center
|
||||||
|
|
||||||
|
if sample.gt_bbox_pixel is not None:
|
||||||
|
prediction["point_in_box"] = point_in_box(
|
||||||
|
pred_center,
|
||||||
|
sample.gt_bbox_pixel,
|
||||||
|
)
|
||||||
|
prediction["bbox_iou"] = bbox_iou(
|
||||||
|
pred_bbox,
|
||||||
|
sample.gt_bbox_pixel,
|
||||||
|
)
|
||||||
|
prediction["normalized_center_error"] = normalized_center_error(
|
||||||
|
pred_center,
|
||||||
|
sample.gt_bbox_pixel,
|
||||||
|
pixel_result["image_width"],
|
||||||
|
pixel_result["image_height"],
|
||||||
|
)
|
||||||
|
except Exception as exc:
|
||||||
|
prediction["visualization_error"] = f"{type(exc).__name__}: {exc}"
|
||||||
|
LOGGER.exception(
|
||||||
|
"[%s] visualization failed for %s",
|
||||||
|
model_name,
|
||||||
|
sample.sample_id,
|
||||||
|
)
|
||||||
|
|
||||||
|
return prediction, timings
|
||||||
|
|
||||||
|
|
||||||
|
def _validate_cuda_environment() -> None:
|
||||||
|
if not torch.cuda.is_available():
|
||||||
|
raise RuntimeError("CUDA is required for the multimodal benchmark")
|
||||||
|
visible_device_count = torch.cuda.device_count()
|
||||||
|
if visible_device_count != 1:
|
||||||
|
raise RuntimeError(
|
||||||
|
"Benchmark requires exactly one visible CUDA device; "
|
||||||
|
f"found {visible_device_count}. Set CUDA_VISIBLE_DEVICES=0."
|
||||||
|
)
|
||||||
96
src/cockpit_grounding/benchmark/statistics.py
Normal file
96
src/cockpit_grounding/benchmark/statistics.py
Normal file
@ -0,0 +1,96 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from collections.abc import Iterable, Sequence
|
||||||
|
from statistics import fmean
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from cockpit_grounding.models.base import InferenceTiming
|
||||||
|
|
||||||
|
|
||||||
|
def percentile(values: Sequence[float], quantile: float) -> float:
|
||||||
|
if not values:
|
||||||
|
raise ValueError("percentile requires at least one value")
|
||||||
|
if not 0.0 <= quantile <= 1.0:
|
||||||
|
raise ValueError("quantile must be between 0 and 1")
|
||||||
|
|
||||||
|
ordered = sorted(values)
|
||||||
|
position = (len(ordered) - 1) * quantile
|
||||||
|
lower = int(position)
|
||||||
|
upper = min(lower + 1, len(ordered) - 1)
|
||||||
|
fraction = position - lower
|
||||||
|
return ordered[lower] + (ordered[upper] - ordered[lower]) * fraction
|
||||||
|
|
||||||
|
|
||||||
|
def summarize_timings(timings: Sequence[InferenceTiming]) -> dict[str, float | None]:
|
||||||
|
if not timings:
|
||||||
|
return {
|
||||||
|
"preprocess_mean_ms": None,
|
||||||
|
"generate_mean_ms": None,
|
||||||
|
"decode_mean_ms": None,
|
||||||
|
"total_mean_ms": None,
|
||||||
|
"total_p50_ms": None,
|
||||||
|
"total_p95_ms": None,
|
||||||
|
}
|
||||||
|
|
||||||
|
totals = [timing.total_ms for timing in timings]
|
||||||
|
return {
|
||||||
|
"preprocess_mean_ms": fmean(t.preprocess_ms for t in timings),
|
||||||
|
"generate_mean_ms": fmean(t.generate_ms for t in timings),
|
||||||
|
"decode_mean_ms": fmean(t.decode_ms for t in timings),
|
||||||
|
"total_mean_ms": fmean(totals),
|
||||||
|
"total_p50_ms": percentile(totals, 0.50),
|
||||||
|
"total_p95_ms": percentile(totals, 0.95),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def summarize_model(
|
||||||
|
*,
|
||||||
|
model_name: str,
|
||||||
|
model_path: str,
|
||||||
|
model_load_seconds: float,
|
||||||
|
memory_metrics: dict[str, float],
|
||||||
|
predictions: Sequence[dict[str, Any]],
|
||||||
|
all_timings: Sequence[InferenceTiming],
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
timing = summarize_timings(all_timings)
|
||||||
|
sample_count = len(predictions)
|
||||||
|
parse_successes = sum(bool(item["parse_success"]) for item in predictions)
|
||||||
|
|
||||||
|
point_values = _present_values(predictions, "point_in_box")
|
||||||
|
iou_values = _present_values(predictions, "bbox_iou")
|
||||||
|
center_error_values = _present_values(predictions, "normalized_center_error")
|
||||||
|
peak_memory_values = _present_values(predictions, "peak_cuda_memory_mb")
|
||||||
|
mean_total = timing["total_mean_ms"]
|
||||||
|
|
||||||
|
return {
|
||||||
|
"model": model_name,
|
||||||
|
"model_path": model_path,
|
||||||
|
"model_load_seconds": model_load_seconds,
|
||||||
|
**memory_metrics,
|
||||||
|
"samples": sample_count,
|
||||||
|
"parse_success_rate": parse_successes / sample_count if sample_count else 0.0,
|
||||||
|
"mean_preprocess_ms": timing["preprocess_mean_ms"],
|
||||||
|
"mean_generate_ms": timing["generate_mean_ms"],
|
||||||
|
"mean_decode_ms": timing["decode_mean_ms"],
|
||||||
|
"mean_total_ms": mean_total,
|
||||||
|
"p50_total_ms": timing["total_p50_ms"],
|
||||||
|
"p95_total_ms": timing["total_p95_ms"],
|
||||||
|
"throughput_samples_per_sec": (
|
||||||
|
1000.0 / mean_total if isinstance(mean_total, float) and mean_total > 0 else None
|
||||||
|
),
|
||||||
|
"peak_cuda_memory_mb": max(peak_memory_values) if peak_memory_values else None,
|
||||||
|
"accuracy_point_in_box": (
|
||||||
|
fmean(point_values) if point_values else None
|
||||||
|
),
|
||||||
|
"mean_bbox_iou": fmean(iou_values) if iou_values else None,
|
||||||
|
"mean_normalized_center_error": (
|
||||||
|
fmean(center_error_values) if center_error_values else None
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _present_values(
|
||||||
|
predictions: Iterable[dict[str, Any]],
|
||||||
|
key: str,
|
||||||
|
) -> list[float]:
|
||||||
|
return [float(item[key]) for item in predictions if item.get(key) is not None]
|
||||||
0
src/cockpit_grounding/grounding/__init__.py
Normal file
0
src/cockpit_grounding/grounding/__init__.py
Normal file
94
src/cockpit_grounding/grounding/predictor.py
Normal file
94
src/cockpit_grounding/grounding/predictor.py
Normal file
@ -0,0 +1,94 @@
|
|||||||
|
import json
|
||||||
|
import re
|
||||||
|
from dataclasses import dataclass
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class GroundingResult:
|
||||||
|
x1: float
|
||||||
|
y1: float
|
||||||
|
x2: float
|
||||||
|
y2: float
|
||||||
|
|
||||||
|
@property
|
||||||
|
def center_relative(self):
|
||||||
|
return (
|
||||||
|
(self.x1 + self.x2) / 2.0,
|
||||||
|
(self.y1 + self.y2) / 2.0,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def build_grounding_prompt(target: str) -> str:
|
||||||
|
return f"""
|
||||||
|
你是一个汽车座舱可操作UI元素定位器。
|
||||||
|
|
||||||
|
输入图像来自真实相机拍摄,而不是系统截图。
|
||||||
|
|
||||||
|
请找到以下目标:
|
||||||
|
|
||||||
|
{target}
|
||||||
|
|
||||||
|
要求:
|
||||||
|
|
||||||
|
1. 找到真正可以被用户点击的控件。
|
||||||
|
2. 不要选择标题、说明文字或者附近的无关区域。
|
||||||
|
3. 返回整个可点击控件的边界框。
|
||||||
|
4. 使用相对坐标:
|
||||||
|
左上角 = (0, 0)
|
||||||
|
右下角 = (1000, 1000)
|
||||||
|
5. x1 < x2,y1 < y2。
|
||||||
|
6. 只返回 JSON,不要解释。
|
||||||
|
|
||||||
|
严格输出:
|
||||||
|
|
||||||
|
{{"bbox_2d": [x1, y1, x2, y2]}}
|
||||||
|
""".strip()
|
||||||
|
|
||||||
|
|
||||||
|
def parse_grounding_output(
|
||||||
|
text: str,
|
||||||
|
) -> GroundingResult:
|
||||||
|
|
||||||
|
# 防止模型偶尔包 markdown code block
|
||||||
|
match = re.search(
|
||||||
|
r'"bbox_2d"\s*:\s*\[\s*'
|
||||||
|
r'([0-9.]+)\s*,\s*'
|
||||||
|
r'([0-9.]+)\s*,\s*'
|
||||||
|
r'([0-9.]+)\s*,\s*'
|
||||||
|
r'([0-9.]+)\s*\]',
|
||||||
|
text,
|
||||||
|
)
|
||||||
|
|
||||||
|
if match is None:
|
||||||
|
raise ValueError(
|
||||||
|
"无法从模型输出解析 bbox_2d:\n"
|
||||||
|
+ text
|
||||||
|
)
|
||||||
|
|
||||||
|
x1, y1, x2, y2 = (
|
||||||
|
float(match.group(i))
|
||||||
|
for i in range(1, 5)
|
||||||
|
)
|
||||||
|
|
||||||
|
# 基础合法性检查
|
||||||
|
values = [x1, y1, x2, y2]
|
||||||
|
|
||||||
|
if not all(
|
||||||
|
0 <= value <= 1000
|
||||||
|
for value in values
|
||||||
|
):
|
||||||
|
raise ValueError(
|
||||||
|
f"坐标超出 0~1000: {values}"
|
||||||
|
)
|
||||||
|
|
||||||
|
if x1 >= x2 or y1 >= y2:
|
||||||
|
raise ValueError(
|
||||||
|
f"非法 bbox: {values}"
|
||||||
|
)
|
||||||
|
|
||||||
|
return GroundingResult(
|
||||||
|
x1=x1,
|
||||||
|
y1=y1,
|
||||||
|
x2=x2,
|
||||||
|
y2=y2,
|
||||||
|
)
|
||||||
0
src/cockpit_grounding/models/__init__.py
Normal file
0
src/cockpit_grounding/models/__init__.py
Normal file
42
src/cockpit_grounding/models/base.py
Normal file
42
src/cockpit_grounding/models/base.py
Normal file
@ -0,0 +1,42 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from typing import Protocol
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class InferenceTiming:
|
||||||
|
preprocess_ms: float
|
||||||
|
generate_ms: float
|
||||||
|
decode_ms: float
|
||||||
|
total_ms: float
|
||||||
|
|
||||||
|
|
||||||
|
class Grounder(Protocol):
|
||||||
|
model_path: str
|
||||||
|
|
||||||
|
def generate(
|
||||||
|
self,
|
||||||
|
image_path: str,
|
||||||
|
prompt: str,
|
||||||
|
max_new_tokens: int = 128,
|
||||||
|
) -> str: ...
|
||||||
|
|
||||||
|
def generate_with_metrics(
|
||||||
|
self,
|
||||||
|
image_path: str,
|
||||||
|
prompt: str,
|
||||||
|
max_new_tokens: int = 128,
|
||||||
|
) -> tuple[str, InferenceTiming, dict[str, float]]: ...
|
||||||
|
|
||||||
|
def cuda_memory_metrics(self) -> dict[str, float]: ...
|
||||||
|
|
||||||
|
def implementation_metadata(self) -> dict[str, str]: ...
|
||||||
|
|
||||||
|
|
||||||
|
class TextGenerator(Protocol):
|
||||||
|
def generate_text(
|
||||||
|
self,
|
||||||
|
prompt: str,
|
||||||
|
max_new_tokens: int = 256,
|
||||||
|
) -> str: ...
|
||||||
20
src/cockpit_grounding/models/factory.py
Normal file
20
src/cockpit_grounding/models/factory.py
Normal file
@ -0,0 +1,20 @@
|
|||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from cockpit_grounding.models.base import Grounder
|
||||||
|
|
||||||
|
SUPPORTED_BACKENDS = ("qwen3vl", "qwen35")
|
||||||
|
|
||||||
|
|
||||||
|
def create_grounder(backend: str, model_path: str | Path) -> Grounder:
|
||||||
|
if backend == "qwen3vl":
|
||||||
|
from cockpit_grounding.models.qwen3vl import Qwen3VLGrounder
|
||||||
|
|
||||||
|
return Qwen3VLGrounder(model_path=str(model_path))
|
||||||
|
if backend == "qwen35":
|
||||||
|
from cockpit_grounding.models.qwen35 import Qwen35Grounder
|
||||||
|
|
||||||
|
return Qwen35Grounder(model_path=str(model_path))
|
||||||
|
raise ValueError(
|
||||||
|
f"Unsupported model backend {backend!r}; "
|
||||||
|
f"expected one of {', '.join(SUPPORTED_BACKENDS)}"
|
||||||
|
)
|
||||||
177
src/cockpit_grounding/models/qwen35.py
Normal file
177
src/cockpit_grounding/models/qwen35.py
Normal file
@ -0,0 +1,177 @@
|
|||||||
|
from pathlib import Path
|
||||||
|
from time import perf_counter
|
||||||
|
|
||||||
|
import torch
|
||||||
|
from transformers import AutoModelForMultimodalLM, AutoProcessor
|
||||||
|
|
||||||
|
from cockpit_grounding.models.base import InferenceTiming
|
||||||
|
|
||||||
|
|
||||||
|
class Qwen35Grounder:
|
||||||
|
def __init__(self, model_path: str) -> None:
|
||||||
|
local_model_path = Path(model_path).expanduser().resolve()
|
||||||
|
if not local_model_path.is_dir():
|
||||||
|
raise FileNotFoundError(
|
||||||
|
f"Local model directory not found: {local_model_path}"
|
||||||
|
)
|
||||||
|
if not torch.cuda.is_available():
|
||||||
|
raise RuntimeError("Qwen35Grounder requires a CUDA device")
|
||||||
|
|
||||||
|
self.model_path = str(local_model_path)
|
||||||
|
print("[Model] Loading local Qwen3.5 model:")
|
||||||
|
print(self.model_path)
|
||||||
|
|
||||||
|
self.model = AutoModelForMultimodalLM.from_pretrained(
|
||||||
|
self.model_path,
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
device_map={"": 0},
|
||||||
|
local_files_only=True,
|
||||||
|
)
|
||||||
|
self.processor = AutoProcessor.from_pretrained(
|
||||||
|
self.model_path,
|
||||||
|
local_files_only=True,
|
||||||
|
)
|
||||||
|
self.model.eval()
|
||||||
|
|
||||||
|
print(f"[Model] Class: {type(self.model).__name__}")
|
||||||
|
print("[Model] Loaded successfully")
|
||||||
|
print("[Model] GPU:", torch.cuda.get_device_name(self.model.device))
|
||||||
|
|
||||||
|
@torch.inference_mode()
|
||||||
|
def generate(
|
||||||
|
self,
|
||||||
|
image_path: str,
|
||||||
|
prompt: str,
|
||||||
|
max_new_tokens: int = 128,
|
||||||
|
) -> str:
|
||||||
|
raw_output, _, _ = self.generate_with_metrics(
|
||||||
|
image_path=image_path,
|
||||||
|
prompt=prompt,
|
||||||
|
max_new_tokens=max_new_tokens,
|
||||||
|
)
|
||||||
|
return raw_output
|
||||||
|
|
||||||
|
@torch.inference_mode()
|
||||||
|
def generate_text(
|
||||||
|
self,
|
||||||
|
prompt: str,
|
||||||
|
max_new_tokens: int = 256,
|
||||||
|
) -> str:
|
||||||
|
messages = [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": [{"type": "text", "text": prompt}],
|
||||||
|
}
|
||||||
|
]
|
||||||
|
inputs = self.processor.apply_chat_template(
|
||||||
|
messages,
|
||||||
|
add_generation_prompt=True,
|
||||||
|
tokenize=True,
|
||||||
|
return_dict=True,
|
||||||
|
return_tensors="pt",
|
||||||
|
)
|
||||||
|
inputs = inputs.to(self.model.device)
|
||||||
|
output_ids = self.model.generate(
|
||||||
|
**inputs,
|
||||||
|
max_new_tokens=max_new_tokens,
|
||||||
|
do_sample=False,
|
||||||
|
)
|
||||||
|
generated_ids = output_ids[:, inputs["input_ids"].shape[-1] :]
|
||||||
|
return self.processor.batch_decode(
|
||||||
|
generated_ids,
|
||||||
|
skip_special_tokens=True,
|
||||||
|
clean_up_tokenization_spaces=False,
|
||||||
|
)[0]
|
||||||
|
|
||||||
|
@torch.inference_mode()
|
||||||
|
def generate_with_metrics(
|
||||||
|
self,
|
||||||
|
image_path: str,
|
||||||
|
prompt: str,
|
||||||
|
max_new_tokens: int = 128,
|
||||||
|
) -> tuple[str, InferenceTiming, dict[str, float]]:
|
||||||
|
image = Path(image_path).expanduser().resolve()
|
||||||
|
if not image.is_file():
|
||||||
|
raise FileNotFoundError(image)
|
||||||
|
|
||||||
|
self._synchronize_cuda()
|
||||||
|
torch.cuda.reset_peak_memory_stats(self.model.device)
|
||||||
|
total_start = perf_counter()
|
||||||
|
preprocess_start = total_start
|
||||||
|
|
||||||
|
messages = [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": [
|
||||||
|
{"type": "image", "path": str(image)},
|
||||||
|
{"type": "text", "text": prompt},
|
||||||
|
],
|
||||||
|
}
|
||||||
|
]
|
||||||
|
inputs = self.processor.apply_chat_template(
|
||||||
|
messages,
|
||||||
|
add_generation_prompt=True,
|
||||||
|
tokenize=True,
|
||||||
|
return_dict=True,
|
||||||
|
return_tensors="pt",
|
||||||
|
)
|
||||||
|
inputs = inputs.to(self.model.device)
|
||||||
|
|
||||||
|
self._synchronize_cuda()
|
||||||
|
preprocess_ms = (perf_counter() - preprocess_start) * 1000.0
|
||||||
|
|
||||||
|
generate_start = perf_counter()
|
||||||
|
output_ids = self.model.generate(
|
||||||
|
**inputs,
|
||||||
|
max_new_tokens=max_new_tokens,
|
||||||
|
do_sample=False,
|
||||||
|
)
|
||||||
|
self._synchronize_cuda()
|
||||||
|
generate_ms = (perf_counter() - generate_start) * 1000.0
|
||||||
|
|
||||||
|
decode_start = perf_counter()
|
||||||
|
generated_ids = output_ids[:, inputs["input_ids"].shape[-1] :]
|
||||||
|
output_text = self.processor.batch_decode(
|
||||||
|
generated_ids,
|
||||||
|
skip_special_tokens=True,
|
||||||
|
clean_up_tokenization_spaces=False,
|
||||||
|
)[0]
|
||||||
|
self._synchronize_cuda()
|
||||||
|
decode_ms = (perf_counter() - decode_start) * 1000.0
|
||||||
|
total_ms = (perf_counter() - total_start) * 1000.0
|
||||||
|
|
||||||
|
timing = InferenceTiming(
|
||||||
|
preprocess_ms=preprocess_ms,
|
||||||
|
generate_ms=generate_ms,
|
||||||
|
decode_ms=decode_ms,
|
||||||
|
total_ms=total_ms,
|
||||||
|
)
|
||||||
|
metrics = {
|
||||||
|
"peak_cuda_memory_mb": self._bytes_to_mb(
|
||||||
|
torch.cuda.max_memory_allocated(self.model.device)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
return output_text, timing, metrics
|
||||||
|
|
||||||
|
def cuda_memory_metrics(self) -> dict[str, float]:
|
||||||
|
return {
|
||||||
|
"model_cuda_allocated_mb": self._bytes_to_mb(
|
||||||
|
torch.cuda.memory_allocated(self.model.device)
|
||||||
|
),
|
||||||
|
"model_cuda_reserved_mb": self._bytes_to_mb(
|
||||||
|
torch.cuda.memory_reserved(self.model.device)
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
def implementation_metadata(self) -> dict[str, str]:
|
||||||
|
return {
|
||||||
|
"transformers_model_class": type(self.model).__name__,
|
||||||
|
"transformers_processor_class": type(self.processor).__name__,
|
||||||
|
}
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _bytes_to_mb(value: int) -> float:
|
||||||
|
return value / (1024.0 * 1024.0)
|
||||||
|
|
||||||
|
def _synchronize_cuda(self) -> None:
|
||||||
|
torch.cuda.synchronize(self.model.device)
|
||||||
204
src/cockpit_grounding/models/qwen3vl.py
Normal file
204
src/cockpit_grounding/models/qwen3vl.py
Normal file
@ -0,0 +1,204 @@
|
|||||||
|
from pathlib import Path
|
||||||
|
from time import perf_counter
|
||||||
|
|
||||||
|
import torch
|
||||||
|
from transformers import (
|
||||||
|
AutoProcessor,
|
||||||
|
Qwen3VLForConditionalGeneration,
|
||||||
|
)
|
||||||
|
from qwen_vl_utils import process_vision_info
|
||||||
|
|
||||||
|
from cockpit_grounding.models.base import InferenceTiming
|
||||||
|
|
||||||
|
|
||||||
|
class Qwen3VLGrounder:
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
model_path: str,
|
||||||
|
) -> None:
|
||||||
|
local_model_path = Path(model_path).expanduser().resolve()
|
||||||
|
if not local_model_path.is_dir():
|
||||||
|
raise FileNotFoundError(
|
||||||
|
f"Local model directory not found: {local_model_path}"
|
||||||
|
)
|
||||||
|
if not torch.cuda.is_available():
|
||||||
|
raise RuntimeError("Qwen3VLGrounder requires a CUDA device")
|
||||||
|
|
||||||
|
self.model_path = str(local_model_path)
|
||||||
|
|
||||||
|
print("[Model] Loading local model:")
|
||||||
|
print(self.model_path)
|
||||||
|
|
||||||
|
self.model = (
|
||||||
|
Qwen3VLForConditionalGeneration
|
||||||
|
.from_pretrained(
|
||||||
|
self.model_path,
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
device_map={"": 0},
|
||||||
|
local_files_only=True,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
self.processor = AutoProcessor.from_pretrained(
|
||||||
|
self.model_path,
|
||||||
|
local_files_only=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
self.model.eval()
|
||||||
|
|
||||||
|
print("[Model] Loaded successfully")
|
||||||
|
print(
|
||||||
|
"[Model] GPU:",
|
||||||
|
torch.cuda.get_device_name(self.model.device)
|
||||||
|
)
|
||||||
|
|
||||||
|
@torch.inference_mode()
|
||||||
|
def generate(
|
||||||
|
self,
|
||||||
|
image_path: str,
|
||||||
|
prompt: str,
|
||||||
|
max_new_tokens: int = 128,
|
||||||
|
) -> str:
|
||||||
|
raw_output, _, _ = self.generate_with_metrics(
|
||||||
|
image_path=image_path,
|
||||||
|
prompt=prompt,
|
||||||
|
max_new_tokens=max_new_tokens,
|
||||||
|
)
|
||||||
|
return raw_output
|
||||||
|
|
||||||
|
@torch.inference_mode()
|
||||||
|
def generate_with_metrics(
|
||||||
|
self,
|
||||||
|
image_path: str,
|
||||||
|
prompt: str,
|
||||||
|
max_new_tokens: int = 128,
|
||||||
|
) -> tuple[str, InferenceTiming, dict[str, float]]:
|
||||||
|
"""Generate a response and report synchronized stage timings."""
|
||||||
|
|
||||||
|
image_path = Path(image_path).resolve()
|
||||||
|
|
||||||
|
if not image_path.exists():
|
||||||
|
raise FileNotFoundError(image_path)
|
||||||
|
|
||||||
|
self._synchronize_cuda()
|
||||||
|
torch.cuda.reset_peak_memory_stats(self.model.device)
|
||||||
|
total_start = perf_counter()
|
||||||
|
preprocess_start = total_start
|
||||||
|
|
||||||
|
messages = [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": [
|
||||||
|
{
|
||||||
|
"type": "image",
|
||||||
|
"image": image_path.as_uri(),
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"type": "text",
|
||||||
|
"text": prompt,
|
||||||
|
},
|
||||||
|
],
|
||||||
|
}
|
||||||
|
]
|
||||||
|
|
||||||
|
text = self.processor.apply_chat_template(
|
||||||
|
messages,
|
||||||
|
tokenize=False,
|
||||||
|
add_generation_prompt=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
images, videos, video_kwargs = process_vision_info(
|
||||||
|
messages,
|
||||||
|
image_patch_size=16,
|
||||||
|
return_video_kwargs=True,
|
||||||
|
return_video_metadata=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
if videos is not None:
|
||||||
|
videos, video_metadatas = zip(*videos)
|
||||||
|
videos = list(videos)
|
||||||
|
video_metadatas = list(video_metadatas)
|
||||||
|
else:
|
||||||
|
video_metadatas = None
|
||||||
|
|
||||||
|
inputs = self.processor(
|
||||||
|
text=text,
|
||||||
|
images=images,
|
||||||
|
videos=videos,
|
||||||
|
video_metadata=video_metadatas,
|
||||||
|
return_tensors="pt",
|
||||||
|
do_resize=False,
|
||||||
|
**video_kwargs,
|
||||||
|
)
|
||||||
|
|
||||||
|
inputs = inputs.to(self.model.device)
|
||||||
|
|
||||||
|
self._synchronize_cuda()
|
||||||
|
preprocess_ms = (perf_counter() - preprocess_start) * 1000.0
|
||||||
|
|
||||||
|
generate_start = perf_counter()
|
||||||
|
generated_ids = self.model.generate(
|
||||||
|
**inputs,
|
||||||
|
max_new_tokens=max_new_tokens,
|
||||||
|
do_sample=False,
|
||||||
|
)
|
||||||
|
self._synchronize_cuda()
|
||||||
|
generate_ms = (perf_counter() - generate_start) * 1000.0
|
||||||
|
|
||||||
|
decode_start = perf_counter()
|
||||||
|
generated_ids_trimmed = [
|
||||||
|
output_ids[len(input_ids):]
|
||||||
|
for input_ids, output_ids
|
||||||
|
in zip(
|
||||||
|
inputs.input_ids,
|
||||||
|
generated_ids,
|
||||||
|
)
|
||||||
|
]
|
||||||
|
|
||||||
|
output_text = self.processor.batch_decode(
|
||||||
|
generated_ids_trimmed,
|
||||||
|
skip_special_tokens=True,
|
||||||
|
clean_up_tokenization_spaces=False,
|
||||||
|
)[0]
|
||||||
|
|
||||||
|
self._synchronize_cuda()
|
||||||
|
decode_ms = (perf_counter() - decode_start) * 1000.0
|
||||||
|
total_ms = (perf_counter() - total_start) * 1000.0
|
||||||
|
peak_cuda_memory_mb = self._bytes_to_mb(
|
||||||
|
torch.cuda.max_memory_allocated(self.model.device)
|
||||||
|
)
|
||||||
|
|
||||||
|
timing = InferenceTiming(
|
||||||
|
preprocess_ms=preprocess_ms,
|
||||||
|
generate_ms=generate_ms,
|
||||||
|
decode_ms=decode_ms,
|
||||||
|
total_ms=total_ms,
|
||||||
|
)
|
||||||
|
extra_metrics = {
|
||||||
|
"peak_cuda_memory_mb": peak_cuda_memory_mb,
|
||||||
|
}
|
||||||
|
return output_text, timing, extra_metrics
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _bytes_to_mb(value: int) -> float:
|
||||||
|
return value / (1024.0 * 1024.0)
|
||||||
|
|
||||||
|
def cuda_memory_metrics(self) -> dict[str, float]:
|
||||||
|
return {
|
||||||
|
"model_cuda_allocated_mb": self._bytes_to_mb(
|
||||||
|
torch.cuda.memory_allocated(self.model.device)
|
||||||
|
),
|
||||||
|
"model_cuda_reserved_mb": self._bytes_to_mb(
|
||||||
|
torch.cuda.memory_reserved(self.model.device)
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
def implementation_metadata(self) -> dict[str, str]:
|
||||||
|
return {
|
||||||
|
"transformers_model_class": type(self.model).__name__,
|
||||||
|
"transformers_processor_class": type(self.processor).__name__,
|
||||||
|
}
|
||||||
|
|
||||||
|
def _synchronize_cuda(self) -> None:
|
||||||
|
torch.cuda.synchronize(self.model.device)
|
||||||
0
src/cockpit_grounding/utils/__init__.py
Normal file
0
src/cockpit_grounding/utils/__init__.py
Normal file
0
src/cockpit_grounding/vision/__init__.py
Normal file
0
src/cockpit_grounding/vision/__init__.py
Normal file
109
src/cockpit_grounding/vision/visualize.py
Normal file
109
src/cockpit_grounding/vision/visualize.py
Normal file
@ -0,0 +1,109 @@
|
|||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import cv2
|
||||||
|
|
||||||
|
from cockpit_grounding.grounding.predictor import (
|
||||||
|
GroundingResult,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def relative_bbox_to_pixels(
|
||||||
|
result: GroundingResult,
|
||||||
|
width: int,
|
||||||
|
height: int,
|
||||||
|
):
|
||||||
|
x1 = round(result.x1 / 1000.0 * width)
|
||||||
|
y1 = round(result.y1 / 1000.0 * height)
|
||||||
|
x2 = round(result.x2 / 1000.0 * width)
|
||||||
|
y2 = round(result.y2 / 1000.0 * height)
|
||||||
|
|
||||||
|
return x1, y1, x2, y2
|
||||||
|
|
||||||
|
|
||||||
|
def visualize_grounding(
|
||||||
|
image_path: str,
|
||||||
|
result: GroundingResult,
|
||||||
|
output_path: str,
|
||||||
|
):
|
||||||
|
image = cv2.imread(image_path)
|
||||||
|
|
||||||
|
if image is None:
|
||||||
|
raise RuntimeError(
|
||||||
|
f"无法读取图片: {image_path}"
|
||||||
|
)
|
||||||
|
|
||||||
|
height, width = image.shape[:2]
|
||||||
|
|
||||||
|
x1, y1, x2, y2 = relative_bbox_to_pixels(
|
||||||
|
result,
|
||||||
|
width,
|
||||||
|
height,
|
||||||
|
)
|
||||||
|
|
||||||
|
# 后面给机器人使用的点
|
||||||
|
u = round((x1 + x2) / 2)
|
||||||
|
v = round((y1 + y2) / 2)
|
||||||
|
|
||||||
|
# bbox
|
||||||
|
cv2.rectangle(
|
||||||
|
image,
|
||||||
|
(x1, y1),
|
||||||
|
(x2, y2),
|
||||||
|
(0, 0, 255),
|
||||||
|
3,
|
||||||
|
)
|
||||||
|
|
||||||
|
# 中心
|
||||||
|
cv2.circle(
|
||||||
|
image,
|
||||||
|
(u, v),
|
||||||
|
12,
|
||||||
|
(0, 0, 255),
|
||||||
|
-1,
|
||||||
|
)
|
||||||
|
|
||||||
|
# 十字线
|
||||||
|
cv2.line(
|
||||||
|
image,
|
||||||
|
(u - 25, v),
|
||||||
|
(u + 25, v),
|
||||||
|
(0, 0, 255),
|
||||||
|
3,
|
||||||
|
)
|
||||||
|
|
||||||
|
cv2.line(
|
||||||
|
image,
|
||||||
|
(u, v - 25),
|
||||||
|
(u, v + 25),
|
||||||
|
(0, 0, 255),
|
||||||
|
3,
|
||||||
|
)
|
||||||
|
|
||||||
|
output_path = Path(output_path)
|
||||||
|
|
||||||
|
output_path.parent.mkdir(
|
||||||
|
parents=True,
|
||||||
|
exist_ok=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
cv2.imwrite(
|
||||||
|
str(output_path),
|
||||||
|
image,
|
||||||
|
)
|
||||||
|
|
||||||
|
return {
|
||||||
|
"image_width": width,
|
||||||
|
"image_height": height,
|
||||||
|
|
||||||
|
"bbox_pixel": [
|
||||||
|
x1,
|
||||||
|
y1,
|
||||||
|
x2,
|
||||||
|
y2,
|
||||||
|
],
|
||||||
|
|
||||||
|
"center_pixel": [
|
||||||
|
u,
|
||||||
|
v,
|
||||||
|
],
|
||||||
|
}
|
||||||
5
src/cockpit_ui_grounding.egg-info/PKG-INFO
Normal file
5
src/cockpit_ui_grounding.egg-info/PKG-INFO
Normal file
@ -0,0 +1,5 @@
|
|||||||
|
Metadata-Version: 2.4
|
||||||
|
Name: cockpit-ui-grounding
|
||||||
|
Version: 0.1.0
|
||||||
|
Summary: Camera-view automotive cockpit UI grounding
|
||||||
|
Requires-Python: >=3.11
|
||||||
12
src/cockpit_ui_grounding.egg-info/SOURCES.txt
Normal file
12
src/cockpit_ui_grounding.egg-info/SOURCES.txt
Normal file
@ -0,0 +1,12 @@
|
|||||||
|
README.md
|
||||||
|
pyproject.toml
|
||||||
|
src/cockpit_grounding/__init__.py
|
||||||
|
src/cockpit_grounding/api/__init__.py
|
||||||
|
src/cockpit_grounding/grounding/__init__.py
|
||||||
|
src/cockpit_grounding/models/__init__.py
|
||||||
|
src/cockpit_grounding/utils/__init__.py
|
||||||
|
src/cockpit_grounding/vision/__init__.py
|
||||||
|
src/cockpit_ui_grounding.egg-info/PKG-INFO
|
||||||
|
src/cockpit_ui_grounding.egg-info/SOURCES.txt
|
||||||
|
src/cockpit_ui_grounding.egg-info/dependency_links.txt
|
||||||
|
src/cockpit_ui_grounding.egg-info/top_level.txt
|
||||||
1
src/cockpit_ui_grounding.egg-info/dependency_links.txt
Normal file
1
src/cockpit_ui_grounding.egg-info/dependency_links.txt
Normal file
@ -0,0 +1 @@
|
|||||||
|
|
||||||
1
src/cockpit_ui_grounding.egg-info/top_level.txt
Normal file
1
src/cockpit_ui_grounding.egg-info/top_level.txt
Normal file
@ -0,0 +1 @@
|
|||||||
|
cockpit_grounding
|
||||||
141
tests/test_benchmark_config.py
Normal file
141
tests/test_benchmark_config.py
Normal file
@ -0,0 +1,141 @@
|
|||||||
|
import json
|
||||||
|
import unittest
|
||||||
|
from pathlib import Path
|
||||||
|
from tempfile import TemporaryDirectory
|
||||||
|
|
||||||
|
from cockpit_grounding.benchmark.config import (
|
||||||
|
load_benchmark_config,
|
||||||
|
load_manifest,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class BenchmarkConfigTest(unittest.TestCase):
|
||||||
|
def test_load_benchmark_config_preserves_model_order(self) -> None:
|
||||||
|
with TemporaryDirectory() as directory:
|
||||||
|
tmp_path = Path(directory)
|
||||||
|
model_a = tmp_path / "model-a"
|
||||||
|
model_b = tmp_path / "model-b"
|
||||||
|
model_a.mkdir()
|
||||||
|
model_b.mkdir()
|
||||||
|
config = tmp_path / "benchmark.toml"
|
||||||
|
config.write_text(
|
||||||
|
"\n".join(
|
||||||
|
(
|
||||||
|
"[benchmark]",
|
||||||
|
"warmup = 1",
|
||||||
|
"repeats = 3",
|
||||||
|
"max_new_tokens = 128",
|
||||||
|
f'output_root = "{tmp_path}"',
|
||||||
|
"[models.first]",
|
||||||
|
f'path = "{model_a}"',
|
||||||
|
"[models.second]",
|
||||||
|
f'path = "{model_b}"',
|
||||||
|
)
|
||||||
|
),
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
|
||||||
|
loaded = load_benchmark_config(config)
|
||||||
|
|
||||||
|
self.assertEqual(
|
||||||
|
[model.name for model in loaded.models],
|
||||||
|
["first", "second"],
|
||||||
|
)
|
||||||
|
self.assertEqual(
|
||||||
|
[model.backend for model in loaded.models],
|
||||||
|
["qwen3vl", "qwen3vl"],
|
||||||
|
)
|
||||||
|
self.assertEqual(loaded.settings.repeats, 3)
|
||||||
|
self.assertTrue(loaded.settings.timestamp_run_directory)
|
||||||
|
|
||||||
|
def test_load_benchmark_config_supports_explicit_backend(self) -> None:
|
||||||
|
with TemporaryDirectory() as directory:
|
||||||
|
tmp_path = Path(directory)
|
||||||
|
model_path = tmp_path / "model"
|
||||||
|
model_path.mkdir()
|
||||||
|
config = tmp_path / "benchmark.toml"
|
||||||
|
config.write_text(
|
||||||
|
"\n".join(
|
||||||
|
(
|
||||||
|
"[benchmark]",
|
||||||
|
"warmup = 0",
|
||||||
|
"repeats = 1",
|
||||||
|
"max_new_tokens = 16",
|
||||||
|
f'output_root = "{tmp_path}"',
|
||||||
|
"[models.qwen35]",
|
||||||
|
'backend = "qwen35"',
|
||||||
|
f'path = "{model_path}"',
|
||||||
|
)
|
||||||
|
),
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
|
||||||
|
loaded = load_benchmark_config(config)
|
||||||
|
|
||||||
|
self.assertEqual(loaded.models[0].backend, "qwen35")
|
||||||
|
|
||||||
|
def test_load_benchmark_config_supports_exact_run_directory(self) -> None:
|
||||||
|
with TemporaryDirectory() as directory:
|
||||||
|
tmp_path = Path(directory)
|
||||||
|
model_path = tmp_path / "model"
|
||||||
|
model_path.mkdir()
|
||||||
|
config = tmp_path / "benchmark.toml"
|
||||||
|
config.write_text(
|
||||||
|
"\n".join(
|
||||||
|
(
|
||||||
|
"[benchmark]",
|
||||||
|
"warmup = 3",
|
||||||
|
"repeats = 5",
|
||||||
|
"max_new_tokens = 128",
|
||||||
|
f'output_root = "{tmp_path}"',
|
||||||
|
"timestamp_run_directory = false",
|
||||||
|
"[models.model]",
|
||||||
|
f'path = "{model_path}"',
|
||||||
|
)
|
||||||
|
),
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
|
||||||
|
loaded = load_benchmark_config(config)
|
||||||
|
|
||||||
|
self.assertFalse(loaded.settings.timestamp_run_directory)
|
||||||
|
|
||||||
|
def test_manifest_supports_optional_gt(self) -> None:
|
||||||
|
with TemporaryDirectory() as directory:
|
||||||
|
tmp_path = Path(directory)
|
||||||
|
image = tmp_path / "image.jpg"
|
||||||
|
image.touch()
|
||||||
|
manifest = tmp_path / "samples.jsonl"
|
||||||
|
entries = (
|
||||||
|
{"id": "without-gt", "image": str(image), "target": "button"},
|
||||||
|
{
|
||||||
|
"id": "with-gt",
|
||||||
|
"image": str(image),
|
||||||
|
"target": "button",
|
||||||
|
"gt_bbox_pixel": [1, 2, 3, 4],
|
||||||
|
},
|
||||||
|
)
|
||||||
|
manifest.write_text(
|
||||||
|
"\n".join(json.dumps(item) for item in entries) + "\n",
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
|
||||||
|
samples = load_manifest(manifest)
|
||||||
|
|
||||||
|
self.assertIsNone(samples[0].gt_bbox_pixel)
|
||||||
|
self.assertEqual(samples[1].gt_bbox_pixel, (1.0, 2.0, 3.0, 4.0))
|
||||||
|
|
||||||
|
def test_manifest_rejects_duplicate_ids(self) -> None:
|
||||||
|
with TemporaryDirectory() as directory:
|
||||||
|
tmp_path = Path(directory)
|
||||||
|
image = tmp_path / "image.jpg"
|
||||||
|
image.touch()
|
||||||
|
manifest = tmp_path / "samples.jsonl"
|
||||||
|
item = {"id": "same", "image": str(image), "target": "button"}
|
||||||
|
manifest.write_text(
|
||||||
|
json.dumps(item) + "\n" + json.dumps(item) + "\n",
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
|
||||||
|
with self.assertRaisesRegex(ValueError, "Duplicate"):
|
||||||
|
load_manifest(manifest)
|
||||||
90
tests/test_benchmark_reporting.py
Normal file
90
tests/test_benchmark_reporting.py
Normal file
@ -0,0 +1,90 @@
|
|||||||
|
import csv
|
||||||
|
import io
|
||||||
|
import unittest
|
||||||
|
from contextlib import redirect_stdout
|
||||||
|
from pathlib import Path
|
||||||
|
from tempfile import TemporaryDirectory
|
||||||
|
|
||||||
|
from cockpit_grounding.benchmark.reporting import (
|
||||||
|
create_run_directory,
|
||||||
|
print_model_selection,
|
||||||
|
write_root_outputs,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class BenchmarkReportingTest(unittest.TestCase):
|
||||||
|
def test_case_results_joins_models_by_case(self) -> None:
|
||||||
|
predictions = [
|
||||||
|
{
|
||||||
|
"model": "qwen3vl_2b",
|
||||||
|
"id": "button",
|
||||||
|
"target": "按钮",
|
||||||
|
"pred_center_pixel": [10, 20],
|
||||||
|
"total_mean_ms": 100.0,
|
||||||
|
"parse_success": True,
|
||||||
|
"pred_bbox_pixel": [1, 2, 19, 38],
|
||||||
|
"image": "/dataset/image.jpg",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"model": "qwen3vl_4b",
|
||||||
|
"id": "button",
|
||||||
|
"target": "按钮",
|
||||||
|
"pred_center_pixel": [11, 21],
|
||||||
|
"total_mean_ms": 120.0,
|
||||||
|
"parse_success": True,
|
||||||
|
"pred_bbox_pixel": [2, 3, 20, 39],
|
||||||
|
"image": "/dataset/image.jpg",
|
||||||
|
},
|
||||||
|
]
|
||||||
|
summaries = {"qwen3vl_2b": {}, "qwen3vl_4b": {}}
|
||||||
|
|
||||||
|
with TemporaryDirectory() as directory:
|
||||||
|
output = Path(directory)
|
||||||
|
write_root_outputs(output, {}, summaries, predictions)
|
||||||
|
with (output / "case_results.csv").open(
|
||||||
|
encoding="utf-8",
|
||||||
|
newline="",
|
||||||
|
) as csv_file:
|
||||||
|
rows = list(csv.DictReader(csv_file))
|
||||||
|
review = (output / "review.md").read_text(encoding="utf-8")
|
||||||
|
|
||||||
|
self.assertEqual(len(rows), 1)
|
||||||
|
self.assertEqual(rows[0]["id"], "button")
|
||||||
|
self.assertEqual(rows[0]["qwen3vl_2b_center_x"], "10")
|
||||||
|
self.assertEqual(rows[0]["qwen3vl_4b_center_y"], "21")
|
||||||
|
self.assertEqual(rows[0]["qwen3vl_2b_latency_ms"], "100.0")
|
||||||
|
self.assertEqual(rows[0]["manual_qwen3vl_2b"], "")
|
||||||
|
self.assertEqual(rows[0]["qwen3vl_4b_bbox_x2"], "20")
|
||||||
|
self.assertIn("qwen3vl_2b/visualizations/button.jpg", review)
|
||||||
|
self.assertIn("qwen3vl_4b/visualizations/button.jpg", review)
|
||||||
|
|
||||||
|
def test_exact_run_directory_uses_run_name_without_timestamp(self) -> None:
|
||||||
|
with TemporaryDirectory() as directory:
|
||||||
|
path = create_run_directory(
|
||||||
|
Path(directory),
|
||||||
|
"selection_run",
|
||||||
|
timestamped=False,
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(path.name, "selection_run")
|
||||||
|
|
||||||
|
def test_model_selection_marks_accuracy_as_manual_without_gt(self) -> None:
|
||||||
|
summary = {
|
||||||
|
"parse_success_rate": 1.0,
|
||||||
|
"mean_total_ms": 100.0,
|
||||||
|
"p50_total_ms": 99.0,
|
||||||
|
"p95_total_ms": 110.0,
|
||||||
|
"peak_cuda_memory_mb": 1000.0,
|
||||||
|
"throughput_samples_per_sec": 10.0,
|
||||||
|
}
|
||||||
|
output = io.StringIO()
|
||||||
|
with redirect_stdout(output):
|
||||||
|
print_model_selection({"model_a": summary, "model_b": summary})
|
||||||
|
|
||||||
|
printed = output.getvalue()
|
||||||
|
self.assertIn("MODEL SELECTION", printed)
|
||||||
|
self.assertIn("N/A - manual review required", printed)
|
||||||
|
accuracy_row = next(
|
||||||
|
line for line in printed.splitlines() if line.startswith("Text UI Accuracy")
|
||||||
|
)
|
||||||
|
self.assertEqual(accuracy_row.count("N/A"), 2)
|
||||||
89
tests/test_benchmark_statistics.py
Normal file
89
tests/test_benchmark_statistics.py
Normal file
@ -0,0 +1,89 @@
|
|||||||
|
import unittest
|
||||||
|
from pathlib import Path
|
||||||
|
from tempfile import TemporaryDirectory
|
||||||
|
|
||||||
|
from cockpit_grounding.benchmark.config import BenchmarkSample
|
||||||
|
from cockpit_grounding.benchmark.runner import _benchmark_sample
|
||||||
|
from cockpit_grounding.benchmark.statistics import (
|
||||||
|
percentile,
|
||||||
|
summarize_model,
|
||||||
|
summarize_timings,
|
||||||
|
)
|
||||||
|
from cockpit_grounding.models.base import InferenceTiming
|
||||||
|
|
||||||
|
|
||||||
|
def _timing(total: float) -> InferenceTiming:
|
||||||
|
return InferenceTiming(
|
||||||
|
preprocess_ms=1.0,
|
||||||
|
generate_ms=total - 2.0,
|
||||||
|
decode_ms=1.0,
|
||||||
|
total_ms=total,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class BenchmarkStatisticsTest(unittest.TestCase):
|
||||||
|
def test_percentile_uses_linear_interpolation(self) -> None:
|
||||||
|
self.assertEqual(percentile([10.0, 20.0, 30.0], 0.5), 20.0)
|
||||||
|
self.assertAlmostEqual(percentile([10.0, 20.0], 0.95), 19.5)
|
||||||
|
|
||||||
|
def test_summarize_timings_handles_empty_input(self) -> None:
|
||||||
|
self.assertIsNone(summarize_timings([])["total_mean_ms"])
|
||||||
|
|
||||||
|
def test_summarize_model_without_gt_uses_null_accuracy(self) -> None:
|
||||||
|
predictions = [
|
||||||
|
{
|
||||||
|
"parse_success": True,
|
||||||
|
"point_in_box": None,
|
||||||
|
"bbox_iou": None,
|
||||||
|
"normalized_center_error": None,
|
||||||
|
"peak_cuda_memory_mb": 123.0,
|
||||||
|
}
|
||||||
|
]
|
||||||
|
summary = summarize_model(
|
||||||
|
model_name="model",
|
||||||
|
model_path="/model",
|
||||||
|
model_load_seconds=2.0,
|
||||||
|
memory_metrics={
|
||||||
|
"model_cuda_allocated_mb": 100.0,
|
||||||
|
"model_cuda_reserved_mb": 120.0,
|
||||||
|
},
|
||||||
|
predictions=predictions,
|
||||||
|
all_timings=[_timing(10.0), _timing(20.0)],
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(summary["parse_success_rate"], 1.0)
|
||||||
|
self.assertEqual(summary["mean_total_ms"], 15.0)
|
||||||
|
self.assertAlmostEqual(summary["throughput_samples_per_sec"], 1000 / 15)
|
||||||
|
self.assertEqual(summary["peak_cuda_memory_mb"], 123.0)
|
||||||
|
self.assertIsNone(summary["accuracy_point_in_box"])
|
||||||
|
|
||||||
|
def test_parse_failure_is_returned_instead_of_raised(self) -> None:
|
||||||
|
class InvalidOutputGrounder:
|
||||||
|
def generate_with_metrics(self, **_: object):
|
||||||
|
return (
|
||||||
|
"not a bbox",
|
||||||
|
_timing(10.0),
|
||||||
|
{"peak_cuda_memory_mb": 100.0},
|
||||||
|
)
|
||||||
|
|
||||||
|
with TemporaryDirectory() as directory:
|
||||||
|
tmp_path = Path(directory)
|
||||||
|
image = tmp_path / "image.jpg"
|
||||||
|
image.touch()
|
||||||
|
prediction, timings = _benchmark_sample(
|
||||||
|
grounder=InvalidOutputGrounder(), # type: ignore[arg-type]
|
||||||
|
model_name="model",
|
||||||
|
sample=BenchmarkSample(
|
||||||
|
sample_id="sample",
|
||||||
|
image=image,
|
||||||
|
target="button",
|
||||||
|
),
|
||||||
|
repeats=3,
|
||||||
|
max_new_tokens=128,
|
||||||
|
visualization_path=tmp_path / "visualization.jpg",
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertFalse(prediction["parse_success"])
|
||||||
|
self.assertIn("ValueError", prediction["parse_error"])
|
||||||
|
self.assertEqual(len(timings), 3)
|
||||||
|
self.assertEqual(prediction["total_mean_ms"], 10.0)
|
||||||
33
tests/test_metrics.py
Normal file
33
tests/test_metrics.py
Normal file
@ -0,0 +1,33 @@
|
|||||||
|
import unittest
|
||||||
|
|
||||||
|
from cockpit_grounding.benchmark.metrics import (
|
||||||
|
bbox_iou,
|
||||||
|
normalized_center_error,
|
||||||
|
point_in_box,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class MetricsTest(unittest.TestCase):
|
||||||
|
def test_point_in_box_includes_boundary(self) -> None:
|
||||||
|
self.assertTrue(point_in_box((10, 20), (10, 20, 30, 40)))
|
||||||
|
self.assertFalse(point_in_box((9, 20), (10, 20, 30, 40)))
|
||||||
|
|
||||||
|
def test_bbox_iou(self) -> None:
|
||||||
|
self.assertAlmostEqual(
|
||||||
|
bbox_iou((0, 0, 10, 10), (5, 5, 15, 15)),
|
||||||
|
25 / 175,
|
||||||
|
)
|
||||||
|
self.assertEqual(bbox_iou((0, 0, 2, 2), (3, 3, 4, 4)), 0.0)
|
||||||
|
|
||||||
|
def test_normalized_center_error_uses_image_diagonal(self) -> None:
|
||||||
|
value = normalized_center_error(
|
||||||
|
pred_center=(5, 5),
|
||||||
|
gt_bbox=(0, 0, 0, 0),
|
||||||
|
image_width=10,
|
||||||
|
image_height=10,
|
||||||
|
)
|
||||||
|
self.assertAlmostEqual(value, 0.5)
|
||||||
|
|
||||||
|
def test_metrics_reject_invalid_bbox(self) -> None:
|
||||||
|
with self.assertRaisesRegex(ValueError, "satisfy"):
|
||||||
|
bbox_iou((10, 0, 0, 10), (0, 0, 10, 10))
|
||||||
9
tests/test_model_factory.py
Normal file
9
tests/test_model_factory.py
Normal file
@ -0,0 +1,9 @@
|
|||||||
|
import unittest
|
||||||
|
|
||||||
|
from cockpit_grounding.models.factory import create_grounder
|
||||||
|
|
||||||
|
|
||||||
|
class ModelFactoryTest(unittest.TestCase):
|
||||||
|
def test_rejects_unknown_backend(self) -> None:
|
||||||
|
with self.assertRaisesRegex(ValueError, "Unsupported model backend"):
|
||||||
|
create_grounder("unknown", "/model")
|
||||||
Loading…
Reference in New Issue
Block a user