76 lines
2.0 KiB
Python
76 lines
2.0 KiB
Python
import argparse
|
|
import logging
|
|
from pathlib import Path
|
|
|
|
from cockpit_grounding.benchmark.config import (
|
|
load_benchmark_config,
|
|
load_manifest,
|
|
)
|
|
from cockpit_grounding.benchmark.reporting import (
|
|
print_comparison,
|
|
print_model_selection,
|
|
)
|
|
from cockpit_grounding.benchmark.runner import run_benchmark
|
|
|
|
|
|
def parse_args() -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(
|
|
description="Benchmark local multimodal models sequentially on one GPU",
|
|
)
|
|
parser.add_argument(
|
|
"--config",
|
|
type=Path,
|
|
required=True,
|
|
help="Benchmark TOML configuration",
|
|
)
|
|
parser.add_argument(
|
|
"--manifest",
|
|
type=Path,
|
|
required=True,
|
|
help="JSONL benchmark manifest",
|
|
)
|
|
parser.add_argument(
|
|
"--run-name",
|
|
help="Optional run directory suffix",
|
|
)
|
|
return parser.parse_args()
|
|
|
|
|
|
def main() -> None:
|
|
args = parse_args()
|
|
logging.basicConfig(
|
|
level=logging.INFO,
|
|
format="%(asctime)s | %(levelname)s | %(message)s",
|
|
)
|
|
|
|
config = load_benchmark_config(args.config)
|
|
samples = load_manifest(args.manifest)
|
|
run_directory, summaries = run_benchmark(
|
|
config=config,
|
|
samples=samples,
|
|
config_path=args.config,
|
|
manifest_path=args.manifest,
|
|
run_name=args.run_name,
|
|
)
|
|
comparison_title = (
|
|
"QWEN3.5 MODEL COMPARISON"
|
|
if all(model.backend == "qwen35" for model in config.models)
|
|
else "MODEL COMPARISON"
|
|
)
|
|
models_by_name = {model.name: model for model in config.models}
|
|
display_summaries = {
|
|
(
|
|
Path(models_by_name[name].path).name
|
|
if models_by_name[name].backend == "qwen35"
|
|
else name
|
|
): summary
|
|
for name, summary in summaries.items()
|
|
}
|
|
print_comparison(display_summaries, title=comparison_title)
|
|
print_model_selection(summaries)
|
|
print(f"Results: {run_directory}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|