- 移除Ollama相关配置和回退机制,统一使用SGLang作为视觉模型提供商 - 添加IMAGE_ANALYSIS类型支持,允许对多张图片进行分析 - 实现视频分析的详细输出模式,支持紧凑和详细两种结果格式 - 更新环境变量配置,添加VIDEO_MODEL_FRAME_LIMIT和MAX_IMAGE_BYTES - 修改compose配置文件中的上下文长度和内存分配参数 - 重构视频采样逻辑,限制单次请求帧数以优化显存使用 - 更新API接口文档,添加mediaUrls参数和详细输出选项说明 - 添加图像分析相关的依赖库opencv-python-headless - 实现结构化JSON响应格式验证和重试机制
466 lines
14 KiB
Python
466 lines
14 KiB
Python
import asyncio
|
|
|
|
import httpx
|
|
|
|
from app import video
|
|
|
|
|
|
class _FakeResponse:
|
|
def __init__(self, body):
|
|
self.body = body
|
|
|
|
def raise_for_status(self):
|
|
return None
|
|
|
|
def json(self):
|
|
return self.body
|
|
|
|
|
|
class _FakeClient:
|
|
responses = []
|
|
payloads = []
|
|
|
|
def __init__(self, **_):
|
|
pass
|
|
|
|
async def __aenter__(self):
|
|
return self
|
|
|
|
async def __aexit__(self, *_):
|
|
return None
|
|
|
|
async def post(self, _, json, headers=None):
|
|
self.payloads.append(json)
|
|
return _FakeResponse(self.responses.pop(0))
|
|
|
|
|
|
def test_vision_request_retries_empty_content(monkeypatch):
|
|
_FakeClient.responses = [
|
|
{
|
|
"choices": [{
|
|
"finish_reason": "length",
|
|
"message": {"content": "", "reasoning_content": "reasoning"},
|
|
}],
|
|
"usage": {"completion_tokens": 1024},
|
|
},
|
|
{
|
|
"choices": [{
|
|
"finish_reason": "stop",
|
|
"message": {"content": '{"passed": true}', "reasoning_content": ""},
|
|
}],
|
|
"usage": {"completion_tokens": 20},
|
|
},
|
|
]
|
|
_FakeClient.payloads = []
|
|
monkeypatch.setattr(httpx, "AsyncClient", _FakeClient)
|
|
payload = {
|
|
"model": "Qwen/Qwen3.8-27B-FP8",
|
|
"think": False,
|
|
"messages": [{"role": "user", "content": "Analyze"}],
|
|
"options": {"num_predict": 1024},
|
|
}
|
|
|
|
content, metadata = asyncio.run(video._request_vision_model(
|
|
payload, base_url="http://sglang-fast:30000/v1"
|
|
))
|
|
|
|
assert content == '{"passed": true}'
|
|
assert metadata["doneReason"] == "stop"
|
|
assert len(_FakeClient.payloads) == 2
|
|
assert _FakeClient.payloads[0]["chat_template_kwargs"]["enable_thinking"] is False
|
|
assert _FakeClient.payloads[1]["max_tokens"] == 4096
|
|
assert _FakeClient.payloads[1]["chat_template_kwargs"]["enable_thinking"] is False
|
|
assert "/no_think" in _FakeClient.payloads[1]["messages"][0]["content"][0]["text"]
|
|
assert metadata["requestedOutputTokens"] == 4096
|
|
|
|
|
|
def test_vision_request_escalates_budget_after_repeated_length_cutoff(monkeypatch):
|
|
_FakeClient.responses = [
|
|
{
|
|
"choices": [{
|
|
"finish_reason": "length",
|
|
"message": {"content": "", "reasoning_content": "first reasoning"},
|
|
}],
|
|
"usage": {"completion_tokens": 768},
|
|
},
|
|
{
|
|
"choices": [{
|
|
"finish_reason": "length",
|
|
"message": {"content": "", "reasoning_content": "more reasoning"},
|
|
}],
|
|
"usage": {"completion_tokens": 4096},
|
|
},
|
|
{
|
|
"choices": [{
|
|
"finish_reason": "stop",
|
|
"message": {"content": '{"passed": true}', "reasoning_content": ""},
|
|
}],
|
|
"usage": {"completion_tokens": 30},
|
|
},
|
|
]
|
|
_FakeClient.payloads = []
|
|
monkeypatch.setattr(httpx, "AsyncClient", _FakeClient)
|
|
payload = {
|
|
"model": "Qwen/Qwen3.8-27B-FP8",
|
|
"think": False,
|
|
"messages": [{"role": "user", "content": "Analyze /no_think"}],
|
|
"options": {"num_predict": 768},
|
|
}
|
|
|
|
content, metadata = asyncio.run(
|
|
video._request_vision_model(
|
|
payload, empty_response_retries=2,
|
|
base_url="http://sglang-fast:30000/v1",
|
|
)
|
|
)
|
|
|
|
assert content == '{"passed": true}'
|
|
assert [_payload["max_tokens"] for _payload in _FakeClient.payloads] == [
|
|
768,
|
|
4096,
|
|
8192,
|
|
]
|
|
assert metadata["attempt"] == 3
|
|
|
|
|
|
def test_vision_request_retries_truncated_structured_content(monkeypatch):
|
|
_FakeClient.responses = [
|
|
{
|
|
"choices": [{
|
|
"finish_reason": "length",
|
|
"message": {"content": '{"passed": true, "events": [', "reasoning_content": ""},
|
|
}],
|
|
"usage": {"completion_tokens": 512},
|
|
},
|
|
{
|
|
"choices": [{
|
|
"finish_reason": "stop",
|
|
"message": {"content": '{"passed": true}', "reasoning_content": ""},
|
|
}],
|
|
"usage": {"completion_tokens": 18},
|
|
},
|
|
]
|
|
_FakeClient.payloads = []
|
|
monkeypatch.setattr(httpx, "AsyncClient", _FakeClient)
|
|
payload = {
|
|
"model": "Qwen/Qwen3.8-27B-FP8",
|
|
"think": False,
|
|
"format": video.VIDEO_RESULT_SCHEMA,
|
|
"messages": [{"role": "user", "content": "Analyze"}],
|
|
"options": {"num_predict": 512},
|
|
}
|
|
|
|
content, metadata = asyncio.run(video._request_vision_model(
|
|
payload,
|
|
empty_response_retries=1,
|
|
base_url="http://sglang-fast:30000/v1",
|
|
))
|
|
|
|
assert content == '{"passed": true}'
|
|
assert len(_FakeClient.payloads) == 2
|
|
assert _FakeClient.payloads[1]["max_tokens"] == 4096
|
|
assert metadata["attempt"] == 2
|
|
|
|
|
|
def test_sglang_payload_uses_openai_multimodal_format():
|
|
payload = video._sglang_payload(
|
|
{
|
|
"model": "Qwen/Qwen3.8-27B-FP8",
|
|
"think": False,
|
|
"messages": [
|
|
{"role": "user", "content": "Analyze", "images": ["abc", "def"]}
|
|
],
|
|
"options": {"temperature": 0.1, "num_predict": 512},
|
|
"format": video.VIDEO_RESULT_SCHEMA,
|
|
}
|
|
)
|
|
|
|
assert payload["model"] == "Qwen/Qwen3.8-27B-FP8"
|
|
assert payload["messages"][0]["content"][0] == {
|
|
"type": "text",
|
|
"text": "Analyze",
|
|
}
|
|
assert payload["messages"][0]["content"][1]["image_url"]["url"] == (
|
|
"data:image/jpeg;base64,abc"
|
|
)
|
|
assert payload["chat_template_kwargs"]["enable_thinking"] is False
|
|
assert payload["response_format"]["json_schema"]["schema"] == video.VIDEO_RESULT_SCHEMA
|
|
|
|
|
|
def test_sglang_response_is_parsed(monkeypatch):
|
|
_FakeClient.responses = [
|
|
{
|
|
"choices": [
|
|
{
|
|
"finish_reason": "stop",
|
|
"message": {"content": '{"passed": true}', "reasoning_content": ""},
|
|
}
|
|
],
|
|
"usage": {"prompt_tokens": 100, "completion_tokens": 20},
|
|
}
|
|
]
|
|
_FakeClient.payloads = []
|
|
monkeypatch.setattr(httpx, "AsyncClient", _FakeClient)
|
|
payload = {
|
|
"model": "Qwen/Qwen3.8-27B-FP8",
|
|
"think": False,
|
|
"messages": [{"role": "user", "content": "Analyze", "images": ["abc"]}],
|
|
"options": {"temperature": 0.1, "num_predict": 512},
|
|
}
|
|
|
|
content, metadata = asyncio.run(
|
|
video._request_vision_model(
|
|
payload,
|
|
provider="sglang",
|
|
base_url="http://sglang-fast:30000/v1",
|
|
)
|
|
)
|
|
|
|
assert content == '{"passed": true}'
|
|
assert metadata["provider"] == "sglang"
|
|
assert metadata["promptEvalCount"] == 100
|
|
assert metadata["evalCount"] == 20
|
|
assert _FakeClient.payloads[0]["messages"][0]["content"][1]["type"] == "image_url"
|
|
|
|
|
|
def test_fast_result_quality_controls_accurate_fallback():
|
|
complete = {
|
|
"passed": True,
|
|
"evidenceSufficient": True,
|
|
"confidence": 0.9,
|
|
"conclusion": "通过",
|
|
"summary": "状态正常",
|
|
"events": [],
|
|
"warnings": [],
|
|
}
|
|
|
|
assert video._fallback_reason(complete, 0.75) is None
|
|
assert (
|
|
video._fallback_reason({**complete, "evidenceSufficient": False}, 0.75)
|
|
== "FAST_RESULT_INSUFFICIENT"
|
|
)
|
|
assert (
|
|
video._fallback_reason({**complete, "confidence": 0.6}, 0.75)
|
|
== "FAST_RESULT_LOW_CONFIDENCE"
|
|
)
|
|
assert video._fallback_reason({"summary": "缺少字段"}, 0.75).startswith(
|
|
"FAST_RESULT_MISSING_FIELDS:"
|
|
)
|
|
|
|
|
|
def test_compact_result_does_not_require_event_evidence_arrays():
|
|
compact = {
|
|
"passed": True,
|
|
"evidenceSufficient": True,
|
|
"confidence": 0.9,
|
|
"conclusion": "通过",
|
|
"summary": "状态变化符合标准",
|
|
}
|
|
|
|
assert video._fallback_reason(compact, 0.75, detailed_output=False) is None
|
|
assert video._fallback_reason(compact, 0.75, detailed_output=True).startswith(
|
|
"FAST_RESULT_MISSING_FIELDS:"
|
|
)
|
|
|
|
|
|
def test_decision_is_fail_closed_when_evidence_is_insufficient():
|
|
result = video._normalize_decision(
|
|
{
|
|
"passed": True,
|
|
"evidenceSufficient": False,
|
|
"confidence": 0.9,
|
|
"conclusion": "可能通过",
|
|
"summary": "没有拍到完整过程",
|
|
"events": [],
|
|
"warnings": [],
|
|
},
|
|
0.75,
|
|
)
|
|
|
|
assert result["passed"] is False
|
|
assert result["decisionReason"] == "INSUFFICIENT_EVIDENCE"
|
|
assert result["conclusion"].startswith("未通过:视频证据不足")
|
|
assert result["result"] == "没有拍到完整过程"
|
|
|
|
|
|
def test_negative_acceptance_criterion_corrects_inconsistent_model_boolean():
|
|
result = video._normalize_decision(
|
|
{
|
|
"passed": False,
|
|
"evidenceSufficient": True,
|
|
"confidence": 1.0,
|
|
"conclusion": "蓝牙图标从有到无",
|
|
"summary": "随着视频推进,蓝牙图标消失,画面内不再显示图标。",
|
|
"events": [
|
|
{"timeRange": "0-3s", "event": "蓝牙图标存在", "confidence": 1.0},
|
|
{"timeRange": "3-6s", "event": "蓝牙图标消失", "confidence": 1.0},
|
|
],
|
|
"warnings": [],
|
|
},
|
|
0.7,
|
|
"仪表蓝牙图标是否消失",
|
|
)
|
|
|
|
assert result["passed"] is True
|
|
assert result["modelPassed"] is False
|
|
assert result["decisionReason"] == "CRITERION_EVIDENCE_CORRECTION"
|
|
|
|
|
|
def test_negative_acceptance_criterion_does_not_pass_when_state_remains():
|
|
result = video._normalize_decision(
|
|
{
|
|
"passed": True,
|
|
"evidenceSufficient": True,
|
|
"confidence": 0.95,
|
|
"conclusion": "蓝牙图标未消失",
|
|
"summary": "视频结束时蓝牙图标仍然显示。",
|
|
"events": [],
|
|
"warnings": [],
|
|
},
|
|
0.7,
|
|
"仪表蓝牙图标是否消失",
|
|
)
|
|
|
|
assert result["passed"] is False
|
|
assert result["modelPassed"] is True
|
|
assert result["decisionReason"] == "CRITERION_EVIDENCE_CORRECTION"
|
|
assert result["conclusion"].startswith("未通过:分析结论未满足通过标准")
|
|
|
|
|
|
def test_disappearance_guidance_requires_first_and_final_frame_comparison():
|
|
guidance = video._temporal_acceptance_guidance("仪表蓝牙图标是否消失")
|
|
|
|
assert "第一张与最后一张" in guidance
|
|
assert "passed必须为true" in guidance
|
|
assert "末段图片中已不再显示仪表蓝牙图标" in guidance
|
|
assert "全程持续存在" in guidance
|
|
|
|
assert video._needs_temporal_removal_verification(
|
|
"仪表蓝牙图标是否消失",
|
|
{"passed": False, "evidenceSufficient": True},
|
|
) is True
|
|
assert video._needs_temporal_removal_verification(
|
|
"仪表蓝牙图标是否出现",
|
|
{"passed": False, "evidenceSufficient": True},
|
|
) is False
|
|
|
|
|
|
def test_video_tuning_is_bounded_and_typed():
|
|
assert video._normalize_tuning(
|
|
{
|
|
"sampleFps": "2",
|
|
"maxWidth": 960,
|
|
"confidenceThreshold": 0.8,
|
|
"fallbackToAccurate": False,
|
|
}
|
|
) == {
|
|
"sampleFps": 2.0,
|
|
"maxWidth": 960,
|
|
"confidenceThreshold": 0.8,
|
|
"fallbackToAccurate": False,
|
|
}
|
|
|
|
try:
|
|
video._normalize_tuning({"sampleFps": 6.5})
|
|
assert False, "out-of-range tuning must fail"
|
|
except ValueError:
|
|
pass
|
|
|
|
|
|
def test_frame_sampling_always_includes_video_tail():
|
|
count, prefix_count, prefix_fps, tail_time, capped = video._frame_sampling_plan(
|
|
6.48, 1.0, 100
|
|
)
|
|
|
|
assert count == 8
|
|
assert prefix_count == 7
|
|
assert prefix_fps == 1.0
|
|
assert round(tail_time, 2) == 6.38
|
|
assert capped is False
|
|
|
|
count, prefix_count, prefix_fps, tail_time, capped = video._frame_sampling_plan(
|
|
20.0, 6.0, 100
|
|
)
|
|
assert count == 100
|
|
assert prefix_count == 99
|
|
assert round(prefix_fps, 2) == 4.95
|
|
assert round(tail_time, 1) == 19.9
|
|
assert capped is True
|
|
|
|
|
|
def test_video_context_is_capped_to_safe_model_budget():
|
|
assert video._video_context_size(16384, 7, 512) == 16384
|
|
assert video._video_context_size(16384, 18, 1536) == 32768
|
|
assert video._video_context_size(16384, 27, 512) == 65536
|
|
assert video._video_context_size(32768, 100, 768) == 65536
|
|
|
|
|
|
def test_model_frame_limit_preserves_global_limit(monkeypatch):
|
|
monkeypatch.setattr(video.settings, "video_sampling_frame_limit", 100)
|
|
monkeypatch.setattr(video.settings, "video_model_frame_limit", 30)
|
|
assert video._model_frame_limit() == 30
|
|
|
|
count, _, _, _, capped = video._frame_sampling_plan(20.0, 3.0, 30)
|
|
assert count == 30
|
|
assert capped is True
|
|
|
|
|
|
def test_auto_mode_falls_back_to_accurate_result(monkeypatch, tmp_path):
|
|
frame_dir = tmp_path / "frames"
|
|
frame_dir.mkdir()
|
|
frame = frame_dir / "frame-001.jpg"
|
|
frame.write_bytes(b"frame")
|
|
calls = []
|
|
|
|
monkeypatch.setattr(
|
|
video,
|
|
"_extract_frames",
|
|
lambda *_: ([frame], 2.0, frame_dir),
|
|
)
|
|
|
|
async def fake_analyze(_, mode, *args):
|
|
calls.append(mode)
|
|
if mode == "FAST":
|
|
return (
|
|
{
|
|
"passed": False,
|
|
"evidenceSufficient": False,
|
|
"confidence": 0.4,
|
|
"conclusion": "证据不足",
|
|
"summary": "快速分析无法判断",
|
|
"events": [],
|
|
"warnings": [],
|
|
},
|
|
{"mode": "FAST", "model": "fast", "sampledFrameCount": 1, "maximumWidth": 896},
|
|
)
|
|
return (
|
|
{
|
|
"passed": True,
|
|
"evidenceSufficient": True,
|
|
"confidence": 0.92,
|
|
"conclusion": "通过",
|
|
"summary": "精确分析确认通过",
|
|
"events": [],
|
|
"warnings": [],
|
|
},
|
|
{"mode": "ACCURATE", "model": "accurate", "sampledFrameCount": 1, "maximumWidth": 1280},
|
|
)
|
|
|
|
monkeypatch.setattr(video, "_analyze_frames", fake_analyze)
|
|
profile_dir = tmp_path / "profile"
|
|
profile_dir.mkdir()
|
|
(profile_dir / "prompt.txt").write_text("Analyze", encoding="utf-8")
|
|
profile = video.Profile(
|
|
"test.video.v1",
|
|
"VIDEO_ANALYSIS",
|
|
profile_dir,
|
|
{"fallbackWhenPassedUnknown": True},
|
|
)
|
|
|
|
result, _, metadata = asyncio.run(video.analyze_video(profile, tmp_path / "video.mp4"))
|
|
|
|
assert calls == ["FAST", "ACCURATE"]
|
|
assert result["passed"] is True
|
|
assert metadata["fallback"] is True
|
|
assert metadata["fallbackReason"] == "FAST_RESULT_INSUFFICIENT"
|