CMVR-AI-ANALYSIS/tests/test_audio_classifier.py
lixiaolong b386f003a0 feat(vision): 集成Qwen3.8视觉模型并优化音视频分析功能
- 集成Qwen3.8-27B-FP8快速模型和Qwen3.8-27B精确模型作为SGLang服务
- 添加SGLang API配置选项(SGLANG_FAST_BASE_URL、SGLANG_ACCURATE_BASE_URL等)
- 实现音频分类中的决策聚合算法(topK、nearestWeight、labelMaxDistance)
- 添加视频采样帧限制(VIDEO_SAMPLING_FRAME_LIMIT)和上下文token限制
- 更新健康检查以监控SGLang服务状态
- 实现视频分析的双模式决策策略(快速+精确)
- 添加音频参考文件导入工具(import_audio_references.py)
- 扩展音频分类标签支持FIND_VEHICLE_HORN类别
- 优化视频分析的帧采样策略,始终包含视频尾部帧
- 添加决策策略参数(tuning、decisionPolicy)支持
- 更新配置类以支持新的SGLang和视频参数
- 修改compose配置以支持Qwen3.8模型部署
- 更新音频分类测试用例验证聚合逻辑
- 重构视频测试以支持SGLang API格式和决策策略
2026-08-19 09:28:53 +08:00

125 lines
4.3 KiB
Python

import json
import wave
from pathlib import Path
import numpy as np
from app.audio import build_audio_profile, classify_audio, classify_sequence
from app.config import settings
from app.profile_store import Profile
def _write_tone(path: Path, frequency: float, duration: float = 1.2) -> None:
sample_rate = 16000
time = np.arange(int(sample_rate * duration)) / sample_rate
envelope = np.minimum(1.0, time * 8) * np.minimum(1.0, (duration - time) * 8)
samples = 0.7 * np.sin(2 * np.pi * frequency * time) * envelope
pcm = (samples * 32767).astype("<i2")
path.parent.mkdir(parents=True, exist_ok=True)
with wave.open(str(path), "wb") as output:
output.setnchannels(1)
output.setsampwidth(2)
output.setframerate(sample_rate)
output.writeframes(pcm.tobytes())
def test_audio_profile_classifies_and_rejects(tmp_path: Path) -> None:
profiles = tmp_path / "profiles"
artifacts = tmp_path / "artifacts"
settings.profiles_dir = profiles
settings.artifacts_dir = artifacts
directory = profiles / "test" / "tones" / "v1"
config = {
"code": "test.tones.v1",
"analysisType": "AUDIO_CLASSIFICATION",
"decision": {
"maxDistance": 0.45,
"minDistanceMargin": 0.02,
"unknownLabel": "UNKNOWN",
},
}
directory.mkdir(parents=True)
(directory / "profile.json").write_text(json.dumps(config), encoding="utf-8")
reference_tones = {
"POWER_ON": 440,
"POWER_OFF": 660,
"ARMED": 880,
"DISARMED": 1100,
"FIND_VEHICLE_HORN": 1320,
}
for label, frequency in reference_tones.items():
_write_tone(directory / "references" / label / f"{label.lower()}.wav", frequency)
profile = Profile("test.tones.v1", "AUDIO_CLASSIFICATION", directory, config)
build_audio_profile(profile)
classified_paths = {}
for label, frequency in reference_tones.items():
path = tmp_path / f"{label.lower()}-test.wav"
_write_tone(path, frequency)
classified_paths[label] = path
other_path = tmp_path / "other-test.wav"
_write_tone(other_path, 1600)
for label, path in classified_paths.items():
assert classify_audio(profile, path)["label"] == label
assert classify_audio(profile, other_path)["label"] == "UNKNOWN"
def test_audio_profile_aggregates_multiple_references(monkeypatch) -> None:
config = {
"decision": {
"maxDistance": 0.45,
"minDistanceMargin": 0.001,
"topK": 3,
"nearestWeight": 0.6,
"unknownLabel": "UNKNOWN",
}
}
profile = Profile("test.aggregate.v1", "AUDIO_CLASSIFICATION", Path("."), config)
references = [np.array([[value]], dtype=np.float32) for value in range(1, 7)]
distances = {1: 0.10, 2: 0.11, 3: 0.12, 4: 0.0, 5: 0.40, 6: 0.40}
monkeypatch.setattr(
"app.audio.sequence_distance",
lambda _query, reference: distances[int(reference[0, 0])],
)
result = classify_sequence(
profile,
np.array([[0.0]], dtype=np.float32),
references,
np.array(["EXPECTED"] * 3 + ["OUTLIER"] * 3),
np.array([f"sample-{index}.wav" for index in range(6)]),
)
assert result["label"] == "EXPECTED"
assert result["evidence"]["neighborCount"] == 3
assert result["thresholds"]["topK"] == 3
def test_audio_profile_applies_label_specific_threshold(monkeypatch) -> None:
config = {
"decision": {
"maxDistance": 0.45,
"minDistanceMargin": 0.015,
"labelMaxDistance": {"FIND_VEHICLE_HORN": 0.35},
"unknownLabel": "UNKNOWN",
}
}
profile = Profile("test.threshold.v1", "AUDIO_CLASSIFICATION", Path("."), config)
references = [np.array([[1.0]], dtype=np.float32), np.array([[2.0]], dtype=np.float32)]
monkeypatch.setattr(
"app.audio.sequence_distance",
lambda _query, reference: 0.36 if int(reference[0, 0]) == 1 else 0.44,
)
result = classify_sequence(
profile,
np.array([[0.0]], dtype=np.float32),
references,
np.array(["FIND_VEHICLE_HORN", "POWER_OFF"]),
np.array(["horn.wav", "power-off.wav"]),
)
assert result["label"] == "UNKNOWN"
assert result["thresholds"]["maxDistance"] == 0.35