2026-08-13 16:56:32 +08:00
|
|
|
import json
|
|
|
|
|
import wave
|
|
|
|
|
from pathlib import Path
|
|
|
|
|
|
|
|
|
|
import numpy as np
|
|
|
|
|
|
2026-08-19 09:28:53 +08:00
|
|
|
from app.audio import build_audio_profile, classify_audio, classify_sequence
|
2026-08-13 16:56:32 +08:00
|
|
|
from app.config import settings
|
|
|
|
|
from app.profile_store import Profile
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _write_tone(path: Path, frequency: float, duration: float = 1.2) -> None:
|
|
|
|
|
sample_rate = 16000
|
|
|
|
|
time = np.arange(int(sample_rate * duration)) / sample_rate
|
|
|
|
|
envelope = np.minimum(1.0, time * 8) * np.minimum(1.0, (duration - time) * 8)
|
|
|
|
|
samples = 0.7 * np.sin(2 * np.pi * frequency * time) * envelope
|
|
|
|
|
pcm = (samples * 32767).astype("<i2")
|
|
|
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
|
|
|
with wave.open(str(path), "wb") as output:
|
|
|
|
|
output.setnchannels(1)
|
|
|
|
|
output.setsampwidth(2)
|
|
|
|
|
output.setframerate(sample_rate)
|
|
|
|
|
output.writeframes(pcm.tobytes())
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_audio_profile_classifies_and_rejects(tmp_path: Path) -> None:
|
|
|
|
|
profiles = tmp_path / "profiles"
|
|
|
|
|
artifacts = tmp_path / "artifacts"
|
|
|
|
|
settings.profiles_dir = profiles
|
|
|
|
|
settings.artifacts_dir = artifacts
|
|
|
|
|
directory = profiles / "test" / "tones" / "v1"
|
|
|
|
|
config = {
|
|
|
|
|
"code": "test.tones.v1",
|
|
|
|
|
"analysisType": "AUDIO_CLASSIFICATION",
|
|
|
|
|
"decision": {
|
|
|
|
|
"maxDistance": 0.45,
|
|
|
|
|
"minDistanceMargin": 0.02,
|
|
|
|
|
"unknownLabel": "UNKNOWN",
|
|
|
|
|
},
|
|
|
|
|
}
|
|
|
|
|
directory.mkdir(parents=True)
|
|
|
|
|
(directory / "profile.json").write_text(json.dumps(config), encoding="utf-8")
|
|
|
|
|
reference_tones = {
|
|
|
|
|
"POWER_ON": 440,
|
|
|
|
|
"POWER_OFF": 660,
|
|
|
|
|
"ARMED": 880,
|
|
|
|
|
"DISARMED": 1100,
|
2026-08-19 09:28:53 +08:00
|
|
|
"FIND_VEHICLE_HORN": 1320,
|
2026-08-13 16:56:32 +08:00
|
|
|
}
|
|
|
|
|
for label, frequency in reference_tones.items():
|
|
|
|
|
_write_tone(directory / "references" / label / f"{label.lower()}.wav", frequency)
|
|
|
|
|
profile = Profile("test.tones.v1", "AUDIO_CLASSIFICATION", directory, config)
|
|
|
|
|
|
|
|
|
|
build_audio_profile(profile)
|
|
|
|
|
classified_paths = {}
|
|
|
|
|
for label, frequency in reference_tones.items():
|
|
|
|
|
path = tmp_path / f"{label.lower()}-test.wav"
|
|
|
|
|
_write_tone(path, frequency)
|
|
|
|
|
classified_paths[label] = path
|
|
|
|
|
other_path = tmp_path / "other-test.wav"
|
|
|
|
|
_write_tone(other_path, 1600)
|
|
|
|
|
|
|
|
|
|
for label, path in classified_paths.items():
|
|
|
|
|
assert classify_audio(profile, path)["label"] == label
|
|
|
|
|
assert classify_audio(profile, other_path)["label"] == "UNKNOWN"
|
2026-08-19 09:28:53 +08:00
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_audio_profile_aggregates_multiple_references(monkeypatch) -> None:
|
|
|
|
|
config = {
|
|
|
|
|
"decision": {
|
|
|
|
|
"maxDistance": 0.45,
|
|
|
|
|
"minDistanceMargin": 0.001,
|
|
|
|
|
"topK": 3,
|
|
|
|
|
"nearestWeight": 0.6,
|
|
|
|
|
"unknownLabel": "UNKNOWN",
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
profile = Profile("test.aggregate.v1", "AUDIO_CLASSIFICATION", Path("."), config)
|
|
|
|
|
references = [np.array([[value]], dtype=np.float32) for value in range(1, 7)]
|
|
|
|
|
distances = {1: 0.10, 2: 0.11, 3: 0.12, 4: 0.0, 5: 0.40, 6: 0.40}
|
|
|
|
|
monkeypatch.setattr(
|
|
|
|
|
"app.audio.sequence_distance",
|
|
|
|
|
lambda _query, reference: distances[int(reference[0, 0])],
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
result = classify_sequence(
|
|
|
|
|
profile,
|
|
|
|
|
np.array([[0.0]], dtype=np.float32),
|
|
|
|
|
references,
|
|
|
|
|
np.array(["EXPECTED"] * 3 + ["OUTLIER"] * 3),
|
|
|
|
|
np.array([f"sample-{index}.wav" for index in range(6)]),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
assert result["label"] == "EXPECTED"
|
|
|
|
|
assert result["evidence"]["neighborCount"] == 3
|
|
|
|
|
assert result["thresholds"]["topK"] == 3
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_audio_profile_applies_label_specific_threshold(monkeypatch) -> None:
|
|
|
|
|
config = {
|
|
|
|
|
"decision": {
|
|
|
|
|
"maxDistance": 0.45,
|
|
|
|
|
"minDistanceMargin": 0.015,
|
|
|
|
|
"labelMaxDistance": {"FIND_VEHICLE_HORN": 0.35},
|
|
|
|
|
"unknownLabel": "UNKNOWN",
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
profile = Profile("test.threshold.v1", "AUDIO_CLASSIFICATION", Path("."), config)
|
|
|
|
|
references = [np.array([[1.0]], dtype=np.float32), np.array([[2.0]], dtype=np.float32)]
|
|
|
|
|
monkeypatch.setattr(
|
|
|
|
|
"app.audio.sequence_distance",
|
|
|
|
|
lambda _query, reference: 0.36 if int(reference[0, 0]) == 1 else 0.44,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
result = classify_sequence(
|
|
|
|
|
profile,
|
|
|
|
|
np.array([[0.0]], dtype=np.float32),
|
|
|
|
|
references,
|
|
|
|
|
np.array(["FIND_VEHICLE_HORN", "POWER_OFF"]),
|
|
|
|
|
np.array(["horn.wav", "power-off.wav"]),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
assert result["label"] == "UNKNOWN"
|
|
|
|
|
assert result["thresholds"]["maxDistance"] == 0.35
|