feat(vision): 集成Qwen3.8视觉模型并优化音视频分析功能

- 集成Qwen3.8-27B-FP8快速模型和Qwen3.8-27B精确模型作为SGLang服务
- 添加SGLang API配置选项(SGLANG_FAST_BASE_URL、SGLANG_ACCURATE_BASE_URL等)
- 实现音频分类中的决策聚合算法(topK、nearestWeight、labelMaxDistance)
- 添加视频采样帧限制(VIDEO_SAMPLING_FRAME_LIMIT)和上下文token限制
- 更新健康检查以监控SGLang服务状态
- 实现视频分析的双模式决策策略(快速+精确)
- 添加音频参考文件导入工具(import_audio_references.py)
- 扩展音频分类标签支持FIND_VEHICLE_HORN类别
- 优化视频分析的帧采样策略,始终包含视频尾部帧
- 添加决策策略参数(tuning、decisionPolicy)支持
- 更新配置类以支持新的SGLang和视频参数
- 修改compose配置以支持Qwen3.8模型部署
- 更新音频分类测试用例验证聚合逻辑
- 重构视频测试以支持SGLang API格式和决策策略
This commit is contained in:
lixiaolong 2026-08-19 09:28:53 +08:00
parent a0d2faac55
commit b386f003a0
15 changed files with 1358 additions and 98 deletions

View File

@ -22,6 +22,7 @@ On the model server the reference directories are:
/data/apps/cmvr-ai-analysis/data/profiles/aima/power-state/v1/references/POWER_OFF /data/apps/cmvr-ai-analysis/data/profiles/aima/power-state/v1/references/POWER_OFF
/data/apps/cmvr-ai-analysis/data/profiles/aima/power-state/v1/references/ARMED /data/apps/cmvr-ai-analysis/data/profiles/aima/power-state/v1/references/ARMED
/data/apps/cmvr-ai-analysis/data/profiles/aima/power-state/v1/references/DISARMED /data/apps/cmvr-ai-analysis/data/profiles/aima/power-state/v1/references/DISARMED
/data/apps/cmvr-ai-analysis/data/profiles/aima/power-state/v1/references/FIND_VEHICLE_HORN
``` ```
After adding or replacing reference files, rebuild the feature library and restart the service profile state: After adding or replacing reference files, rebuild the feature library and restart the service profile state:
@ -44,6 +45,14 @@ under `references`, add its display name to `labelNames`, upload the reference a
and rebuild the profile. The classifier and platform workflow component do not need and rebuild the profile. The classifier and platform workflow component do not need
another code change. another code change.
Audio profiles can reduce single-sample false positives with decision aggregation:
- `topK` selects how many nearest references contribute to each label score.
- `nearestWeight` controls the nearest reference weight; the remaining weight is
assigned to the mean of the selected references.
- `labelMaxDistance` defines stricter distance limits for labels that otherwise have
a higher false-positive risk.
## API ## API
`POST /api/v1/analysis/run` accepts `requestId`, `analysisType`, `profileCode`, `mediaUrl`, `options`, and `context`. `POST /api/v1/analysis/run` accepts `requestId`, `analysisType`, `profileCode`, `mediaUrl`, `options`, and `context`.
@ -59,3 +68,24 @@ Video requests may set `options.analysisMode` to one of:
- `ACCURATE`: use the high-accuracy model only. - `ACCURATE`: use the high-accuracy model only.
Existing workflows without this option are treated as `AUTO`. Existing workflows without this option are treated as `AUTO`.
## Qwen3.8 video models
The model server runs the video models as separate SGLang services:
- `Qwen/Qwen3.8-27B-FP8` on port `14081` for fast analysis.
- `Qwen/Qwen3.8-27B` on port `14082` for accurate analysis.
Both services use a 262K context window and FP8 KV cache. ModelScope downloads are
kept under `/data/apps/cmvr-ai-analysis/models/modelscope`, outside the containers.
Start or inspect them with:
```bash
cd /data/apps/cmvr-ai-analysis
docker compose -f deploy/compose.sglang.yaml up -d
docker compose -f deploy/compose.sglang.yaml ps
```
The analysis service calls their OpenAI-compatible APIs. When either SGLang endpoint
is unavailable, it falls back to the existing Ollama model for the corresponding
mode, provided `VISION_OLLAMA_FALLBACK_ENABLED=true`.

View File

@ -195,19 +195,45 @@ def classify_sequence(
labels: np.ndarray, labels: np.ndarray,
files: np.ndarray, files: np.ndarray,
) -> dict[str, Any]: ) -> dict[str, Any]:
decision = profile.config.get("decision", {})
top_k = max(1, int(decision.get("topK", 1)))
nearest_weight = float(decision.get("nearestWeight", 1.0))
nearest_weight = max(0.0, min(1.0, nearest_weight))
best_by_label: dict[str, dict[str, Any]] = {} best_by_label: dict[str, dict[str, Any]] = {}
for reference, label, file_name in zip(reference_sequences, labels, files): for reference, label, file_name in zip(reference_sequences, labels, files):
distance = sequence_distance(query, reference) distance = sequence_distance(query, reference)
current = best_by_label.get(label) label = str(label)
if current is None or distance < current["distance"]: current = best_by_label.setdefault(label, {"distances": []})
best_by_label[label] = {"distance": distance, "reference": file_name} current["distances"].append((distance, str(file_name)))
for item in best_by_label.values():
ranked_distances = sorted(item.pop("distances"), key=lambda row: row[0])
nearest_distance, nearest_reference = ranked_distances[0]
neighbors = ranked_distances[:top_k]
aggregate_distance = (
nearest_weight * nearest_distance
+ (1.0 - nearest_weight) * float(np.mean([row[0] for row in neighbors]))
)
item.update(
{
"distance": aggregate_distance,
"nearestDistance": nearest_distance,
"reference": nearest_reference,
"neighborCount": len(neighbors),
}
)
ranking = sorted(best_by_label.items(), key=lambda item: item[1]["distance"]) ranking = sorted(best_by_label.items(), key=lambda item: item[1]["distance"])
winner_label, winner = ranking[0] winner_label, winner = ranking[0]
second_distance = ranking[1][1]["distance"] if len(ranking) > 1 else 1.0 second_distance = ranking[1][1]["distance"] if len(ranking) > 1 else 1.0
distance_margin = second_distance - winner["distance"] distance_margin = second_distance - winner["distance"]
decision = profile.config.get("decision", {}) default_maximum_distance = float(decision.get("maxDistance", 0.45))
maximum_distance = float(decision.get("maxDistance", 0.45)) label_maximum_distances = decision.get("labelMaxDistance", {})
maximum_distance = float(
label_maximum_distances.get(winner_label, default_maximum_distance)
if isinstance(label_maximum_distances, dict)
else default_maximum_distance
)
minimum_margin = float(decision.get("minDistanceMargin", 0.015)) minimum_margin = float(decision.get("minDistanceMargin", 0.015))
matched = winner["distance"] <= maximum_distance and distance_margin >= minimum_margin matched = winner["distance"] <= maximum_distance and distance_margin >= minimum_margin
output_label = winner_label if matched else str(decision.get("unknownLabel", "UNKNOWN")) output_label = winner_label if matched else str(decision.get("unknownLabel", "UNKNOWN"))
@ -221,10 +247,17 @@ def classify_sequence(
label: round(max(0.0, min(1.0, 1.0 - item["distance"])), 6) label: round(max(0.0, min(1.0, 1.0 - item["distance"])), 6)
for label, item in ranking for label, item in ranking
}, },
"evidence": {"reference": winner["reference"] if matched else None}, "evidence": {
"reference": winner["reference"] if matched else None,
"nearestDistance": round(winner["nearestDistance"], 6),
"aggregateDistance": round(winner["distance"], 6),
"neighborCount": winner["neighborCount"],
},
"thresholds": { "thresholds": {
"maxDistance": maximum_distance, "maxDistance": maximum_distance,
"minDistanceMargin": minimum_margin, "minDistanceMargin": minimum_margin,
"topK": top_k,
"nearestWeight": nearest_weight,
}, },
} }

View File

@ -14,9 +14,15 @@ class Settings(BaseSettings):
ollama_model: str = "qwen3-vl:32b" ollama_model: str = "qwen3-vl:32b"
media_timeout_seconds: int = 120 media_timeout_seconds: int = 120
ollama_timeout_seconds: int = 600 ollama_timeout_seconds: int = 600
sglang_fast_base_url: str = ""
sglang_accurate_base_url: str = ""
sglang_api_key: str = ""
sglang_timeout_seconds: int = 600
vision_ollama_fallback_enabled: bool = True
max_audio_bytes: int = 100 * 1024 * 1024 max_audio_bytes: int = 100 * 1024 * 1024
max_video_bytes: int = 2 * 1024 * 1024 * 1024 max_video_bytes: int = 2 * 1024 * 1024 * 1024
max_video_frames: int = 24 video_sampling_frame_limit: int = 100
max_video_context_tokens: int = 262144
settings = Settings() settings = Settings()

View File

@ -1,3 +1,4 @@
import asyncio
import shutil import shutil
import time import time
from contextlib import asynccontextmanager from contextlib import asynccontextmanager
@ -33,17 +34,35 @@ app = FastAPI(title="CMVR Media Analysis Service", version="1.0.0", lifespan=lif
@app.get("/health") @app.get("/health")
async def health() -> dict: async def health() -> dict:
ollama = "DOWN" async def probe(url: str) -> str:
try: try:
async with httpx.AsyncClient(timeout=3) as client: async with httpx.AsyncClient(timeout=2) as client:
response = await client.get(f"{settings.ollama_base_url.rstrip('/')}/api/version") response = await client.get(url)
response.raise_for_status() response.raise_for_status()
ollama = "UP" return "UP"
except Exception: except Exception:
pass return "DOWN"
sglang_urls = {
"fast": settings.sglang_fast_base_url,
"accurate": settings.sglang_accurate_base_url,
}
checks = [probe(f"{settings.ollama_base_url.rstrip('/')}/api/version")]
configured_names = [name for name, url in sglang_urls.items() if url]
checks.extend(
probe(f"{sglang_urls[name].rstrip('/')}/models") for name in configured_names
)
statuses = await asyncio.gather(*checks)
ollama = statuses[0]
configured_statuses = dict(zip(configured_names, statuses[1:]))
sglang = {
name: configured_statuses.get(name, "NOT_CONFIGURED")
for name in sglang_urls
}
return { return {
"status": "UP", "status": "UP",
"ollama": ollama, "ollama": ollama,
"sglang": sglang,
"visionModel": settings.ollama_model, "visionModel": settings.ollama_model,
"profiles": profile_store.status(), "profiles": profile_store.status(),
} }
@ -74,6 +93,8 @@ async def run_analysis(request: AnalysisRequest) -> AnalysisResponse:
media_path, media_path,
str(request.options.get("instruction", "")), str(request.options.get("instruction", "")),
str(request.options.get("analysisMode", "AUTO")), str(request.options.get("analysisMode", "AUTO")),
request.options.get("tuning"),
str(request.options.get("decisionPolicy", "FAIL_CLOSED")),
) )
evidence = result.get("evidence") evidence = result.get("evidence")
if not isinstance(evidence, dict): if not isinstance(evidence, dict):
@ -82,6 +103,10 @@ async def run_analysis(request: AnalysisRequest) -> AnalysisResponse:
evidence.update( evidence.update(
{ {
"sampledFrameCount": video_metadata["sampledFrameCount"], "sampledFrameCount": video_metadata["sampledFrameCount"],
"sampleFps": video_metadata["effectiveSampleFps"],
"samplingFrameLimit": video_metadata["samplingFrameLimit"],
"samplingCapped": video_metadata["samplingCapped"],
"numCtx": video_metadata["numCtx"],
"maximumWidth": video_metadata["maximumWidth"], "maximumWidth": video_metadata["maximumWidth"],
"durationSeconds": duration, "durationSeconds": duration,
} }
@ -89,9 +114,13 @@ async def run_analysis(request: AnalysisRequest) -> AnalysisResponse:
result["analysisMode"] = video_metadata["mode"] result["analysisMode"] = video_metadata["mode"]
result["fallback"] = video_metadata["fallback"] result["fallback"] = video_metadata["fallback"]
result["fallbackReason"] = video_metadata["fallbackReason"] result["fallbackReason"] = video_metadata["fallbackReason"]
result["effectiveTuning"] = video_metadata["tuning"]
model = { model = {
"provider": "Ollama", "provider": video_metadata["provider"],
"name": video_metadata["model"], "name": video_metadata["model"],
"primaryProvider": video_metadata["primaryProvider"],
"primaryModel": video_metadata["primaryModel"],
"providerFallback": video_metadata["providerFallback"],
"requestedMode": video_metadata["requestedMode"], "requestedMode": video_metadata["requestedMode"],
"usedMode": video_metadata["mode"], "usedMode": video_metadata["mode"],
"fallback": video_metadata["fallback"], "fallback": video_metadata["fallback"],

View File

@ -1,6 +1,7 @@
import base64 import base64
import copy import copy
import json import json
import math
import re import re
import subprocess import subprocess
import tempfile import tempfile
@ -20,7 +21,16 @@ class VisionModelError(RuntimeError):
VIDEO_RESULT_SCHEMA = { VIDEO_RESULT_SCHEMA = {
"type": "object", "type": "object",
"properties": { "properties": {
"passed": {"type": ["boolean", "null"]}, "passed": {
"type": "boolean",
"description": (
"Whether the user's acceptance criterion is satisfied. For negative "
"criteria such as disappearance or shutdown, return true when that "
"negative state is visibly achieved."
),
},
"evidenceSufficient": {"type": "boolean"},
"confidence": {"type": "number", "minimum": 0, "maximum": 1},
"conclusion": {"type": "string"}, "conclusion": {"type": "string"},
"summary": {"type": "string"}, "summary": {"type": "string"},
"events": { "events": {
@ -37,7 +47,21 @@ VIDEO_RESULT_SCHEMA = {
}, },
"warnings": {"type": "array", "items": {"type": "string"}}, "warnings": {"type": "array", "items": {"type": "string"}},
}, },
"required": ["passed", "conclusion", "summary", "events", "warnings"], "required": [
"passed",
"evidenceSufficient",
"confidence",
"conclusion",
"summary",
"events",
"warnings",
],
}
VIDEO_TUNING_LIMITS = {
"sampleFps": (0.25, 6.0),
"maxWidth": (640, 1280),
"confidenceThreshold": (0.5, 0.95),
} }
@ -66,25 +90,40 @@ def _duration(path: Path) -> float:
return value return value
def _frame_sampling_plan(
duration: float, sample_fps: float, maximum_frames: int
) -> tuple[int, int, float, float, bool]:
requested_count = max(2, math.ceil(duration * sample_fps) + 1)
count = min(requested_count, max(2, maximum_frames))
prefix_count = count - 1
capped = count < requested_count
prefix_fps = prefix_count / duration if capped else sample_fps
end_margin = min(0.1, max(0.001, duration / 2))
tail_time = max(0.0, duration - end_margin)
return count, prefix_count, prefix_fps, tail_time, capped
def _extract_frames( def _extract_frames(
path: Path, maximum: int, maximum_width: int path: Path, sample_fps: float, maximum_frames: int, maximum_width: int
) -> tuple[list[Path], float, Path]: ) -> tuple[list[Path], float, Path]:
duration = _duration(path) duration = _duration(path)
count = max(2, min(maximum, int(duration) + 1)) count, prefix_count, prefix_fps, tail_time, _ = _frame_sampling_plan(
fps = count / duration duration, sample_fps, maximum_frames
)
directory = Path(tempfile.mkdtemp(prefix="frames-", dir=settings.jobs_dir)) directory = Path(tempfile.mkdtemp(prefix="frames-", dir=settings.jobs_dir))
output = directory / "frame-%03d.jpg" output = directory / "frame-%03d.jpg"
process = subprocess.run( prefix_process = subprocess.run(
[ [
"ffmpeg", "ffmpeg",
"-y",
"-v", "-v",
"error", "error",
"-i", "-i",
str(path), str(path),
"-vf", "-vf",
f"fps={fps:.8f},scale='min({maximum_width},iw)':-2", f"fps={prefix_fps:.8f},scale='min({maximum_width},iw)':-2",
"-frames:v", "-frames:v",
str(count), str(prefix_count),
"-q:v", "-q:v",
"3", "3",
str(output), str(output),
@ -92,10 +131,39 @@ def _extract_frames(
capture_output=True, capture_output=True,
check=False, check=False,
) )
tail_output = directory / f"frame-{count:03d}.jpg"
tail_process = subprocess.run(
[
"ffmpeg",
"-y",
"-v",
"error",
"-i",
str(path),
"-ss",
f"{tail_time:.6f}",
"-vf",
f"scale='min({maximum_width},iw)':-2",
"-frames:v",
"1",
"-q:v",
"3",
str(tail_output),
],
capture_output=True,
check=False,
)
frames = sorted(directory.glob("frame-*.jpg")) frames = sorted(directory.glob("frame-*.jpg"))
if process.returncode != 0 or not frames: if prefix_process.returncode != 0 or tail_process.returncode != 0 or len(frames) < 2:
error = process.stderr.decode("utf-8", errors="replace")[-500:] error = (
raise ValueError(f"Unable to extract video frames: {error}") prefix_process.stderr.decode("utf-8", errors="replace")
+ tail_process.stderr.decode("utf-8", errors="replace")
)[-500:]
detail = error or (
f"prefixExit={prefix_process.returncode}, tailExit={tail_process.returncode}, "
f"expectedAtMost={count}, actual={len(frames)}"
)
raise ValueError(f"Unable to extract video frames: {detail}")
return frames, duration, directory return frames, duration, directory
@ -108,6 +176,15 @@ def _select_frames(frames: list[Path], maximum: int) -> list[Path]:
return [frames[index] for index in indices] return [frames[index] for index in indices]
def _video_context_size(base_context: int, image_count: int, output_tokens: int) -> int:
estimated_tokens = 4096 + image_count * 1400 + output_tokens
required_context = 1 << max(1, estimated_tokens - 1).bit_length()
return min(
settings.max_video_context_tokens,
max(base_context, required_context),
)
def _encode_image(path: Path, maximum_width: int, extracted_width: int) -> str: def _encode_image(path: Path, maximum_width: int, extracted_width: int) -> str:
if maximum_width >= extracted_width: if maximum_width >= extracted_width:
image = path.read_bytes() image = path.read_bytes()
@ -151,10 +228,20 @@ def _parse_json(content: str) -> dict[str, Any]:
return {"summary": content, "structured": False} return {"summary": content, "structured": False}
def _fallback_reason(result: dict[str, Any], fallback_on_unknown: bool) -> str | None: def _fallback_reason(
result: dict[str, Any], confidence_threshold: float
) -> str | None:
if result.get("structured") is False: if result.get("structured") is False:
return "FAST_RESULT_NOT_STRUCTURED" return "FAST_RESULT_NOT_STRUCTURED"
required = ("passed", "conclusion", "summary", "events", "warnings") required = (
"passed",
"evidenceSufficient",
"confidence",
"conclusion",
"summary",
"events",
"warnings",
)
missing = [field for field in required if field not in result] missing = [field for field in required if field not in result]
if missing: if missing:
return "FAST_RESULT_MISSING_FIELDS:" + ",".join(missing) return "FAST_RESULT_MISSING_FIELDS:" + ",".join(missing)
@ -164,64 +251,402 @@ def _fallback_reason(result: dict[str, Any], fallback_on_unknown: bool) -> str |
return "FAST_RESULT_EMPTY_SUMMARY" return "FAST_RESULT_EMPTY_SUMMARY"
if not isinstance(result.get("events"), list) or not isinstance(result.get("warnings"), list): if not isinstance(result.get("events"), list) or not isinstance(result.get("warnings"), list):
return "FAST_RESULT_INVALID_COLLECTIONS" return "FAST_RESULT_INVALID_COLLECTIONS"
if fallback_on_unknown and result.get("passed") is None: if result.get("evidenceSufficient") is not True:
return "FAST_RESULT_INSUFFICIENT" return "FAST_RESULT_INSUFFICIENT"
if result.get("passed") is not True and result.get("passed") is not False and result.get("passed") is not None: if result.get("passed") is not True and result.get("passed") is not False:
return "FAST_RESULT_INVALID_DECISION" return "FAST_RESULT_INVALID_DECISION"
try:
confidence = float(result.get("confidence"))
except (TypeError, ValueError):
return "FAST_RESULT_INVALID_CONFIDENCE"
if confidence < confidence_threshold:
return "FAST_RESULT_LOW_CONFIDENCE"
return None return None
def _bounded_number(
value: Any, name: str, minimum: float, maximum: float, integer: bool = False
) -> int | float:
try:
number = float(value)
except (TypeError, ValueError) as exc:
raise ValueError(f"Video tuning parameter {name} must be numeric") from exc
if number < minimum or number > maximum:
raise ValueError(
f"Video tuning parameter {name} must be between {minimum} and {maximum}"
)
if integer and not number.is_integer():
raise ValueError(f"Video tuning parameter {name} must be an integer")
return int(number) if integer else number
def _normalize_tuning(raw: dict[str, Any] | None) -> dict[str, Any]:
tuning = dict(raw or {})
normalized: dict[str, Any] = {}
if "sampleFps" in tuning:
normalized["sampleFps"] = _bounded_number(
tuning["sampleFps"], "sampleFps", *VIDEO_TUNING_LIMITS["sampleFps"]
)
if "maxWidth" in tuning:
normalized["maxWidth"] = _bounded_number(
tuning["maxWidth"], "maxWidth", *VIDEO_TUNING_LIMITS["maxWidth"], True
)
if "confidenceThreshold" in tuning:
normalized["confidenceThreshold"] = _bounded_number(
tuning["confidenceThreshold"],
"confidenceThreshold",
*VIDEO_TUNING_LIMITS["confidenceThreshold"],
)
if "fallbackToAccurate" in tuning:
value = tuning["fallbackToAccurate"]
if not isinstance(value, bool):
raise ValueError("Video tuning parameter fallbackToAccurate must be boolean")
normalized["fallbackToAccurate"] = value
return normalized
def _apply_tuning(strategy: dict[str, Any], tuning: dict[str, Any]) -> dict[str, Any]:
configured = dict(strategy)
for name in ("sampleFps", "maxWidth"):
if name in tuning:
configured[name] = tuning[name]
return configured
_CRITERION_TRANSITIONS = (
("消失", ("消失", "不再显示", "从有到无", "由有变无")),
("熄灭", ("熄灭", "从亮到灭", "由亮变灭")),
("关闭", ("已关闭", "从开到关", "由开变关")),
("停止", ("已停止", "停止运行")),
("断开", ("已断开", "连接断开")),
("出现", ("出现", "从无到有", "由无变有")),
("点亮", ("点亮", "亮起", "从灭到亮", "由灭变亮")),
("开启", ("已开启", "从关到开", "由关变开")),
("启动", ("已启动", "开始运行")),
("连接", ("已连接", "连接成功")),
)
_REMOVAL_CRITERION_TERMS = (
"消失", "熄灭", "关闭", "停止", "断开", "移除", "不再显示",
"disappear", "turn off", "shut down", "disconnect", "remove", "absent",
)
_APPEARANCE_CRITERION_TERMS = (
"出现", "点亮", "开启", "启动", "连接", "显示",
"appear", "turn on", "start", "connect", "visible",
)
def _temporal_acceptance_guidance(acceptance_criterion: str) -> str:
original_criterion = str(acceptance_criterion or "").strip()
criterion = original_criterion.lower()
guidance = (
f"当前通过标准是【{original_criterion}】。图片严格按时间从前到后排列,"
"第一张是初始状态,最后一张是结束状态。判定 passed 前必须逐帧检查,"
"并明确比较第一张与最后一张图片。"
)
if any(term in criterion for term in _REMOVAL_CRITERION_TERMS):
target = original_criterion
for term in _REMOVAL_CRITERION_TERMS:
index = criterion.find(term)
if index >= 0:
target = original_criterion[:index]
break
target = re.sub(r"(?:是否|有没有|有无)$", "", target).strip(" ::,,。??") or "目标"
return guidance + (
f"针对当前消失、移除、关闭或断开的标准:如果{target}在前段图片中存在,"
f"但在最后一张及末段图片中已不再显示{target},就说明状态变化成功,"
"passed必须为true。不能因为第一张图片中存在,就判断全程持续存在。"
)
if any(term in criterion for term in _APPEARANCE_CRITERION_TERMS):
return guidance + (
"针对出现、点亮、开启、启动或连接的标准:如果目标在前段图片中不存在,"
"但在最后一张及末段图片中已出现,就说明状态变化成功,passed 必须为 true。"
)
return guidance
def _needs_temporal_removal_verification(
acceptance_criterion: str, result: dict[str, Any]
) -> bool:
criterion = str(acceptance_criterion or "").strip().lower()
return (
any(term in criterion for term in _REMOVAL_CRITERION_TERMS)
and result.get("passed") is False
and result.get("evidenceSufficient") is True
)
def _criterion_verdict(result: dict[str, Any], acceptance_criterion: str) -> bool | None:
criterion = str(acceptance_criterion or "").strip()
if not criterion:
return None
conclusion = str(result.get("conclusion") or "").strip()
summary = str(result.get("summary") or "").strip()
events = result.get("events") if isinstance(result.get("events"), list) else []
evidence_text = " ".join(
[conclusion, summary]
+ [
str(event.get("event") if isinstance(event, dict) else event)
for event in events
]
)
explicit_failure = re.search(
r"(?:不符合|未满足|不满足|未达到|没有达到|判定失败|未通过).{0,12}(?:标准|要求|条件)?",
evidence_text,
)
if explicit_failure:
return False
explicit_success = re.search(
r"(?:符合|满足|达到|已完成).{0,12}(?:标准|要求|条件)",
evidence_text,
)
if explicit_success:
return True
for criterion_word, observed_phrases in _CRITERION_TRANSITIONS:
if criterion_word not in criterion:
continue
negated = re.search(
rf"(?:未|没有|尚未|未能|不曾).{{0,4}}{re.escape(criterion_word)}",
evidence_text,
)
if negated:
return False
if criterion_word in {"消失", "熄灭", "关闭", "停止", "断开"} and re.search(
r"(?:仍然|仍|依然|依旧).{0,8}(?:显示|存在|亮起|开启|运行|连接)",
evidence_text,
):
return False
if any(phrase in evidence_text for phrase in observed_phrases):
return True
return None
def _normalize_decision(
result: dict[str, Any], confidence_threshold: float, acceptance_criterion: str = ""
) -> dict[str, Any]:
normalized = dict(result or {})
warnings = normalized.get("warnings")
if not isinstance(warnings, list):
warnings = []
normalized["warnings"] = warnings
normalized["events"] = normalized.get("events") if isinstance(normalized.get("events"), list) else []
try:
confidence = min(1.0, max(0.0, float(normalized.get("confidence", 0))))
except (TypeError, ValueError):
confidence = 0.0
evidence_sufficient = normalized.get("evidenceSufficient") is True
model_passed = normalized.get("passed") is True
criterion_verdict = _criterion_verdict(normalized, acceptance_criterion)
raw_passed = model_passed if criterion_verdict is None else criterion_verdict
decision_reason = "MODEL_DECISION"
if normalized.get("structured") is False:
evidence_sufficient = False
decision_reason = "UNSTRUCTURED_RESULT"
elif not evidence_sufficient:
decision_reason = "INSUFFICIENT_EVIDENCE"
elif confidence < confidence_threshold:
decision_reason = "LOW_CONFIDENCE"
elif criterion_verdict is not None and criterion_verdict != model_passed:
decision_reason = "CRITERION_EVIDENCE_CORRECTION"
passed = raw_passed and evidence_sufficient and confidence >= confidence_threshold
if not passed and decision_reason != "MODEL_DECISION":
if decision_reason == "LOW_CONFIDENCE":
reason_text = "分析置信度不足"
elif decision_reason == "CRITERION_EVIDENCE_CORRECTION":
reason_text = "分析结论未满足通过标准"
else:
reason_text = "视频证据不足"
conclusion = str(normalized.get("conclusion") or "").strip()
normalized["conclusion"] = f"未通过:{reason_text}" + (f";{conclusion}" if conclusion else "")
if reason_text not in warnings:
warnings.append(reason_text)
normalized["passed"] = passed
if criterion_verdict is not None and criterion_verdict != model_passed:
normalized["modelPassed"] = model_passed
normalized["evidenceSufficient"] = evidence_sufficient
normalized["confidence"] = confidence
normalized["decisionReason"] = decision_reason
normalized["confidenceThreshold"] = confidence_threshold
normalized.setdefault("conclusion", "通过" if passed else "未通过")
normalized.setdefault("summary", normalized["conclusion"])
normalized.pop("structured", None)
return normalized
def _strategy(profile: Profile, name: str) -> dict[str, Any]: def _strategy(profile: Profile, name: str) -> dict[str, Any]:
defaults = { defaults = {
"FAST": { "FAST": {
"model": profile.config.get("model", settings.ollama_model), "model": profile.config.get("model", settings.ollama_model),
"maxFrames": 6, "sampleFps": 1.0,
"maxWidth": 896, "maxWidth": 896,
"maxOutputTokens": 512, "maxOutputTokens": 512,
"numCtx": 16384, "numCtx": 16384,
"emptyResponseRetries": 0, "emptyResponseRetries": 0,
"maxRetryOutputTokens": 4096,
}, },
"ACCURATE": { "ACCURATE": {
"model": profile.config.get("model", settings.ollama_model), "model": profile.config.get("model", settings.ollama_model),
"maxFrames": 12, "sampleFps": 3.0,
"maxWidth": 1280, "maxWidth": 1280,
"maxOutputTokens": 768, "maxOutputTokens": 768,
"numCtx": 32768, "numCtx": 32768,
"emptyResponseRetries": 1, "emptyResponseRetries": 2,
"maxRetryOutputTokens": 8192,
}, },
} }
configured = profile.config.get("strategies", {}).get(name.lower(), {}) configured = profile.config.get("strategies", {}).get(name.lower(), {})
return {**defaults[name], **configured} strategy = {**defaults[name], **configured}
strategy["provider"] = str(strategy.get("provider", "ollama")).strip().lower()
if strategy["provider"] == "sglang":
strategy["baseUrl"] = (
settings.sglang_fast_base_url
if name == "FAST"
else settings.sglang_accurate_base_url
)
return strategy
def _sglang_payload(payload: dict[str, Any]) -> dict[str, Any]:
message = payload.get("messages", [{}])[0]
content: list[dict[str, Any]] = [
{"type": "text", "text": str(message.get("content") or "")}
]
content.extend(
{
"type": "image_url",
"image_url": {"url": f"data:image/jpeg;base64,{image}"},
}
for image in message.get("images") or []
)
options = payload.get("options") or {}
return {
"model": payload["model"],
"messages": [{"role": "user", "content": content}],
"stream": False,
"temperature": float(options.get("temperature", 0.1)),
"max_tokens": int(options.get("num_predict", 512)),
"response_format": {
"type": "json_schema",
"json_schema": {
"name": "video_analysis",
"strict": True,
"schema": VIDEO_RESULT_SCHEMA,
},
},
"chat_template_kwargs": {
"enable_thinking": bool(payload.get("think", False)),
"preserve_thinking": False,
},
}
def _request_image_count(payload: dict[str, Any], provider: str) -> int:
message = payload.get("messages", [{}])[0]
if provider == "ollama":
return len(message.get("images") or [])
content = message.get("content") or []
return sum(
1 for item in content
if isinstance(item, dict) and item.get("type") == "image_url"
)
async def _request_vision_model( async def _request_vision_model(
payload: dict[str, Any], empty_response_retries: int = 1 payload: dict[str, Any], empty_response_retries: int = 1,
max_retry_output_tokens: int = 8192,
provider: str = "ollama",
base_url: str | None = None,
) -> tuple[str, dict[str, Any]]: ) -> tuple[str, dict[str, Any]]:
provider = str(provider or "ollama").strip().lower()
if provider not in {"ollama", "sglang"}:
raise ValueError(f"Unsupported vision model provider: {provider}")
if provider == "sglang" and not str(base_url or "").strip():
raise VisionModelError("SGLang vision model endpoint is not configured")
last_metadata: dict[str, Any] = {} last_metadata: dict[str, Any] = {}
timeout = httpx.Timeout(settings.ollama_timeout_seconds) timeout_seconds = (
settings.sglang_timeout_seconds
if provider == "sglang"
else settings.ollama_timeout_seconds
)
timeout = httpx.Timeout(timeout_seconds, connect=5.0)
async with httpx.AsyncClient(timeout=timeout) as client: async with httpx.AsyncClient(timeout=timeout) as client:
for attempt in range(empty_response_retries + 1): for attempt in range(empty_response_retries + 1):
request_payload = copy.deepcopy(payload) request_payload = copy.deepcopy(payload)
if attempt: if attempt:
options = request_payload.setdefault("options", {}) options = request_payload.setdefault("options", {})
options["num_predict"] = max(int(options.get("num_predict", 0)), 1536) previous_limit = max(
request_payload["messages"][0]["content"] += ( 1,
"\n\nReturn the final JSON now. Do not return reasoning without a final answer." int(options.get("num_predict", 0)),
int(last_metadata.get("requestedOutputTokens") or 0),
) )
response = await client.post( previous_eval_count = int(last_metadata.get("evalCount") or 0)
f"{settings.ollama_base_url.rstrip('/')}/api/chat", json=request_payload retry_limit = max(4096, previous_limit * 2, previous_eval_count + 2048)
options["num_predict"] = min(max_retry_output_tokens, retry_limit)
request_payload["think"] = False
request_payload["messages"][0]["content"] += (
"\n\n/no_think\nReturn only the final JSON now. Do not output reasoning."
)
provider_payload = (
_sglang_payload(request_payload)
if provider == "sglang"
else request_payload
) )
response.raise_for_status() url = (
f"{str(base_url).rstrip('/')}/chat/completions"
if provider == "sglang"
else f"{settings.ollama_base_url.rstrip('/')}/api/chat"
)
headers = {}
if provider == "sglang" and settings.sglang_api_key:
headers["Authorization"] = f"Bearer {settings.sglang_api_key}"
response = await client.post(url, json=provider_payload, headers=headers)
try:
response.raise_for_status()
except httpx.HTTPStatusError as exc:
detail = response.text.strip()[:1000]
raise VisionModelError(
"Vision model request rejected "
f"(provider={provider}, status={response.status_code}, "
f"imageCount={_request_image_count(provider_payload, provider)}, "
f"numCtx={request_payload.get('options', {}).get('num_ctx')}, "
f"detail={detail})"
) from exc
body = response.json() body = response.json()
message = body.get("message") or {} if provider == "sglang":
content = str(message.get("content") or "").strip() choices = body.get("choices") or []
metadata = { choice = choices[0] if choices else {}
"doneReason": body.get("done_reason"), message = choice.get("message") or {}
"promptEvalCount": body.get("prompt_eval_count"), usage = body.get("usage") or {}
"evalCount": body.get("eval_count"), content = str(message.get("content") or "").strip()
"totalDurationNs": body.get("total_duration"), metadata = {
"thinkingLength": len(str(message.get("thinking") or "")), "doneReason": choice.get("finish_reason"),
} "promptEvalCount": usage.get("prompt_tokens"),
"evalCount": usage.get("completion_tokens"),
"totalDurationNs": None,
"thinkingLength": len(str(message.get("reasoning_content") or "")),
"attempt": attempt + 1,
"requestedOutputTokens": provider_payload.get("max_tokens"),
}
else:
message = body.get("message") or {}
content = str(message.get("content") or "").strip()
metadata = {
"doneReason": body.get("done_reason"),
"promptEvalCount": body.get("prompt_eval_count"),
"evalCount": body.get("eval_count"),
"totalDurationNs": body.get("total_duration"),
"thinkingLength": len(str(message.get("thinking") or "")),
"attempt": attempt + 1,
"requestedOutputTokens": request_payload.get("options", {}).get("num_predict"),
}
metadata["provider"] = provider
if content: if content:
return content, metadata return content, metadata
last_metadata = metadata last_metadata = metadata
@ -238,36 +663,82 @@ async def _analyze_frames(
duration: float, duration: float,
extracted_width: int, extracted_width: int,
prompt: str, prompt: str,
tuning: dict[str, Any] | None = None,
) -> tuple[dict[str, Any], dict[str, Any]]: ) -> tuple[dict[str, Any], dict[str, Any]]:
strategy = _strategy(profile, strategy_name) strategy = _apply_tuning(_strategy(profile, strategy_name), tuning or {})
selected = _select_frames(frames, max(2, int(strategy["maxFrames"]))) desired_count, _, _, _, sampling_capped = _frame_sampling_plan(
duration,
float(strategy["sampleFps"]),
settings.video_sampling_frame_limit,
)
selected = _select_frames(frames, desired_count)
maximum_width = max(320, int(strategy["maxWidth"])) maximum_width = max(320, int(strategy["maxWidth"]))
images = [_encode_image(path, maximum_width, extracted_width) for path in selected] images = [_encode_image(path, maximum_width, extracted_width) for path in selected]
strategy_prompt = prompt + ( strategy_prompt = prompt + (
f"\n\nThe video duration is {duration:.3f} seconds. " f"\n\nThe video duration is {duration:.3f} seconds. "
f"The following {len(selected)} images are ordered frames sampled across the video." f"The following {len(selected)} images are ordered frames sampled across the video."
) )
think = bool(strategy.get("think", profile.config.get("think", False)))
if not think:
strategy_prompt += "\n\n/no_think\nReturn only the requested JSON object without reasoning."
output_tokens = int(strategy["maxOutputTokens"])
context_tokens = _video_context_size(
int(strategy["numCtx"]), len(selected), output_tokens
)
payload = { payload = {
"model": strategy["model"], "model": strategy["model"],
"stream": False, "stream": False,
"think": bool(strategy.get("think", profile.config.get("think", False))), "think": think,
"format": VIDEO_RESULT_SCHEMA, "format": VIDEO_RESULT_SCHEMA,
"keep_alive": str(strategy.get("keepAlive", "30m")), "keep_alive": str(strategy.get("keepAlive", "30m")),
"messages": [{"role": "user", "content": strategy_prompt, "images": images}], "messages": [{"role": "user", "content": strategy_prompt, "images": images}],
"options": { "options": {
"temperature": float(strategy.get("temperature", profile.config.get("temperature", 0.1))), "temperature": float(strategy.get("temperature", profile.config.get("temperature", 0.1))),
"num_predict": int(strategy["maxOutputTokens"]), "num_predict": output_tokens,
"num_ctx": int(strategy["numCtx"]), "num_ctx": context_tokens,
}, },
} }
content, metrics = await _request_vision_model( provider = str(strategy.get("provider", "ollama"))
payload, int(strategy.get("emptyResponseRetries", 0)) actual_model = str(strategy["model"])
) provider_fallback = False
try:
content, metrics = await _request_vision_model(
payload,
int(strategy.get("emptyResponseRetries", 0)),
int(strategy.get("maxRetryOutputTokens", 8192)),
provider,
strategy.get("baseUrl"),
)
except (VisionModelError, httpx.HTTPError):
fallback_model = str(strategy.get("fallbackModel") or "").strip()
if (
provider != "sglang"
or not settings.vision_ollama_fallback_enabled
or not fallback_model
):
raise
fallback_payload = copy.deepcopy(payload)
fallback_payload["model"] = fallback_model
content, metrics = await _request_vision_model(
fallback_payload,
int(strategy.get("emptyResponseRetries", 0)),
int(strategy.get("maxRetryOutputTokens", 8192)),
)
provider_fallback = True
actual_model = fallback_model
metadata = { metadata = {
"mode": strategy_name, "mode": strategy_name,
"model": strategy["model"], "model": actual_model,
"provider": metrics.get("provider", provider),
"primaryProvider": provider,
"primaryModel": strategy["model"],
"providerFallback": provider_fallback,
"sampledFrameCount": len(selected), "sampledFrameCount": len(selected),
"maximumWidth": maximum_width, "maximumWidth": maximum_width,
"effectiveSampleFps": float(strategy["sampleFps"]),
"samplingFrameLimit": settings.video_sampling_frame_limit,
"samplingCapped": sampling_capped,
"numCtx": context_tokens,
**metrics, **metrics,
} }
return _parse_json(content), metadata return _parse_json(content), metadata
@ -278,55 +749,137 @@ async def analyze_video(
media_path: Path, media_path: Path,
extra_instruction: str = "", extra_instruction: str = "",
analysis_mode: str = "AUTO", analysis_mode: str = "AUTO",
tuning: dict[str, Any] | None = None,
decision_policy: str = "FAIL_CLOSED",
) -> tuple[dict[str, Any], float, dict[str, Any]]: ) -> tuple[dict[str, Any], float, dict[str, Any]]:
requested_mode = str(analysis_mode or "AUTO").strip().upper() requested_mode = str(analysis_mode or "AUTO").strip().upper()
if requested_mode not in {"AUTO", "FAST", "ACCURATE"}: if requested_mode not in {"AUTO", "FAST", "ACCURATE"}:
raise ValueError("Video analysis mode must be AUTO, FAST, or ACCURATE") raise ValueError("Video analysis mode must be AUTO, FAST, or ACCURATE")
if str(decision_policy or "FAIL_CLOSED").strip().upper() != "FAIL_CLOSED":
raise ValueError("Video decision policy must be FAIL_CLOSED")
fast = _strategy(profile, "FAST") effective_tuning = _normalize_tuning(tuning)
accurate = _strategy(profile, "ACCURATE") confidence_threshold = float(effective_tuning.get("confidenceThreshold", 0.75))
fallback_to_accurate = bool(effective_tuning.get("fallbackToAccurate", True))
fast = _apply_tuning(_strategy(profile, "FAST"), effective_tuning)
accurate = _apply_tuning(_strategy(profile, "ACCURATE"), effective_tuning)
extraction_strategy = fast if requested_mode == "FAST" else accurate extraction_strategy = fast if requested_mode == "FAST" else accurate
extraction_frames = max(2, int(extraction_strategy["maxFrames"])) extraction_sample_fps = float(extraction_strategy["sampleFps"])
extraction_width = max(320, int(extraction_strategy["maxWidth"])) extraction_width = max(320, int(extraction_strategy["maxWidth"]))
frames, duration, frame_dir = _extract_frames( frames, duration, frame_dir = _extract_frames(
media_path, extraction_frames, extraction_width media_path,
extraction_sample_fps,
settings.video_sampling_frame_limit,
extraction_width,
) )
try: try:
prompt_path = profile.directory / str(profile.config.get("promptFile", "prompt.txt")) prompt_path = profile.directory / str(profile.config.get("promptFile", "prompt.txt"))
prompt = prompt_path.read_text(encoding="utf-8") prompt = prompt_path.read_text(encoding="utf-8")
if extra_instruction.strip(): if extra_instruction.strip():
prompt += "\n\nAdditional inspection requirement:\n" + extra_instruction.strip() temporal_guidance = _temporal_acceptance_guidance(extra_instruction)
prompt += (
"\n\nAcceptance criterion (passed=true exactly when this criterion is "
"satisfied; this is not merely an object-detection question):\n"
+ "通过标准:"
+ extra_instruction.strip()
+ "。"
+ temporal_guidance
)
async def finalize_fast_result(
result: dict[str, Any], metadata: dict[str, Any], mode_label: str,
fallback_reason: str | None = None,
) -> tuple[dict[str, Any], dict[str, Any]]:
normalized = _normalize_decision(
result, confidence_threshold, extra_instruction
)
if _needs_temporal_removal_verification(extra_instruction, normalized):
focused_prompt = prompt + (
"\n\n首尾帧精确复核:本次只提供视频第一张和最后一张图片。"
"必须比较具体目标在两张图片中的存在状态,再按通过标准输出结论。"
)
verified, verified_metadata = await _analyze_frames(
profile,
"ACCURATE",
[frames[0], frames[-1]],
duration,
extraction_width,
focused_prompt,
effective_tuning,
)
verified = _normalize_decision(
verified, confidence_threshold, extra_instruction
)
verified_metadata.update(
{
"requestedMode": mode_label,
"fallback": True,
"fallbackReason": "TEMPORAL_REMOVAL_VERIFICATION",
"tuning": effective_tuning,
}
)
return verified, verified_metadata
metadata.update(
{
"requestedMode": mode_label,
"fallback": False,
"fallbackReason": fallback_reason,
"tuning": effective_tuning,
}
)
return normalized, metadata
if requested_mode in {"FAST", "ACCURATE"}: if requested_mode in {"FAST", "ACCURATE"}:
result, metadata = await _analyze_frames( result, metadata = await _analyze_frames(
profile, requested_mode, frames, duration, extraction_width, prompt profile, requested_mode, frames, duration, extraction_width, prompt,
) effective_tuning,
metadata.update(
{"requestedMode": requested_mode, "fallback": False, "fallbackReason": None}
) )
if requested_mode == "FAST":
result, metadata = await finalize_fast_result(
result, metadata, requested_mode
)
else:
result = _normalize_decision(
result, confidence_threshold, extra_instruction
)
metadata.update(
{
"requestedMode": requested_mode,
"fallback": False,
"fallbackReason": None,
"tuning": effective_tuning,
}
)
return result, duration, metadata return result, duration, metadata
fallback_reason = None fallback_reason = None
try: try:
result, metadata = await _analyze_frames( result, metadata = await _analyze_frames(
profile, "FAST", frames, duration, extraction_width, prompt profile, "FAST", frames, duration, extraction_width, prompt,
effective_tuning,
) )
fallback_reason = _fallback_reason( fallback_reason = _fallback_reason(result, confidence_threshold)
result, bool(profile.config.get("fallbackWhenPassedUnknown", True)) if fallback_reason is None or not fallback_to_accurate:
) result, metadata = await finalize_fast_result(
if fallback_reason is None: result, metadata, "AUTO", fallback_reason
metadata.update(
{"requestedMode": "AUTO", "fallback": False, "fallbackReason": None}
) )
return result, duration, metadata return result, duration, metadata
except (VisionModelError, httpx.HTTPError) as exc: except (VisionModelError, httpx.HTTPError) as exc:
fallback_reason = f"FAST_MODEL_ERROR:{type(exc).__name__}" fallback_reason = f"FAST_MODEL_ERROR:{type(exc).__name__}"
result, metadata = await _analyze_frames( result, metadata = await _analyze_frames(
profile, "ACCURATE", frames, duration, extraction_width, prompt profile, "ACCURATE", frames, duration, extraction_width, prompt,
effective_tuning,
) )
result = _normalize_decision(result, confidence_threshold, extra_instruction)
metadata.update( metadata.update(
{"requestedMode": "AUTO", "fallback": True, "fallbackReason": fallback_reason} {
"requestedMode": "AUTO",
"fallback": True,
"fallbackReason": fallback_reason,
"tuning": effective_tuning,
}
) )
return result, duration, metadata return result, duration, metadata
finally: finally:

View File

@ -1,6 +1,6 @@
{ {
"code": "aima.power_state.v1", "code": "aima.power_state.v1",
"name": "爱玛车辆声音事件识别", "name": "爱玛车辆声音类型检测",
"analysisType": "AUDIO_CLASSIFICATION", "analysisType": "AUDIO_CLASSIFICATION",
"minimumReferencesPerLabel": 3, "minimumReferencesPerLabel": 3,
"maximumAudioSeconds": 30, "maximumAudioSeconds": 30,
@ -9,11 +9,17 @@
"POWER_OFF": "关机", "POWER_OFF": "关机",
"ARMED": "设防", "ARMED": "设防",
"DISARMED": "解防", "DISARMED": "解防",
"FIND_VEHICLE_HORN": "鸣笛寻车",
"UNKNOWN": "其他声音" "UNKNOWN": "其他声音"
}, },
"decision": { "decision": {
"maxDistance": 0.45, "maxDistance": 0.45,
"minDistanceMargin": 0.015, "minDistanceMargin": 0.015,
"topK": 3,
"nearestWeight": 0.6,
"labelMaxDistance": {
"FIND_VEHICLE_HORN": 0.35
},
"unknownLabel": "UNKNOWN" "unknownLabel": "UNKNOWN"
} }
} }

View File

@ -6,24 +6,29 @@
"promptFile": "prompt.txt", "promptFile": "prompt.txt",
"think": false, "think": false,
"temperature": 0.1, "temperature": 0.1,
"fallbackWhenPassedUnknown": true,
"strategies": { "strategies": {
"fast": { "fast": {
"model": "qwen3-vl:8b-instruct", "provider": "sglang",
"maxFrames": 6, "model": "Qwen/Qwen3.8-27B-FP8",
"fallbackModel": "qwen3-vl:8b-instruct",
"sampleFps": 1.0,
"maxWidth": 896, "maxWidth": 896,
"maxOutputTokens": 512, "maxOutputTokens": 512,
"numCtx": 16384, "numCtx": 16384,
"emptyResponseRetries": 0, "emptyResponseRetries": 0,
"maxRetryOutputTokens": 4096,
"keepAlive": "30m" "keepAlive": "30m"
}, },
"accurate": { "accurate": {
"model": "qwen3-vl:32b", "provider": "sglang",
"maxFrames": 12, "model": "Qwen/Qwen3.8-27B",
"maxWidth": 1280, "fallbackModel": "qwen3-vl:32b",
"sampleFps": 3.0,
"maxWidth": 1120,
"maxOutputTokens": 768, "maxOutputTokens": 768,
"numCtx": 32768, "numCtx": 32768,
"emptyResponseRetries": 1, "emptyResponseRetries": 2,
"maxRetryOutputTokens": 8192,
"keepAlive": "30m" "keepAlive": "30m"
} }
} }

View File

@ -1 +1 @@
You are an industrial test video analyst. Analyze only visible evidence in the ordered video frames. Do not invent events that are not visible. Return one JSON object with these fields: passed (boolean or null when the criterion is insufficient), conclusion (short string), summary (string), events (array of objects containing timeRange, event, confidence), and warnings (array of strings). Use Chinese for all explanatory text. You are an industrial test video analyst. Analyze only visible evidence in the ordered video frames. Do not invent events that are not visible. Apply the user's acceptance criterion and always make a binary decision. The passed field means whether the acceptance criterion is satisfied; it does not mean whether the observed object exists. For a negative or state-removal criterion such as "the icon disappears", "the light turns off", "the device shuts down", or "the connection is disconnected", passed must be true when that requested state change is visibly completed. Evidence that is missing, incomplete, ambiguous, or below the required confidence must be treated as failed, never as unknown. Return one JSON object with these fields: passed (boolean), evidenceSufficient (boolean), confidence (number from 0 to 1), conclusion (short string), summary (string), events (array of objects containing timeRange, event, confidence), and warnings (array of strings). Use Chinese for all explanatory text.

View File

@ -1,6 +1,12 @@
API_KEY=replace-with-a-strong-random-token API_KEY=replace-with-a-strong-random-token
MEDIA_TIMEOUT_SECONDS=120 MEDIA_TIMEOUT_SECONDS=120
OLLAMA_TIMEOUT_SECONDS=600 OLLAMA_TIMEOUT_SECONDS=600
SGLANG_FAST_BASE_URL=http://192.168.28.10:14081/v1
SGLANG_ACCURATE_BASE_URL=http://192.168.28.10:14082/v1
SGLANG_API_KEY=
SGLANG_TIMEOUT_SECONDS=600
VISION_OLLAMA_FALLBACK_ENABLED=true
MAX_AUDIO_BYTES=104857600 MAX_AUDIO_BYTES=104857600
MAX_VIDEO_BYTES=2147483648 MAX_VIDEO_BYTES=2147483648
MAX_VIDEO_FRAMES=24 VIDEO_SAMPLING_FRAME_LIMIT=100
MAX_VIDEO_CONTEXT_TOKENS=262144

View File

@ -0,0 +1,82 @@
services:
qwen38-fast:
container_name: cmvr-qwen38-fast
image: quay.io/gpustack/runner:cuda13.0-sglang0.5.15.post1
restart: unless-stopped
ipc: host
shm_size: 32gb
ports:
- "192.168.28.10:14081:30000"
environment:
SGLANG_USE_MODELSCOPE: "true"
MODELSCOPE_CACHE: /root/.cache/modelscope
MODELSCOPE_DOWNLOAD_PARALLEL_WORKERS: "4"
volumes:
- ../models/modelscope:/root/.cache/modelscope
gpus:
- driver: nvidia
device_ids: ["2"]
capabilities: [gpu]
command:
- python
- -m
- sglang.launch_server
- --model-path
- Qwen/Qwen3.8-27B-FP8
- --served-model-name
- Qwen/Qwen3.8-27B-FP8
- --host
- 0.0.0.0
- --port
- "30000"
- --tp-size
- "1"
- --context-length
- "262144"
- --kv-cache-dtype
- fp8_e4m3
- --mem-fraction-static
- "0.88"
- --reasoning-parser
- qwen3
qwen38-accurate:
container_name: cmvr-qwen38-accurate
image: quay.io/gpustack/runner:cuda13.0-sglang0.5.15.post1
restart: unless-stopped
ipc: host
shm_size: 32gb
ports:
- "192.168.28.10:14082:30000"
environment:
SGLANG_USE_MODELSCOPE: "true"
MODELSCOPE_CACHE: /root/.cache/modelscope
MODELSCOPE_DOWNLOAD_PARALLEL_WORKERS: "4"
volumes:
- ../models/modelscope:/root/.cache/modelscope
gpus:
- driver: nvidia
device_ids: ["3"]
capabilities: [gpu]
command:
- python
- -m
- sglang.launch_server
- --model-path
- Qwen/Qwen3.8-27B
- --served-model-name
- Qwen/Qwen3.8-27B
- --host
- 0.0.0.0
- --port
- "30000"
- --tp-size
- "1"
- --context-length
- "262144"
- --kv-cache-dtype
- fp8_e4m3
- --mem-fraction-static
- "0.92"
- --reasoning-parser
- qwen3

View File

@ -1,7 +1,9 @@
name: cmvr-ai-analysis-qwen38
services: services:
analysis-service: analysis-service:
container_name: cmvr-ai-analysis container_name: cmvr-ai-analysis
image: cmvr-ai-analysis:1.0.0 image: cmvr-ai-analysis:1.2.0-qwen38
build: build:
context: .. context: ..
dockerfile: Dockerfile dockerfile: Dockerfile
@ -17,6 +19,10 @@ services:
JOBS_DIR: /data/jobs JOBS_DIR: /data/jobs
OLLAMA_BASE_URL: http://host.docker.internal:11434 OLLAMA_BASE_URL: http://host.docker.internal:11434
OLLAMA_MODEL: qwen3-vl:32b OLLAMA_MODEL: qwen3-vl:32b
SGLANG_FAST_BASE_URL: http://192.168.28.10:14081/v1
SGLANG_ACCURATE_BASE_URL: http://192.168.28.10:14082/v1
SGLANG_TIMEOUT_SECONDS: "600"
VISION_OLLAMA_FALLBACK_ENABLED: "true"
extra_hosts: extra_hosts:
- "host.docker.internal:host-gateway" - "host.docker.internal:host-gateway"
volumes: volumes:

View File

@ -4,7 +4,7 @@ from pathlib import Path
import numpy as np import numpy as np
from app.audio import build_audio_profile, classify_audio from app.audio import build_audio_profile, classify_audio, classify_sequence
from app.config import settings from app.config import settings
from app.profile_store import Profile from app.profile_store import Profile
@ -45,6 +45,7 @@ def test_audio_profile_classifies_and_rejects(tmp_path: Path) -> None:
"POWER_OFF": 660, "POWER_OFF": 660,
"ARMED": 880, "ARMED": 880,
"DISARMED": 1100, "DISARMED": 1100,
"FIND_VEHICLE_HORN": 1320,
} }
for label, frequency in reference_tones.items(): for label, frequency in reference_tones.items():
_write_tone(directory / "references" / label / f"{label.lower()}.wav", frequency) _write_tone(directory / "references" / label / f"{label.lower()}.wav", frequency)
@ -62,3 +63,62 @@ def test_audio_profile_classifies_and_rejects(tmp_path: Path) -> None:
for label, path in classified_paths.items(): for label, path in classified_paths.items():
assert classify_audio(profile, path)["label"] == label assert classify_audio(profile, path)["label"] == label
assert classify_audio(profile, other_path)["label"] == "UNKNOWN" assert classify_audio(profile, other_path)["label"] == "UNKNOWN"
def test_audio_profile_aggregates_multiple_references(monkeypatch) -> None:
config = {
"decision": {
"maxDistance": 0.45,
"minDistanceMargin": 0.001,
"topK": 3,
"nearestWeight": 0.6,
"unknownLabel": "UNKNOWN",
}
}
profile = Profile("test.aggregate.v1", "AUDIO_CLASSIFICATION", Path("."), config)
references = [np.array([[value]], dtype=np.float32) for value in range(1, 7)]
distances = {1: 0.10, 2: 0.11, 3: 0.12, 4: 0.0, 5: 0.40, 6: 0.40}
monkeypatch.setattr(
"app.audio.sequence_distance",
lambda _query, reference: distances[int(reference[0, 0])],
)
result = classify_sequence(
profile,
np.array([[0.0]], dtype=np.float32),
references,
np.array(["EXPECTED"] * 3 + ["OUTLIER"] * 3),
np.array([f"sample-{index}.wav" for index in range(6)]),
)
assert result["label"] == "EXPECTED"
assert result["evidence"]["neighborCount"] == 3
assert result["thresholds"]["topK"] == 3
def test_audio_profile_applies_label_specific_threshold(monkeypatch) -> None:
config = {
"decision": {
"maxDistance": 0.45,
"minDistanceMargin": 0.015,
"labelMaxDistance": {"FIND_VEHICLE_HORN": 0.35},
"unknownLabel": "UNKNOWN",
}
}
profile = Profile("test.threshold.v1", "AUDIO_CLASSIFICATION", Path("."), config)
references = [np.array([[1.0]], dtype=np.float32), np.array([[2.0]], dtype=np.float32)]
monkeypatch.setattr(
"app.audio.sequence_distance",
lambda _query, reference: 0.36 if int(reference[0, 0]) == 1 else 0.44,
)
result = classify_sequence(
profile,
np.array([[0.0]], dtype=np.float32),
references,
np.array(["FIND_VEHICLE_HORN", "POWER_OFF"]),
np.array(["horn.wav", "power-off.wav"]),
)
assert result["label"] == "UNKNOWN"
assert result["thresholds"]["maxDistance"] == 0.35

View File

@ -29,7 +29,7 @@ class _FakeClient:
async def __aexit__(self, *_): async def __aexit__(self, *_):
return None return None
async def post(self, _, json): async def post(self, _, json, headers=None):
self.payloads.append(json) self.payloads.append(json)
return _FakeResponse(self.responses.pop(0)) return _FakeResponse(self.responses.pop(0))
@ -61,26 +61,266 @@ def test_vision_request_retries_empty_content(monkeypatch):
assert metadata["doneReason"] == "stop" assert metadata["doneReason"] == "stop"
assert len(_FakeClient.payloads) == 2 assert len(_FakeClient.payloads) == 2
assert _FakeClient.payloads[0]["think"] is False assert _FakeClient.payloads[0]["think"] is False
assert _FakeClient.payloads[1]["options"]["num_predict"] == 1536 assert _FakeClient.payloads[1]["options"]["num_predict"] == 4096
assert _FakeClient.payloads[1]["think"] is False
assert "/no_think" in _FakeClient.payloads[1]["messages"][0]["content"]
assert metadata["requestedOutputTokens"] == 4096
def test_vision_request_escalates_budget_after_repeated_length_cutoff(monkeypatch):
_FakeClient.responses = [
{
"done_reason": "length",
"eval_count": 768,
"message": {"content": "", "thinking": "first reasoning"},
},
{
"done_reason": "length",
"eval_count": 4096,
"message": {"content": "", "thinking": "more reasoning"},
},
{
"done_reason": "stop",
"eval_count": 30,
"message": {"content": '{"passed": true}', "thinking": ""},
},
]
_FakeClient.payloads = []
monkeypatch.setattr(httpx, "AsyncClient", _FakeClient)
payload = {
"think": False,
"messages": [{"role": "user", "content": "Analyze /no_think"}],
"options": {"num_predict": 768},
}
content, metadata = asyncio.run(
video._request_vision_model(payload, empty_response_retries=2)
)
assert content == '{"passed": true}'
assert [_payload["options"]["num_predict"] for _payload in _FakeClient.payloads] == [
768,
4096,
8192,
]
assert metadata["attempt"] == 3
def test_sglang_payload_uses_openai_multimodal_format():
payload = video._sglang_payload(
{
"model": "Qwen/Qwen3.8-27B-FP8",
"think": False,
"messages": [
{"role": "user", "content": "Analyze", "images": ["abc", "def"]}
],
"options": {"temperature": 0.1, "num_predict": 512},
}
)
assert payload["model"] == "Qwen/Qwen3.8-27B-FP8"
assert payload["messages"][0]["content"][0] == {
"type": "text",
"text": "Analyze",
}
assert payload["messages"][0]["content"][1]["image_url"]["url"] == (
"data:image/jpeg;base64,abc"
)
assert payload["chat_template_kwargs"]["enable_thinking"] is False
assert payload["response_format"]["json_schema"]["schema"] == video.VIDEO_RESULT_SCHEMA
def test_sglang_response_is_parsed(monkeypatch):
_FakeClient.responses = [
{
"choices": [
{
"finish_reason": "stop",
"message": {"content": '{"passed": true}', "reasoning_content": ""},
}
],
"usage": {"prompt_tokens": 100, "completion_tokens": 20},
}
]
_FakeClient.payloads = []
monkeypatch.setattr(httpx, "AsyncClient", _FakeClient)
payload = {
"model": "Qwen/Qwen3.8-27B-FP8",
"think": False,
"messages": [{"role": "user", "content": "Analyze", "images": ["abc"]}],
"options": {"temperature": 0.1, "num_predict": 512},
}
content, metadata = asyncio.run(
video._request_vision_model(
payload,
provider="sglang",
base_url="http://sglang-fast:30000/v1",
)
)
assert content == '{"passed": true}'
assert metadata["provider"] == "sglang"
assert metadata["promptEvalCount"] == 100
assert metadata["evalCount"] == 20
assert _FakeClient.payloads[0]["messages"][0]["content"][1]["type"] == "image_url"
def test_fast_result_quality_controls_accurate_fallback(): def test_fast_result_quality_controls_accurate_fallback():
complete = { complete = {
"passed": True, "passed": True,
"evidenceSufficient": True,
"confidence": 0.9,
"conclusion": "通过", "conclusion": "通过",
"summary": "状态正常", "summary": "状态正常",
"events": [], "events": [],
"warnings": [], "warnings": [],
} }
assert video._fallback_reason(complete, True) is None assert video._fallback_reason(complete, 0.75) is None
assert video._fallback_reason({**complete, "passed": None}, True) == "FAST_RESULT_INSUFFICIENT" assert (
assert video._fallback_reason({**complete, "passed": None}, False) is None video._fallback_reason({**complete, "evidenceSufficient": False}, 0.75)
assert video._fallback_reason({"summary": "缺少字段"}, True).startswith( == "FAST_RESULT_INSUFFICIENT"
)
assert (
video._fallback_reason({**complete, "confidence": 0.6}, 0.75)
== "FAST_RESULT_LOW_CONFIDENCE"
)
assert video._fallback_reason({"summary": "缺少字段"}, 0.75).startswith(
"FAST_RESULT_MISSING_FIELDS:" "FAST_RESULT_MISSING_FIELDS:"
) )
def test_decision_is_fail_closed_when_evidence_is_insufficient():
result = video._normalize_decision(
{
"passed": True,
"evidenceSufficient": False,
"confidence": 0.9,
"conclusion": "可能通过",
"summary": "没有拍到完整过程",
"events": [],
"warnings": [],
},
0.75,
)
assert result["passed"] is False
assert result["decisionReason"] == "INSUFFICIENT_EVIDENCE"
assert result["conclusion"].startswith("未通过:视频证据不足")
def test_negative_acceptance_criterion_corrects_inconsistent_model_boolean():
result = video._normalize_decision(
{
"passed": False,
"evidenceSufficient": True,
"confidence": 1.0,
"conclusion": "蓝牙图标从有到无",
"summary": "随着视频推进,蓝牙图标消失,画面内不再显示图标。",
"events": [
{"timeRange": "0-3s", "event": "蓝牙图标存在", "confidence": 1.0},
{"timeRange": "3-6s", "event": "蓝牙图标消失", "confidence": 1.0},
],
"warnings": [],
},
0.7,
"仪表蓝牙图标是否消失",
)
assert result["passed"] is True
assert result["modelPassed"] is False
assert result["decisionReason"] == "CRITERION_EVIDENCE_CORRECTION"
def test_negative_acceptance_criterion_does_not_pass_when_state_remains():
result = video._normalize_decision(
{
"passed": True,
"evidenceSufficient": True,
"confidence": 0.95,
"conclusion": "蓝牙图标未消失",
"summary": "视频结束时蓝牙图标仍然显示。",
"events": [],
"warnings": [],
},
0.7,
"仪表蓝牙图标是否消失",
)
assert result["passed"] is False
assert result["modelPassed"] is True
assert result["decisionReason"] == "CRITERION_EVIDENCE_CORRECTION"
assert result["conclusion"].startswith("未通过:分析结论未满足通过标准")
def test_disappearance_guidance_requires_first_and_final_frame_comparison():
guidance = video._temporal_acceptance_guidance("仪表蓝牙图标是否消失")
assert "第一张与最后一张" in guidance
assert "passed必须为true" in guidance
assert "末段图片中已不再显示仪表蓝牙图标" in guidance
assert "全程持续存在" in guidance
assert video._needs_temporal_removal_verification(
"仪表蓝牙图标是否消失",
{"passed": False, "evidenceSufficient": True},
) is True
assert video._needs_temporal_removal_verification(
"仪表蓝牙图标是否出现",
{"passed": False, "evidenceSufficient": True},
) is False
def test_video_tuning_is_bounded_and_typed():
assert video._normalize_tuning(
{
"sampleFps": "2",
"maxWidth": 960,
"confidenceThreshold": 0.8,
"fallbackToAccurate": False,
}
) == {
"sampleFps": 2.0,
"maxWidth": 960,
"confidenceThreshold": 0.8,
"fallbackToAccurate": False,
}
try:
video._normalize_tuning({"sampleFps": 6.5})
assert False, "out-of-range tuning must fail"
except ValueError:
pass
def test_frame_sampling_always_includes_video_tail():
count, prefix_count, prefix_fps, tail_time, capped = video._frame_sampling_plan(
6.48, 1.0, 100
)
assert count == 8
assert prefix_count == 7
assert prefix_fps == 1.0
assert round(tail_time, 2) == 6.38
assert capped is False
count, prefix_count, prefix_fps, tail_time, capped = video._frame_sampling_plan(
20.0, 6.0, 100
)
assert count == 100
assert prefix_count == 99
assert round(prefix_fps, 2) == 4.95
assert round(tail_time, 1) == 19.9
assert capped is True
def test_video_context_expands_with_frame_count():
assert video._video_context_size(16384, 7, 512) == 16384
assert video._video_context_size(16384, 27, 512) == 65536
assert video._video_context_size(32768, 100, 768) == 262144
def test_auto_mode_falls_back_to_accurate_result(monkeypatch, tmp_path): def test_auto_mode_falls_back_to_accurate_result(monkeypatch, tmp_path):
frame_dir = tmp_path / "frames" frame_dir = tmp_path / "frames"
frame_dir.mkdir() frame_dir.mkdir()
@ -99,7 +339,9 @@ def test_auto_mode_falls_back_to_accurate_result(monkeypatch, tmp_path):
if mode == "FAST": if mode == "FAST":
return ( return (
{ {
"passed": None, "passed": False,
"evidenceSufficient": False,
"confidence": 0.4,
"conclusion": "证据不足", "conclusion": "证据不足",
"summary": "快速分析无法判断", "summary": "快速分析无法判断",
"events": [], "events": [],
@ -110,6 +352,8 @@ def test_auto_mode_falls_back_to_accurate_result(monkeypatch, tmp_path):
return ( return (
{ {
"passed": True, "passed": True,
"evidenceSufficient": True,
"confidence": 0.92,
"conclusion": "通过", "conclusion": "通过",
"summary": "精确分析确认通过", "summary": "精确分析确认通过",
"events": [], "events": [],

View File

@ -0,0 +1,199 @@
import argparse
import csv
import hashlib
import json
import shutil
import subprocess
from collections import Counter
from pathlib import Path
from urllib.parse import urlparse
import httpx
SUPPORTED_LABELS = {
"POWER_ON",
"POWER_OFF",
"ARMED",
"DISARMED",
"FIND_VEHICLE_HORN",
}
def file_sha256(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as stream:
for chunk in iter(lambda: stream.read(1024 * 1024), b""):
digest.update(chunk)
return digest.hexdigest()
def audio_duration(path: Path) -> float:
process = subprocess.run(
[
"ffprobe",
"-v",
"error",
"-show_entries",
"format=duration",
"-of",
"default=noprint_wrappers=1:nokey=1",
str(path),
],
capture_output=True,
text=True,
check=False,
)
if process.returncode != 0:
detail = process.stderr.strip()[-500:]
raise ValueError(f"ffprobe failed: {detail or process.returncode}")
return float(process.stdout.strip())
def download(client: httpx.Client, url: str, destination: Path) -> None:
with client.stream("GET", url) as response:
response.raise_for_status()
with destination.open("wb") as stream:
for chunk in response.iter_bytes():
stream.write(chunk)
def media_suffix(url: str) -> str:
suffix = Path(urlparse(url).path).suffix.lower()
return suffix if suffix and len(suffix) <= 10 else ".audio"
def existing_reference_hashes(reference_root: Path) -> dict[str, str]:
hashes: dict[str, str] = {}
for label_dir in sorted(reference_root.iterdir()):
if not label_dir.is_dir():
continue
for path in sorted(label_dir.iterdir()):
if path.is_file():
hashes.setdefault(file_sha256(path), label_dir.name)
return hashes
def import_references(
manifest: Path,
profile_dir: Path,
import_dir: Path,
minimum_duration: float,
commit: bool,
) -> dict:
reference_root = profile_dir / "references"
download_dir = import_dir / "downloads"
accepted_dir = import_dir / "accepted"
download_dir.mkdir(parents=True, exist_ok=True)
accepted_dir.mkdir(parents=True, exist_ok=True)
known_hashes = existing_reference_hashes(reference_root)
batch_hashes: dict[str, str] = {}
audit_rows: list[dict[str, str]] = []
counts: Counter[str] = Counter()
with manifest.open("r", encoding="utf-8-sig", newline="") as stream:
rows = list(csv.DictReader(stream, delimiter="\t"))
timeout = httpx.Timeout(120, connect=10)
with httpx.Client(timeout=timeout, follow_redirects=True) as client:
for row in rows:
log_id = str(row.get("log_id") or "").strip()
label = str(row.get("trusted_label") or "").strip().upper()
url = str(row.get("media_url") or "").strip()
audit = dict(row)
audit.update({"duration_seconds": "", "sha256": "", "status": ""})
if label not in SUPPORTED_LABELS:
audit["status"] = "SKIPPED_NO_TRUSTED_LABEL"
counts[audit["status"]] += 1
audit_rows.append(audit)
continue
if not url.startswith(("http://", "https://")):
audit["status"] = "SKIPPED_INVALID_URL"
counts[audit["status"]] += 1
audit_rows.append(audit)
continue
downloaded = download_dir / f"{log_id}{media_suffix(url)}"
try:
if not downloaded.exists():
download(client, url, downloaded)
duration = audio_duration(downloaded)
digest = file_sha256(downloaded)
audit["duration_seconds"] = f"{duration:.3f}"
audit["sha256"] = digest
except (httpx.HTTPError, OSError, ValueError) as exc:
audit["status"] = f"FAILED:{type(exc).__name__}:{str(exc)[:200]}"
counts["FAILED"] += 1
audit_rows.append(audit)
continue
if duration < minimum_duration:
audit["status"] = "SKIPPED_TOO_SHORT"
elif digest in known_hashes:
audit["status"] = f"SKIPPED_EXISTING_REFERENCE:{known_hashes[digest]}"
elif digest in batch_hashes:
audit["status"] = f"SKIPPED_DUPLICATE_IMPORT:{batch_hashes[digest]}"
else:
batch_hashes[digest] = label
staged = accepted_dir / label / downloaded.name
staged.parent.mkdir(parents=True, exist_ok=True)
if not staged.exists():
shutil.copy2(downloaded, staged)
if commit:
destination = reference_root / label / downloaded.name
destination.parent.mkdir(parents=True, exist_ok=True)
if not destination.exists():
shutil.copy2(staged, destination)
audit["status"] = "IMPORTED"
else:
audit["status"] = "READY"
counts[audit["status"].split(":", 1)[0]] += 1
audit_rows.append(audit)
audit_path = import_dir / "audit.csv"
fieldnames = list(audit_rows[0].keys()) if audit_rows else ["status"]
with audit_path.open("w", encoding="utf-8-sig", newline="") as stream:
writer = csv.DictWriter(stream, fieldnames=fieldnames)
writer.writeheader()
writer.writerows(audit_rows)
ready_by_label = Counter(
row.get("trusted_label", "")
for row in audit_rows
if row["status"] in {"READY", "IMPORTED"}
)
return {
"manifestRows": len(rows),
"minimumDurationSeconds": minimum_duration,
"committed": commit,
"statusCounts": dict(sorted(counts.items())),
"acceptedByLabel": dict(sorted(ready_by_label.items())),
"auditPath": str(audit_path),
}
def main() -> None:
parser = argparse.ArgumentParser(
description="Import trusted audio references from an exported log manifest"
)
parser.add_argument("--manifest", required=True, type=Path)
parser.add_argument("--profile-dir", required=True, type=Path)
parser.add_argument("--import-dir", required=True, type=Path)
parser.add_argument("--minimum-duration", type=float, default=4.0)
parser.add_argument("--commit", action="store_true")
args = parser.parse_args()
result = import_references(
args.manifest,
args.profile_dir,
args.import_dir,
args.minimum_duration,
args.commit,
)
print(json.dumps(result, ensure_ascii=False, indent=2))
if __name__ == "__main__":
main()