Files
DPM/backend/app/vision_llm.py
T
Pine 107727ea5e feat: Refactor DpmOverlays to use ShowCard for rendering cards and add ToolStatusToast for operation status notifications
- Moved card rendering logic from DpmOverlays to a new ShowCard component for better reusability.
- Introduced ToolStatusToast to display real-time operation statuses in the top right corner.
- Updated PageHeader to conditionally render credits based on the current path.
- Modified PromptPanel to change tool names and update prompt titles.
- Enhanced ScreenLayout to include ToolStatusToast.
- Updated styles for new components and adjusted existing styles for consistency.
- Implemented statusBus utility for dispatching tool status events.
- Updated useMqttControl to integrate tool status notifications during navigation and card display actions.
2026-08-19 12:28:35 +08:00

84 lines
3.0 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
"""对话开场画面识别 —— 用多模态 LLM(qwen3-vl-flash)识别实时画面中的人数、性别等,
并把结果作为上下文注入对话,让 AI 了解当前在场观众。
与 vision_yoloYOLO 数人脸/姿态)互补:这里用 LLM 做语义级理解(人数、性别构成、场景描述)。
"""
import json
import logging
import ssl
import urllib.request
import certifi
from .config import settings
log = logging.getLogger("dpm.vision")
_SSL_CTX = ssl.create_default_context(cafile=certifi.where())
_DEFAULT_PROMPT = (
"请识别这张实时画面,只输出一个 JSON 对象(不要输出任何其他文字):"
'{"people": 画面中人数(int), "males": 其中男性人数(int), "females": 其中女性人数(int), '
'"desc": 一句话中文描述画面(含大致人数、性别构成、人物大致状态,如年龄/坐站/是否看屏幕)。'
"若画面无人或不确定,则 people=0、males=0、females=0desc='画面中暂时没有人'"
)
def analyze_scene(jpeg_b64: str, prompt: str = "") -> dict:
"""调用多模态 LLM 识别画面中人数/性别,返回 {ok, people, males, females, desc}。"""
payload = {
"model": settings.VISION_LLM_MODEL,
"messages": [{
"role": "user",
"content": [
{"type": "text", "text": prompt or _DEFAULT_PROMPT},
{"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{jpeg_b64}"}},
],
}],
}
req = urllib.request.Request(
f"{settings.LLM_BASE_URL}/chat/completions",
data=json.dumps(payload, ensure_ascii=False).encode("utf-8"),
headers={
"Content-Type": "application/json",
"Authorization": f"Bearer {settings.DASHSCOPE_API_KEY}",
},
)
with urllib.request.urlopen(req, timeout=30, context=_SSL_CTX) as resp:
data = json.loads(resp.read().decode("utf-8"))
content = data["choices"][0]["message"].get("content") or ""
return _parse(content)
def _parse(text: str) -> dict:
"""从模型输出中抽取 JSON 对象并规范化为结果 dict。"""
text = (text or "").strip()
if text.startswith("```"):
lines = [l for l in text.splitlines() if not l.strip().startswith("```")]
text = "\n".join(lines).strip()
obj: dict = {}
i, j = text.find("{"), text.rfind("}")
if i != -1 and j > i:
try:
obj = json.loads(text[i:j + 1])
except Exception: # noqa: BLE001
log.warning("vision llm: 返回 JSON 解析失败: %s", text[:120])
people = _as_int(obj.get("people"))
males = _as_int(obj.get("males"))
females = _as_int(obj.get("females"))
return {
"ok": True,
"people": people,
"males": males,
"females": females,
"desc": str(obj.get("desc") or "").strip(),
}
def _as_int(v):
try:
return int(float(v))
except (TypeError, ValueError):
return 0