init the project
This commit is contained in:
@@ -0,0 +1,85 @@
|
||||
# Vision LLM wrapper for analyzing document images (type + description).
|
||||
|
||||
import logging
|
||||
from typing import Optional
|
||||
|
||||
from server.core.llm_provider.router import router
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
PROMPT_IMAGE = """请分析这张图片,判断类型并输出文字描述。
|
||||
|
||||
## 判断图片类型
|
||||
|
||||
如果是 **流程图 / 架构图 / 状态图 / 时序图 / 活动图**,详细描述:
|
||||
- 图中所有节点/步骤/状态/组件的名称
|
||||
- 所有连线/箭头/转换关系及其方向
|
||||
- 所有分支条件、判断逻辑和判断结果
|
||||
- 所有文字标注、注释、标签
|
||||
- 图的整体结构和逻辑流程
|
||||
- 如果图片包含多个子图,拆解描述
|
||||
|
||||
如果是 **其他类型**(UI原型图 / 界面截图 / 设计稿 / 手机屏幕截图 / 网页截图等),简要描述图片内容。
|
||||
|
||||
## 输出格式
|
||||
|
||||
**1. 类型标签(单独一行):**
|
||||
type: <flowchart|architecture|state|sequence|activity|other>
|
||||
|
||||
**2. 文字描述:**
|
||||
该图片的详细文字描述。
|
||||
|
||||
不要输出 ---YAML--- 分隔符或 YAML 内容,不要添加任何额外的解释或问候语。"""
|
||||
|
||||
|
||||
class ImageParser:
|
||||
"""Analyze document images using Qwen VL model.
|
||||
|
||||
Usage::
|
||||
|
||||
parser = ImageParser()
|
||||
result = parser.parse_image("images/flow.png")
|
||||
# {"type": "flowchart", "description": "..."}
|
||||
"""
|
||||
|
||||
_VALID_TYPES = {"flowchart", "architecture", "state", "sequence", "activity", "other"}
|
||||
|
||||
def __init__(self):
|
||||
self._llm = router.get_image_client()
|
||||
|
||||
@property
|
||||
def usage(self) -> dict:
|
||||
return self._llm.usage
|
||||
|
||||
def parse_image(self, image_path: str) -> Optional[dict]:
|
||||
"""Parse an image and return {type, description}."""
|
||||
logger.info("Parsing image: %s", image_path)
|
||||
|
||||
try:
|
||||
content = self._llm.chat_with_image(
|
||||
model=router.image_model,
|
||||
image_path=image_path,
|
||||
prompt=PROMPT_IMAGE,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error("Image parsing failed: %s", e)
|
||||
return {"type": "other", "description": "", "error": str(e)}
|
||||
|
||||
return self._parse_response(content)
|
||||
|
||||
def _parse_response(self, content: str) -> dict:
|
||||
"""Extract (type, description) from vision model response."""
|
||||
content = content.strip()
|
||||
parsed_type = "other"
|
||||
desc_lines: list[str] = []
|
||||
|
||||
for line in content.splitlines():
|
||||
stripped = line.strip()
|
||||
if (stripped.startswith("type:") or stripped.startswith("类型:")):
|
||||
type_val = stripped.split(":", 1)[1].strip().lower()
|
||||
if type_val in self._VALID_TYPES:
|
||||
parsed_type = type_val
|
||||
else:
|
||||
desc_lines.append(line)
|
||||
|
||||
return {"type": parsed_type, "description": "\n".join(desc_lines).strip()}
|
||||
Reference in New Issue
Block a user