aps-agent/server/aps_domain/drawing_ai.py

249 lines
10 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

# ============================================================
# AI 精确解析(moduleId: domain-drawing-ai, 可重生 ✅)
# 优先 LLM 结构化提取;未配置/失败/非 JSON -> fail-closed 回退确定性算法,绝不编造。
# ============================================================
from __future__ import annotations
import json
import re
from typing import Any
AI_VERSION = "drawing-ai.v1"
_SYSTEM_PROMPT = """你是工业图纸结构化解析器。只能根据用户提供的图纸证据提取信息,禁止编造。
输出严格 JSON 对象:
{
"drawingNumber": str|null,
"revision": str|null,
"title": str|null,
"scale": str|null,
"format": str|null,
"material": str|null,
"surfaceTreatment": str|null,
"processSteps": [{"code": str, "name": str, "description": str, "confidence": number}],
"bomRows": [{"position": str|null, "reference": str|null, "name": str|null, "quantity": number|null, "unit": str|null}],
"details": {"thread": str|null, "roughness": str|null, "tolerance": str|null, "depthMm": number|null},
"blocks": [{"blockCode": str, "blockName": str|null}],
"assemblies": [{"assemblyName": str}],
"locations": [{"location": str}],
"sequenceSteps": [{"stepIndex": number, "sentence": str, "markers": [str]}],
"confidence": number,
"warnings": [str]
}
提取不到的值一律 null;数量/工时/资源没有证据时不得猜测。"""
def _evidence_text(parsed: dict[str, Any]) -> str:
parts: list[str] = []
asset = parsed.get("asset") or {}
drawing = parsed.get("drawing") or {}
parts.append(f"文件名: {asset.get('filename')}")
parts.append(f"图号: {drawing.get('drawingNumber')} 版本: {drawing.get('revision')}")
for row in parsed.get("fieldCandidates") or []:
parts.append(f"字段 {row.get('field')}: {row.get('value')} (置信度 {row.get('confidence')})")
parts.append("文本:")
for row in (parsed.get("texts") or [])[:200]:
parts.append(str(row.get("text") or ""))
refs = parsed.get("itemReferences") or []
if refs:
parts.append("零件引用: " + ", ".join(str(row.get("reference")) for row in refs[:100]))
return "\n".join(parts)
def _deterministic_summary(parsed: dict[str, Any]) -> dict[str, Any]:
fields = {row.get("field"): row.get("value") for row in parsed.get("fieldCandidates") or []}
from server.aps_domain.drawing_process import recognize_process_candidates
process = recognize_process_candidates(parsed)
return {
"drawingNumber": fields.get("drawingNumber"),
"revision": fields.get("revision"),
"title": fields.get("title"),
"scale": fields.get("scale"),
"format": fields.get("format"),
"material": fields.get("material") or fields.get("materialSpecification"),
"processSteps": [
{
"code": row["operationCode"],
"name": row["operationName"],
"description": (row.get("evidence") or {}).get("text"),
"confidence": row["confidence"],
}
for row in process.get("operations") or []
],
"bomRows": [],
"details": {
"thread": fields.get("threadSpecification"),
"roughness": fields.get("surfaceRoughness"),
"tolerance": fields.get("tolerance"),
"depthMm": fields.get("depthMm"),
},
"blocks": [{"blockCode": row["blockCode"], "blockName": row.get("blockName")} for row in process.get("blocks") or []],
"assemblies": [{"assemblyName": row["assemblyName"]} for row in process.get("assemblies") or []],
"locations": [{"location": row["location"]} for row in process.get("locations") or []],
"sequenceSteps": [
{"stepIndex": row["stepIndex"], "sentence": row["sentence"], "markers": row.get("markers") or []}
for row in process.get("sequenceSteps") or []
],
"confidence": 0.0,
"warnings": ["AI 未启用或不可用,使用确定性解析结果;字段仍需人工复核。"],
}
def _coerce_analysis(raw: Any) -> dict[str, Any] | None:
"""宽容解析模型返回:dict / JSON 文本 / markdown 围栏 / 双重编码 JSON。"""
if isinstance(raw, dict):
return raw
if not isinstance(raw, str):
return None
text = re.sub(r"^```(?:json)?\s*|\s*```$", "", raw.strip(), flags=re.IGNORECASE).strip()
attempts: list[Any] = [text]
try:
attempts.append(json.loads(text))
except Exception: # noqa: BLE001, S110 - 交给后续解析
pass
match = re.search(r"\{.*\}", text, re.DOTALL)
if match:
attempts.append(match.group(0))
for candidate in attempts:
if isinstance(candidate, str):
try:
candidate = json.loads(candidate)
except Exception: # noqa: BLE001, S112 - 继续尝试
continue
if isinstance(candidate, dict):
return candidate
return None
def _candidate_evidence(
parsed: dict[str, Any],
candidate_type: str,
row: dict[str, Any],
) -> dict[str, Any]:
asset = parsed.get("asset") or {}
drawing = parsed.get("drawing") or {}
evidence = {
"source": "drawingAiAnalysis",
"candidateType": candidate_type,
"drawingAssetId": asset.get("id"),
"sourceSha256": asset.get("sha256"),
"filename": asset.get("filename"),
"drawingNumber": drawing.get("drawingNumber"),
"revision": drawing.get("revision"),
}
supplied = row.get("evidence")
if supplied:
evidence["modelEvidence"] = supplied
return {key: value for key, value in evidence.items() if value not in (None, "")}
def _normalize_review_candidates(
parsed: dict[str, Any],
analysis: dict[str, Any],
) -> dict[str, Any]:
"""把模型抽取统一降为有证据、待审核且不自动补工时/资源的候选。"""
normalized = dict(analysis)
def normalize_group(
key: str,
candidate_type: str,
*,
prohibit_schedule_values: bool = False,
) -> list[dict[str, Any]]:
rows: list[dict[str, Any]] = []
for raw_row in normalized.get(key) or []:
if not isinstance(raw_row, dict):
continue
row = dict(raw_row)
row["candidateType"] = candidate_type
row["status"] = "PENDING_REVIEW"
row["reviewRequired"] = True
if prohibit_schedule_values:
row["standardTime"] = None
row["resourceCode"] = None
unknown = list(row.get("unknownFields") or [])
for field in ("standardTime", "resourceCode"):
if field not in unknown:
unknown.append(field)
row["unknownFields"] = unknown
row["evidence"] = _candidate_evidence(parsed, candidate_type, row)
rows.append(row)
normalized[key] = rows
return rows
normalize_group("processSteps", "ROUTING_OPERATION", prohibit_schedule_values=True)
if "routingOperations" in normalized:
normalize_group("routingOperations", "ROUTING_OPERATION", prohibit_schedule_values=True)
normalize_group("bomRows", "BOM_ROW")
if "bomCandidates" in normalized:
normalize_group("bomCandidates", "BOM_ROW")
material_rows = normalized.get("materialCandidates")
if not isinstance(material_rows, list):
material_rows = normalized.get("materials")
if not isinstance(material_rows, list):
material = normalized.get("material")
material_rows = [{"value": material}] if material not in (None, "") else []
normalized["materialCandidates"] = material_rows
normalize_group("materialCandidates", "MATERIAL")
if "materials" in normalized:
normalized["materials"] = [dict(row) for row in normalized["materialCandidates"]]
normalized["status"] = "PENDING_REVIEW"
normalized["reviewRequired"] = True
normalized["writesMasterData"] = False
normalized["prohibitedAutoFill"] = ["routing.standardTime", "routing.resourceCode"]
return normalized
async def analyze_drawing_ai(
parsed: dict[str, Any],
*,
provider: Any | None = None,
) -> dict[str, Any]:
deterministic = _normalize_review_candidates(parsed, _deterministic_summary(parsed))
if provider is None or not getattr(provider, "enabled", False):
return {
"contractVersion": AI_VERSION,
"status": "AI_UNAVAILABLE",
"reason": "ai-provider-not-configured",
"message": "未配置 AI 模型服务,已展示确定性解析结果",
"analysis": deterministic,
"deterministic": True,
}
provider_error: str | None = None
analysis: Any = None
try:
analysis = await provider.chat_json(_SYSTEM_PROMPT, _evidence_text(parsed), timeout=30.0)
except Exception as exc: # noqa: BLE001 - provider 异常一律 fail-closed
provider_error = f"provider-error:{type(exc).__name__}"
analysis = _coerce_analysis(analysis)
if not isinstance(analysis, dict) and hasattr(provider, "chat_text"):
try:
raw_text = await provider.chat_text(_SYSTEM_PROMPT, _evidence_text(parsed), timeout=45.0)
analysis = _coerce_analysis(raw_text)
except Exception: # noqa: BLE001 - 兜底失败继续回退
analysis = None
if not isinstance(analysis, dict):
return {
"contractVersion": AI_VERSION,
"status": "AI_UNAVAILABLE",
"reason": provider_error or "provider-invalid-json",
"message": "模型返回格式无效或服务异常,已回退确定性解析结果",
"analysis": deterministic,
"deterministic": True,
}
analysis = _normalize_review_candidates(parsed, analysis)
analysis.setdefault("warnings", [])
analysis.setdefault("confidence", 0.0)
return {
"contractVersion": AI_VERSION,
"status": "AI_ANALYZED",
"analysis": analysis,
"deterministic": False,
}
__all__ = ["AI_VERSION", "analyze_drawing_ai"]