aps-agent/server/aps_domain/drawing_pdf.py

213 lines
8.3 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

# ============================================================
# PDF 图纸解析与预览(moduleId: domain-drawing-pdf, 可重生 ✅)
# R71.1 扩展:PDF 与 DXF 共用 drawing-master-candidates.v1 候选契约。
# 诚实边界:只提取文本/表格/页数,不做几何推断;数量/工时/资源保持 None。
# ============================================================
from __future__ import annotations
import re
from collections import Counter
from datetime import UTC, datetime
from pathlib import Path
from typing import Any
from server.aps_domain.drawing_dxf import (
_ITEM_REF_RE,
_clean_text,
_field_candidate,
_file_digest,
_filename_metadata,
_stable_id,
)
MAX_PDF_BYTES = 50 * 1024 * 1024
MAX_PDF_PAGES = 200
MAX_PDF_TEXT_BLOCKS = 2_000
_TITLE_FIELD_PATTERNS: tuple[tuple[str, re.Pattern[str]], ...] = (
("drawingTitle", re.compile(r"(?:TITLE|图名|图纸名称)\s*[::]?\s*(.+)", re.IGNORECASE)),
("material", re.compile(r"(?:MATERIAL|材料|材质)\s*[::]?\s*(.+)", re.IGNORECASE)),
("scale", re.compile(r"(?:SCALE|比例)\s*[::]?\s*(\d+\s*[::]\s*\d+)", re.IGNORECASE)),
("drawnBy", re.compile(r"(?:DRAWN BY|DESIGNED BY|设计|制图)\s*[::]?\s*(.+)", re.IGNORECASE)),
("date", re.compile(r"(?:DATE|日期)\s*[::]?\s*(\d{4}[-/.]\d{1,2}[-/.]\d{1,2})", re.IGNORECASE)),
)
def _utc_now() -> str:
return datetime.now(UTC).isoformat().replace("+00:00", "Z")
def _safe_path(path: str | Path) -> Path:
p = Path(path).expanduser().resolve(strict=True)
if not p.is_file() or p.suffix.lower() != ".pdf":
raise ValueError("Only an existing .pdf file can be inspected")
size = p.stat().st_size
if size <= 0:
raise ValueError("PDF file is empty")
if size > MAX_PDF_BYTES:
raise ValueError(f"PDF file exceeds {MAX_PDF_BYTES} bytes")
return p
def inspect_pdf(path: str | Path) -> dict[str, Any]:
"""Parse a PDF into the same stable review-only evidence contract as DXF."""
p = _safe_path(path)
digest = _file_digest(p)
try:
import pdfplumber
except ImportError as exc: # pragma: no cover - environment guard
raise RuntimeError("PDF parsing requires the 'pdfplumber' package") from exc
try:
with pdfplumber.open(p) as pdf:
page_count = len(pdf.pages)
if page_count > MAX_PDF_PAGES:
raise ValueError(f"PDF 页数超过 {MAX_PDF_PAGES} 页上限")
texts: list[dict[str, Any]] = []
tables: list[dict[str, Any]] = []
item_refs: list[dict[str, Any]] = []
stats: Counter[str] = Counter()
for page_no, page in enumerate(pdf.pages, start=1):
if page_no > MAX_PDF_PAGES:
break
raw = page.extract_text() or ""
for line_no, line in enumerate(_clean_text(raw).splitlines()):
if len(texts) >= MAX_PDF_TEXT_BLOCKS:
break
texts.append({
"text": line,
"page": page_no,
"line": line_no + 1,
"evidence": {"page": page_no, "line": line_no + 1, "source": "text"},
})
for match in _ITEM_REF_RE.finditer(line):
item_refs.append({
"reference": match.group(0),
"page": page_no,
"line": line_no + 1,
"confidence": 0.62,
"evidence": {"page": page_no, "line": line_no + 1, "text": line},
})
try:
extracted_tables = page.extract_tables() or []
except Exception: # noqa: BLE001 - 表结构不稳定的 PDF 不阻断文本解析
extracted_tables = []
for table in extracted_tables:
tables.append({
"page": page_no,
"rows": len(table),
"columns": max((len(row) for row in table), default=0),
})
stats["page"] = page_count
stats["text_block"] = len(texts)
stats["table"] = len(tables)
except ValueError:
raise
except Exception as exc:
raise ValueError(f"Invalid or unsupported PDF: {exc}") from exc
name_meta = _filename_metadata(p)
drawing_no = str(name_meta.get("drawingNumber") or p.stem)
revision = name_meta.get("revision")
text_blob = "\n".join(str(row.get("text") or "") for row in texts)
in_text = drawing_no in text_blob
field_candidates = [
_field_candidate(
"drawingNumber", drawing_no, 0.93 if in_text else 0.62,
{"filename": p.name, "source": "filename+text" if in_text else "filename", "textFound": in_text},
review_required=not in_text,
),
_field_candidate(
"revision", revision, 0.90 if revision else 0.40,
{"filename": p.name, "source": "filename"},
review_required=not revision,
),
]
for field, pattern in _TITLE_FIELD_PATTERNS:
match = pattern.search(text_blob)
if not match:
continue
value = match.group(1).strip() if match.lastindex else match.group(0).strip()
field_candidates.append(
_field_candidate(field, value, 0.70, {"source": "text", "text": match.group(0)},
review_required=True)
)
asset_id = _stable_id("drawing", digest)
parsed = {
"asset": {
"id": asset_id,
"kind": "PDF",
"filename": p.name,
"sourcePath": str(p),
"sha256": digest,
"sizeBytes": p.stat().st_size,
"mimeType": "application/pdf",
},
"drawing": {
"drawingNumber": drawing_no,
"revision": revision,
"pageCount": page_count,
"modelspaceEntityCount": None,
"paperSpaceEntityCount": None,
"metadata": {"source": "pdf", "pageCount": page_count},
},
"fieldCandidates": field_candidates,
"texts": texts,
"itemReferences": item_refs,
"dimensions": [],
"layers": [],
"entityStatistics": dict(stats),
"tables": tables,
"warnings": [
"PDF 文本与表格仅为候选证据,数量/单位/工时/资源不得从表格自动落库。",
"扫描件/图片型 PDF 无文本层时,需要真实 OCR(R71.6)接入。",
],
"parsedAt": _utc_now(),
}
from server.aps_domain.drawing_process import recognize_process_candidates
process = recognize_process_candidates(parsed)
for detail in process.get("details") or []:
parsed["fieldCandidates"].append(_field_candidate(
detail["field"], detail["value"], float(detail["confidence"]),
detail["evidence"], review_required=True,
))
return parsed
def render_pdf_preview(path: str | Path, *, scale: float = 1.5, page_index: int = 0) -> dict[str, Any]:
"""Render the first PDF page as a PNG data URL for the drawing canvas."""
p = _safe_path(path)
try:
import base64
import io
import pypdfium2 as pdfium
except ImportError as exc: # pragma: no cover - environment guard
raise RuntimeError("PDF preview requires the 'pypdfium2' package") from exc
try:
document = pdfium.PdfDocument(str(p))
if len(document) == 0:
raise ValueError("PDF 没有可渲染页面")
index = max(0, min(int(page_index), len(document) - 1))
bitmap = document[index].render(scale=max(0.25, float(scale)))
image = bitmap.to_pil().convert("RGB")
buffer = io.BytesIO()
image.save(buffer, format="PNG")
encoded = base64.b64encode(buffer.getvalue()).decode("ascii")
return {
"mimeType": "image/png",
"dataUrl": f"data:image/png;base64,{encoded}",
"pageCount": len(document),
"pageIndex": index,
"width": image.width,
"height": image.height,
}
except ValueError:
raise
except Exception as exc:
raise ValueError(f"PDF 预览渲染失败:{exc}") from exc
__all__ = ["inspect_pdf", "render_pdf_preview"]