feat: add pcf analysis web app and api
This commit is contained in:
@@ -0,0 +1 @@
|
||||
"""PCF 焊接数据导入生成工具。"""
|
||||
@@ -0,0 +1,64 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import yaml
|
||||
|
||||
|
||||
ROOT_DIR = Path(__file__).resolve().parents[1]
|
||||
CONFIG_PATH = ROOT_DIR / "config" / "mapping.yaml"
|
||||
|
||||
|
||||
@dataclass
|
||||
class AppConfig:
|
||||
path: Path
|
||||
data: dict[str, Any]
|
||||
|
||||
@classmethod
|
||||
def load(cls, path: Path = CONFIG_PATH) -> "AppConfig":
|
||||
with path.open("r", encoding="utf-8") as f:
|
||||
data = yaml.safe_load(f) or {}
|
||||
return cls(path=path, data=data)
|
||||
|
||||
def save(self) -> None:
|
||||
with self.path.open("w", encoding="utf-8") as f:
|
||||
yaml.safe_dump(self.data, f, allow_unicode=True, sort_keys=False)
|
||||
|
||||
@property
|
||||
def defaults(self) -> dict[str, str]:
|
||||
return self.data.setdefault("defaults", {})
|
||||
|
||||
@property
|
||||
def field_mappings(self) -> dict[str, str]:
|
||||
return self.data.setdefault("field_mappings", {})
|
||||
|
||||
@property
|
||||
def dynamic_field_mappings(self) -> dict[str, str]:
|
||||
return self.data.setdefault("dynamic_field_mappings", {})
|
||||
|
||||
@property
|
||||
def ignored_fields(self) -> set[str]:
|
||||
return set(self.data.setdefault("ignored_fields", []))
|
||||
|
||||
@property
|
||||
def bore_map(self) -> dict[str, dict[str, str]]:
|
||||
return self.data.setdefault("bore_map", {})
|
||||
|
||||
@property
|
||||
def deepseek(self) -> dict[str, Any]:
|
||||
return self.data.setdefault("deepseek", {})
|
||||
|
||||
@property
|
||||
def template_path(self) -> Path:
|
||||
configured = self.data.get("template", {}).get("path", "焊接数据导入模版.xls")
|
||||
return ROOT_DIR / configured
|
||||
|
||||
@property
|
||||
def output_sheet(self) -> str:
|
||||
return self.data.get("template", {}).get("output_sheet", "Sheet1")
|
||||
|
||||
@property
|
||||
def confidence_threshold(self) -> float:
|
||||
return float(self.deepseek.get("confidence_threshold", 0.85))
|
||||
+114
@@ -0,0 +1,114 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
from .config import AppConfig
|
||||
from .models import MappingDecision, UnknownFieldCandidate
|
||||
from .pcf_parser import STANDARD_TARGET_FIELDS
|
||||
|
||||
|
||||
class DeepSeekMapper:
|
||||
def __init__(self, config: AppConfig):
|
||||
self.config = config
|
||||
self.api_key = str(config.deepseek.get("api_key", "") or "").strip()
|
||||
self.base_url = config.deepseek.get("base_url", "https://api.deepseek.com").rstrip("/")
|
||||
self.model = config.deepseek.get("model", "deepseek-v4-flash")
|
||||
|
||||
@property
|
||||
def enabled(self) -> bool:
|
||||
return bool(self.api_key)
|
||||
|
||||
async def map_unknown_fields(
|
||||
self, candidates: list[UnknownFieldCandidate]
|
||||
) -> list[MappingDecision]:
|
||||
if not self.enabled or not candidates:
|
||||
return []
|
||||
|
||||
payload = {
|
||||
"standard_fields": sorted(STANDARD_TARGET_FIELDS),
|
||||
"candidates": [
|
||||
{
|
||||
"source_field": c.source_field,
|
||||
"section": c.section,
|
||||
"sample_values": c.sample_values,
|
||||
"context": c.context,
|
||||
}
|
||||
for c in candidates[:50]
|
||||
],
|
||||
}
|
||||
prompt = (
|
||||
"你是 PCF 管道文件字段映射助手。请把未知 PCF 字段映射到标准字段。"
|
||||
"只在语义和样例值明确时映射;不确定时 target_field 返回空字符串,confidence 低于 0.85。"
|
||||
"必须返回 JSON,格式为 {\"mappings\": [{\"source_field\": str, \"target_field\": str, "
|
||||
"\"confidence\": number, \"reason\": str, \"sample_values\": [str]}]}。\n\n"
|
||||
f"输入:{json.dumps(payload, ensure_ascii=False)}"
|
||||
)
|
||||
|
||||
response = await self._chat_json(prompt)
|
||||
raw_mappings = response.get("mappings", response if isinstance(response, list) else [])
|
||||
decisions: list[MappingDecision] = []
|
||||
for item in raw_mappings:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
try:
|
||||
decision = MappingDecision(
|
||||
source_field=str(item.get("source_field", "")).strip(),
|
||||
target_field=str(item.get("target_field", "")).strip(),
|
||||
confidence=float(item.get("confidence", 0)),
|
||||
reason=str(item.get("reason", "")).strip(),
|
||||
sample_values=[str(v) for v in item.get("sample_values", [])],
|
||||
)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
if decision.is_usable:
|
||||
decisions.append(decision)
|
||||
return decisions
|
||||
|
||||
async def _chat_json(self, prompt: str) -> dict[str, Any] | list[Any]:
|
||||
url = f"{self.base_url}/chat/completions"
|
||||
headers = {
|
||||
"Authorization": f"Bearer {self.api_key}",
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
body = {
|
||||
"model": self.model,
|
||||
"messages": [
|
||||
{"role": "system", "content": "只输出合法 JSON,不要输出 Markdown。"},
|
||||
{"role": "user", "content": prompt},
|
||||
],
|
||||
"response_format": {"type": "json_object"},
|
||||
"temperature": 0,
|
||||
}
|
||||
async with httpx.AsyncClient(timeout=60) as client:
|
||||
resp = await client.post(url, headers=headers, json=body)
|
||||
resp.raise_for_status()
|
||||
content = resp.json()["choices"][0]["message"]["content"]
|
||||
return json.loads(content)
|
||||
|
||||
|
||||
def persist_high_confidence_mappings(
|
||||
config: AppConfig,
|
||||
decisions: list[MappingDecision],
|
||||
threshold: float | None = None,
|
||||
) -> dict[str, str]:
|
||||
threshold = config.confidence_threshold if threshold is None else threshold
|
||||
accepted: dict[str, str] = {}
|
||||
changed = False
|
||||
for decision in decisions:
|
||||
if (
|
||||
decision.confidence >= threshold
|
||||
and decision.target_field in STANDARD_TARGET_FIELDS
|
||||
and decision.source_field not in config.field_mappings
|
||||
):
|
||||
current = config.dynamic_field_mappings.get(decision.source_field)
|
||||
if current and current != decision.target_field:
|
||||
continue
|
||||
config.dynamic_field_mappings[decision.source_field] = decision.target_field
|
||||
accepted[decision.source_field] = decision.target_field
|
||||
changed = True
|
||||
if changed:
|
||||
config.save()
|
||||
return accepted
|
||||
@@ -0,0 +1,185 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import xlrd
|
||||
from xlutils.copy import copy as copy_workbook
|
||||
|
||||
from .config import AppConfig
|
||||
from .models import WeldRecord
|
||||
|
||||
|
||||
EXCEL_FIELD_TO_RECORD = {
|
||||
"管线代号": "line_no",
|
||||
"焊口代号": "weld_no",
|
||||
"焊接区域(安装/预制)": "weld_area",
|
||||
"规格(mm)": "spec",
|
||||
"壁厚": "wall_thickness",
|
||||
"单线图号": "pipeline_reference",
|
||||
"管线等级代号": "piping_spec",
|
||||
"外径": "outside_diameter",
|
||||
}
|
||||
|
||||
EXCEL_FIELD_TO_ENGLISH = {
|
||||
"单位代码": "unit_code",
|
||||
"装置工区编号": "plant_area_code",
|
||||
"管线代号": "line_no",
|
||||
"焊口代号": "weld_no",
|
||||
"材质类型代号": "material_type_code",
|
||||
"材质1代号": "material_1_code",
|
||||
"材质2代号": "material_2_code",
|
||||
"材料1": "material_1",
|
||||
"材料2": "material_2",
|
||||
"探伤比例代号": "ndt_ratio_code",
|
||||
"焊缝类型代号": "weld_type_code",
|
||||
"焊接区域(安装/预制)": "weld_area",
|
||||
"焊口属性(固定、活动)": "weld_property",
|
||||
"达因数": "dia_inch",
|
||||
"规格(mm)": "spec",
|
||||
"壁厚": "wall_thickness",
|
||||
"焊接方法代码": "welding_method_code",
|
||||
"试验压力": "test_pressure",
|
||||
"焊条代号": "welding_rod_code",
|
||||
"焊丝代号": "welding_wire_code",
|
||||
"介质代号": "medium_code",
|
||||
"单线图号": "isometric_no",
|
||||
"设计压力": "design_pressure",
|
||||
"设计温度": "design_temperature",
|
||||
"坡口代号": "groove_code",
|
||||
"管线等级代号": "piping_class_code",
|
||||
"组件一代号": "component_1_code",
|
||||
"组件二代号": "component_2_code",
|
||||
"炉批号一": "heat_no_1",
|
||||
"炉批号二": "heat_no_2",
|
||||
"所属管段": "pipe_spool",
|
||||
"预热温度": "preheat_temperature",
|
||||
"是否需热处理(是,否)": "need_heat_treatment",
|
||||
"热处理编号": "heat_treatment_no",
|
||||
"焊接位置(1G/2G/3G/4G/5G/6G)": "welding_position",
|
||||
"外径": "outside_diameter",
|
||||
"硬度检测比例(数值)": "hardness_test_ratio",
|
||||
"焊接气体保护": "shielding_gas",
|
||||
"是否非标(是/否)": "is_non_standard",
|
||||
"壁板号": "plate_no",
|
||||
"延长米": "extended_meter",
|
||||
"管道长度": "pipe_length",
|
||||
}
|
||||
|
||||
|
||||
class XlsTemplateWriter:
|
||||
def __init__(self, config: AppConfig):
|
||||
self.config = config
|
||||
|
||||
def write(self, records: list[WeldRecord], output_path: Path) -> None:
|
||||
template_path = self.config.template_path
|
||||
if not template_path.exists():
|
||||
raise FileNotFoundError(f"模板不存在:{template_path}")
|
||||
|
||||
rb = xlrd.open_workbook(
|
||||
str(template_path),
|
||||
formatting_info=True,
|
||||
on_demand=False,
|
||||
)
|
||||
sheet_index = rb.sheet_names().index(self.config.output_sheet)
|
||||
rs = rb.sheet_by_index(sheet_index)
|
||||
wb = copy_workbook(rb)
|
||||
ws = wb.get_sheet(sheet_index)
|
||||
|
||||
headers = [str(rs.cell_value(0, c)).strip() for c in range(rs.ncols)]
|
||||
template_style_by_col = self._template_styles(rb, rs, headers)
|
||||
|
||||
# 清空模板中已有示例数据,避免生成记录少于示例时残留旧值。
|
||||
for r in range(1, rs.nrows):
|
||||
for c in range(rs.ncols):
|
||||
ws.write(r, c, "", template_style_by_col[c])
|
||||
|
||||
for row_offset, record in enumerate(records, start=1):
|
||||
row_values = self.row_values(headers, record)
|
||||
for col_idx, value in enumerate(row_values):
|
||||
ws.write(row_offset, col_idx, value, template_style_by_col[col_idx])
|
||||
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
wb.save(str(output_path))
|
||||
|
||||
def _template_styles(self, rb: xlrd.book.Book, rs: xlrd.sheet.Sheet, headers: list[str]):
|
||||
styles = []
|
||||
for col_idx, _header in enumerate(headers):
|
||||
source_row = 1 if rs.nrows > 1 else 0
|
||||
xf_idx = rs.cell_xf_index(source_row, col_idx)
|
||||
styles.append(rb.xf_list[xf_idx])
|
||||
# xlutils 复制后需要使用目标工作簿内部样式对象。
|
||||
# 经验上直接使用 rb.xf_list 会失败,因此回退到 xlwt 默认样式时由调用方兜底。
|
||||
return [self._xf_to_xlwt_style(rb, xf) for xf in styles]
|
||||
|
||||
def _xf_to_xlwt_style(self, rb: xlrd.book.Book, xf: xlrd.formatting.XF):
|
||||
import xlwt
|
||||
|
||||
style = xlwt.XFStyle()
|
||||
font = rb.font_list[xf.font_index]
|
||||
style.font.name = font.name
|
||||
style.font.bold = font.bold
|
||||
style.font.italic = font.italic
|
||||
style.font.height = font.height
|
||||
style.font.colour_index = font.colour_index
|
||||
|
||||
fmt = rb.format_map.get(xf.format_key)
|
||||
if fmt:
|
||||
style.num_format_str = fmt.format_str
|
||||
|
||||
alignment = style.alignment
|
||||
alignment.horz = xf.alignment.hor_align
|
||||
alignment.vert = xf.alignment.vert_align
|
||||
alignment.wrap = xf.alignment.text_wrapped
|
||||
|
||||
borders = style.borders
|
||||
borders.left = xf.border.left_line_style
|
||||
borders.right = xf.border.right_line_style
|
||||
borders.top = xf.border.top_line_style
|
||||
borders.bottom = xf.border.bottom_line_style
|
||||
borders.left_colour = xf.border.left_colour_index
|
||||
borders.right_colour = xf.border.right_colour_index
|
||||
borders.top_colour = xf.border.top_colour_index
|
||||
borders.bottom_colour = xf.border.bottom_colour_index
|
||||
|
||||
pattern = style.pattern
|
||||
pattern.pattern = xf.background.fill_pattern
|
||||
pattern.pattern_fore_colour = xf.background.pattern_colour_index
|
||||
pattern.pattern_back_colour = xf.background.background_colour_index
|
||||
return style
|
||||
|
||||
def headers(self) -> list[str]:
|
||||
rb = xlrd.open_workbook(str(self.config.template_path), formatting_info=False, on_demand=True)
|
||||
try:
|
||||
rs = rb.sheet_by_name(self.config.output_sheet)
|
||||
return [str(rs.cell_value(0, c)).strip() for c in range(rs.ncols)]
|
||||
finally:
|
||||
rb.release_resources()
|
||||
|
||||
def build_import_rows(
|
||||
self, records: list[WeldRecord]
|
||||
) -> tuple[list[str], list[dict[str, str]], list[dict[str, str]]]:
|
||||
headers = self.headers()
|
||||
columns = self.import_columns(headers)
|
||||
keys = [column["key"] for column in columns]
|
||||
rows = [dict(zip(keys, self.row_values(headers, record))) for record in records]
|
||||
return headers, columns, rows
|
||||
|
||||
def import_columns(self, headers: list[str]) -> list[dict[str, str]]:
|
||||
used: set[str] = set()
|
||||
columns: list[dict[str, str]] = []
|
||||
for index, header in enumerate(headers, start=1):
|
||||
key = EXCEL_FIELD_TO_ENGLISH.get(header) or f"column_{index:02d}"
|
||||
if key in used:
|
||||
key = f"{key}_{index:02d}"
|
||||
used.add(key)
|
||||
columns.append({"key": key, "label": header})
|
||||
return columns
|
||||
|
||||
def row_values(self, headers: list[str], record: WeldRecord) -> list[str]:
|
||||
values: list[str] = []
|
||||
for header in headers:
|
||||
if header in EXCEL_FIELD_TO_RECORD:
|
||||
values.append(str(getattr(record, EXCEL_FIELD_TO_RECORD[header], "") or ""))
|
||||
else:
|
||||
values.append(str(self.config.defaults.get(header, "") or ""))
|
||||
return values
|
||||
@@ -0,0 +1,465 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import csv
|
||||
import json
|
||||
import shutil
|
||||
import uuid
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
from urllib.parse import unquote, urlparse
|
||||
from urllib.request import url2pathname
|
||||
|
||||
import httpx
|
||||
from fastapi import UploadFile
|
||||
|
||||
from .config import AppConfig, ROOT_DIR
|
||||
from .deepseek import DeepSeekMapper, persist_high_confidence_mappings
|
||||
from .excel_writer import XlsTemplateWriter
|
||||
from .models import JobResult, WeldRecord
|
||||
from .pcf_parser import PcfParser
|
||||
|
||||
|
||||
JOBS_DIR = ROOT_DIR / "outputs" / "jobs"
|
||||
UPLOADS_DIR = ROOT_DIR / "outputs" / "uploads"
|
||||
MAX_ATTACHMENT_BYTES = 100 * 1024 * 1024
|
||||
|
||||
|
||||
class JobStore:
|
||||
def __init__(self) -> None:
|
||||
self.results: dict[str, JobResult] = {}
|
||||
|
||||
def set(self, result: JobResult) -> None:
|
||||
self.results[result.job_id] = result
|
||||
|
||||
def get(self, job_id: str) -> JobResult | None:
|
||||
return self.results.get(job_id)
|
||||
|
||||
|
||||
job_store = JobStore()
|
||||
|
||||
|
||||
def safe_filename(name: str) -> str:
|
||||
keep = []
|
||||
for ch in Path(name).name:
|
||||
if ch.isalnum() or ch in ".-_() ":
|
||||
keep.append(ch)
|
||||
else:
|
||||
keep.append("_")
|
||||
return "".join(keep) or "attachment.pcf"
|
||||
|
||||
|
||||
async def process_uploads(files: list[UploadFile]) -> JobResult:
|
||||
job_id = uuid.uuid4().hex[:12]
|
||||
upload_dir = UPLOADS_DIR / job_id
|
||||
upload_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
saved_files: list[Path] = []
|
||||
messages: list[str] = []
|
||||
for upload in files:
|
||||
filename = safe_filename(upload.filename or "upload.pcf")
|
||||
target = upload_dir / filename
|
||||
if target.suffix.lower() != ".pcf":
|
||||
messages.append(f"拒绝非 PCF 文件:{filename}")
|
||||
continue
|
||||
with target.open("wb") as f:
|
||||
shutil.copyfileobj(upload.file, f)
|
||||
saved_files.append(target)
|
||||
|
||||
return await process_pcf_paths(job_id=job_id, pcf_files=saved_files, initial_messages=messages)
|
||||
|
||||
|
||||
async def analyze_attachment_urls(urls: list[str]) -> dict:
|
||||
job_id = uuid.uuid4().hex[:12]
|
||||
download_dir = UPLOADS_DIR / job_id
|
||||
download_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
pcf_files, messages = await download_attachment_urls(urls, download_dir)
|
||||
analysis = await analyze_pcf_paths(pcf_files=pcf_files, initial_messages=messages)
|
||||
return analysis_to_json_payload(job_id=job_id, input_files=len(pcf_files), analysis=analysis)
|
||||
|
||||
|
||||
async def process_pcf_paths(
|
||||
job_id: str,
|
||||
pcf_files: list[Path],
|
||||
initial_messages: list[str] | None = None,
|
||||
) -> JobResult:
|
||||
job_dir = JOBS_DIR / job_id
|
||||
job_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
analysis = await analyze_pcf_paths(pcf_files=pcf_files, initial_messages=initial_messages)
|
||||
config = analysis["config"]
|
||||
records = analysis["records"]
|
||||
candidates = analysis["candidates"]
|
||||
decisions = analysis["decisions"]
|
||||
messages = analysis["messages"]
|
||||
import_headers = analysis["import_headers"]
|
||||
import_columns = analysis["import_columns"]
|
||||
import_rows = analysis["import_rows"]
|
||||
issue_rows = analysis["issue_rows"]
|
||||
mapper_enabled = analysis["llm_enabled"]
|
||||
|
||||
output_xls = job_dir / "焊接数据导入结果.xls"
|
||||
issues_csv = job_dir / "问题清单.csv"
|
||||
report_json = job_dir / "解析报告.json"
|
||||
|
||||
writer = XlsTemplateWriter(config)
|
||||
writer.write(records, output_xls)
|
||||
write_issues_csv(issue_rows, issues_csv)
|
||||
write_report_json(
|
||||
records=records,
|
||||
candidates=candidates,
|
||||
decisions=decisions,
|
||||
accepted_mappings=analysis["accepted_mappings"],
|
||||
messages=messages,
|
||||
report_path=report_json,
|
||||
llm_enabled=mapper_enabled,
|
||||
)
|
||||
|
||||
warning_count = analysis["warning_count"]
|
||||
error_count = analysis["error_count"]
|
||||
|
||||
result = JobResult(
|
||||
job_id=job_id,
|
||||
input_files=len(pcf_files),
|
||||
weld_count=len(records),
|
||||
warning_count=warning_count,
|
||||
error_count=error_count,
|
||||
output_xls=output_xls,
|
||||
issues_csv=issues_csv,
|
||||
report_json=report_json,
|
||||
messages=messages,
|
||||
llm_enabled=mapper_enabled,
|
||||
llm_mappings=decisions,
|
||||
import_headers=import_headers,
|
||||
import_columns=import_columns,
|
||||
import_rows=import_rows,
|
||||
issue_rows=issue_rows,
|
||||
)
|
||||
job_store.set(result)
|
||||
return result
|
||||
|
||||
|
||||
async def analyze_pcf_paths(
|
||||
pcf_files: list[Path],
|
||||
initial_messages: list[str] | None = None,
|
||||
) -> dict:
|
||||
messages = list(initial_messages or [])
|
||||
config = AppConfig.load()
|
||||
parser = PcfParser(config)
|
||||
records, candidates, parse_messages = parser.parse_files(pcf_files)
|
||||
messages.extend(parse_messages)
|
||||
|
||||
mapper = DeepSeekMapper(config)
|
||||
decisions = []
|
||||
accepted_mappings = {}
|
||||
if mapper.enabled and candidates:
|
||||
try:
|
||||
decisions = await mapper.map_unknown_fields(candidates)
|
||||
accepted_mappings = persist_high_confidence_mappings(config, decisions)
|
||||
if accepted_mappings:
|
||||
config = AppConfig.load()
|
||||
parser = PcfParser(config)
|
||||
records, _, parse_messages = parser.parse_files(pcf_files, accepted_mappings)
|
||||
messages.extend(parse_messages)
|
||||
except Exception as exc:
|
||||
messages.append(f"DeepSeek 字段映射失败,已降级为规则解析:{exc}")
|
||||
elif candidates:
|
||||
messages.append("未在 config/mapping.yaml 配置 deepseek.api_key,已跳过非标准字段智能映射")
|
||||
|
||||
writer = XlsTemplateWriter(config)
|
||||
import_headers, import_columns, import_rows = writer.build_import_rows(records)
|
||||
issue_rows = build_issue_rows(records, candidates, decisions, config.confidence_threshold)
|
||||
|
||||
warning_count = sum(len(r.issues) for r in records) + len(messages)
|
||||
error_count = 0 if records else 1
|
||||
if not records:
|
||||
messages.append("未解析到任何焊口记录")
|
||||
|
||||
return {
|
||||
"config": config,
|
||||
"records": records,
|
||||
"candidates": candidates,
|
||||
"decisions": decisions,
|
||||
"accepted_mappings": accepted_mappings,
|
||||
"messages": messages,
|
||||
"llm_enabled": mapper.enabled,
|
||||
"import_headers": import_headers,
|
||||
"import_columns": import_columns,
|
||||
"import_rows": import_rows,
|
||||
"issue_rows": issue_rows,
|
||||
"warning_count": warning_count,
|
||||
"error_count": error_count,
|
||||
}
|
||||
|
||||
|
||||
async def download_attachment_urls(urls: list[str], target_dir: Path) -> tuple[list[Path], list[str]]:
|
||||
pcf_files: list[Path] = []
|
||||
messages: list[str] = []
|
||||
async with httpx.AsyncClient(follow_redirects=True, timeout=60) as client:
|
||||
for index, url in enumerate(urls, start=1):
|
||||
parsed = urlparse(url)
|
||||
if is_local_attachment(url, parsed):
|
||||
local_files, local_messages = collect_local_attachment(url, target_dir, index)
|
||||
pcf_files.extend(local_files)
|
||||
messages.extend(local_messages)
|
||||
continue
|
||||
|
||||
if parsed.scheme not in {"http", "https"}:
|
||||
messages.append(f"附件引用协议不支持:{url}")
|
||||
continue
|
||||
|
||||
downloaded, download_messages = await download_remote_attachment(client, url, target_dir, index)
|
||||
pcf_files.extend(downloaded)
|
||||
messages.extend(download_messages)
|
||||
return pcf_files, messages
|
||||
|
||||
|
||||
def is_local_attachment(raw: str, parsed) -> bool:
|
||||
if parsed.scheme == "file":
|
||||
return True
|
||||
if parsed.scheme in {"http", "https"}:
|
||||
return False
|
||||
if len(raw) >= 3 and raw[1] == ":" and raw[2] in {"\\", "/"}:
|
||||
return True
|
||||
if raw.startswith("\\\\"):
|
||||
return True
|
||||
return parsed.scheme == ""
|
||||
|
||||
|
||||
def collect_local_attachment(raw: str, target_dir: Path, index: int) -> tuple[list[Path], list[str]]:
|
||||
messages: list[str] = []
|
||||
pcf_files: list[Path] = []
|
||||
source = local_path_from_reference(raw)
|
||||
|
||||
if not source.exists() or not source.is_file():
|
||||
messages.append(f"本地附件不存在或不是文件:{raw}")
|
||||
return pcf_files, messages
|
||||
if source.stat().st_size > MAX_ATTACHMENT_BYTES:
|
||||
messages.append(f"本地附件超过大小限制:{source}")
|
||||
return pcf_files, messages
|
||||
|
||||
target = target_dir / safe_filename(source.name or f"attachment_{index}.pcf")
|
||||
if source.resolve() != target.resolve():
|
||||
shutil.copyfile(source, target)
|
||||
else:
|
||||
target = source
|
||||
|
||||
if target.suffix.lower() == ".pcf":
|
||||
pcf_files.append(target)
|
||||
elif target.suffix.lower() == ".zip":
|
||||
extracted, extract_messages = extract_pcf_zip(target, target_dir / f"zip_{index}")
|
||||
pcf_files.extend(extracted)
|
||||
messages.extend(extract_messages)
|
||||
else:
|
||||
messages.append(f"本地附件不是 PCF 或 ZIP,已跳过:{source.name}")
|
||||
return pcf_files, messages
|
||||
|
||||
|
||||
def local_path_from_reference(raw: str) -> Path:
|
||||
parsed = urlparse(raw)
|
||||
if parsed.scheme == "file":
|
||||
if parsed.netloc and parsed.netloc not in {"localhost", "127.0.0.1"}:
|
||||
return Path(f"//{parsed.netloc}{url2pathname(parsed.path)}")
|
||||
return Path(url2pathname(unquote(parsed.path)))
|
||||
return Path(raw)
|
||||
|
||||
|
||||
async def download_remote_attachment(
|
||||
client: httpx.AsyncClient,
|
||||
url: str,
|
||||
target_dir: Path,
|
||||
index: int,
|
||||
) -> tuple[list[Path], list[str]]:
|
||||
pcf_files: list[Path] = []
|
||||
messages: list[str] = []
|
||||
try:
|
||||
response = await client.get(url)
|
||||
response.raise_for_status()
|
||||
except Exception as exc:
|
||||
messages.append(f"附件下载失败:{url},原因:{exc}")
|
||||
return pcf_files, messages
|
||||
|
||||
content = response.content
|
||||
if len(content) > MAX_ATTACHMENT_BYTES:
|
||||
messages.append(f"附件超过大小限制:{url}")
|
||||
return pcf_files, messages
|
||||
|
||||
filename = filename_from_response(url, response.headers, index)
|
||||
saved = target_dir / filename
|
||||
saved.write_bytes(content)
|
||||
|
||||
if saved.suffix.lower() == ".pcf":
|
||||
pcf_files.append(saved)
|
||||
elif saved.suffix.lower() == ".zip":
|
||||
extracted, extract_messages = extract_pcf_zip(saved, target_dir / f"zip_{index}")
|
||||
pcf_files.extend(extracted)
|
||||
messages.extend(extract_messages)
|
||||
else:
|
||||
messages.append(f"附件不是 PCF 或 ZIP,已跳过:{filename}")
|
||||
return pcf_files, messages
|
||||
|
||||
|
||||
def filename_from_response(url: str, headers: httpx.Headers, index: int) -> str:
|
||||
disposition = headers.get("content-disposition", "")
|
||||
marker = "filename="
|
||||
if marker in disposition:
|
||||
raw = disposition.split(marker, 1)[1].strip().strip('"')
|
||||
return safe_filename(unquote(raw))
|
||||
parsed_name = Path(unquote(urlparse(url).path)).name
|
||||
return safe_filename(parsed_name or f"attachment_{index}.pcf")
|
||||
|
||||
|
||||
def extract_pcf_zip(zip_path: Path, target_dir: Path) -> tuple[list[Path], list[str]]:
|
||||
target_dir.mkdir(parents=True, exist_ok=True)
|
||||
pcf_files: list[Path] = []
|
||||
messages: list[str] = []
|
||||
try:
|
||||
with zipfile.ZipFile(zip_path) as archive:
|
||||
for member in archive.infolist():
|
||||
if member.is_dir():
|
||||
continue
|
||||
name = safe_filename(Path(member.filename).name)
|
||||
if not name.lower().endswith(".pcf"):
|
||||
continue
|
||||
target = target_dir / name
|
||||
with archive.open(member) as src, target.open("wb") as dst:
|
||||
shutil.copyfileobj(src, dst)
|
||||
pcf_files.append(target)
|
||||
except zipfile.BadZipFile:
|
||||
messages.append(f"ZIP 附件无法解压:{zip_path.name}")
|
||||
if not pcf_files:
|
||||
messages.append(f"ZIP 附件中未找到 PCF 文件:{zip_path.name}")
|
||||
return pcf_files, messages
|
||||
|
||||
|
||||
def analysis_to_json_payload(job_id: str, input_files: int, analysis: dict) -> dict:
|
||||
records: list[WeldRecord] = analysis["records"]
|
||||
return {
|
||||
"job_id": job_id,
|
||||
"input_files": input_files,
|
||||
"weld_count": len(records),
|
||||
"warning_count": analysis["warning_count"],
|
||||
"error_count": analysis["error_count"],
|
||||
"messages": analysis["messages"],
|
||||
"llm_enabled": analysis["llm_enabled"],
|
||||
"tables": {
|
||||
"import_headers": analysis["import_headers"],
|
||||
"import_columns": analysis["import_columns"],
|
||||
"import_rows": analysis["import_rows"],
|
||||
"issue_rows": analysis["issue_rows"],
|
||||
},
|
||||
"report": build_report_payload(
|
||||
records=records,
|
||||
candidates=analysis["candidates"],
|
||||
decisions=analysis["decisions"],
|
||||
accepted_mappings=analysis["accepted_mappings"],
|
||||
messages=analysis["messages"],
|
||||
llm_enabled=analysis["llm_enabled"],
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def build_issue_rows(
|
||||
records: list[WeldRecord],
|
||||
candidates,
|
||||
decisions,
|
||||
threshold: float,
|
||||
) -> list[dict[str, str]]:
|
||||
rows: list[dict[str, str]] = []
|
||||
for record in records:
|
||||
for issue in record.issues:
|
||||
rows.append(
|
||||
{
|
||||
"类型": "解析警告",
|
||||
"文件": record.source_file,
|
||||
"焊口代号": record.weld_no,
|
||||
"字段": "",
|
||||
"内容": issue,
|
||||
"处理状态": "需复核",
|
||||
}
|
||||
)
|
||||
high_conf = {d.source_field for d in decisions if d.confidence >= threshold}
|
||||
for candidate in candidates:
|
||||
rows.append(
|
||||
{
|
||||
"类型": "未知字段",
|
||||
"文件": "",
|
||||
"焊口代号": "",
|
||||
"字段": candidate.source_field,
|
||||
"内容": "; ".join(candidate.sample_values),
|
||||
"处理状态": "已自动映射" if candidate.source_field in high_conf else "未映射/需复核",
|
||||
}
|
||||
)
|
||||
return rows
|
||||
|
||||
|
||||
def write_issues_csv(issue_rows: list[dict[str, str]], path: Path) -> None:
|
||||
with path.open("w", encoding="utf-8-sig", newline="") as f:
|
||||
writer = csv.writer(f)
|
||||
headers = ["类型", "文件", "焊口代号", "字段", "内容", "处理状态"]
|
||||
writer.writerow(headers)
|
||||
for row in issue_rows:
|
||||
writer.writerow([row.get(header, "") for header in headers])
|
||||
|
||||
|
||||
def write_report_json(
|
||||
records: list[WeldRecord],
|
||||
candidates,
|
||||
decisions,
|
||||
accepted_mappings: dict[str, str],
|
||||
messages: list[str],
|
||||
report_path: Path,
|
||||
llm_enabled: bool,
|
||||
) -> None:
|
||||
payload = build_report_payload(
|
||||
records=records,
|
||||
candidates=candidates,
|
||||
decisions=decisions,
|
||||
accepted_mappings=accepted_mappings,
|
||||
messages=messages,
|
||||
llm_enabled=llm_enabled,
|
||||
)
|
||||
with report_path.open("w", encoding="utf-8") as f:
|
||||
json.dump(payload, f, ensure_ascii=False, indent=2)
|
||||
|
||||
|
||||
def build_report_payload(
|
||||
records: list[WeldRecord],
|
||||
candidates,
|
||||
decisions,
|
||||
accepted_mappings: dict[str, str],
|
||||
messages: list[str],
|
||||
llm_enabled: bool,
|
||||
) -> dict:
|
||||
return {
|
||||
"summary": {
|
||||
"weld_count": len(records),
|
||||
"source_files": sorted({r.source_file for r in records}),
|
||||
"llm_enabled": llm_enabled,
|
||||
"unknown_field_count": len(candidates),
|
||||
},
|
||||
"messages": messages,
|
||||
"accepted_mappings": accepted_mappings,
|
||||
"llm_mappings": [
|
||||
{
|
||||
"source_field": d.source_field,
|
||||
"target_field": d.target_field,
|
||||
"confidence": d.confidence,
|
||||
"reason": d.reason,
|
||||
"sample_values": d.sample_values,
|
||||
}
|
||||
for d in decisions
|
||||
],
|
||||
"records_preview": [
|
||||
{
|
||||
"source_file": r.source_file,
|
||||
"pipeline_reference": r.pipeline_reference,
|
||||
"line_no": r.line_no,
|
||||
"weld_no": r.weld_no,
|
||||
"diameter": r.diameter,
|
||||
"weld_area": r.weld_area,
|
||||
"issues": r.issues,
|
||||
}
|
||||
for r in records[:20]
|
||||
],
|
||||
}
|
||||
+77
@@ -0,0 +1,77 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from fastapi import FastAPI, File, HTTPException, Request, UploadFile
|
||||
from fastapi.responses import FileResponse, HTMLResponse
|
||||
from fastapi.staticfiles import StaticFiles
|
||||
from fastapi.templating import Jinja2Templates
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from .config import ROOT_DIR
|
||||
from .job_service import analyze_attachment_urls, job_store, process_uploads
|
||||
|
||||
|
||||
app = FastAPI(title="PCF 焊接数据导入生成工具")
|
||||
templates = Jinja2Templates(directory=str(ROOT_DIR / "templates"))
|
||||
app.mount("/static", StaticFiles(directory=str(ROOT_DIR / "static")), name="static")
|
||||
|
||||
|
||||
class ParseUrlRequest(BaseModel):
|
||||
attachment_url: str | None = Field(default=None, description="单个附件 URL,支持 .pcf 或 .zip")
|
||||
attachment_urls: list[str] | None = Field(default=None, description="多个附件 URL,支持 .pcf 或 .zip")
|
||||
|
||||
def urls(self) -> list[str]:
|
||||
urls: list[str] = []
|
||||
if self.attachment_url:
|
||||
urls.append(self.attachment_url)
|
||||
if self.attachment_urls:
|
||||
urls.extend(self.attachment_urls)
|
||||
return [url for url in urls if url]
|
||||
|
||||
|
||||
@app.get("/", response_class=HTMLResponse)
|
||||
async def index(request: Request):
|
||||
return templates.TemplateResponse("index.html", {"request": request})
|
||||
|
||||
|
||||
@app.post("/api/jobs")
|
||||
async def create_job(files: list[UploadFile] = File(...)):
|
||||
if not files:
|
||||
raise HTTPException(status_code=400, detail="请至少上传一个 PCF 文件")
|
||||
result = await process_uploads(files)
|
||||
return result.to_public_dict()
|
||||
|
||||
|
||||
@app.post("/api/AnalysisPcf")
|
||||
async def analysis_pcf(request: ParseUrlRequest):
|
||||
urls = request.urls()
|
||||
if not urls:
|
||||
raise HTTPException(status_code=400, detail="请提供 attachment_url 或 attachment_urls")
|
||||
return await analyze_attachment_urls(urls)
|
||||
|
||||
|
||||
@app.get("/api/jobs/{job_id}")
|
||||
async def get_job(job_id: str):
|
||||
result = job_store.get(job_id)
|
||||
if not result:
|
||||
raise HTTPException(status_code=404, detail="任务不存在或服务已重启")
|
||||
return result.to_public_dict()
|
||||
|
||||
|
||||
@app.get("/api/jobs/{job_id}/download/{kind}")
|
||||
async def download(job_id: str, kind: str):
|
||||
result = job_store.get(job_id)
|
||||
if not result:
|
||||
raise HTTPException(status_code=404, detail="任务不存在或服务已重启")
|
||||
files = {
|
||||
"xls": (result.output_xls, "application/vnd.ms-excel"),
|
||||
"issues": (result.issues_csv, "text/csv; charset=utf-8"),
|
||||
"report": (result.report_json, "application/json"),
|
||||
}
|
||||
if kind not in files:
|
||||
raise HTTPException(status_code=404, detail="下载类型不存在")
|
||||
path, media_type = files[kind]
|
||||
if not Path(path).exists():
|
||||
raise HTTPException(status_code=404, detail="文件不存在")
|
||||
return FileResponse(path, media_type=media_type, filename=Path(path).name)
|
||||
+115
@@ -0,0 +1,115 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
@dataclass
|
||||
class PcfBlock:
|
||||
"""PCF 中的一个顶层段。"""
|
||||
|
||||
block_type: str
|
||||
value: str = ""
|
||||
fields: dict[str, list[str]] = field(default_factory=dict)
|
||||
line_no: int = 0
|
||||
|
||||
def first(self, key: str, default: str = "") -> str:
|
||||
values = self.fields.get(key, [])
|
||||
return values[0] if values else default
|
||||
|
||||
|
||||
@dataclass
|
||||
class WeldRecord:
|
||||
"""内部标准焊口记录。"""
|
||||
|
||||
source_file: str
|
||||
pipeline_reference: str = ""
|
||||
piping_spec: str = ""
|
||||
line_no: str = ""
|
||||
weld_no: str = ""
|
||||
diameter: str = ""
|
||||
outside_diameter: str = ""
|
||||
wall_thickness: str = ""
|
||||
spec: str = ""
|
||||
weld_area_raw: str = ""
|
||||
weld_area: str = ""
|
||||
contractor_raw: str = ""
|
||||
component_identifier: str = ""
|
||||
master_component_identifier: str = ""
|
||||
skey: str = ""
|
||||
uci: str = ""
|
||||
material_1: str = ""
|
||||
material_2: str = ""
|
||||
issues: list[str] = field(default_factory=list)
|
||||
raw_fields: dict[str, list[str]] = field(default_factory=dict)
|
||||
|
||||
def set_if_empty(self, field_name: str, value: str) -> None:
|
||||
if hasattr(self, field_name) and not getattr(self, field_name) and value:
|
||||
setattr(self, field_name, value)
|
||||
|
||||
|
||||
@dataclass
|
||||
class UnknownFieldCandidate:
|
||||
"""等待 DeepSeek 识别的非标准字段候选。"""
|
||||
|
||||
source_field: str
|
||||
section: str
|
||||
sample_values: list[str]
|
||||
context: str
|
||||
|
||||
|
||||
@dataclass
|
||||
class MappingDecision:
|
||||
"""DeepSeek 字段映射结果。"""
|
||||
|
||||
source_field: str
|
||||
target_field: str
|
||||
confidence: float
|
||||
reason: str = ""
|
||||
sample_values: list[str] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def is_usable(self) -> bool:
|
||||
return bool(self.source_field and self.target_field and self.confidence >= 0)
|
||||
|
||||
|
||||
@dataclass
|
||||
class JobResult:
|
||||
job_id: str
|
||||
input_files: int
|
||||
weld_count: int
|
||||
warning_count: int
|
||||
error_count: int
|
||||
output_xls: Path
|
||||
issues_csv: Path
|
||||
report_json: Path
|
||||
messages: list[str] = field(default_factory=list)
|
||||
llm_enabled: bool = False
|
||||
llm_mappings: list[MappingDecision] = field(default_factory=list)
|
||||
import_headers: list[str] = field(default_factory=list)
|
||||
import_columns: list[dict[str, str]] = field(default_factory=list)
|
||||
import_rows: list[dict[str, str]] = field(default_factory=list)
|
||||
issue_rows: list[dict[str, str]] = field(default_factory=list)
|
||||
|
||||
def to_public_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"job_id": self.job_id,
|
||||
"input_files": self.input_files,
|
||||
"weld_count": self.weld_count,
|
||||
"warning_count": self.warning_count,
|
||||
"error_count": self.error_count,
|
||||
"messages": self.messages,
|
||||
"llm_enabled": self.llm_enabled,
|
||||
"tables": {
|
||||
"import_headers": self.import_headers,
|
||||
"import_columns": self.import_columns,
|
||||
"import_rows": self.import_rows,
|
||||
"issue_rows": self.issue_rows,
|
||||
},
|
||||
"downloads": {
|
||||
"xls": f"/api/jobs/{self.job_id}/download/xls",
|
||||
"issues": f"/api/jobs/{self.job_id}/download/issues",
|
||||
"report": f"/api/jobs/{self.job_id}/download/report",
|
||||
},
|
||||
}
|
||||
@@ -0,0 +1,279 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import Counter, defaultdict
|
||||
from pathlib import Path
|
||||
|
||||
from .config import AppConfig
|
||||
from .models import PcfBlock, UnknownFieldCandidate, WeldRecord
|
||||
|
||||
|
||||
KNOWN_COMPONENT_BLOCKS = {
|
||||
"WELD",
|
||||
"PIPE",
|
||||
"FLANGE",
|
||||
"GASKET",
|
||||
"BOLT",
|
||||
"VALVE",
|
||||
"ELBOW",
|
||||
"TEE",
|
||||
"TEE-STUB",
|
||||
"OLET",
|
||||
"SUPPORT",
|
||||
"INSTRUMENT",
|
||||
"REDUCER-ECCENTRIC",
|
||||
"REDUCER-CONCENTRIC",
|
||||
"FILTER-OFFSET",
|
||||
"END-CONNECTION-PIPELINE",
|
||||
"END-CONNECTION-EQUIPMENT",
|
||||
"END-POSITION-CLOSED",
|
||||
"END-POSITION-OPEN",
|
||||
"FLOW-ARROW",
|
||||
}
|
||||
|
||||
HEADER_BLOCKS = {"PIPELINE-REFERENCE"}
|
||||
|
||||
STANDARD_TARGET_FIELDS = {
|
||||
"pipeline_reference",
|
||||
"piping_spec",
|
||||
"line_no",
|
||||
"weld_no",
|
||||
"diameter",
|
||||
"weld_area_raw",
|
||||
"contractor_raw",
|
||||
"component_identifier",
|
||||
"master_component_identifier",
|
||||
"skey",
|
||||
"uci",
|
||||
"material_1",
|
||||
"material_2",
|
||||
"wall_thickness",
|
||||
"outside_diameter",
|
||||
}
|
||||
|
||||
|
||||
def read_text_best_effort(path: Path) -> str:
|
||||
data = path.read_bytes()
|
||||
for encoding in ("utf-8-sig", "utf-8", "gbk", "latin-1"):
|
||||
try:
|
||||
return data.decode(encoding)
|
||||
except UnicodeDecodeError:
|
||||
continue
|
||||
return data.decode("utf-8", errors="ignore")
|
||||
|
||||
|
||||
def parse_blocks(text: str) -> list[PcfBlock]:
|
||||
blocks: list[PcfBlock] = []
|
||||
current: PcfBlock | None = None
|
||||
for idx, raw_line in enumerate(text.splitlines(), start=1):
|
||||
if not raw_line.strip():
|
||||
continue
|
||||
|
||||
if raw_line[:1].isspace():
|
||||
if current is None:
|
||||
continue
|
||||
stripped = raw_line.strip()
|
||||
parts = stripped.split(None, 1)
|
||||
key = parts[0].strip()
|
||||
value = parts[1].strip() if len(parts) > 1 else ""
|
||||
current.fields.setdefault(key, []).append(value)
|
||||
continue
|
||||
|
||||
if current is not None:
|
||||
blocks.append(current)
|
||||
|
||||
parts = raw_line.strip().split(None, 1)
|
||||
block_type = parts[0].strip()
|
||||
value = parts[1].strip() if len(parts) > 1 else ""
|
||||
current = PcfBlock(block_type=block_type, value=value, line_no=idx)
|
||||
|
||||
if current is not None:
|
||||
blocks.append(current)
|
||||
return blocks
|
||||
|
||||
|
||||
def normalize_weld_area(raw: str) -> str:
|
||||
lower = raw.lower()
|
||||
if "shop weld" in lower:
|
||||
return "预制"
|
||||
if "field weld" in lower or "job site" in lower:
|
||||
return "安装"
|
||||
return ""
|
||||
|
||||
|
||||
def build_spec(diameter: str, outside_diameter: str, wall_thickness: str) -> str:
|
||||
if outside_diameter and wall_thickness:
|
||||
return f"φ{outside_diameter}x{wall_thickness}"
|
||||
if outside_diameter:
|
||||
return f"φ{outside_diameter}"
|
||||
if diameter:
|
||||
return f"DN{diameter}"
|
||||
return ""
|
||||
|
||||
|
||||
class PcfParser:
|
||||
def __init__(self, config: AppConfig):
|
||||
self.config = config
|
||||
|
||||
def parse_files(
|
||||
self, files: list[Path], runtime_mappings: dict[str, str] | None = None
|
||||
) -> tuple[list[WeldRecord], list[UnknownFieldCandidate], list[str]]:
|
||||
runtime_mappings = runtime_mappings or {}
|
||||
records: list[WeldRecord] = []
|
||||
candidates_by_key: dict[tuple[str, str], UnknownFieldCandidate] = {}
|
||||
messages: list[str] = []
|
||||
|
||||
for path in sorted(files, key=lambda p: p.name):
|
||||
if path.suffix.lower() != ".pcf":
|
||||
messages.append(f"跳过非 PCF 文件:{path.name}")
|
||||
continue
|
||||
try:
|
||||
file_records, file_candidates = self.parse_file(path, runtime_mappings)
|
||||
records.extend(file_records)
|
||||
for c in file_candidates:
|
||||
key = (c.section, c.source_field)
|
||||
if key not in candidates_by_key:
|
||||
candidates_by_key[key] = c
|
||||
except Exception as exc:
|
||||
messages.append(f"{path.name} 解析失败:{exc}")
|
||||
|
||||
self._assign_weld_numbers(records)
|
||||
return records, list(candidates_by_key.values()), messages
|
||||
|
||||
def parse_file(
|
||||
self, path: Path, runtime_mappings: dict[str, str] | None = None
|
||||
) -> tuple[list[WeldRecord], list[UnknownFieldCandidate]]:
|
||||
runtime_mappings = runtime_mappings or {}
|
||||
text = read_text_best_effort(path)
|
||||
blocks = parse_blocks(text)
|
||||
|
||||
header = self._extract_header(blocks)
|
||||
component_blocks = {
|
||||
b.first("COMPONENT-IDENTIFIER"): b
|
||||
for b in blocks
|
||||
if b.block_type in KNOWN_COMPONENT_BLOCKS and b.first("COMPONENT-IDENTIFIER")
|
||||
}
|
||||
|
||||
records: list[WeldRecord] = []
|
||||
unknown_counter: dict[str, Counter[str]] = defaultdict(Counter)
|
||||
known_sources = (
|
||||
set(self.config.field_mappings)
|
||||
| set(self.config.dynamic_field_mappings)
|
||||
| set(runtime_mappings)
|
||||
| self.config.ignored_fields
|
||||
)
|
||||
|
||||
for block in blocks:
|
||||
if block.block_type != "WELD":
|
||||
continue
|
||||
record = self._record_from_weld(path, block, header, component_blocks)
|
||||
self._apply_dynamic_mappings(record, block, runtime_mappings)
|
||||
records.append(record)
|
||||
|
||||
for field_name, values in block.fields.items():
|
||||
if field_name not in known_sources:
|
||||
for value in values[:3]:
|
||||
unknown_counter[field_name][value] += 1
|
||||
|
||||
candidates = [
|
||||
UnknownFieldCandidate(
|
||||
source_field=field_name,
|
||||
section="WELD",
|
||||
sample_values=[value for value, _ in counter.most_common(5)],
|
||||
context=f"{path.name} 的 WELD 段中出现未知字段 {field_name}",
|
||||
)
|
||||
for field_name, counter in unknown_counter.items()
|
||||
]
|
||||
return records, candidates
|
||||
|
||||
def _extract_header(self, blocks: list[PcfBlock]) -> dict[str, str]:
|
||||
header: dict[str, str] = {}
|
||||
for block in blocks:
|
||||
if block.block_type == "PIPELINE-REFERENCE":
|
||||
header["PIPELINE-REFERENCE"] = block.value
|
||||
for key, values in block.fields.items():
|
||||
if values:
|
||||
header[key] = values[0]
|
||||
break
|
||||
return header
|
||||
|
||||
def _record_from_weld(
|
||||
self,
|
||||
path: Path,
|
||||
block: PcfBlock,
|
||||
header: dict[str, str],
|
||||
component_blocks: dict[str, PcfBlock],
|
||||
) -> WeldRecord:
|
||||
diameter = block.first("WELD-ATTRIBUTE2").strip()
|
||||
bore_info = self.config.bore_map.get(diameter, {})
|
||||
outside_diameter = str(bore_info.get("outside_diameter", "") or "")
|
||||
wall_thickness = str(bore_info.get("wall_thickness", "") or "")
|
||||
weld_area_raw = block.first("WELD-ATTRIBUTE3")
|
||||
master_id = block.first("MASTER-COMPONENT-IDENTIFIER")
|
||||
|
||||
record = WeldRecord(
|
||||
source_file=path.name,
|
||||
pipeline_reference=header.get("PIPELINE-REFERENCE", ""),
|
||||
piping_spec=header.get("PIPING-SPEC", ""),
|
||||
line_no=block.first("WELD-ATTRIBUTE1"),
|
||||
diameter=diameter,
|
||||
outside_diameter=outside_diameter,
|
||||
wall_thickness=wall_thickness,
|
||||
spec=build_spec(diameter, outside_diameter, wall_thickness),
|
||||
weld_area_raw=weld_area_raw,
|
||||
weld_area=normalize_weld_area(weld_area_raw),
|
||||
contractor_raw=block.first("WELD-ATTRIBUTE4"),
|
||||
component_identifier=block.first("COMPONENT-IDENTIFIER"),
|
||||
master_component_identifier=master_id,
|
||||
skey=block.first("SKEY"),
|
||||
uci=block.first("UCI"),
|
||||
raw_fields={k: list(v) for k, v in block.fields.items()},
|
||||
)
|
||||
|
||||
master = component_blocks.get(master_id)
|
||||
if master:
|
||||
record.raw_fields["MASTER-BLOCK-TYPE"] = [master.block_type]
|
||||
if master.first("ITEM-DESCRIPTION"):
|
||||
record.raw_fields["MASTER-ITEM-DESCRIPTION"] = [master.first("ITEM-DESCRIPTION")]
|
||||
|
||||
if not record.weld_area:
|
||||
record.issues.append(f"无法识别焊接区域:{record.weld_area_raw}")
|
||||
if not record.wall_thickness:
|
||||
record.issues.append("PCF 未提供可可靠填充的壁厚")
|
||||
if not record.line_no:
|
||||
record.issues.append("缺少管线代号")
|
||||
if not record.diameter:
|
||||
record.issues.append("缺少口径")
|
||||
return record
|
||||
|
||||
def _apply_dynamic_mappings(
|
||||
self,
|
||||
record: WeldRecord,
|
||||
block: PcfBlock,
|
||||
runtime_mappings: dict[str, str],
|
||||
) -> None:
|
||||
merged = {}
|
||||
merged.update(self.config.dynamic_field_mappings)
|
||||
merged.update(runtime_mappings)
|
||||
|
||||
for source_field, target_field in merged.items():
|
||||
if target_field not in STANDARD_TARGET_FIELDS:
|
||||
continue
|
||||
value = block.first(source_field)
|
||||
if value:
|
||||
record.set_if_empty(target_field, value)
|
||||
|
||||
if record.diameter and not record.outside_diameter:
|
||||
bore_info = self.config.bore_map.get(record.diameter, {})
|
||||
record.outside_diameter = str(bore_info.get("outside_diameter", "") or "")
|
||||
record.wall_thickness = str(bore_info.get("wall_thickness", "") or "")
|
||||
record.spec = build_spec(record.diameter, record.outside_diameter, record.wall_thickness)
|
||||
if record.weld_area_raw and not record.weld_area:
|
||||
record.weld_area = normalize_weld_area(record.weld_area_raw)
|
||||
|
||||
def _assign_weld_numbers(self, records: list[WeldRecord]) -> None:
|
||||
counters: dict[str, int] = defaultdict(int)
|
||||
for record in records:
|
||||
group_key = record.pipeline_reference or record.line_no or "UNKNOWN"
|
||||
counters[group_key] += 1
|
||||
if not record.weld_no:
|
||||
record.weld_no = str(counters[group_key])
|
||||
Reference in New Issue
Block a user