280 lines
9.6 KiB
Python
280 lines
9.6 KiB
Python
|
|
from __future__ import annotations
|
||
|
|
|
||
|
|
from collections import Counter, defaultdict
|
||
|
|
from pathlib import Path
|
||
|
|
|
||
|
|
from .config import AppConfig
|
||
|
|
from .models import PcfBlock, UnknownFieldCandidate, WeldRecord
|
||
|
|
|
||
|
|
|
||
|
|
KNOWN_COMPONENT_BLOCKS = {
|
||
|
|
"WELD",
|
||
|
|
"PIPE",
|
||
|
|
"FLANGE",
|
||
|
|
"GASKET",
|
||
|
|
"BOLT",
|
||
|
|
"VALVE",
|
||
|
|
"ELBOW",
|
||
|
|
"TEE",
|
||
|
|
"TEE-STUB",
|
||
|
|
"OLET",
|
||
|
|
"SUPPORT",
|
||
|
|
"INSTRUMENT",
|
||
|
|
"REDUCER-ECCENTRIC",
|
||
|
|
"REDUCER-CONCENTRIC",
|
||
|
|
"FILTER-OFFSET",
|
||
|
|
"END-CONNECTION-PIPELINE",
|
||
|
|
"END-CONNECTION-EQUIPMENT",
|
||
|
|
"END-POSITION-CLOSED",
|
||
|
|
"END-POSITION-OPEN",
|
||
|
|
"FLOW-ARROW",
|
||
|
|
}
|
||
|
|
|
||
|
|
HEADER_BLOCKS = {"PIPELINE-REFERENCE"}
|
||
|
|
|
||
|
|
STANDARD_TARGET_FIELDS = {
|
||
|
|
"pipeline_reference",
|
||
|
|
"piping_spec",
|
||
|
|
"line_no",
|
||
|
|
"weld_no",
|
||
|
|
"diameter",
|
||
|
|
"weld_area_raw",
|
||
|
|
"contractor_raw",
|
||
|
|
"component_identifier",
|
||
|
|
"master_component_identifier",
|
||
|
|
"skey",
|
||
|
|
"uci",
|
||
|
|
"material_1",
|
||
|
|
"material_2",
|
||
|
|
"wall_thickness",
|
||
|
|
"outside_diameter",
|
||
|
|
}
|
||
|
|
|
||
|
|
|
||
|
|
def read_text_best_effort(path: Path) -> str:
|
||
|
|
data = path.read_bytes()
|
||
|
|
for encoding in ("utf-8-sig", "utf-8", "gbk", "latin-1"):
|
||
|
|
try:
|
||
|
|
return data.decode(encoding)
|
||
|
|
except UnicodeDecodeError:
|
||
|
|
continue
|
||
|
|
return data.decode("utf-8", errors="ignore")
|
||
|
|
|
||
|
|
|
||
|
|
def parse_blocks(text: str) -> list[PcfBlock]:
|
||
|
|
blocks: list[PcfBlock] = []
|
||
|
|
current: PcfBlock | None = None
|
||
|
|
for idx, raw_line in enumerate(text.splitlines(), start=1):
|
||
|
|
if not raw_line.strip():
|
||
|
|
continue
|
||
|
|
|
||
|
|
if raw_line[:1].isspace():
|
||
|
|
if current is None:
|
||
|
|
continue
|
||
|
|
stripped = raw_line.strip()
|
||
|
|
parts = stripped.split(None, 1)
|
||
|
|
key = parts[0].strip()
|
||
|
|
value = parts[1].strip() if len(parts) > 1 else ""
|
||
|
|
current.fields.setdefault(key, []).append(value)
|
||
|
|
continue
|
||
|
|
|
||
|
|
if current is not None:
|
||
|
|
blocks.append(current)
|
||
|
|
|
||
|
|
parts = raw_line.strip().split(None, 1)
|
||
|
|
block_type = parts[0].strip()
|
||
|
|
value = parts[1].strip() if len(parts) > 1 else ""
|
||
|
|
current = PcfBlock(block_type=block_type, value=value, line_no=idx)
|
||
|
|
|
||
|
|
if current is not None:
|
||
|
|
blocks.append(current)
|
||
|
|
return blocks
|
||
|
|
|
||
|
|
|
||
|
|
def normalize_weld_area(raw: str) -> str:
|
||
|
|
lower = raw.lower()
|
||
|
|
if "shop weld" in lower:
|
||
|
|
return "预制"
|
||
|
|
if "field weld" in lower or "job site" in lower:
|
||
|
|
return "安装"
|
||
|
|
return ""
|
||
|
|
|
||
|
|
|
||
|
|
def build_spec(diameter: str, outside_diameter: str, wall_thickness: str) -> str:
|
||
|
|
if outside_diameter and wall_thickness:
|
||
|
|
return f"φ{outside_diameter}x{wall_thickness}"
|
||
|
|
if outside_diameter:
|
||
|
|
return f"φ{outside_diameter}"
|
||
|
|
if diameter:
|
||
|
|
return f"DN{diameter}"
|
||
|
|
return ""
|
||
|
|
|
||
|
|
|
||
|
|
class PcfParser:
|
||
|
|
def __init__(self, config: AppConfig):
|
||
|
|
self.config = config
|
||
|
|
|
||
|
|
def parse_files(
|
||
|
|
self, files: list[Path], runtime_mappings: dict[str, str] | None = None
|
||
|
|
) -> tuple[list[WeldRecord], list[UnknownFieldCandidate], list[str]]:
|
||
|
|
runtime_mappings = runtime_mappings or {}
|
||
|
|
records: list[WeldRecord] = []
|
||
|
|
candidates_by_key: dict[tuple[str, str], UnknownFieldCandidate] = {}
|
||
|
|
messages: list[str] = []
|
||
|
|
|
||
|
|
for path in sorted(files, key=lambda p: p.name):
|
||
|
|
if path.suffix.lower() != ".pcf":
|
||
|
|
messages.append(f"跳过非 PCF 文件:{path.name}")
|
||
|
|
continue
|
||
|
|
try:
|
||
|
|
file_records, file_candidates = self.parse_file(path, runtime_mappings)
|
||
|
|
records.extend(file_records)
|
||
|
|
for c in file_candidates:
|
||
|
|
key = (c.section, c.source_field)
|
||
|
|
if key not in candidates_by_key:
|
||
|
|
candidates_by_key[key] = c
|
||
|
|
except Exception as exc:
|
||
|
|
messages.append(f"{path.name} 解析失败:{exc}")
|
||
|
|
|
||
|
|
self._assign_weld_numbers(records)
|
||
|
|
return records, list(candidates_by_key.values()), messages
|
||
|
|
|
||
|
|
def parse_file(
|
||
|
|
self, path: Path, runtime_mappings: dict[str, str] | None = None
|
||
|
|
) -> tuple[list[WeldRecord], list[UnknownFieldCandidate]]:
|
||
|
|
runtime_mappings = runtime_mappings or {}
|
||
|
|
text = read_text_best_effort(path)
|
||
|
|
blocks = parse_blocks(text)
|
||
|
|
|
||
|
|
header = self._extract_header(blocks)
|
||
|
|
component_blocks = {
|
||
|
|
b.first("COMPONENT-IDENTIFIER"): b
|
||
|
|
for b in blocks
|
||
|
|
if b.block_type in KNOWN_COMPONENT_BLOCKS and b.first("COMPONENT-IDENTIFIER")
|
||
|
|
}
|
||
|
|
|
||
|
|
records: list[WeldRecord] = []
|
||
|
|
unknown_counter: dict[str, Counter[str]] = defaultdict(Counter)
|
||
|
|
known_sources = (
|
||
|
|
set(self.config.field_mappings)
|
||
|
|
| set(self.config.dynamic_field_mappings)
|
||
|
|
| set(runtime_mappings)
|
||
|
|
| self.config.ignored_fields
|
||
|
|
)
|
||
|
|
|
||
|
|
for block in blocks:
|
||
|
|
if block.block_type != "WELD":
|
||
|
|
continue
|
||
|
|
record = self._record_from_weld(path, block, header, component_blocks)
|
||
|
|
self._apply_dynamic_mappings(record, block, runtime_mappings)
|
||
|
|
records.append(record)
|
||
|
|
|
||
|
|
for field_name, values in block.fields.items():
|
||
|
|
if field_name not in known_sources:
|
||
|
|
for value in values[:3]:
|
||
|
|
unknown_counter[field_name][value] += 1
|
||
|
|
|
||
|
|
candidates = [
|
||
|
|
UnknownFieldCandidate(
|
||
|
|
source_field=field_name,
|
||
|
|
section="WELD",
|
||
|
|
sample_values=[value for value, _ in counter.most_common(5)],
|
||
|
|
context=f"{path.name} 的 WELD 段中出现未知字段 {field_name}",
|
||
|
|
)
|
||
|
|
for field_name, counter in unknown_counter.items()
|
||
|
|
]
|
||
|
|
return records, candidates
|
||
|
|
|
||
|
|
def _extract_header(self, blocks: list[PcfBlock]) -> dict[str, str]:
|
||
|
|
header: dict[str, str] = {}
|
||
|
|
for block in blocks:
|
||
|
|
if block.block_type == "PIPELINE-REFERENCE":
|
||
|
|
header["PIPELINE-REFERENCE"] = block.value
|
||
|
|
for key, values in block.fields.items():
|
||
|
|
if values:
|
||
|
|
header[key] = values[0]
|
||
|
|
break
|
||
|
|
return header
|
||
|
|
|
||
|
|
def _record_from_weld(
|
||
|
|
self,
|
||
|
|
path: Path,
|
||
|
|
block: PcfBlock,
|
||
|
|
header: dict[str, str],
|
||
|
|
component_blocks: dict[str, PcfBlock],
|
||
|
|
) -> WeldRecord:
|
||
|
|
diameter = block.first("WELD-ATTRIBUTE2").strip()
|
||
|
|
bore_info = self.config.bore_map.get(diameter, {})
|
||
|
|
outside_diameter = str(bore_info.get("outside_diameter", "") or "")
|
||
|
|
wall_thickness = str(bore_info.get("wall_thickness", "") or "")
|
||
|
|
weld_area_raw = block.first("WELD-ATTRIBUTE3")
|
||
|
|
master_id = block.first("MASTER-COMPONENT-IDENTIFIER")
|
||
|
|
|
||
|
|
record = WeldRecord(
|
||
|
|
source_file=path.name,
|
||
|
|
pipeline_reference=header.get("PIPELINE-REFERENCE", ""),
|
||
|
|
piping_spec=header.get("PIPING-SPEC", ""),
|
||
|
|
line_no=block.first("WELD-ATTRIBUTE1"),
|
||
|
|
diameter=diameter,
|
||
|
|
outside_diameter=outside_diameter,
|
||
|
|
wall_thickness=wall_thickness,
|
||
|
|
spec=build_spec(diameter, outside_diameter, wall_thickness),
|
||
|
|
weld_area_raw=weld_area_raw,
|
||
|
|
weld_area=normalize_weld_area(weld_area_raw),
|
||
|
|
contractor_raw=block.first("WELD-ATTRIBUTE4"),
|
||
|
|
component_identifier=block.first("COMPONENT-IDENTIFIER"),
|
||
|
|
master_component_identifier=master_id,
|
||
|
|
skey=block.first("SKEY"),
|
||
|
|
uci=block.first("UCI"),
|
||
|
|
raw_fields={k: list(v) for k, v in block.fields.items()},
|
||
|
|
)
|
||
|
|
|
||
|
|
master = component_blocks.get(master_id)
|
||
|
|
if master:
|
||
|
|
record.raw_fields["MASTER-BLOCK-TYPE"] = [master.block_type]
|
||
|
|
if master.first("ITEM-DESCRIPTION"):
|
||
|
|
record.raw_fields["MASTER-ITEM-DESCRIPTION"] = [master.first("ITEM-DESCRIPTION")]
|
||
|
|
|
||
|
|
if not record.weld_area:
|
||
|
|
record.issues.append(f"无法识别焊接区域:{record.weld_area_raw}")
|
||
|
|
if not record.wall_thickness:
|
||
|
|
record.issues.append("PCF 未提供可可靠填充的壁厚")
|
||
|
|
if not record.line_no:
|
||
|
|
record.issues.append("缺少管线代号")
|
||
|
|
if not record.diameter:
|
||
|
|
record.issues.append("缺少口径")
|
||
|
|
return record
|
||
|
|
|
||
|
|
def _apply_dynamic_mappings(
|
||
|
|
self,
|
||
|
|
record: WeldRecord,
|
||
|
|
block: PcfBlock,
|
||
|
|
runtime_mappings: dict[str, str],
|
||
|
|
) -> None:
|
||
|
|
merged = {}
|
||
|
|
merged.update(self.config.dynamic_field_mappings)
|
||
|
|
merged.update(runtime_mappings)
|
||
|
|
|
||
|
|
for source_field, target_field in merged.items():
|
||
|
|
if target_field not in STANDARD_TARGET_FIELDS:
|
||
|
|
continue
|
||
|
|
value = block.first(source_field)
|
||
|
|
if value:
|
||
|
|
record.set_if_empty(target_field, value)
|
||
|
|
|
||
|
|
if record.diameter and not record.outside_diameter:
|
||
|
|
bore_info = self.config.bore_map.get(record.diameter, {})
|
||
|
|
record.outside_diameter = str(bore_info.get("outside_diameter", "") or "")
|
||
|
|
record.wall_thickness = str(bore_info.get("wall_thickness", "") or "")
|
||
|
|
record.spec = build_spec(record.diameter, record.outside_diameter, record.wall_thickness)
|
||
|
|
if record.weld_area_raw and not record.weld_area:
|
||
|
|
record.weld_area = normalize_weld_area(record.weld_area_raw)
|
||
|
|
|
||
|
|
def _assign_weld_numbers(self, records: list[WeldRecord]) -> None:
|
||
|
|
counters: dict[str, int] = defaultdict(int)
|
||
|
|
for record in records:
|
||
|
|
group_key = record.pipeline_reference or record.line_no or "UNKNOWN"
|
||
|
|
counters[group_key] += 1
|
||
|
|
if not record.weld_no:
|
||
|
|
record.weld_no = str(counters[group_key])
|