# -*- coding: utf-8 -*- """마스터 데이터 검사 (2) 본문 — `check_master.py` 가 부름. 표형 요소의 출처 절을 본문 md 에서 찾아 수를 맞댐 · 결손(본문 표에 있고 요소에 없음)·허구(요소에 있고 본문 절에 없음). """ from __future__ import annotations import re import sys from decimal import Decimal from pathlib import Path HERE = Path(__file__).resolve().parent ROOT = HERE.parents[2] COST = ROOT / "resources/knowledge/original/원가계산" sys.path.insert(0, str(HERE)) sys.path.insert(0, str(ROOT / "resources/knowledge/original/_pipeline")) import master_formula as mf # noqa: E402 from pum_md_tool import _NUM, md_lines # noqa: E402 DIVISIONS = { "공통": "01_공통부문", "토목": "02_토목부문", "건축": "03_건축부문", "기계설비": "04_기계설비부문", "유지관리": "05_유지관리부문", } _IDENT = re.compile(r"\d+-\d+(?:-\d+)*") def _key(parts: str) -> tuple[int, ...]: return tuple(int(p) for p in parts.split("-")) def section_of(source: str) -> tuple[Path | None, str, str]: """(본문 md, 절 번호, 절 글) — 못 찾으면 md 나 글이 빔.""" ident = _IDENT.search(source) if not ident: return None, "", "" ident = ident.group() parts = _key(ident) if source.startswith("산림품셈"): chapters = COST / "산림_표준품셈/본문" elif source.startswith("건설품셈"): chapters = COST / "건설공사_표준품셈/본문" / DIVISIONS.get(source.split()[1], "") else: return None, ident, "" best = None for md in chapters.glob(f"제{parts[0]:02d}장_*/*.md"): head = md.name.split("_", 1)[0] if not _IDENT.fullmatch(head): continue own = _key(head) if ( own[0] and own == parts[: len(own)] and (best is None or len(own) > len(_key(best.name.split("_", 1)[0]))) ): best = md if best is None: return None, ident, "" lines = best.read_text(encoding="utf-8").split("\n") heading = re.compile(rf"^(#+)\s*{re.escape(ident)}\.?(\s|$)") for at, line in enumerate(lines): m = heading.match(line) if m: level, end = len(m.group(1)), len(lines) for later in range(at + 1, len(lines)): h = re.match(r"^(#+)\s", lines[later]) if h and len(h.group(1)) <= level: end = later break return best, ident, "\n".join(lines[at:end]) return best, ident, "" def _canon(token: str) -> str: return format(Decimal(token.replace(",", "")).normalize(), "f") def numbers_in(text: str) -> set[str]: # 칸 안에서 줄바뀐 천 단위 수(「5,1
50」)는 이어 붙임 text = re.sub(r"(?<=\d,\d)
(?=\d{2})", "", text) return {_canon(t) for _, line in md_lines(text) for t in _NUM.findall(line)} def _flat(value) -> set[str]: if isinstance(value, Decimal): return {_canon(str(abs(value)))} if isinstance(value, str): return {_canon(t) for t in _NUM.findall(value)} if isinstance(value, list): return set().union(*map(_flat, value)) if value else set() if isinstance(value, dict): return set().union(*map(_flat, value.values())) if value else set() return set() def check_body(files: dict[str, dict]) -> list[str]: """표마다 허구 · 절마다 결손.""" out, by_section = [], {} for name, data in files.items(): for table in (data.get("표") or []) if data.get("그룹") in mf.TABLE_GROUPS else []: where = f"{name} · {table.get('키')} {table.get('원문번호')}" source = str(table.get("출처", "")) if not source.startswith(("산림품셈", "건설품셈")): continue # 품셈 본문이 아닌 출처(질의회신 등) — 절 대조 대상 아님 md, ident, text = section_of(source) if not text: out.append(f"{where} · 본문 절 못 찾음 「{table.get('출처')}」") continue mine = ( _flat(table.get("줄")) | _flat(table.get("기준")) | _flat(table.get("주")) | _flat(table.get("값칸")) ) fake = mine - numbers_in(text) if fake: out.append(f"{where} · 허구 {len(fake)} {sorted(fake, key=Decimal)}") slot = by_section.setdefault((md, ident), [text, set(), []]) slot[1] |= mine slot[2].append(where) _elements_into(files, by_section) for (md, ident), (text, mine, tables) in by_section.items(): for (md2, ident2), (_, mine2, tables2) in by_section.items(): if md2 == md and ident2.startswith(ident + "-"): mine = mine | mine2 tables = tables + tables2 table_text = "\n".join(line for line in text.split("\n") if line.lstrip().startswith("|")) lack = numbers_in(table_text) - mine if lack: out.append( f"{md.name} {ident} · 결손 {len(lack)} {sorted(lack, key=Decimal)} ← {' · '.join(tables)}" ) return out def _elements_into(files: dict[str, dict], by_section: dict) -> None: """요소 줄(기계 운전경비처럼 표 대신 요소로 올린 절)의 수도 그 절 결손 대조에 넣음.""" seen: dict[str, tuple] = {} for data in files.values(): if data.get("그룹") in (*mf.TABLE_GROUPS, "로직"): continue for row in data.get("줄") or []: source = row.get("출처") for text in source.values() if isinstance(source, dict) else [source]: if not isinstance(text, str) or not text.startswith(("건설품셈", "산림품셈")): continue if text not in seen: seen[text] = section_of(text)[:2] slot = by_section.get(seen[text]) if slot: slot[1] |= _flat(row.get("값")) | _flat( [row.get(k) for k in ("이름", "규격", "원문번호")] )