- (2) 본문 검사(section_of · numbers_in · check_body 등)를 그대로 옮김 · check_master 는 569줄 - cm.check_body · cm.DIVISIONS 는 check_master 에서 그대로 부름 - 직접 실행(전부 · 계산전부) 결과가 나누기 전과 같음 · 시험 254 통과(전후 같음) Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_019qWfH6MMhLwjhtLYnDVUXv
159 lines
6.1 KiB
Python
159 lines
6.1 KiB
Python
# -*- coding: utf-8 -*-
|
|
"""마스터 데이터 검사 (2) 본문 — `check_master.py` 가 부름.
|
|
|
|
표형 요소의 출처 절을 본문 md 에서 찾아 수를 맞댐 · 결손(본문 표에 있고 요소에 없음)·허구(요소에 있고 본문 절에 없음).
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
import sys
|
|
from decimal import Decimal
|
|
from pathlib import Path
|
|
|
|
HERE = Path(__file__).resolve().parent
|
|
ROOT = HERE.parents[2]
|
|
COST = ROOT / "resources/knowledge/original/원가계산"
|
|
sys.path.insert(0, str(HERE))
|
|
sys.path.insert(0, str(ROOT / "resources/knowledge/original/_pipeline"))
|
|
|
|
import master_formula as mf # noqa: E402
|
|
from pum_md_tool import _NUM, md_lines # noqa: E402
|
|
|
|
DIVISIONS = {
|
|
"공통": "01_공통부문",
|
|
"토목": "02_토목부문",
|
|
"건축": "03_건축부문",
|
|
"기계설비": "04_기계설비부문",
|
|
"유지관리": "05_유지관리부문",
|
|
}
|
|
_IDENT = re.compile(r"\d+-\d+(?:-\d+)*")
|
|
|
|
|
|
def _key(parts: str) -> tuple[int, ...]:
|
|
return tuple(int(p) for p in parts.split("-"))
|
|
|
|
|
|
def section_of(source: str) -> tuple[Path | None, str, str]:
|
|
"""(본문 md, 절 번호, 절 글) — 못 찾으면 md 나 글이 빔."""
|
|
ident = _IDENT.search(source)
|
|
if not ident:
|
|
return None, "", ""
|
|
ident = ident.group()
|
|
parts = _key(ident)
|
|
if source.startswith("산림품셈"):
|
|
chapters = COST / "산림_표준품셈/본문"
|
|
elif source.startswith("건설품셈"):
|
|
chapters = COST / "건설공사_표준품셈/본문" / DIVISIONS.get(source.split()[1], "")
|
|
else:
|
|
return None, ident, ""
|
|
best = None
|
|
for md in chapters.glob(f"제{parts[0]:02d}장_*/*.md"):
|
|
head = md.name.split("_", 1)[0]
|
|
if not _IDENT.fullmatch(head):
|
|
continue
|
|
own = _key(head)
|
|
if (
|
|
own[0]
|
|
and own == parts[: len(own)]
|
|
and (best is None or len(own) > len(_key(best.name.split("_", 1)[0])))
|
|
):
|
|
best = md
|
|
if best is None:
|
|
return None, ident, ""
|
|
lines = best.read_text(encoding="utf-8").split("\n")
|
|
heading = re.compile(rf"^(#+)\s*{re.escape(ident)}\.?(\s|$)")
|
|
for at, line in enumerate(lines):
|
|
m = heading.match(line)
|
|
if m:
|
|
level, end = len(m.group(1)), len(lines)
|
|
for later in range(at + 1, len(lines)):
|
|
h = re.match(r"^(#+)\s", lines[later])
|
|
if h and len(h.group(1)) <= level:
|
|
end = later
|
|
break
|
|
return best, ident, "\n".join(lines[at:end])
|
|
return best, ident, ""
|
|
|
|
|
|
def _canon(token: str) -> str:
|
|
return format(Decimal(token.replace(",", "")).normalize(), "f")
|
|
|
|
|
|
def numbers_in(text: str) -> set[str]:
|
|
# 칸 안에서 줄바뀐 천 단위 수(「5,1<br>50」)는 이어 붙임
|
|
text = re.sub(r"(?<=\d,\d)<br>(?=\d{2})", "", text)
|
|
return {_canon(t) for _, line in md_lines(text) for t in _NUM.findall(line)}
|
|
|
|
|
|
def _flat(value) -> set[str]:
|
|
if isinstance(value, Decimal):
|
|
return {_canon(str(abs(value)))}
|
|
if isinstance(value, str):
|
|
return {_canon(t) for t in _NUM.findall(value)}
|
|
if isinstance(value, list):
|
|
return set().union(*map(_flat, value)) if value else set()
|
|
if isinstance(value, dict):
|
|
return set().union(*map(_flat, value.values())) if value else set()
|
|
return set()
|
|
|
|
|
|
def check_body(files: dict[str, dict]) -> list[str]:
|
|
"""표마다 허구 · 절마다 결손."""
|
|
out, by_section = [], {}
|
|
for name, data in files.items():
|
|
for table in (data.get("표") or []) if data.get("그룹") in mf.TABLE_GROUPS else []:
|
|
where = f"{name} · {table.get('키')} {table.get('원문번호')}"
|
|
source = str(table.get("출처", ""))
|
|
if not source.startswith(("산림품셈", "건설품셈")):
|
|
continue # 품셈 본문이 아닌 출처(질의회신 등) — 절 대조 대상 아님
|
|
md, ident, text = section_of(source)
|
|
if not text:
|
|
out.append(f"{where} · 본문 절 못 찾음 「{table.get('출처')}」")
|
|
continue
|
|
mine = (
|
|
_flat(table.get("줄"))
|
|
| _flat(table.get("기준"))
|
|
| _flat(table.get("주"))
|
|
| _flat(table.get("값칸"))
|
|
)
|
|
fake = mine - numbers_in(text)
|
|
if fake:
|
|
out.append(f"{where} · 허구 {len(fake)} {sorted(fake, key=Decimal)}")
|
|
slot = by_section.setdefault((md, ident), [text, set(), []])
|
|
slot[1] |= mine
|
|
slot[2].append(where)
|
|
_elements_into(files, by_section)
|
|
for (md, ident), (text, mine, tables) in by_section.items():
|
|
for (md2, ident2), (_, mine2, tables2) in by_section.items():
|
|
if md2 == md and ident2.startswith(ident + "-"):
|
|
mine = mine | mine2
|
|
tables = tables + tables2
|
|
table_text = "\n".join(line for line in text.split("\n") if line.lstrip().startswith("|"))
|
|
lack = numbers_in(table_text) - mine
|
|
if lack:
|
|
out.append(
|
|
f"{md.name} {ident} · 결손 {len(lack)} {sorted(lack, key=Decimal)} ← {' · '.join(tables)}"
|
|
)
|
|
return out
|
|
|
|
|
|
def _elements_into(files: dict[str, dict], by_section: dict) -> None:
|
|
"""요소 줄(기계 운전경비처럼 표 대신 요소로 올린 절)의 수도 그 절 결손 대조에 넣음."""
|
|
seen: dict[str, tuple] = {}
|
|
for data in files.values():
|
|
if data.get("그룹") in (*mf.TABLE_GROUPS, "로직"):
|
|
continue
|
|
for row in data.get("줄") or []:
|
|
source = row.get("출처")
|
|
for text in source.values() if isinstance(source, dict) else [source]:
|
|
if not isinstance(text, str) or not text.startswith(("건설품셈", "산림품셈")):
|
|
continue
|
|
if text not in seen:
|
|
seen[text] = section_of(text)[:2]
|
|
slot = by_section.get(seen[text])
|
|
if slot:
|
|
slot[1] |= _flat(row.get("값")) | _flat(
|
|
[row.get(k) for k in ("이름", "규격", "원문번호")]
|
|
)
|