diff --git a/resources/master_data/scripts/check_master.py b/resources/master_data/scripts/check_master.py index cc45db3d..28e14b02 100644 --- a/resources/master_data/scripts/check_master.py +++ b/resources/master_data/scripts/check_master.py @@ -5,7 +5,7 @@ ./venv/Scripts/python.exe resources/master_data/scripts/check_master.py 계산 GF000123 뒷길이=35 돌=견치돌 … (1) 틀 — 파일 이름 · 머리 칸 · 줄 칸 · 출처 모양 · 키 모양·겹침 · 대장 -(2) 본문 — 표형 요소의 출처 절을 본문 md 에서 찾아 수를 맞댐 · 결손(본문 표에 있고 요소에 없음)·허구(요소에 있고 본문 절에 없음) +(2) 본문(`check_master_body.py`) — 표형 요소의 출처 절을 본문 md 에서 찾아 수를 맞댐 · 결손(본문 표에 있고 요소에 없음)·허구(요소에 있고 본문 절에 없음) (3) 로직 — 변수가 가리키는 요소·표·칸 · 찾기 조건 이름 · 로직 입력 · 재료 고르기 조건 · 돌고 도는 참조 · 끊긴 키 (4) 조합 — 일위대가 조합이 담은 로직 키 · 같은 로직 두 번 · 빈 조합 · 조합을 담음 (`master_combo.py`) (5) 계산 — 로직 키와 입력값으로 호표 줄별 수량·단가·금액 · 비목 합 (`master_formula.py`) @@ -31,7 +31,9 @@ import master_combo as mcb # noqa: E402 import master_formula as mf # noqa: E402 import master_keys as mk # noqa: E402 import master_material_combo as mmc # noqa: E402 -from pum_md_tool import _NUM, md_lines # noqa: E402 + +# (2) 본문은 따로 모듈 — 부르는 쪽은 `cm.check_body` · `cm.DIVISIONS` 그대로 +from check_master_body import DIVISIONS, check_body # noqa: E402, F401 BOOKS = ( "산림품셈", @@ -53,13 +55,6 @@ BOOKS = ( "공간정보법", "자체", ) -DIVISIONS = { - "공통": "01_공통부문", - "토목": "02_토목부문", - "건축": "03_건축부문", - "기계설비": "04_기계설비부문", - "유지관리": "05_유지관리부문", -} _FILE = re.compile( rf"^(?:(인력|기계)|({'|'.join(mf.GROUPS)})_({'|'.join(BOOKS)})(?:_(\d\d)장_(.+))?)\.json$" ) @@ -68,7 +63,6 @@ SPECIAL_FILES = {"기준수량없음_산림품셈_2026-01-01.json"} _SOURCE = re.compile( rf"^(?:자체|(?:{'|'.join(b for b in BOOKS if b not in ('건설품셈', '자체'))})(?:\s\S.*)?|건설품셈 (?:{'|'.join(DIVISIONS)}) \S.*)$" ) -_IDENT = re.compile(r"\d+-\d+(?:-\d+)*") ROW_KEYS = { "요소": ("키", "원문번호", "이름", "값", "출처"), "표": ( @@ -430,135 +424,6 @@ def _check_table(where: str, table: dict) -> list[str]: return out -# ── (2) 본문 ────────────────────────────────────────────────────────── -def _key(parts: str) -> tuple[int, ...]: - return tuple(int(p) for p in parts.split("-")) - - -def section_of(source: str) -> tuple[Path | None, str, str]: - """(본문 md, 절 번호, 절 글) — 못 찾으면 md 나 글이 빔.""" - ident = _IDENT.search(source) - if not ident: - return None, "", "" - ident = ident.group() - parts = _key(ident) - if source.startswith("산림품셈"): - chapters = COST / "산림_표준품셈/본문" - elif source.startswith("건설품셈"): - chapters = COST / "건설공사_표준품셈/본문" / DIVISIONS.get(source.split()[1], "") - else: - return None, ident, "" - best = None - for md in chapters.glob(f"제{parts[0]:02d}장_*/*.md"): - head = md.name.split("_", 1)[0] - if not _IDENT.fullmatch(head): - continue - own = _key(head) - if ( - own[0] - and own == parts[: len(own)] - and (best is None or len(own) > len(_key(best.name.split("_", 1)[0]))) - ): - best = md - if best is None: - return None, ident, "" - lines = best.read_text(encoding="utf-8").split("\n") - heading = re.compile(rf"^(#+)\s*{re.escape(ident)}\.?(\s|$)") - for at, line in enumerate(lines): - m = heading.match(line) - if m: - level, end = len(m.group(1)), len(lines) - for later in range(at + 1, len(lines)): - h = re.match(r"^(#+)\s", lines[later]) - if h and len(h.group(1)) <= level: - end = later - break - return best, ident, "\n".join(lines[at:end]) - return best, ident, "" - - -def _canon(token: str) -> str: - return format(Decimal(token.replace(",", "")).normalize(), "f") - - -def numbers_in(text: str) -> set[str]: - # 칸 안에서 줄바뀐 천 단위 수(「5,1
50」)는 이어 붙임 - text = re.sub(r"(?<=\d,\d)
(?=\d{2})", "", text) - return {_canon(t) for _, line in md_lines(text) for t in _NUM.findall(line)} - - -def _flat(value) -> set[str]: - if isinstance(value, Decimal): - return {_canon(str(abs(value)))} - if isinstance(value, str): - return {_canon(t) for t in _NUM.findall(value)} - if isinstance(value, list): - return set().union(*map(_flat, value)) if value else set() - if isinstance(value, dict): - return set().union(*map(_flat, value.values())) if value else set() - return set() - - -def check_body(files: dict[str, dict]) -> list[str]: - """표마다 허구 · 절마다 결손.""" - out, by_section = [], {} - for name, data in files.items(): - for table in (data.get("표") or []) if data.get("그룹") in mf.TABLE_GROUPS else []: - where = f"{name} · {table.get('키')} {table.get('원문번호')}" - source = str(table.get("출처", "")) - if not source.startswith(("산림품셈", "건설품셈")): - continue # 품셈 본문이 아닌 출처(질의회신 등) — 절 대조 대상 아님 - md, ident, text = section_of(source) - if not text: - out.append(f"{where} · 본문 절 못 찾음 「{table.get('출처')}」") - continue - mine = ( - _flat(table.get("줄")) - | _flat(table.get("기준")) - | _flat(table.get("주")) - | _flat(table.get("값칸")) - ) - fake = mine - numbers_in(text) - if fake: - out.append(f"{where} · 허구 {len(fake)} {sorted(fake, key=Decimal)}") - slot = by_section.setdefault((md, ident), [text, set(), []]) - slot[1] |= mine - slot[2].append(where) - _elements_into(files, by_section) - for (md, ident), (text, mine, tables) in by_section.items(): - for (md2, ident2), (_, mine2, tables2) in by_section.items(): - if md2 == md and ident2.startswith(ident + "-"): - mine = mine | mine2 - tables = tables + tables2 - table_text = "\n".join(line for line in text.split("\n") if line.lstrip().startswith("|")) - lack = numbers_in(table_text) - mine - if lack: - out.append( - f"{md.name} {ident} · 결손 {len(lack)} {sorted(lack, key=Decimal)} ← {' · '.join(tables)}" - ) - return out - - -def _elements_into(files: dict[str, dict], by_section: dict) -> None: - """요소 줄(기계 운전경비처럼 표 대신 요소로 올린 절)의 수도 그 절 결손 대조에 넣음.""" - seen: dict[str, tuple] = {} - for data in files.values(): - if data.get("그룹") in (*mf.TABLE_GROUPS, "로직"): - continue - for row in data.get("줄") or []: - source = row.get("출처") - for text in source.values() if isinstance(source, dict) else [source]: - if not isinstance(text, str) or not text.startswith(("건설품셈", "산림품셈")): - continue - if text not in seen: - seen[text] = section_of(text)[:2] - slot = by_section.get(seen[text]) - if slot: - slot[1] |= _flat(row.get("값")) | _flat( - [row.get(k) for k in ("이름", "규격", "원문번호")] - ) - - # ── (3) 로직 ────────────────────────────────────────────────────────── def check_logics( files: dict[str, dict], whole: mf.Master, book: dict | None = None, narrow=None diff --git a/resources/master_data/scripts/check_master_body.py b/resources/master_data/scripts/check_master_body.py new file mode 100644 index 00000000..a6bab790 --- /dev/null +++ b/resources/master_data/scripts/check_master_body.py @@ -0,0 +1,158 @@ +# -*- coding: utf-8 -*- +"""마스터 데이터 검사 (2) 본문 — `check_master.py` 가 부름. + +표형 요소의 출처 절을 본문 md 에서 찾아 수를 맞댐 · 결손(본문 표에 있고 요소에 없음)·허구(요소에 있고 본문 절에 없음). +""" + +from __future__ import annotations + +import re +import sys +from decimal import Decimal +from pathlib import Path + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[2] +COST = ROOT / "resources/knowledge/original/원가계산" +sys.path.insert(0, str(HERE)) +sys.path.insert(0, str(ROOT / "resources/knowledge/original/_pipeline")) + +import master_formula as mf # noqa: E402 +from pum_md_tool import _NUM, md_lines # noqa: E402 + +DIVISIONS = { + "공통": "01_공통부문", + "토목": "02_토목부문", + "건축": "03_건축부문", + "기계설비": "04_기계설비부문", + "유지관리": "05_유지관리부문", +} +_IDENT = re.compile(r"\d+-\d+(?:-\d+)*") + + +def _key(parts: str) -> tuple[int, ...]: + return tuple(int(p) for p in parts.split("-")) + + +def section_of(source: str) -> tuple[Path | None, str, str]: + """(본문 md, 절 번호, 절 글) — 못 찾으면 md 나 글이 빔.""" + ident = _IDENT.search(source) + if not ident: + return None, "", "" + ident = ident.group() + parts = _key(ident) + if source.startswith("산림품셈"): + chapters = COST / "산림_표준품셈/본문" + elif source.startswith("건설품셈"): + chapters = COST / "건설공사_표준품셈/본문" / DIVISIONS.get(source.split()[1], "") + else: + return None, ident, "" + best = None + for md in chapters.glob(f"제{parts[0]:02d}장_*/*.md"): + head = md.name.split("_", 1)[0] + if not _IDENT.fullmatch(head): + continue + own = _key(head) + if ( + own[0] + and own == parts[: len(own)] + and (best is None or len(own) > len(_key(best.name.split("_", 1)[0]))) + ): + best = md + if best is None: + return None, ident, "" + lines = best.read_text(encoding="utf-8").split("\n") + heading = re.compile(rf"^(#+)\s*{re.escape(ident)}\.?(\s|$)") + for at, line in enumerate(lines): + m = heading.match(line) + if m: + level, end = len(m.group(1)), len(lines) + for later in range(at + 1, len(lines)): + h = re.match(r"^(#+)\s", lines[later]) + if h and len(h.group(1)) <= level: + end = later + break + return best, ident, "\n".join(lines[at:end]) + return best, ident, "" + + +def _canon(token: str) -> str: + return format(Decimal(token.replace(",", "")).normalize(), "f") + + +def numbers_in(text: str) -> set[str]: + # 칸 안에서 줄바뀐 천 단위 수(「5,1
50」)는 이어 붙임 + text = re.sub(r"(?<=\d,\d)
(?=\d{2})", "", text) + return {_canon(t) for _, line in md_lines(text) for t in _NUM.findall(line)} + + +def _flat(value) -> set[str]: + if isinstance(value, Decimal): + return {_canon(str(abs(value)))} + if isinstance(value, str): + return {_canon(t) for t in _NUM.findall(value)} + if isinstance(value, list): + return set().union(*map(_flat, value)) if value else set() + if isinstance(value, dict): + return set().union(*map(_flat, value.values())) if value else set() + return set() + + +def check_body(files: dict[str, dict]) -> list[str]: + """표마다 허구 · 절마다 결손.""" + out, by_section = [], {} + for name, data in files.items(): + for table in (data.get("표") or []) if data.get("그룹") in mf.TABLE_GROUPS else []: + where = f"{name} · {table.get('키')} {table.get('원문번호')}" + source = str(table.get("출처", "")) + if not source.startswith(("산림품셈", "건설품셈")): + continue # 품셈 본문이 아닌 출처(질의회신 등) — 절 대조 대상 아님 + md, ident, text = section_of(source) + if not text: + out.append(f"{where} · 본문 절 못 찾음 「{table.get('출처')}」") + continue + mine = ( + _flat(table.get("줄")) + | _flat(table.get("기준")) + | _flat(table.get("주")) + | _flat(table.get("값칸")) + ) + fake = mine - numbers_in(text) + if fake: + out.append(f"{where} · 허구 {len(fake)} {sorted(fake, key=Decimal)}") + slot = by_section.setdefault((md, ident), [text, set(), []]) + slot[1] |= mine + slot[2].append(where) + _elements_into(files, by_section) + for (md, ident), (text, mine, tables) in by_section.items(): + for (md2, ident2), (_, mine2, tables2) in by_section.items(): + if md2 == md and ident2.startswith(ident + "-"): + mine = mine | mine2 + tables = tables + tables2 + table_text = "\n".join(line for line in text.split("\n") if line.lstrip().startswith("|")) + lack = numbers_in(table_text) - mine + if lack: + out.append( + f"{md.name} {ident} · 결손 {len(lack)} {sorted(lack, key=Decimal)} ← {' · '.join(tables)}" + ) + return out + + +def _elements_into(files: dict[str, dict], by_section: dict) -> None: + """요소 줄(기계 운전경비처럼 표 대신 요소로 올린 절)의 수도 그 절 결손 대조에 넣음.""" + seen: dict[str, tuple] = {} + for data in files.values(): + if data.get("그룹") in (*mf.TABLE_GROUPS, "로직"): + continue + for row in data.get("줄") or []: + source = row.get("출처") + for text in source.values() if isinstance(source, dict) else [source]: + if not isinstance(text, str) or not text.startswith(("건설품셈", "산림품셈")): + continue + if text not in seen: + seen[text] = section_of(text)[:2] + slot = by_section.get(seen[text]) + if slot: + slot[1] |= _flat(row.get("값")) | _flat( + [row.get(k) for k in ("이름", "규격", "원문번호")] + )