"""산림사업 표준품셈 476표 → 공종 마스터 정규화 (B08 일감 1번 · 공종 축). 무엇을 만드나 `resources/data_work_item_master/work_item_master_.json` — 공종 계층 + 표 귀속. 같은 폴더에 `form_undetermined_*.json`(형태 판정 실패분)과 `_manifest.json` 을 함께 낸다. 왜 이렇게 나누나 (PLAN 8-7 담당 경계) 이 파일은 **공종 축만** 만든다. 자원 축(`resource_kind`·`resource_code`·`amount`)은 단가표를 아는 B09 가 뒤 패스로 채운다. 그래서 각 표의 **원문 셀(`raw_row`)을 그대로 실어 보낸다** — 버리면 B09 가 476표를 다시 열어야 한다. 공종의 정체 = 품셈의 **절 번호**다 품셈은 「9-3-1. 인력」처럼 절 제목이 공종이고, 표의 행은 그 공종의 **조건별 변형** (토질·암종·규격)이다. 그래서 계층·정렬은 목차표(`F0001`)에서 나오고, 표는 그 절에 붙는다. ⚠ 최대 함정 — `pum_form` (PLAN 8-6) 같은 숫자라도 뜻이 반대다. productivity(생산량형) : 「㎥/hr」·「㎥/1인/1일」 → 품 = 1 ÷ 값 requirement(소요량형) : 「100㎥당 x인」 → 품 = 값 ÷ 밑수 태그가 없으면 뒤집힌 값이 조용히 들어간다. **판정 못 한 표는 빈칸으로 두지 않고 `form_undetermined` 목록으로 뽑아** 사람이 보게 한다. 실행 ./venv/Scripts/python.exe B08_Quantity/B08_Quantity_Build_WorkItemMaster.py """ from __future__ import annotations import hashlib import json import re from datetime import datetime, timezone from pathlib import Path from typing import Any # 표 한 장 읽기(형태·밑수·깃발)는 `_Table` — 옛 이름은 여기서도 그대로 부름(시험·다른 파일이 씀). from B08_Quantity.B08_Quantity_Build_WorkItemMaster_Table import ( # noqa: F401 REFERENCE_MARKS, basis_from_name, basis_from_source, basis_quantity_is_grouped, capacity_formula_pending, crew_table, detect_basis, detect_form, expression_cells, formula_rows, new_report, norm, person_basis_ok, resource_shares, spaced_names, special_glyphs, table_entry, ) ROOT = Path(__file__).resolve().parent.parent SOURCE = ROOT / "resources" / "data_cost_input_value" / "pum_forest_2026.json" OUT_DIR = ROOT / "resources" / "data_work_item_master" #: 불변 열쇠 장부 — **판이 붙지 않는다.** 판마다 새로 짓는 산출물이 아니라 대대로 잇는 장부다. KEY_REGISTRY = OUT_DIR / "work_item_keys.json" SCHEMA_VERSION = "1.0" CODE_PREFIX = "FP" # Forest Pumsem — 공종코드 접두. 품셈 절 번호를 그대로 싣는다. def sha256_of(path: Path) -> str: return hashlib.sha256(path.read_bytes()).hexdigest() def parse_toc(rows: list[list[str]]) -> list[dict[str, Any]]: """목차표(F0001) → 계층 목록. 장(제n장)·절(n-n)·항(n-n-n)·목(n-n-n-n).""" nodes: list[dict[str, Any]] = [] order = 0 for raw in rows: cells = [norm(c) for c in raw] cells = [c for c in cells if c] if not cells: continue key = cells[0] name = cells[1] if len(cells) > 1 else "" if m := re.fullmatch(r"제(\d+)장", key): number = m.group(1) elif re.fullmatch(r"\d+(?:-\d+){1,3}", key): number = key else: continue # 부록 등 — 공종 계층이 아니다. order += 256 # STmate 의 SORTCODE 관례(256 간격) — 중간 삽입 여유. nodes.append({"number": number, "name": name, "sort_order": order, "tables": []}) _set_number(nodes[-1], number) return nodes def _set_number(node: dict[str, Any], number: str) -> None: """번호에서 코드·단계·윗줄을 셈 — 목차를 읽을 때와 본문 번호로 바로잡을 때 한 벌.""" parts = number.split("-") node.update( { "work_item_code": f"{CODE_PREFIX}-" + "-".join(p.zfill(2) for p in parts), "number": number, "level": len(parts), "parent_code": ( f"{CODE_PREFIX}-" + "-".join(p.zfill(2) for p in parts[:-1]) if len(parts) > 1 else None ), } ) #: 본문 절 제목 — 「12-17-3. 무근진동기 제외」 꼴(번호 + 마침표 + 이름)만. 문장 조각은 안 봄. _BODY_TITLE_RE = re.compile(r"^\s*(\d+(?:-\d+){0,3})\.\s+(.+)$") def renumber_from_body( nodes: list[dict[str, Any]], tables: list[dict[str, Any]] ) -> list[dict[str, str]]: """⚠ **목차 오기**를 본문 절 번호로 바로잡음 — 원문이 또렷하면 원문(2026-09-17 코덱스 원문 대조 · 브레인). 품셈 목차가 「12-17-2 무근진동기 제외」(본문 12-17-3) · 「12-24-1 지수판 설치」(본문 12-27-1) 로 **같은 번호를 두 번** 적었음. 목차만 믿으면 뒤 줄이 앞 표를 덮어써 이름과 표가 어긋난 줄이 서고 진짜 표(F0364·F0375)가 orphan 으로 사라짐 — 재료·노무·장비가 있는 원가 공종이라 금액이 빠짐. ⚠ 겹친 번호의 줄이고 · 본문에 같은 이름 제목이 있고 · 그 본문 번호가 목차에 없을 때만 고침. 셋 중 하나라도 아니면 그대로 둠(겹친 채 남아 `toc_duplicate_rows` 와 시험이 드러냄). """ body: dict[str, set[str]] = {} for table in tables: if found := _BODY_TITLE_RE.match(norm(table.get("section"))): body.setdefault("".join(found.group(2).split()), set()).add(found.group(1)) count: dict[str, int] = {} for node in nodes: count[node["number"]] = count.get(node["number"], 0) + 1 fixed = [] for node in nodes: numbers = body.get("".join(node["name"].split()), set()) if count[node["number"]] < 2 or node["number"] in numbers: continue candidates = sorted(numbers - set(count)) if len(candidates) != 1: continue printed = node["number"] _set_number(node, candidates[0]) count[printed] -= 1 count[node["number"]] = 1 fixed.append({"name": node["name"], "toc_number": printed, "body_number": node["number"]}) return fixed def _is_section_title(section: str, by_number: dict[str, dict[str, Any]]) -> bool: """진짜 절 제목인가 — 「번호. 이름」, 또는 마침표 없이 번호 뒤 글이 목차 절 이름과 같음.""" if re.match(r"^\s*\d+(?:-\d+){0,3}\.\s", section): return True number = section_number(section) node = by_number.get(number) if number else None if node is None: return False tail = section.strip()[len(number) :] return "".join(tail.split()) == "".join(str(node.get("name") or "").split()) def section_number(section: str) -> str | None: """`"9-3-1. 인력"` → `"9-3-1"`. 번호가 없으면 `None`.""" m = re.match(r"^\s*(\d+(?:-\d+){0,3})[.\s]", section + " ") return m.group(1) if m else None def build() -> tuple[ dict[str, Any], list[dict[str, Any]], list[dict[str, Any]], tuple[dict[str, Any], list[dict[str, str]]], ]: data = json.loads(SOURCE.read_text(encoding="utf-8")) # 사람 판정 선언(표 형태·부모 모양)은 절 번호로 적혀 판에 묶임 — 판이 다르면 멈춤(명세 17장). from common_util.common_util_pum_edition import require_code_edition require_code_edition(data["effective_date"], "공종 마스터 빌드 — 표 형태·부모 모양 판정") tables = data["variables"]["pum"]["tables"] # 원문을 함께 연다 — 밑수가 표 밖(본문)에 있기 때문이다. 못 열면 밑수 없이 간다. source_lines: list[str] = [] for entry in data.get("sources") or []: candidate = ROOT / str(entry.get("path") or "") if candidate.is_file(): source_lines = candidate.read_text(encoding="utf-8").splitlines() break toc_table = next(t for t in tables if t["table_id"] == "F0001") nodes = parse_toc(toc_table["rows"]) toc_corrections = renumber_from_body(nodes, tables) by_number = {n["number"]: n for n in nodes} attached = 0 report = new_report() orphans: list[dict[str, Any]] = [] last_section = "" for table in tables: if table["table_id"] == "F0001": continue # 목차 자신은 공종이 아니다. section = norm(table.get("section")) # ⚠ 원본 자료가 [주] 문장 조각을 절 제목으로 읽은 표가 있음 — 「12-2 표면 마무리를 따른다.」 # (12-17-1 펌프카 타설 [주]) 가 표 7장을 12-2 에 붙여 가짜 갈래가 섰음(2026-09-14). # 절 제목은 「번호. 이름」 이거나(마침표 없는 원문 제목 「7-15-2 숲가꾸기, …」 도 있음) # 번호 뒤 글이 **목차의 그 절 이름과 같음** — 둘 다 아니면 문장 조각이라 앞 절을 이어받음. if _is_section_title(section, by_number): last_section = section elif last_section: section = last_section number = section_number(section) chapter = number.split("-")[0] if number else None node = by_number.get(number) if number else None entry = table_entry( table, section, number, chapter, source_lines, node and node["name"], report ) if node is None: orphans.append({"pum_table_id": table["table_id"], "section": section}) continue node["tables"].append(entry) attached += 1 # 부모 모양(명세 2장) — 기본 choose_one, 합산형은 사람이 적은 것만. from B08_Quantity.B08_Quantity_Build_WorkItemMaster_Parents import apply_parent_modes apply_parent_modes(nodes) # 갈래 키(명세 14장 정정) — 표 모양 가르기는 B09 표 읽기 한 벌을 그대로 돌려 씀. from B08_Quantity.B08_Quantity_Build_WorkItemMaster_Variants import attach_variant_keys attach_variant_keys(nodes, data["effective_date"]) return finish( data, source=SOURCE, dataset_id="work_item_master_forest", nodes=nodes, tables_total=len(tables) - 1, attached=attached, orphans=orphans, report=report, registry_path=KEY_REGISTRY, toc_corrections=toc_corrections, ) def finish( data: dict[str, Any], *, source: Path, dataset_id: str, nodes: list[dict[str, Any]], tables_total: int, attached: int, orphans: list[dict[str, Any]], report: dict[str, Any], registry_path: Path, toc_corrections: list[dict[str, str]], key_prefix: str = "FW", general_roots: tuple[str, ...] | None = None, policy: dict[str, Any] | None = None, ) -> tuple[ dict[str, Any], list[dict[str, Any]], list[dict[str, Any]], tuple[dict[str, Any], list[dict[str, str]]], ]: """열쇠·목차 칸을 얹고 산출 한 벌을 짬 — 산림·건설 같은 꼴(`_Const` 도 이리 옴).""" # 불변 열쇠(`FW-00001`) — 목차 번호를 열쇠에서 칸으로 내린다(2026-09-17 브레인 ①). # ⚠ 장부에 없는 목차 코드는 **새 열쇠**를 받는다. 짐작으로 옛 열쇠를 물려주지 않는다. from B08_Quantity.B08_Quantity_Build_WorkItemMaster_Keys import ( assign_keys, decorate, load_registry, toc_edition, ) edition = toc_edition(data.get("sources"), data["effective_date"]) registry = load_registry(registry_path, key_prefix) newly_issued, toc_duplicates = assign_keys(nodes, registry, edition=edition) decorate(nodes, edition=edition, roots=general_roots) # None = 산림 잣대(`axis_policy.json`) role_counts: dict[str, int] = {} for node in nodes: role_counts[node["axis_role"]] = role_counts.get(node["axis_role"], 0) + 1 src_meta = { "dataset_id": data["dataset_id"], "effective_date": data["effective_date"], "sha256": sha256_of(source), "file": source.name, } return ( { "schema_version": SCHEMA_VERSION, "dataset_id": dataset_id, "effective_date": data["effective_date"], # 공종 코드(`FP-*` 절 번호)가 묶인 품셈 판 — 자원 축·인계 줄이 이 값을 싣는다. "pum_edition": data["effective_date"], # 목차 번호가 묶인 고시 판. `work_item_code`(=`toc_code`)는 **이 판의 목차 번호**다. "toc_edition": edition, "generated_at": datetime.now(timezone.utc).astimezone().isoformat(timespec="seconds"), "dataset_version": src_meta, "policy": { "axis": "work_item_only", "resource_axis_owner": "B09", "no_invented_values": True, "raw_row_preserved": True, # 줄의 열쇠는 `work_item_key`(`FW-…`)다. `work_item_code`·`toc_number` 는 # **그 판의 목차 번호**라 개정 때 밀린다 — 열쇠로 쓰지 말 것. "row_key": "work_item_key", "toc_is_column_not_key": True, **(policy or {}), }, "stats": { "toc_nodes": len(nodes), "tables_total": tables_total, "tables_attached": attached, "tables_orphan": len(orphans), "form_undetermined": len(report["undetermined"]), "basis_found": report["basis_found"], "basis_missing": len(report["basis_missing"]), "basis_grouped": report["basis_grouped"], # 줄의 구실 — 공종(`work_item`) · 묶는 마디(`group`) · 총칙(`general_provision`) · # 표도 아래도 없는 잎(`empty`). 화면·조합은 공종 축만 보면 된다. "axis_roles": role_counts, "keys_newly_issued": len(newly_issued), "toc_duplicate_rows": len(toc_duplicates), "toc_corrections": len(toc_corrections), }, # 목차 오기를 본문 절 번호로 바로잡은 줄 — 목차에 적힌 번호(`toc_number`)도 남김. "toc_corrections": toc_corrections, # ⚠ 품셈 목차가 **같은 번호를 두 번 적은 자리**. 목차 번호를 열쇠로 쓰면 이 줄들이 # 서로 덮는다 — 열쇠를 따로 두는 까닭의 실물이다. "toc_duplicate_rows": toc_duplicates, "orphan_tables": orphans, "work_items": nodes, }, report["undetermined"], report["basis_missing"], (registry, newly_issued), ) def main() -> None: from B08_Quantity.B08_Quantity_Build_WorkItemMaster_Const import ( CONST_KEY_REGISTRY, FILE_TAG, build_const, ) write_outputs(build(), KEY_REGISTRY, "") write_outputs(build_const(), CONST_KEY_REGISTRY, f"{FILE_TAG}_") def write_outputs(result: tuple, registry_path: Path, tag: str) -> None: """산출 넷(마스터·미판정·밑수 미확보·판 지문) + 장부. `tag` — 건설은 `const_`(산림 파일과 안 부딪힘).""" master, undetermined, basis_missing, (registry, newly_issued) = result OUT_DIR.mkdir(parents=True, exist_ok=True) from B08_Quantity.B08_Quantity_Build_WorkItemMaster_Keys import save_registry save_registry(registry_path, registry) date = master["effective_date"] master_path = OUT_DIR / f"{tag}work_item_master_{date}.json" undet_path = OUT_DIR / f"{tag}form_undetermined_{date}.json" basis_path = OUT_DIR / f"{tag}basis_missing_{date}.json" master_path.write_text(json.dumps(master, ensure_ascii=False, indent=1), encoding="utf-8") undet_path.write_text( json.dumps( { "schema_version": SCHEMA_VERSION, "dataset_id": "work_item_master_form_undetermined", "effective_date": date, "note": "형태를 못 정한 표. 사람이 보고 productivity/requirement/coefficient 로 확정할 것.", "items": undetermined, }, ensure_ascii=False, indent=1, ), encoding="utf-8", ) basis_path.write_text( json.dumps( { "schema_version": SCHEMA_VERSION, "dataset_id": "work_item_master_basis_missing", "effective_date": date, "note": ( "밑수(「10㎡당」 같은 기준 수량)를 못 찾은 표. **1 단위당으로 단정하지 말 것** — " "곱셈이 10배·100배 틀린다. 값을 곱해야 하는 형태(requirement·productivity)만 담는다." ), "items": basis_missing, }, ensure_ascii=False, indent=1, ), encoding="utf-8", ) manifest = { "schema_version": SCHEMA_VERSION, "dataset_id": "data_work_item_master_manifest", "generated_at": master["generated_at"], "built_by": "B08_Quantity/B08_Quantity_Build_WorkItemMaster.py", "source": master["dataset_version"], "files": [ { "file": p.name, "sha256": sha256_of(p), "size_bytes": p.stat().st_size, } for p in (master_path, undet_path, basis_path) ], } (OUT_DIR / (f"_manifest_{tag.rstrip('_')}.json" if tag else "_manifest.json")).write_text( json.dumps(manifest, ensure_ascii=False, indent=1), encoding="utf-8" ) s = master["stats"] print(f"── {master['dataset_id']}") print(f"목차 계층 {s['toc_nodes']}") print( f"표 귀속 {s['tables_attached']} / {s['tables_total']} (미귀속 {s['tables_orphan']})" ) print(f"형태 미판정 {s['form_undetermined']}") print(f"밑수 확보 {s['basis_found']} (묶음 기준 {s['basis_grouped']})") print(f"밑수 미확보 {s['basis_missing']} → {basis_path.name}") roles = s["axis_roles"] print( "줄 구실 공종 {} · 묶는 마디 {} · 총칙 {} · 빈 잎 {}".format( roles.get("work_item", 0), roles.get("group", 0), roles.get("general_provision", 0), roles.get("empty", 0), ) ) print(f"열쇠 새로 발급 {len(newly_issued)} (장부 {registry_path.name})") if s["toc_duplicate_rows"]: print(f"목차 번호 겹침 {s['toc_duplicate_rows']} (열쇠는 안 겹침)") print(f"산출 {master_path.relative_to(ROOT)}") if __name__ == "__main__": main()