"""B09 원가계산 — **열이 자원인 표** 읽기 (자원 축 보조, 2026-09-08). 품셈 표에는 자원이 **행**이 아니라 **열 머리**에 오는 모양이 따로 있다 (39 표). 구 분 | 콘크리트공(인) | 보통인부(인) 무근구조물 | 0.12 | 0.15 철근구조물 | 0.14 | 0.16 행은 **규격 갈래**(무근·철근·소형구조물)이고 갈래마다 품이 다르다. 이 모양을 행-자원 표로 읽으면 통째로 안 맞는다 — 콘크리트 타설(12-1)이 그래서 하나도 안 서고 있었다. ⚠ **자리 밀림을 고쳐 읽지 않는다.** 첫 칸이 병합된 표는 값이 한 칸씩 밀려 오는데 (목재틀흙막이가 「건축목공 8.760」 자리에 등급 글자를 두어 단가가 503만원으로 섰다), 밀린 행은 **버리고 `unmatched` 에 남긴다.** 어느 칸이 어느 자원인지 단정할 수 없다. `B09_Estimation_ResourceAxis` 가 700줄 제한에 걸려 이 표 모양만 떼어 낸 파일이다. """ from __future__ import annotations from decimal import Decimal from typing import Any from B09_Estimation.B09_Estimation_ResourceAxis import ( AxisResult, ResourceCatalog, ResourceRow, UnmatchedRow, parse_amount, split_name_and_spec, ) #: 갈래 이름 자리에서 걸러 낼 말 — **합계 줄만**이다. 넓게 잡으면 등급이 지워진다. _TOTAL_LABELS = ("계", "합계", "소계", "총계", "구분") def _normalize_label(text: str) -> str: return "".join(str(text).split()) #: 「계」 열 — **가공 + 조립을 이미 더한 값**이다. 같이 읽으면 두 번 센다(㉤ 열 방향). _SUM_GROUP_LABELS = ("계", "합계", "소계", "총계") def _sum_group_positions(headers: list, resource_count: int) -> set: """「계」 묶음이 차지하는 열 번호. 2단 머리에서 묶음 하나가 여러 열을 먹는다. 첫 줄이 `구조별 | 가공 | 조립 | 계` 이고 둘째 줄이 `철근공 | 보통인부` × 3 벌이면 묶음 하나가 **2열씩** 차지한다. 「계」 묶음의 열은 통째로 뺀다. """ groups = [str(h).strip() for h in headers[1:]] if not groups or resource_count % len(groups) != 0: return set() per_group = resource_count // len(groups) blocked = set() for index, group in enumerate(groups): if "".join(group.split()) in _SUM_GROUP_LABELS: start = index * per_group blocked.update(range(start, start + per_group)) return blocked def second_row_columns(table: dict, catalog: ResourceCatalog): """**첫 자료 행이 진짜 열 머리**인 2단 표를 읽는다. `condition_note` 가 공정(「가공·조립·계」)뿐이고 자원 이름이 그 아래 줄에 오는 표다 (철근 현장가공 및 조립 12-3). 자원을 못 찾으면 빈 목록을 돌려준다. """ rows = table.get("raw_row") or [] if not rows: return [], 0 header_row = [str(c).strip() for c in rows[0]] found = [] for position, cell in enumerate(header_row): if not cell: continue name, spec = split_name_and_spec(cell) entry = catalog.resolve(name, spec) if entry is None and not spec: candidates = catalog.by_name(name) entry = candidates[0] if len(candidates) == 1 else None if entry is not None: found.append((position, entry)) if len(found) < 2: return [], 0 # ⚠ 「계」 묶음은 뺀다 — 가공 + 조립을 이미 더한 값이라 같이 읽으면 두 번 센다. blocked = _sum_group_positions(table.get("condition_note") or [], len(found)) # ⚠ 머리 줄에는 **갈래 칸이 없다** — 자료 행은 첫 칸이 갈래 이름이라 한 칸 밀려 있다. # 이 보정을 빼면 갈래 이름 칸을 자원 값으로 읽는다. kept = [ (position + 1, entry) for order, (position, entry) in enumerate(found) if order not in blocked ] return kept, 1 def transposed_columns(table: dict[str, Any], catalog: ResourceCatalog) -> list[tuple[int, Any]]: """열 머리에서 자원을 찾는다. `[(열 번호, 카탈로그 줄)]`. 첫 칸은 갈래 이름(「구 분」)이라 **1번 열부터** 본다. 카탈로그에 있는 이름만 자원으로 본다 — 필터로 거르지 않는다(넓은 필터가 정상 자원을 지운 전례). """ headers = table.get("condition_note") or [] found: list[tuple[int, Any]] = [] for position, header in enumerate(headers[1:], start=1): name, spec = split_name_and_spec(str(header)) entry = catalog.resolve(name, spec) if entry is None and not spec: candidates = catalog.by_name(name) entry = candidates[0] if len(candidates) == 1 else None if entry is not None: found.append((position, entry)) return found def match_transposed_table( node: dict[str, Any], table: dict[str, Any], catalog: ResourceCatalog, result: AxisResult, basis_quantity: Decimal | None, unit: str, ) -> bool: """열이 자원인 표를 읽는다. 그런 표가 아니면 `False` 를 돌려 원래 길로 보낸다. 행마다 **규격 갈래 하나**가 되므로 `variant` 를 달아 둔다 — 「무근구조물」과 「철근구조물」은 품이 달라 **한 일위대가로 뭉치면 안 된다**. """ columns = transposed_columns(table, catalog) skip_rows = 0 if not columns: # 자원 이름이 **둘째 줄**에 오는 2단 표일 수 있다. columns, skip_rows = second_row_columns(table, catalog) if not columns: return False # ⚠ **좌우로 두 판이 붙은 표는 통째로 버린다.** 열 머리가 되풀이되면 # (「거리 | 보통인부 | 거리 | 보통인부」) 오른쪽 판의 **거리값이 인원으로** 읽힌다 # — 2026-09-08 실측: 소운반이 「보통인부 60인」이 되어 단가가 1,553만원으로 섰다. # 판 경계를 짐작해 읽지 않는다. # ⚠ 2단 표에서는 **같은 직종이 공정마다 되풀이되는 것이 정상**이다 # (철근 12-3: 가공 철근공 + 조립 철근공 = 합쳐야 맞는 값). 되풀이 금지는 # **1단 표에만** 건다 — 거기서만 「두 판이 좌우로 붙은 표」를 뜻한다. codes = [entry.code for _, entry in columns] if skip_rows == 0 and len(codes) != len(set(codes)): result.unmatched.append( UnmatchedRow( work_item_code=node.get("work_item_code", ""), pum_table_id=str(table.get("pum_table_id", "")), cell=" | ".join(str(c) for c in (table.get("condition_note") or [])), reason="열 머리가 되풀이되는 두 판 짜리 표 — 자리를 단정할 수 없어 버렸습니다.", ) ) return True # ⚠ **갈래 이름이 되풀이되면 그 줄들을 버린다.** 첫 칸이 병합된 표에서 상위 등급이 # 떨어져 나가면 「상」이 두 번 나오고, 그대로 두면 서로 다른 등급의 품이 **합산**된다 # (2026-09-08 실측: 목재틀흙막이 「상」이 8.760 + 13.767 = 22.5 인이 되어 단가가 # 667만원으로 섰다). 어느 등급인지 단정할 수 없으므로 고쳐 읽지 않는다. labels = [str(row[0]).strip() for row in (table.get("raw_row") or [])[skip_rows:] if row] repeated = {label for label in labels if label and labels.count(label) > 1} work_item_code = node.get("work_item_code", "") table_id = str(table.get("pum_table_id", "")) form = str(table.get("pum_form", "")) matched_any = False for index, row in enumerate(table.get("raw_row", [])): if index < skip_rows: continue # 그 줄은 자료가 아니라 **열 머리**다 cells = [str(c) for c in row] if not cells: continue variant = cells[0].strip() # ⚠ 첫 칸은 **갈래 이름**이지 자원 이름이 아니다 — 자원용 머리글 필터를 여기 쓰면 # 「중」·「상」 같은 정상 등급이 통째로 지워진다(2026-09-08 실측: 목재틀흙막이의 # 중·상 등급이 사라지고 상등구조만 남았다). 합계 줄만 걸러 낸다. if not variant or _normalize_label(variant) in _TOTAL_LABELS: continue if variant in repeated: result.unmatched.append( UnmatchedRow( work_item_code=work_item_code, pum_table_id=table_id, cell=variant, reason="같은 갈래 이름이 두 번 나오는 표 — 등급을 단정할 수 없어 버렸습니다.", ) ) continue # ⚠ **자리 밀림 검사** — 자원 열 가운데 하나라도 수가 아니면 그 행은 밀린 것이다. # 첫 칸이 병합된 표에서 값이 한 칸씩 밀려 들어온다(2026-09-08 실측: 목재틀흙막이가 # 「건축목공 8.760」 자리에 등급 글자를 두어 단가가 503만원으로 섰다). # **밀린 행은 고쳐 읽지 않고 버린다** — 어느 칸이 어느 자원인지 단정할 수 없다. readable = [ parse_amount(cells[position]) if position < len(cells) else None for position, _ in columns ] if any(value is None for value in readable) and any( value is not None for value in readable ): result.unmatched.append( UnmatchedRow( work_item_code=work_item_code, pum_table_id=table_id, cell=variant, reason="자원 열의 값이 한 칸 밀린 행 — 자리를 단정할 수 없어 버렸습니다.", ) ) continue for position, entry in columns: if position >= len(cells): # 칸이 모자란 행 — **자리를 밀어 읽지 않는다**. 밀려 읽으면 다른 직종의 # 품이 붙는다(기계경비 표에서 실제로 겪은 사고). result.unmatched.append( UnmatchedRow( work_item_code=work_item_code, pum_table_id=table_id, cell=f"{variant} / {entry.name}", reason="칸 수가 열 머리와 안 맞아 버렸습니다(자리 밀림 방지).", ) ) continue amount = parse_amount(cells[position]) if amount is None: continue if basis_quantity not in (None, 0, Decimal(1)): amount = amount / basis_quantity result.rows.append( ResourceRow( work_item_code=work_item_code, pum_table_id=table_id, pum_form=form, resource_kind=entry.kind, resource_code=entry.code, resource_name=entry.name, resource_spec=entry.spec, amount=amount, amount_unit=unit, raw_row_index=index, variant=variant, ) ) matched_any = True return matched_any