import os, sys, re, json, glob, fitz sys.stdout.reconfigure(encoding='utf-8') pdf_path = 'resources/knowledge/original/원가계산/건설공사_표준품셈/2026년_건설공사_표준품셈.pdf' doc = fitz.open(pdf_path) def verify_file(fpath): fname = os.path.basename(fpath) rel_path = os.path.relpath(fpath) print(f"\n========================================================") print(f" Verifying: {fname} ({rel_path})") print(f"========================================================") with open(fpath, 'r', encoding='utf-8') as f: lines = f.readlines() current_sec_num = 'UNKNOWN' current_sec_title = 'UNKNOWN' in_table = False headers = [] table_rows = [] start_line = 0 file_risk_tables = [] for idx, line in enumerate(lines): line_s = line.strip() sec_m = re.search(r'(\d+-\d+(?:-\d+)?)(?:\s+([^\n(]+))?', line_s) if sec_m and not line_s.startswith('|'): current_sec_num = sec_m.group(1).strip() current_sec_title = sec_m.group(0).strip() if line_s.startswith('|') and not line_s.startswith('|---'): cols = [c.strip() for c in line_s.split('|')[1:-1]] if not in_table: headers = cols in_table = True start_line = idx + 1 table_rows = [] continue if in_table: table_rows.append((idx + 1, cols)) else: if in_table and (line_s == '' or not line_s.startswith('|')): header_text = " ".join(headers) col0_vals = [r[1][0] for r in table_rows if len(r[1]) > 0] # Check Risk criteria: is_risk = False reasons = [] # C1: Efficiency / coefficient / factor if any(k in header_text for k in ['효율', '작업효율', '계수', '할증', '보정', '손실률', '환산', '증감']) or any(k in current_sec_title for k in ['효율', '계수', '할증']): is_risk = True reasons.append('효율/계수/할증') # C2: Multiple rows with remarks column if any('비고' in h or '비 고' in h for h in headers) and len(table_rows) >= 2: is_risk = True reasons.append('비고 열 다중행') # C3: Condition table (no workers) if any(k in header_text for k in ['조건', '지형', '토질', '수심', '암질', '두께', '경사', '규격', '차단', '구분']) and not any(k in header_text for k in ['직종', '인부', '기술자', '기능사']): if len(table_rows) >= 2: is_risk = True reasons.append('조건 분기 표') # C4: Multi-level / empty col0 if any(v == '' for v in col0_vals) or any('→' in v for v in col0_vals): is_risk = True reasons.append('Col 0 빈칸/병합/연결') if is_risk: file_risk_tables.append({ 'sec_num': current_sec_num, 'sec_title': current_sec_title, 'start_line': start_line, 'rows': len(table_rows), 'headers': headers, 'table_rows': table_rows, 'reasons': reasons }) in_table = False headers = [] table_rows = [] print(f"Total Risk Tables in {fname}: {len(file_risk_tables)}") # Check each risk table in PDF findings = [] for rt in file_risk_tables: sec = rt['sec_num'] # Find in PDF matched_pages = [] for p in range(len(doc)): if p < 45: continue # skip TOC t = doc[p].get_text() if sec in t: matched_pages.append(p) # Compare with PDF tables print(f"\n [위험 표] L{rt['start_line']} | {rt['sec_title']} ({', '.join(rt['reasons'])})") print(f" MD Headers: {rt['headers']}") print(f" MD Rows ({rt['rows']}): {[r[1][0] if len(r[1])>0 else '' for r in rt['table_rows'][:4]]} ...") print(f" PDF Matched Pages: {[p+1 for p in matched_pages]}") # Check table extraction on matched pages pdf_table_found = False for p in matched_pages: page = doc[p] tabs = page.find_tables() for t_idx, tab in enumerate(tabs): df = tab.extract() if not df or len(df) < 2: continue # Match table header or contents df_flat = " ".join([c for row in df for c in row if c]) # Check if headers overlap overlap = sum(1 for h in rt['headers'] if h and h in df_flat) if overlap >= max(1, len(rt['headers']) // 2): pdf_table_found = True # Check row count and Col 0 structures pdf_r_cnt = len(df) - 1 md_r_cnt = rt['rows'] # Inspect Col 0 in PDF vs MD pdf_col0 = [r[0] if r and len(r)>0 and r[0] else '' for r in df[1:]] md_col0 = [r[1][0] if len(r[1])>0 else '' for r in rt['table_rows']] # Detect potential mismatch # Multi-line in pdf col0 vs split rows in md pdf_col0_lines = sum(len(c.split('\n')) for c in pdf_col0 if c) # Print comparison snippet print(f" -> Matched PDF Table on Page {p+1}: PDF rows={pdf_r_cnt} vs MD rows={md_r_cnt}") # If mismatch in row counts or Col 0 text if pdf_r_cnt != md_r_cnt: print(f" [!] ROW COUNT DIFF: PDF {pdf_r_cnt} vs MD {md_r_cnt}") print(f" PDF Col 0: {pdf_col0[:3]}") print(f" MD Col 0: {md_col0[:3]}") findings.append({ 'sec': sec, 'line': rt['start_line'], 'page': p + 1, 'type': 'Row count mismatch', 'detail': f"PDF rows={pdf_r_cnt}, MD rows={md_r_cnt}" }) break if pdf_table_found: break return len(file_risk_tables), findings if __name__ == '__main__': target = sys.argv[1] if len(sys.argv) > 1 else 'resources/knowledge/original/원가계산/건설공사_표준품셈/02_토목부문/제1장_도로포장공사.md' verify_file(target)