knowledge (구 Aislo-law 독립 저장소 → resources/knowledge 이관, 저장소 폐지): - 법령·행정규칙·표준시방서·교본 원문 + 기술문서 55건 + 실무 분석·종합비교 - 루트 지침 체계: README(지도)·00_운영지침·01_수집지침·02_분석지침· 03_미결_및_확인사항(교본 충돌 리스트 포함)·04_참조_법령기준_목록 - 기술문서 55건 원문 전수 검증 완료 (사방 설계홍수량 법정 기준 등 반영) - 정리: CAD·오피스 잔재 142건, 중복 zip 7건(413MB), 빈 폴더 30개 제거 resources 그룹 재편 (이름순 그룹핑): - app_branding(구 prog_icon.jpg)·app_policies(구 legal)· data_global_contours(구 grobal_contours)·data_rainfall_idf_cache(구 wamis_contours)· template_2dDrawing(구 dwg_analysis/templete — 오타 교정, 상수·경로 동기화) - dwg_analysis(분석 완료 1.8GB)·templates(빈 폴더)·templete_calc_cost.xlsx 삭제 - 참조 코드 5파일 경로 수정 + 프론트 재빌드 (구 경로 잔존 0) - .gitignore: resources 추적 전환, national_contours.gpkg(22GB) 영구 제외 - .env: knowledge 수집용 API 정보 주석 통합 (KCSC·법령센터·조달청 제비율) Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
150 lines
5.3 KiB
Python
150 lines
5.3 KiB
Python
# -*- coding: utf-8 -*-
|
|
"""법령·기준 md 전수 품질 점검.
|
|
|
|
검출: 테이블깨짐 / 사진누락 / 내용누락 / 띄어쓰기소실 / 줄바꿈
|
|
소스 대조: 별표·첨부 md ↔ 같은 이름 PDF, 현행 본문 md ↔ 같은 이름 XML.
|
|
결과를 qc_report.json 으로 저장하고 카테고리별 요약 출력.
|
|
"""
|
|
import json, re
|
|
from pathlib import Path
|
|
import pymupdf
|
|
|
|
ROOT = Path(__file__).resolve().parent.parent
|
|
OUT = Path(__file__).resolve().parent / "data"
|
|
|
|
def norm(s):
|
|
return re.sub(r"[^가-힣0-9A-Za-z%㎞㎡㎥℃]", "", s)
|
|
|
|
def strip_fenced(text):
|
|
"""``` 코드펜스 안을 빈 줄로 치환(위치 보존)."""
|
|
out, infence = [], False
|
|
for l in text.split(chr(10)):
|
|
if l.lstrip().startswith("```"):
|
|
infence = not infence
|
|
out.append("")
|
|
continue
|
|
out.append("" if infence else l)
|
|
return chr(10).join(out)
|
|
|
|
def ncols(line):
|
|
s = line.strip()
|
|
if s.startswith("|"):
|
|
s = s[1:]
|
|
if s.endswith("|"):
|
|
s = s[:-1]
|
|
return len(s.split("|"))
|
|
|
|
def check_tables(text):
|
|
"""마크다운 표 유효성. (문제 리스트) 반환."""
|
|
issues = []
|
|
lines = strip_fenced(text).split("\n")
|
|
i = 0
|
|
while i < len(lines):
|
|
if not lines[i].lstrip().startswith("|"):
|
|
i += 1
|
|
continue
|
|
start = i
|
|
block = []
|
|
while i < len(lines) and lines[i].lstrip().startswith("|"):
|
|
block.append(lines[i])
|
|
i += 1
|
|
head = ncols(block[0])
|
|
if len(block) < 2 or not set(block[1].replace("|", "").replace(" ", "").replace(":", "")) <= set("-"):
|
|
issues.append(f"L{start+1} 구분선 없음/이상")
|
|
continue
|
|
for j, b in enumerate(block):
|
|
if j == 1:
|
|
continue
|
|
if ncols(b) != head:
|
|
issues.append(f"L{start+j+1} 열수 {ncols(b)}≠{head}")
|
|
return issues
|
|
|
|
def check_space(text):
|
|
"""프로즈(표·펜스 제외)의 한글 12자 이상 연속 비율."""
|
|
prose = [l for l in strip_fenced(text).split(chr(10)) if not l.lstrip().startswith("|")]
|
|
t = chr(10).join(prose)
|
|
kor = len(re.findall(r"[가-힣]", t))
|
|
if kor < 400:
|
|
return 0.0
|
|
runs = re.findall(r"[가-힣]{12,}", t)
|
|
return round(sum(len(x) for x in runs) / kor, 3)
|
|
|
|
def check_linebreak(text):
|
|
"""줄바꿈 결함: 표 앞 빈 줄 없음, 헤딩 직후 표 붙음."""
|
|
issues = []
|
|
lines = strip_fenced(text).split("\n")
|
|
for i in range(1, len(lines)):
|
|
s = lines[i].strip()
|
|
prev = lines[i-1].strip()
|
|
# 표 시작인데 앞 줄이 텍스트(표/빈줄/헤딩 아님)
|
|
if s.startswith("|") and prev and not prev.startswith("|") and not prev.startswith("#") and not prev.startswith(">"):
|
|
issues.append(f"L{i+1} 표 앞 빈 줄 없음")
|
|
return issues[:5]
|
|
|
|
def pdf_stats(pdf):
|
|
d = pymupdf.open(pdf)
|
|
txt = "\n".join(p.get_text() for p in d)
|
|
imgs = sum(len(p.get_images()) for p in d)
|
|
return txt, imgs
|
|
|
|
def run():
|
|
report = []
|
|
targets = []
|
|
for pat in ("법률", "행정규칙", "표준시방서"):
|
|
base = ROOT / pat
|
|
targets += [p for p in base.rglob("*.md")]
|
|
targets += [p for p in (ROOT / "KS").glob("*.md")]
|
|
targets = [p for p in targets if p.name not in ("_목록.md", "_meta.md")]
|
|
|
|
for md in sorted(targets):
|
|
rel = str(md.relative_to(ROOT)).replace("\\", "/")
|
|
text = md.read_text(encoding="utf-8")
|
|
rec = {"file": rel, "issues": {}}
|
|
|
|
t = check_tables(text)
|
|
if t:
|
|
rec["issues"]["테이블"] = t[:6]
|
|
sp = check_space(text)
|
|
if sp > 0.15:
|
|
rec["issues"]["띄어쓰기"] = sp
|
|
lb = check_linebreak(text)
|
|
if lb:
|
|
rec["issues"]["줄바꿈"] = lb
|
|
|
|
# 소스 대조 (별표/첨부 → PDF). 단, HWP/HWPX 재추출본은 PDF가 소스가 아니므로 제외.
|
|
pdf = md.with_suffix(".pdf")
|
|
head = text[:200]
|
|
reextracted = ("재추출" in head) or (".hwp" in head) or (".hwpx" in head)
|
|
if pdf.exists() and not reextracted:
|
|
try:
|
|
ptxt, pimgs = pdf_stats(pdf)
|
|
pn, mn = norm(ptxt), norm(text)
|
|
if pn and len(mn) / len(pn) < 0.98:
|
|
rec["issues"]["내용누락"] = f"{len(pn)}→{len(mn)} ({len(mn)/len(pn):.2f})"
|
|
mimg = text.count("〔그림〕") + text.count("![")
|
|
if pimgs > 0 and mimg == 0:
|
|
rec["issues"]["사진누락"] = f"PDF 이미지 {pimgs}개 / md 0"
|
|
except Exception as e:
|
|
rec["issues"]["PDF오류"] = str(e)[:50]
|
|
|
|
if rec["issues"]:
|
|
report.append(rec)
|
|
|
|
json.dump(report, open(OUT / "qc_report.json", "w", encoding="utf-8"), ensure_ascii=False, indent=1)
|
|
|
|
# 요약
|
|
cat = {}
|
|
for r in report:
|
|
for k in r["issues"]:
|
|
cat[k] = cat.get(k, 0) + 1
|
|
print(f"점검 {len(targets)}개 / 문제 파일 {len(report)}개")
|
|
print("카테고리별:", cat)
|
|
print("\n=== 심각(내용누락·사진누락·테이블) 상위 ===")
|
|
sev = [r for r in report if set(r["issues"]) & {"내용누락", "사진누락", "테이블"}]
|
|
for r in sev[:30]:
|
|
ks = ", ".join(f"{k}={v if not isinstance(v,list) else len(v)}" for k, v in r["issues"].items())
|
|
print(f" {r['file'][-64:]} [{ks}]")
|
|
|
|
if __name__ == "__main__":
|
|
run()
|