Files
Aislo/B03_FileInput/B03_FileInput_Engine_Analyze.py
T
eomsangdonandClaude Opus 5 fe72ea041b feat(B03): 계획노선 shapefile 입력과 PRJ 2개 분리를 지원한다
원청 정식 계획노선이 shapefile(UTM-K)로, 지형이 별도 PRJ(동부원점 Bessel)로
들어오는데 입력 경로가 shapefile 확장자를 막고 PRJ를 프로젝트당 1개로 전제했다.

- 업로드 허용에 .shp/.shx/.dbf/.cpg 추가, 한 번에 보낼 파일 수 5 -> 10
- B03_FileInput_Engine_Shapefile: ESRI 규격 직접 파싱(GDAL 미사용). 형제 파일이
  아직 안 왔어도 .shp 하나로 기하를 읽는다. .cpg 내용이 949뿐인 실물을 CP949로
  정규화해 한글 속성을 살린다.
- 노선 판독을 read_planned_route로 일원화(CSV/shapefile), PlannedRoute에
  crs_input 추가 - 변환 입력은 EPSG 코드가 아니라 crs_input_from_prj가 주는
  값(EPSG:n 또는 원문 WKT)이다. 실물 PRJ 2종 모두 to_epsg가 None이다.
- shapefile 세트를 input/shp/ 한 폴더에 모은다(GDAL 요건). 노선 PRJ가 그 안에
  남으므로 지형 PRJ(input/prj/)와 파일명 정렬 운에 기대지 않고 갈린다.
  find_project_prj가 지형 PRJ를 프로젝트 좌표계로 고른다.
- 필수 세트를 노선 1종(csv 또는 shp) + prj + tfw로 완화, shp면 shx/dbf 동반 필수.
- UI: 확장자 단독 슬롯 매칭을 basename 그룹핑으로 바꿔 노선 PRJ와 지형 PRJ가
  같은 슬롯을 다투지 않게 하고, 노선 슬롯이 파일 한 벌을 담아 함께 전송한다.

자체검증: tmp/tests/test_route_shapefile_input.py 9개 통과, tsc --noEmit 통과,
ruff check/format 통과. 전체 스위트 잔여 실패 11건은 HEAD 사본(git archive)에서
동일하게 재현되는 기존 실패다.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-31 19:28:34 +09:00

416 lines
16 KiB
Python

"""B03 원본 입력 파일 메타데이터 분석."""
import csv
import logging
import math
import re
from pathlib import Path
from threading import get_ident
from typing import Any
import laspy
import numpy as np
import rasterio
from pyproj import CRS
logger = logging.getLogger(__name__)
_CUSTOM_VERTICAL_AUTHORITY_PATTERN = re.compile(
r',?\s*AUTHORITY\["EPSG","(9995|99999)"\]',
re.IGNORECASE,
)
_CUSTOM_VERTICAL_WARNING_PATTERN = re.compile(
r"proj_create_from_database: crs not found: EPSG:(9995|99999)",
re.IGNORECASE,
)
class _CustomVerticalCrsWarningFilter(logging.Filter):
"""알려진 사설 수직 CRS 경고만 B03 안내 로그로 변환한다."""
def __init__(self, source: Path) -> None:
super().__init__()
self.source = source
self.thread_id = get_ident()
self.logged_codes: set[str] = set()
def filter(self, record: logging.LogRecord) -> bool:
if get_ident() != self.thread_id:
return True
match = _CUSTOM_VERTICAL_WARNING_PATTERN.search(record.getMessage())
if match is None:
return True
code = f"EPSG:{match.group(1)}"
if code not in self.logged_codes:
logger.info(
"사용자 정의 수직 CRS 코드를 보존합니다: file=%s code=%s",
self.source.name,
code,
)
self.logged_codes.add(code)
return False
def _component_metadata(crs: CRS) -> dict[str, Any]:
"""CRS 구성 요소를 JSON 저장 가능한 메타데이터로 변환한다."""
authority = crs.to_authority()
epsg = crs.to_epsg()
if epsg is None and (crs.is_projected or crs.is_geographic):
# DB 대조 실패(비표준 TOWGS84·AUTHORITY 없는 ESRI WKT 등) — 파라미터
# 지문으로 라벨을 보강한다. 변환 정본은 여전히 원문 WKT다 (2026-08-31).
from common_util.common_util_crs import identify_epsg
epsg = identify_epsg(crs)
return {
"name": crs.name,
"type": crs.type_name,
"epsg": epsg,
"authority": (
{"name": authority[0], "code": authority[1]} if authority is not None else None
),
}
def normalize_crs_metadata(crs: Any | None) -> dict[str, Any]:
"""복합 CRS를 수평·수직 구성 요소로 분리해 일관된 상태를 반환한다."""
if crs is None:
return {
"crs": None,
"epsg": None,
"horizontal_crs": None,
"vertical_crs": None,
"crs_status": "missing_crs",
}
parsed = crs if isinstance(crs, CRS) else CRS.from_user_input(crs)
# TOWGS84가 붙은 WKT는 BoundCRS로 감싸져 is_projected가 False가 된다 —
# 벗겨야 수평 성분 탐색과 EPSG 라벨이 동작한다 (2026-08-31).
from common_util.common_util_crs import strip_bound
flattened = strip_bound(parsed)
components = [strip_bound(item) for item in (flattened.sub_crs_list or [flattened])]
horizontal = next(
(item for item in components if item.is_projected or item.is_geographic),
None,
)
vertical = next((item for item in components if item.is_vertical), None)
horizontal_metadata = _component_metadata(horizontal) if horizontal is not None else None
vertical_metadata = _component_metadata(vertical) if vertical is not None else None
horizontal_epsg = horizontal_metadata["epsg"] if horizontal_metadata is not None else None
if horizontal_epsg is None:
status = "unknown_horizontal_crs"
elif vertical_metadata is not None and vertical_metadata["epsg"] is None:
status = "custom_vertical_crs"
else:
status = "identified"
return {
"crs": crs.to_string(),
"epsg": horizontal_epsg,
"horizontal_crs": horizontal_metadata,
"vertical_crs": vertical_metadata,
"crs_status": status,
}
def _log_crs_status(source: Path, metadata: dict[str, Any]) -> None:
"""정상화된 CRS 상태를 B03 도메인 로그로 남긴다."""
if metadata["crs_status"] == "custom_vertical_crs":
logger.info(
"사용자 정의 수직 CRS를 보존합니다: file=%s horizontal_epsg=%s vertical=%s",
source.name,
metadata["epsg"],
metadata["vertical_crs"]["name"],
)
elif metadata["crs_status"] == "unknown_horizontal_crs":
logger.warning("수평 CRS를 EPSG로 식별하지 못했습니다: file=%s", source.name)
def _prepare_prj_wkt(text: str) -> tuple[str, list[str]]:
"""KNGeoid24 사설 EPSG 표식만 파싱용 WKT에서 분리한다."""
if "KNGeoid24" not in text:
return text, []
codes = sorted({f"EPSG:{code}" for code in _CUSTOM_VERTICAL_AUTHORITY_PATTERN.findall(text)})
return _CUSTOM_VERTICAL_AUTHORITY_PATTERN.sub("", text), codes
def analyze_las_metadata(path: str | Path) -> dict[str, Any]:
"""LAS/LAZ 헤더와 분류 통계를 메모리에 전체 적재하지 않고 분석한다."""
source = Path(path)
with laspy.open(source) as las_file:
header = las_file.header
point_format = header.point_format
dimension_names = list(point_format.dimension_names)
point_count = int(header.point_count)
crs = header.parse_crs()
crs_metadata = normalize_crs_metadata(crs)
_log_crs_status(source, crs_metadata)
metadata: dict[str, Any] = {
"file": source.name,
"version": f"{header.version.major}.{header.version.minor}",
"point_format": {
"id": point_format.id,
"dimensions": dimension_names,
},
"point_count": point_count,
"bounds": {
"x": [float(header.mins[0]), float(header.maxs[0])],
"y": [float(header.mins[1]), float(header.maxs[1])],
"z": [float(header.mins[2]), float(header.maxs[2])],
},
"scale": [float(value) for value in header.scales],
"offset": [float(value) for value in header.offsets],
"has_crs": crs is not None,
**crs_metadata,
"has_classification": "classification" in dimension_names,
"has_rgb": all(name in dimension_names for name in ("red", "green", "blue")),
"has_intensity": "intensity" in dimension_names,
"has_return_number": "return_number" in dimension_names,
}
if metadata["has_classification"] and point_count > 0:
classification_counts: dict[int, int] = {}
for chunk in las_file.chunk_iterator(500_000):
values, counts = np.unique(
np.asarray(chunk.classification, dtype=np.uint8),
return_counts=True,
)
for value, count in zip(values.tolist(), counts.tolist(), strict=True):
classification_counts[value] = classification_counts.get(value, 0) + count
metadata["classification_summary"] = {
str(key): value for key, value in sorted(classification_counts.items())
}
return metadata
def analyze_prj_metadata(path: str | Path) -> dict[str, Any]:
"""PRJ WKT에서 좌표계 식별자와 명칭을 추출한다."""
source = Path(path)
text = source.read_text(encoding="utf-8", errors="replace").strip()
metadata: dict[str, Any] = {
"file": source.name,
"text_preview": text[:500],
"epsg": None,
"name": None,
"authority": None,
"is_valid": False,
}
if not text:
metadata["error"] = "PRJ 파일이 비어 있습니다."
return metadata
parse_text, custom_authority_codes = _prepare_prj_wkt(text)
try:
crs = CRS.from_wkt(parse_text)
except Exception as exc:
metadata["error"] = str(exc)
return metadata
crs_metadata = normalize_crs_metadata(crs)
_log_crs_status(source, crs_metadata)
metadata.update(
{
**crs_metadata,
"name": crs.name,
"authority": crs.to_authority(),
"custom_authority_codes": custom_authority_codes,
"is_valid": True,
}
)
return metadata
def analyze_tfw_metadata(path: str | Path) -> dict[str, Any]:
"""TFW의 affine 변환 계수와 유효성을 분석한다."""
source = Path(path)
values = [
float(line.strip())
for line in source.read_text(encoding="utf-8", errors="replace").splitlines()
if line.strip()
]
if any(not math.isfinite(value) for value in values):
raise ValueError("TFW 변환 계수는 유한한 숫자여야 합니다.")
return {
"file": source.name,
"values": values,
"pixel_size_x": values[0] if len(values) > 0 else None,
"rotation_y": values[1] if len(values) > 1 else None,
"rotation_x": values[2] if len(values) > 2 else None,
"pixel_size_y": values[3] if len(values) > 3 else None,
"origin_x": values[4] if len(values) > 4 else None,
"origin_y": values[5] if len(values) > 5 else None,
"is_valid": len(values) == 6,
}
def analyze_tif_metadata(path: str | Path) -> dict[str, Any]:
"""TIF/GeoTIFF 데이터셋의 공간 및 밴드 메타데이터를 분석한다."""
source = Path(path)
rasterio_logger = logging.getLogger("rasterio._env")
warning_filter = _CustomVerticalCrsWarningFilter(source)
rasterio_logger.addFilter(warning_filter)
try:
with rasterio.open(source) as dataset:
crs = dataset.crs
crs_metadata = normalize_crs_metadata(crs)
_log_crs_status(source, crs_metadata)
bounds = dataset.bounds
return {
"file": source.name,
"width": int(dataset.width),
"height": int(dataset.height),
"count": int(dataset.count),
"dtypes": list(dataset.dtypes),
"nodata": float(dataset.nodata) if dataset.nodata is not None else None,
**crs_metadata,
"bounds": {
"left": float(bounds.left),
"bottom": float(bounds.bottom),
"right": float(bounds.right),
"top": float(bounds.top),
},
"transform": [float(value) for value in list(dataset.transform)[:6]],
"resolution": [float(value) for value in dataset.res],
"likely_type": "dem" if dataset.count == 1 else "image",
}
finally:
rasterio_logger.removeFilter(warning_filter)
_PLANNED_ROUTE_COLUMNS = ("route_name", "sequence", "x", "y", "z", "crs_epsg")
def _parse_route_integer(value: str, *, field: str, row_number: int) -> int:
normalized = value.strip()
if not re.fullmatch(r"[0-9]+", normalized):
raise ValueError(f"CSV {row_number}행의 {field} 값은 양의 정수여야 합니다.")
parsed = int(normalized)
if parsed <= 0:
raise ValueError(f"CSV {row_number}행의 {field} 값은 양의 정수여야 합니다.")
return parsed
def _parse_route_coordinate(value: str, *, field: str, row_number: int) -> float:
try:
parsed = float(value.strip())
except (AttributeError, ValueError) as exc:
raise ValueError(f"CSV {row_number}행의 {field} 값은 숫자여야 합니다.") from exc
if not math.isfinite(parsed):
raise ValueError(f"CSV {row_number}행의 {field} 값은 유한한 숫자여야 합니다.")
return parsed
def analyze_planned_route_csv(path: str | Path) -> dict[str, Any]:
"""원청 계획노선 CSV를 검증하고 경로 메타데이터를 반환한다."""
source = Path(path)
with source.open("r", encoding="utf-8-sig", newline="") as csv_file:
reader = csv.DictReader(csv_file)
if reader.fieldnames is None:
raise ValueError("계획노선 CSV 헤더를 찾을 수 없습니다.")
normalized_headers = [header.strip() for header in reader.fieldnames]
if len(set(normalized_headers)) != len(normalized_headers):
raise ValueError("계획노선 CSV 헤더에 중복된 열이 있습니다.")
header_map = dict(zip(normalized_headers, reader.fieldnames, strict=True))
missing = [column for column in _PLANNED_ROUTE_COLUMNS if column not in header_map]
if missing:
raise ValueError(f"계획노선 CSV 필수 열이 없습니다: {', '.join(missing)}")
route_name: str | None = None
crs_epsg: int | None = None
points: list[tuple[float, float, float]] = []
for expected_sequence, row in enumerate(reader, start=1):
row_number = expected_sequence + 1
current_name = (row.get(header_map["route_name"]) or "").strip()
if not current_name:
raise ValueError(f"CSV {row_number}행의 route_name 값이 비어 있습니다.")
if route_name is None:
route_name = current_name
elif current_name != route_name:
raise ValueError("계획노선 CSV에는 하나의 route_name만 사용할 수 있습니다.")
sequence = _parse_route_integer(
row.get(header_map["sequence"]) or "",
field="sequence",
row_number=row_number,
)
if sequence != expected_sequence:
raise ValueError(
f"CSV {row_number}행의 sequence는 {expected_sequence}이어야 합니다."
)
current_epsg = _parse_route_integer(
row.get(header_map["crs_epsg"]) or "",
field="crs_epsg",
row_number=row_number,
)
if crs_epsg is None:
crs_epsg = current_epsg
elif current_epsg != crs_epsg:
raise ValueError("계획노선 CSV의 crs_epsg는 모든 행에서 같아야 합니다.")
points.append(
tuple(
_parse_route_coordinate(
row.get(header_map[field]) or "",
field=field,
row_number=row_number,
)
for field in ("x", "y", "z")
)
)
if len(points) < 2:
raise ValueError("계획노선 CSV에는 좌표가 2개 이상 있어야 합니다.")
xs, ys, zs = zip(*points, strict=True)
return {
"file": source.name,
"extension": "csv",
"size_bytes": source.stat().st_size,
"purpose": "planned_route",
"route_name": route_name,
"point_count": len(points),
"epsg": crs_epsg,
"columns": list(_PLANNED_ROUTE_COLUMNS),
"bounds": {
"x_min": min(xs),
"x_max": max(xs),
"y_min": min(ys),
"y_max": max(ys),
"z_min": min(zs),
"z_max": max(zs),
},
"start_point": list(points[0]),
"end_point": list(points[-1]),
}
def analyze_input_metadata(path: str | Path) -> dict[str, Any]:
"""입력 파일 확장자에 맞는 B03 메타데이터 분석 함수를 호출한다."""
source = Path(path)
extension = source.suffix.lower()
if extension == ".csv":
return analyze_planned_route_csv(source)
if extension == ".shp":
from B03_FileInput.B03_FileInput_Engine_Shapefile import analyze_shapefile_metadata
return analyze_shapefile_metadata(source)
if extension in {".las", ".laz"}:
return analyze_las_metadata(source)
if extension == ".prj":
return analyze_prj_metadata(source)
if extension == ".tfw":
return analyze_tfw_metadata(source)
if extension in {".tif", ".tiff"}:
return analyze_tif_metadata(source)
return {
"file": source.name,
"extension": extension.lstrip("."),
"size_bytes": source.stat().st_size,
}