"""B03 원본 입력 파일 메타데이터 분석.""" import logging import math import re from pathlib import Path from threading import get_ident from typing import Any import laspy import numpy as np import rasterio from pyproj import CRS logger = logging.getLogger(__name__) _CUSTOM_VERTICAL_AUTHORITY_PATTERN = re.compile( r',?\s*AUTHORITY\["EPSG","(9995|99999)"\]', re.IGNORECASE, ) _CUSTOM_VERTICAL_WARNING_PATTERN = re.compile( r"proj_create_from_database: crs not found: EPSG:(9995|99999)", re.IGNORECASE, ) class _CustomVerticalCrsWarningFilter(logging.Filter): """알려진 사설 수직 CRS 경고만 B03 안내 로그로 변환한다.""" def __init__(self, source: Path) -> None: super().__init__() self.source = source self.thread_id = get_ident() self.logged_codes: set[str] = set() def filter(self, record: logging.LogRecord) -> bool: if get_ident() != self.thread_id: return True match = _CUSTOM_VERTICAL_WARNING_PATTERN.search(record.getMessage()) if match is None: return True code = f"EPSG:{match.group(1)}" if code not in self.logged_codes: logger.info( "사용자 정의 수직 CRS 코드를 보존합니다: file=%s code=%s", self.source.name, code, ) self.logged_codes.add(code) return False def _component_metadata(crs: CRS) -> dict[str, Any]: """CRS 구성 요소를 JSON 저장 가능한 메타데이터로 변환한다.""" authority = crs.to_authority() epsg = crs.to_epsg() return { "name": crs.name, "type": crs.type_name, "epsg": epsg, "authority": ( {"name": authority[0], "code": authority[1]} if authority is not None else None ), } def normalize_crs_metadata(crs: Any | None) -> dict[str, Any]: """복합 CRS를 수평·수직 구성 요소로 분리해 일관된 상태를 반환한다.""" if crs is None: return { "crs": None, "epsg": None, "horizontal_crs": None, "vertical_crs": None, "crs_status": "missing_crs", } parsed = crs if isinstance(crs, CRS) else CRS.from_user_input(crs) components = parsed.sub_crs_list or [parsed] horizontal = next( (item for item in components if item.is_projected or item.is_geographic), None, ) vertical = next((item for item in components if item.is_vertical), None) horizontal_metadata = _component_metadata(horizontal) if horizontal is not None else None vertical_metadata = _component_metadata(vertical) if vertical is not None else None horizontal_epsg = horizontal_metadata["epsg"] if horizontal_metadata is not None else None if horizontal_epsg is None: status = "unknown_horizontal_crs" elif vertical_metadata is not None and vertical_metadata["epsg"] is None: status = "custom_vertical_crs" else: status = "identified" return { "crs": crs.to_string(), "epsg": horizontal_epsg, "horizontal_crs": horizontal_metadata, "vertical_crs": vertical_metadata, "crs_status": status, } def _log_crs_status(source: Path, metadata: dict[str, Any]) -> None: """정상화된 CRS 상태를 B03 도메인 로그로 남긴다.""" if metadata["crs_status"] == "custom_vertical_crs": logger.info( "사용자 정의 수직 CRS를 보존합니다: file=%s horizontal_epsg=%s vertical=%s", source.name, metadata["epsg"], metadata["vertical_crs"]["name"], ) elif metadata["crs_status"] == "unknown_horizontal_crs": logger.warning("수평 CRS를 EPSG로 식별하지 못했습니다: file=%s", source.name) def _prepare_prj_wkt(text: str) -> tuple[str, list[str]]: """KNGeoid24 사설 EPSG 표식만 파싱용 WKT에서 분리한다.""" if "KNGeoid24" not in text: return text, [] codes = sorted({f"EPSG:{code}" for code in _CUSTOM_VERTICAL_AUTHORITY_PATTERN.findall(text)}) return _CUSTOM_VERTICAL_AUTHORITY_PATTERN.sub("", text), codes def analyze_las_metadata(path: str | Path) -> dict[str, Any]: """LAS/LAZ 헤더와 분류 통계를 메모리에 전체 적재하지 않고 분석한다.""" source = Path(path) with laspy.open(source) as las_file: header = las_file.header point_format = header.point_format dimension_names = list(point_format.dimension_names) point_count = int(header.point_count) crs = header.parse_crs() crs_metadata = normalize_crs_metadata(crs) _log_crs_status(source, crs_metadata) metadata: dict[str, Any] = { "file": source.name, "version": f"{header.version.major}.{header.version.minor}", "point_format": { "id": point_format.id, "dimensions": dimension_names, }, "point_count": point_count, "bounds": { "x": [float(header.mins[0]), float(header.maxs[0])], "y": [float(header.mins[1]), float(header.maxs[1])], "z": [float(header.mins[2]), float(header.maxs[2])], }, "scale": [float(value) for value in header.scales], "offset": [float(value) for value in header.offsets], "has_crs": crs is not None, **crs_metadata, "has_classification": "classification" in dimension_names, "has_rgb": all(name in dimension_names for name in ("red", "green", "blue")), "has_intensity": "intensity" in dimension_names, "has_return_number": "return_number" in dimension_names, } if metadata["has_classification"] and point_count > 0: classification_counts: dict[int, int] = {} for chunk in las_file.chunk_iterator(500_000): values, counts = np.unique( np.asarray(chunk.classification, dtype=np.uint8), return_counts=True, ) for value, count in zip(values.tolist(), counts.tolist(), strict=True): classification_counts[value] = classification_counts.get(value, 0) + count metadata["classification_summary"] = { str(key): value for key, value in sorted(classification_counts.items()) } return metadata def analyze_prj_metadata(path: str | Path) -> dict[str, Any]: """PRJ WKT에서 좌표계 식별자와 명칭을 추출한다.""" source = Path(path) text = source.read_text(encoding="utf-8", errors="replace").strip() metadata: dict[str, Any] = { "file": source.name, "text_preview": text[:500], "epsg": None, "name": None, "authority": None, "is_valid": False, } if not text: metadata["error"] = "PRJ 파일이 비어 있습니다." return metadata parse_text, custom_authority_codes = _prepare_prj_wkt(text) try: crs = CRS.from_wkt(parse_text) except Exception as exc: metadata["error"] = str(exc) return metadata crs_metadata = normalize_crs_metadata(crs) _log_crs_status(source, crs_metadata) metadata.update( { **crs_metadata, "name": crs.name, "authority": crs.to_authority(), "custom_authority_codes": custom_authority_codes, "is_valid": True, } ) return metadata def analyze_tfw_metadata(path: str | Path) -> dict[str, Any]: """TFW의 affine 변환 계수와 유효성을 분석한다.""" source = Path(path) values = [ float(line.strip()) for line in source.read_text(encoding="utf-8", errors="replace").splitlines() if line.strip() ] if any(not math.isfinite(value) for value in values): raise ValueError("TFW 변환 계수는 유한한 숫자여야 합니다.") return { "file": source.name, "values": values, "pixel_size_x": values[0] if len(values) > 0 else None, "rotation_y": values[1] if len(values) > 1 else None, "rotation_x": values[2] if len(values) > 2 else None, "pixel_size_y": values[3] if len(values) > 3 else None, "origin_x": values[4] if len(values) > 4 else None, "origin_y": values[5] if len(values) > 5 else None, "is_valid": len(values) == 6, } def analyze_tif_metadata(path: str | Path) -> dict[str, Any]: """TIF/GeoTIFF 데이터셋의 공간 및 밴드 메타데이터를 분석한다.""" source = Path(path) rasterio_logger = logging.getLogger("rasterio._env") warning_filter = _CustomVerticalCrsWarningFilter(source) rasterio_logger.addFilter(warning_filter) try: with rasterio.open(source) as dataset: crs = dataset.crs crs_metadata = normalize_crs_metadata(crs) _log_crs_status(source, crs_metadata) bounds = dataset.bounds return { "file": source.name, "width": int(dataset.width), "height": int(dataset.height), "count": int(dataset.count), "dtypes": list(dataset.dtypes), "nodata": float(dataset.nodata) if dataset.nodata is not None else None, **crs_metadata, "bounds": { "left": float(bounds.left), "bottom": float(bounds.bottom), "right": float(bounds.right), "top": float(bounds.top), }, "transform": [float(value) for value in list(dataset.transform)[:6]], "resolution": [float(value) for value in dataset.res], "likely_type": "dem" if dataset.count == 1 else "image", } finally: rasterio_logger.removeFilter(warning_filter) def analyze_input_metadata(path: str | Path) -> dict[str, Any]: """입력 파일 확장자에 맞는 B03 메타데이터 분석 함수를 호출한다.""" source = Path(path) extension = source.suffix.lower() if extension in {".las", ".laz"}: return analyze_las_metadata(source) if extension == ".prj": return analyze_prj_metadata(source) if extension == ".tfw": return analyze_tfw_metadata(source) if extension in {".tif", ".tiff"}: return analyze_tif_metadata(source) return { "file": source.name, "extension": extension.lstrip("."), "size_bytes": source.stat().st_size, }