289 lines
10 KiB
Python
289 lines
10 KiB
Python
"""B03 원본 입력 파일 메타데이터 분석."""
|
|
|
|
import logging
|
|
import math
|
|
import re
|
|
from pathlib import Path
|
|
from threading import get_ident
|
|
from typing import Any
|
|
|
|
import laspy
|
|
import numpy as np
|
|
import rasterio
|
|
from pyproj import CRS
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
_CUSTOM_VERTICAL_AUTHORITY_PATTERN = re.compile(
|
|
r',?\s*AUTHORITY\["EPSG","(9995|99999)"\]',
|
|
re.IGNORECASE,
|
|
)
|
|
_CUSTOM_VERTICAL_WARNING_PATTERN = re.compile(
|
|
r"proj_create_from_database: crs not found: EPSG:(9995|99999)",
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
|
|
class _CustomVerticalCrsWarningFilter(logging.Filter):
|
|
"""알려진 사설 수직 CRS 경고만 B03 안내 로그로 변환한다."""
|
|
|
|
def __init__(self, source: Path) -> None:
|
|
super().__init__()
|
|
self.source = source
|
|
self.thread_id = get_ident()
|
|
self.logged_codes: set[str] = set()
|
|
|
|
def filter(self, record: logging.LogRecord) -> bool:
|
|
if get_ident() != self.thread_id:
|
|
return True
|
|
match = _CUSTOM_VERTICAL_WARNING_PATTERN.search(record.getMessage())
|
|
if match is None:
|
|
return True
|
|
code = f"EPSG:{match.group(1)}"
|
|
if code not in self.logged_codes:
|
|
logger.info(
|
|
"사용자 정의 수직 CRS 코드를 보존합니다: file=%s code=%s",
|
|
self.source.name,
|
|
code,
|
|
)
|
|
self.logged_codes.add(code)
|
|
return False
|
|
|
|
|
|
def _component_metadata(crs: CRS) -> dict[str, Any]:
|
|
"""CRS 구성 요소를 JSON 저장 가능한 메타데이터로 변환한다."""
|
|
authority = crs.to_authority()
|
|
epsg = crs.to_epsg()
|
|
return {
|
|
"name": crs.name,
|
|
"type": crs.type_name,
|
|
"epsg": epsg,
|
|
"authority": (
|
|
{"name": authority[0], "code": authority[1]} if authority is not None else None
|
|
),
|
|
}
|
|
|
|
|
|
def normalize_crs_metadata(crs: Any | None) -> dict[str, Any]:
|
|
"""복합 CRS를 수평·수직 구성 요소로 분리해 일관된 상태를 반환한다."""
|
|
if crs is None:
|
|
return {
|
|
"crs": None,
|
|
"epsg": None,
|
|
"horizontal_crs": None,
|
|
"vertical_crs": None,
|
|
"crs_status": "missing_crs",
|
|
}
|
|
|
|
parsed = crs if isinstance(crs, CRS) else CRS.from_user_input(crs)
|
|
components = parsed.sub_crs_list or [parsed]
|
|
horizontal = next(
|
|
(item for item in components if item.is_projected or item.is_geographic),
|
|
None,
|
|
)
|
|
vertical = next((item for item in components if item.is_vertical), None)
|
|
|
|
horizontal_metadata = _component_metadata(horizontal) if horizontal is not None else None
|
|
vertical_metadata = _component_metadata(vertical) if vertical is not None else None
|
|
horizontal_epsg = horizontal_metadata["epsg"] if horizontal_metadata is not None else None
|
|
|
|
if horizontal_epsg is None:
|
|
status = "unknown_horizontal_crs"
|
|
elif vertical_metadata is not None and vertical_metadata["epsg"] is None:
|
|
status = "custom_vertical_crs"
|
|
else:
|
|
status = "identified"
|
|
|
|
return {
|
|
"crs": crs.to_string(),
|
|
"epsg": horizontal_epsg,
|
|
"horizontal_crs": horizontal_metadata,
|
|
"vertical_crs": vertical_metadata,
|
|
"crs_status": status,
|
|
}
|
|
|
|
|
|
def _log_crs_status(source: Path, metadata: dict[str, Any]) -> None:
|
|
"""정상화된 CRS 상태를 B03 도메인 로그로 남긴다."""
|
|
if metadata["crs_status"] == "custom_vertical_crs":
|
|
logger.info(
|
|
"사용자 정의 수직 CRS를 보존합니다: file=%s horizontal_epsg=%s vertical=%s",
|
|
source.name,
|
|
metadata["epsg"],
|
|
metadata["vertical_crs"]["name"],
|
|
)
|
|
elif metadata["crs_status"] == "unknown_horizontal_crs":
|
|
logger.warning("수평 CRS를 EPSG로 식별하지 못했습니다: file=%s", source.name)
|
|
|
|
|
|
def _prepare_prj_wkt(text: str) -> tuple[str, list[str]]:
|
|
"""KNGeoid24 사설 EPSG 표식만 파싱용 WKT에서 분리한다."""
|
|
if "KNGeoid24" not in text:
|
|
return text, []
|
|
codes = sorted({f"EPSG:{code}" for code in _CUSTOM_VERTICAL_AUTHORITY_PATTERN.findall(text)})
|
|
return _CUSTOM_VERTICAL_AUTHORITY_PATTERN.sub("", text), codes
|
|
|
|
|
|
def analyze_las_metadata(path: str | Path) -> dict[str, Any]:
|
|
"""LAS/LAZ 헤더와 분류 통계를 메모리에 전체 적재하지 않고 분석한다."""
|
|
source = Path(path)
|
|
with laspy.open(source) as las_file:
|
|
header = las_file.header
|
|
point_format = header.point_format
|
|
dimension_names = list(point_format.dimension_names)
|
|
point_count = int(header.point_count)
|
|
crs = header.parse_crs()
|
|
crs_metadata = normalize_crs_metadata(crs)
|
|
_log_crs_status(source, crs_metadata)
|
|
metadata: dict[str, Any] = {
|
|
"file": source.name,
|
|
"version": f"{header.version.major}.{header.version.minor}",
|
|
"point_format": {
|
|
"id": point_format.id,
|
|
"dimensions": dimension_names,
|
|
},
|
|
"point_count": point_count,
|
|
"bounds": {
|
|
"x": [float(header.mins[0]), float(header.maxs[0])],
|
|
"y": [float(header.mins[1]), float(header.maxs[1])],
|
|
"z": [float(header.mins[2]), float(header.maxs[2])],
|
|
},
|
|
"scale": [float(value) for value in header.scales],
|
|
"offset": [float(value) for value in header.offsets],
|
|
"has_crs": crs is not None,
|
|
**crs_metadata,
|
|
"has_classification": "classification" in dimension_names,
|
|
"has_rgb": all(name in dimension_names for name in ("red", "green", "blue")),
|
|
"has_intensity": "intensity" in dimension_names,
|
|
"has_return_number": "return_number" in dimension_names,
|
|
}
|
|
|
|
if metadata["has_classification"] and point_count > 0:
|
|
classification_counts: dict[int, int] = {}
|
|
for chunk in las_file.chunk_iterator(500_000):
|
|
values, counts = np.unique(
|
|
np.asarray(chunk.classification, dtype=np.uint8),
|
|
return_counts=True,
|
|
)
|
|
for value, count in zip(values.tolist(), counts.tolist(), strict=True):
|
|
classification_counts[value] = classification_counts.get(value, 0) + count
|
|
metadata["classification_summary"] = {
|
|
str(key): value for key, value in sorted(classification_counts.items())
|
|
}
|
|
|
|
return metadata
|
|
|
|
|
|
def analyze_prj_metadata(path: str | Path) -> dict[str, Any]:
|
|
"""PRJ WKT에서 좌표계 식별자와 명칭을 추출한다."""
|
|
source = Path(path)
|
|
text = source.read_text(encoding="utf-8", errors="replace").strip()
|
|
metadata: dict[str, Any] = {
|
|
"file": source.name,
|
|
"text_preview": text[:500],
|
|
"epsg": None,
|
|
"name": None,
|
|
"authority": None,
|
|
"is_valid": False,
|
|
}
|
|
if not text:
|
|
metadata["error"] = "PRJ 파일이 비어 있습니다."
|
|
return metadata
|
|
|
|
parse_text, custom_authority_codes = _prepare_prj_wkt(text)
|
|
try:
|
|
crs = CRS.from_wkt(parse_text)
|
|
except Exception as exc:
|
|
metadata["error"] = str(exc)
|
|
return metadata
|
|
|
|
crs_metadata = normalize_crs_metadata(crs)
|
|
_log_crs_status(source, crs_metadata)
|
|
metadata.update(
|
|
{
|
|
**crs_metadata,
|
|
"name": crs.name,
|
|
"authority": crs.to_authority(),
|
|
"custom_authority_codes": custom_authority_codes,
|
|
"is_valid": True,
|
|
}
|
|
)
|
|
return metadata
|
|
|
|
|
|
def analyze_tfw_metadata(path: str | Path) -> dict[str, Any]:
|
|
"""TFW의 affine 변환 계수와 유효성을 분석한다."""
|
|
source = Path(path)
|
|
values = [
|
|
float(line.strip())
|
|
for line in source.read_text(encoding="utf-8", errors="replace").splitlines()
|
|
if line.strip()
|
|
]
|
|
if any(not math.isfinite(value) for value in values):
|
|
raise ValueError("TFW 변환 계수는 유한한 숫자여야 합니다.")
|
|
|
|
return {
|
|
"file": source.name,
|
|
"values": values,
|
|
"pixel_size_x": values[0] if len(values) > 0 else None,
|
|
"rotation_y": values[1] if len(values) > 1 else None,
|
|
"rotation_x": values[2] if len(values) > 2 else None,
|
|
"pixel_size_y": values[3] if len(values) > 3 else None,
|
|
"origin_x": values[4] if len(values) > 4 else None,
|
|
"origin_y": values[5] if len(values) > 5 else None,
|
|
"is_valid": len(values) == 6,
|
|
}
|
|
|
|
|
|
def analyze_tif_metadata(path: str | Path) -> dict[str, Any]:
|
|
"""TIF/GeoTIFF 데이터셋의 공간 및 밴드 메타데이터를 분석한다."""
|
|
source = Path(path)
|
|
rasterio_logger = logging.getLogger("rasterio._env")
|
|
warning_filter = _CustomVerticalCrsWarningFilter(source)
|
|
rasterio_logger.addFilter(warning_filter)
|
|
try:
|
|
with rasterio.open(source) as dataset:
|
|
crs = dataset.crs
|
|
crs_metadata = normalize_crs_metadata(crs)
|
|
_log_crs_status(source, crs_metadata)
|
|
bounds = dataset.bounds
|
|
return {
|
|
"file": source.name,
|
|
"width": int(dataset.width),
|
|
"height": int(dataset.height),
|
|
"count": int(dataset.count),
|
|
"dtypes": list(dataset.dtypes),
|
|
"nodata": float(dataset.nodata) if dataset.nodata is not None else None,
|
|
**crs_metadata,
|
|
"bounds": {
|
|
"left": float(bounds.left),
|
|
"bottom": float(bounds.bottom),
|
|
"right": float(bounds.right),
|
|
"top": float(bounds.top),
|
|
},
|
|
"transform": [float(value) for value in list(dataset.transform)[:6]],
|
|
"resolution": [float(value) for value in dataset.res],
|
|
"likely_type": "dem" if dataset.count == 1 else "image",
|
|
}
|
|
finally:
|
|
rasterio_logger.removeFilter(warning_filter)
|
|
|
|
|
|
def analyze_input_metadata(path: str | Path) -> dict[str, Any]:
|
|
"""입력 파일 확장자에 맞는 B03 메타데이터 분석 함수를 호출한다."""
|
|
source = Path(path)
|
|
extension = source.suffix.lower()
|
|
if extension in {".las", ".laz"}:
|
|
return analyze_las_metadata(source)
|
|
if extension == ".prj":
|
|
return analyze_prj_metadata(source)
|
|
if extension == ".tfw":
|
|
return analyze_tfw_metadata(source)
|
|
if extension in {".tif", ".tiff"}:
|
|
return analyze_tif_metadata(source)
|
|
return {
|
|
"file": source.name,
|
|
"extension": extension.lstrip("."),
|
|
"size_bytes": source.stat().st_size,
|
|
}
|