Files
Aislo/B03_FileInput/B03_FileInput_Engine_Analyze.py
T
eomsangdonandClaude Opus 5 9019ca28ef feat(B03): 파일 카드 미리보기 확대 — 전폭·고정 높이, 라이다 점 그림·GeoTIFF 썸네일
- 미리보기를 카드 한 행 전폭·높이 120px 고정으로 키움.
- 라이다: 분류 통계를 훑는 그 길에 XY 를 성기게 주워 탑뷰 점 그림(상한 5천 점,
  파일 재열람 없음). 분류가 없는 파일은 앞 3청크만 봄.
- GeoTIFF: 오버뷰가 있을 때만 128px 흑백 썸네일(2~98 백분위 대비), 없으면 종전 범위 사각형.
- 부속 파일(.shx·.dbf·.cpg·.prj·.tfw)은 그림 대신 구분되는 값 표시 — 머리글만 읽는
  가벼운 분석기 추가.
- 값이 없는 완료 카드는 「보여 줄 값이 없음」 한 줄.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-04 19:17:00 +09:00

552 lines
22 KiB
Python

"""B03 원본 입력 파일 메타데이터 분석."""
import base64
import csv
import io
import logging
import math
import re
from pathlib import Path
from threading import get_ident
from typing import Any
import laspy
import numpy as np
import rasterio
from PIL import Image
from pyproj import CRS
logger = logging.getLogger(__name__)
_CUSTOM_VERTICAL_AUTHORITY_PATTERN = re.compile(
r',?\s*AUTHORITY\["EPSG","(9995|99999)"\]',
re.IGNORECASE,
)
_CUSTOM_VERTICAL_WARNING_PATTERN = re.compile(
r"proj_create_from_database: crs not found: EPSG:(9995|99999)",
re.IGNORECASE,
)
class _CustomVerticalCrsWarningFilter(logging.Filter):
"""알려진 사설 수직 CRS 경고만 B03 안내 로그로 변환한다."""
def __init__(self, source: Path) -> None:
super().__init__()
self.source = source
self.thread_id = get_ident()
self.logged_codes: set[str] = set()
def filter(self, record: logging.LogRecord) -> bool:
if get_ident() != self.thread_id:
return True
match = _CUSTOM_VERTICAL_WARNING_PATTERN.search(record.getMessage())
if match is None:
return True
code = f"EPSG:{match.group(1)}"
if code not in self.logged_codes:
logger.info(
"사용자 정의 수직 CRS 코드를 보존합니다: file=%s code=%s",
self.source.name,
code,
)
self.logged_codes.add(code)
return False
def _component_metadata(crs: CRS) -> dict[str, Any]:
"""CRS 구성 요소를 JSON 저장 가능한 메타데이터로 변환한다."""
authority = crs.to_authority()
epsg = crs.to_epsg()
if epsg is None and (crs.is_projected or crs.is_geographic):
# DB 대조 실패(비표준 TOWGS84·AUTHORITY 없는 ESRI WKT 등) — 파라미터
# 지문으로 라벨을 보강한다. 변환 정본은 여전히 원문 WKT다 (2026-08-31).
from common_util.common_util_crs import identify_epsg
epsg = identify_epsg(crs)
return {
"name": crs.name,
"type": crs.type_name,
"epsg": epsg,
"authority": (
{"name": authority[0], "code": authority[1]} if authority is not None else None
),
}
def normalize_crs_metadata(crs: Any | None) -> dict[str, Any]:
"""복합 CRS를 수평·수직 구성 요소로 분리해 일관된 상태를 반환한다."""
if crs is None:
return {
"crs": None,
"epsg": None,
"horizontal_crs": None,
"vertical_crs": None,
"crs_status": "missing_crs",
}
parsed = crs if isinstance(crs, CRS) else CRS.from_user_input(crs)
# TOWGS84가 붙은 WKT는 BoundCRS로 감싸져 is_projected가 False가 된다 —
# 벗겨야 수평 성분 탐색과 EPSG 라벨이 동작한다 (2026-08-31).
from common_util.common_util_crs import strip_bound
flattened = strip_bound(parsed)
components = [strip_bound(item) for item in (flattened.sub_crs_list or [flattened])]
horizontal = next(
(item for item in components if item.is_projected or item.is_geographic),
None,
)
vertical = next((item for item in components if item.is_vertical), None)
horizontal_metadata = _component_metadata(horizontal) if horizontal is not None else None
vertical_metadata = _component_metadata(vertical) if vertical is not None else None
horizontal_epsg = horizontal_metadata["epsg"] if horizontal_metadata is not None else None
if horizontal_epsg is None:
status = "unknown_horizontal_crs"
elif vertical_metadata is not None and vertical_metadata["epsg"] is None:
status = "custom_vertical_crs"
else:
status = "identified"
return {
"crs": crs.to_string(),
"epsg": horizontal_epsg,
"horizontal_crs": horizontal_metadata,
"vertical_crs": vertical_metadata,
"crs_status": status,
}
def _log_crs_status(source: Path, metadata: dict[str, Any]) -> None:
"""정상화된 CRS 상태를 B03 도메인 로그로 남긴다."""
if metadata["crs_status"] == "custom_vertical_crs":
logger.info(
"사용자 정의 수직 CRS를 보존합니다: file=%s horizontal_epsg=%s vertical=%s",
source.name,
metadata["epsg"],
metadata["vertical_crs"]["name"],
)
elif metadata["crs_status"] == "unknown_horizontal_crs":
logger.warning("수평 CRS를 EPSG로 식별하지 못했습니다: file=%s", source.name)
def _prepare_prj_wkt(text: str) -> tuple[str, list[str]]:
"""KNGeoid24 사설 EPSG 표식만 파싱용 WKT에서 분리한다."""
if "KNGeoid24" not in text:
return text, []
codes = sorted({f"EPSG:{code}" for code in _CUSTOM_VERTICAL_AUTHORITY_PATTERN.findall(text)})
return _CUSTOM_VERTICAL_AUTHORITY_PATTERN.sub("", text), codes
# 카드 미리보기 점 그림의 점 수 상한과, 분류가 없는 파일에서 볼 청크 수 (2026-09-04).
_LAS_PREVIEW_MAX_POINTS = 5000
_LAS_PREVIEW_MAX_CHUNKS = 3
def analyze_las_metadata(path: str | Path) -> dict[str, Any]:
"""LAS/LAZ 헤더와 분류 통계를 메모리에 전체 적재하지 않고 분석한다."""
source = Path(path)
with laspy.open(source) as las_file:
header = las_file.header
point_format = header.point_format
dimension_names = list(point_format.dimension_names)
point_count = int(header.point_count)
crs = header.parse_crs()
crs_metadata = normalize_crs_metadata(crs)
_log_crs_status(source, crs_metadata)
metadata: dict[str, Any] = {
"file": source.name,
"version": f"{header.version.major}.{header.version.minor}",
"point_format": {
"id": point_format.id,
"dimensions": dimension_names,
},
"point_count": point_count,
"bounds": {
"x": [float(header.mins[0]), float(header.maxs[0])],
"y": [float(header.mins[1]), float(header.maxs[1])],
"z": [float(header.mins[2]), float(header.maxs[2])],
},
"scale": [float(value) for value in header.scales],
"offset": [float(value) for value in header.offsets],
"has_crs": crs is not None,
**crs_metadata,
"has_classification": "classification" in dimension_names,
"has_rgb": all(name in dimension_names for name in ("red", "green", "blue")),
"has_intensity": "intensity" in dimension_names,
"has_return_number": "return_number" in dimension_names,
}
# 카드 미리보기용 탑뷰 점 그림 (2026-09-04 사용자 지시).
# 분류 통계를 훑는 **그 길에** XY 를 성기게 주워 둔다 — 파일을 다시 읽지 않으므로
# 업로드 시간이 늘지 않는다. 분류가 없어 훑지 않는 파일만 앞 몇 청크를 본다.
preview_points: list[list[int]] = []
stride = max(1, point_count // _LAS_PREVIEW_MAX_POINTS) if point_count else 1
def _collect(chunk: Any) -> None:
if len(preview_points) >= _LAS_PREVIEW_MAX_POINTS:
return
xs = np.asarray(chunk.x, dtype=np.float64)[::stride]
ys = np.asarray(chunk.y, dtype=np.float64)[::stride]
room = _LAS_PREVIEW_MAX_POINTS - len(preview_points)
# 카드 안 작은 그림이라 1m 눈금이면 충분하다 — 정수로 줄여 저장 용량을 아낀다.
for x, y in zip(xs[:room].tolist(), ys[:room].tolist(), strict=True):
preview_points.append([round(x), round(y)])
if metadata["has_classification"] and point_count > 0:
classification_counts: dict[int, int] = {}
for chunk in las_file.chunk_iterator(500_000):
values, counts = np.unique(
np.asarray(chunk.classification, dtype=np.uint8),
return_counts=True,
)
for value, count in zip(values.tolist(), counts.tolist(), strict=True):
classification_counts[value] = classification_counts.get(value, 0) + count
_collect(chunk)
metadata["classification_summary"] = {
str(key): value for key, value in sorted(classification_counts.items())
}
elif point_count > 0:
# 분류가 없으면 전 점을 훑을 이유가 없다 — 앞 몇 청크만 보고 끝낸다.
for index, chunk in enumerate(las_file.chunk_iterator(500_000)):
_collect(chunk)
if (
index + 1 >= _LAS_PREVIEW_MAX_CHUNKS
or len(preview_points) >= _LAS_PREVIEW_MAX_POINTS
):
break
if preview_points:
metadata["preview_points"] = preview_points
return metadata
def analyze_prj_metadata(path: str | Path) -> dict[str, Any]:
"""PRJ WKT에서 좌표계 식별자와 명칭을 추출한다."""
source = Path(path)
text = source.read_text(encoding="utf-8", errors="replace").strip()
metadata: dict[str, Any] = {
"file": source.name,
"text_preview": text[:500],
"epsg": None,
"name": None,
"authority": None,
"is_valid": False,
}
if not text:
metadata["error"] = "PRJ 파일이 비어 있습니다."
return metadata
parse_text, custom_authority_codes = _prepare_prj_wkt(text)
try:
crs = CRS.from_wkt(parse_text)
except Exception as exc:
metadata["error"] = str(exc)
return metadata
crs_metadata = normalize_crs_metadata(crs)
_log_crs_status(source, crs_metadata)
metadata.update(
{
**crs_metadata,
"name": crs.name,
"authority": crs.to_authority(),
"custom_authority_codes": custom_authority_codes,
"is_valid": True,
}
)
return metadata
def analyze_tfw_metadata(path: str | Path) -> dict[str, Any]:
"""TFW의 affine 변환 계수와 유효성을 분석한다."""
source = Path(path)
values = [
float(line.strip())
for line in source.read_text(encoding="utf-8", errors="replace").splitlines()
if line.strip()
]
if any(not math.isfinite(value) for value in values):
raise ValueError("TFW 변환 계수는 유한한 숫자여야 합니다.")
return {
"file": source.name,
"values": values,
"pixel_size_x": values[0] if len(values) > 0 else None,
"rotation_y": values[1] if len(values) > 1 else None,
"rotation_x": values[2] if len(values) > 2 else None,
"pixel_size_y": values[3] if len(values) > 3 else None,
"origin_x": values[4] if len(values) > 4 else None,
"origin_y": values[5] if len(values) > 5 else None,
"is_valid": len(values) == 6,
}
# 카드 미리보기 썸네일 한 변 크기(px) — 128 이면 카드 폭에서 충분히 읽힌다 (2026-09-04).
_TIF_THUMBNAIL_PX = 128
def _geotiff_thumbnail(dataset: Any) -> str | None:
"""저해상 흑백 썸네일을 data URL(PNG) 로 만든다. 만들 수 없으면 None.
오버뷰(피라미드)가 있는 파일만 대상으로 한다 — 오버뷰가 없으면 원본을 훑어 축소해야
해서 업로드가 눈에 띄게 느려진다(사용자 제약 「오래 걸리면 안 됨」).
"""
try:
if not any(dataset.overviews(index) for index in dataset.indexes):
return None
band = dataset.read(
1,
out_shape=(_TIF_THUMBNAIL_PX, _TIF_THUMBNAIL_PX),
masked=True,
)
finite = band.compressed()
if finite.size == 0:
return None
low = float(np.percentile(finite, 2))
high = float(np.percentile(finite, 98))
if high <= low:
return None
# 2~98 백분위로 늘려 대비를 준다 — DEM 은 값 폭이 좁아 그냥 펴면 밋밋하다.
scaled = np.clip((band.filled(low) - low) / (high - low), 0.0, 1.0)
grey = (scaled * 255).astype(np.uint8)
image = Image.fromarray(grey, mode="L")
buffer = io.BytesIO()
image.save(buffer, format="PNG", optimize=True)
return "data:image/png;base64," + base64.b64encode(buffer.getvalue()).decode("ascii")
except Exception as exc: # 썸네일은 곁가지다 — 실패해도 분석을 막지 않는다.
logger.warning("GeoTIFF 썸네일 생성 실패: %s", exc)
return None
def analyze_tif_metadata(path: str | Path) -> dict[str, Any]:
"""TIF/GeoTIFF 데이터셋의 공간 및 밴드 메타데이터를 분석한다."""
source = Path(path)
rasterio_logger = logging.getLogger("rasterio._env")
warning_filter = _CustomVerticalCrsWarningFilter(source)
rasterio_logger.addFilter(warning_filter)
try:
with rasterio.open(source) as dataset:
crs = dataset.crs
crs_metadata = normalize_crs_metadata(crs)
_log_crs_status(source, crs_metadata)
bounds = dataset.bounds
return {
"file": source.name,
# 카드 미리보기용 흑백 썸네일 (2026-09-04 사용자 지시).
# **오버뷰가 있을 때만** 만든다 — 없는 큰 파일을 축소 읽으면 수 초가 걸린다.
"preview_thumbnail": _geotiff_thumbnail(dataset),
"width": int(dataset.width),
"height": int(dataset.height),
"count": int(dataset.count),
"dtypes": list(dataset.dtypes),
"nodata": float(dataset.nodata) if dataset.nodata is not None else None,
**crs_metadata,
"bounds": {
"left": float(bounds.left),
"bottom": float(bounds.bottom),
"right": float(bounds.right),
"top": float(bounds.top),
},
"transform": [float(value) for value in list(dataset.transform)[:6]],
"resolution": [float(value) for value in dataset.res],
"likely_type": "dem" if dataset.count == 1 else "image",
}
finally:
rasterio_logger.removeFilter(warning_filter)
_PLANNED_ROUTE_COLUMNS = ("route_name", "sequence", "x", "y", "z", "crs_epsg")
def _parse_route_integer(value: str, *, field: str, row_number: int) -> int:
normalized = value.strip()
if not re.fullmatch(r"[0-9]+", normalized):
raise ValueError(f"CSV {row_number}행의 {field} 값은 양의 정수여야 합니다.")
parsed = int(normalized)
if parsed <= 0:
raise ValueError(f"CSV {row_number}행의 {field} 값은 양의 정수여야 합니다.")
return parsed
def _parse_route_coordinate(value: str, *, field: str, row_number: int) -> float:
try:
parsed = float(value.strip())
except (AttributeError, ValueError) as exc:
raise ValueError(f"CSV {row_number}행의 {field} 값은 숫자여야 합니다.") from exc
if not math.isfinite(parsed):
raise ValueError(f"CSV {row_number}행의 {field} 값은 유한한 숫자여야 합니다.")
return parsed
def analyze_planned_route_csv(path: str | Path) -> dict[str, Any]:
"""원청 계획노선 CSV를 검증하고 경로 메타데이터를 반환한다."""
source = Path(path)
with source.open("r", encoding="utf-8-sig", newline="") as csv_file:
reader = csv.DictReader(csv_file)
if reader.fieldnames is None:
raise ValueError("계획노선 CSV 헤더를 찾을 수 없습니다.")
normalized_headers = [header.strip() for header in reader.fieldnames]
if len(set(normalized_headers)) != len(normalized_headers):
raise ValueError("계획노선 CSV 헤더에 중복된 열이 있습니다.")
header_map = dict(zip(normalized_headers, reader.fieldnames, strict=True))
missing = [column for column in _PLANNED_ROUTE_COLUMNS if column not in header_map]
if missing:
raise ValueError(f"계획노선 CSV 필수 열이 없습니다: {', '.join(missing)}")
route_name: str | None = None
crs_epsg: int | None = None
points: list[tuple[float, float, float]] = []
for expected_sequence, row in enumerate(reader, start=1):
row_number = expected_sequence + 1
current_name = (row.get(header_map["route_name"]) or "").strip()
if not current_name:
raise ValueError(f"CSV {row_number}행의 route_name 값이 비어 있습니다.")
if route_name is None:
route_name = current_name
elif current_name != route_name:
raise ValueError("계획노선 CSV에는 하나의 route_name만 사용할 수 있습니다.")
sequence = _parse_route_integer(
row.get(header_map["sequence"]) or "",
field="sequence",
row_number=row_number,
)
if sequence != expected_sequence:
raise ValueError(
f"CSV {row_number}행의 sequence는 {expected_sequence}이어야 합니다."
)
current_epsg = _parse_route_integer(
row.get(header_map["crs_epsg"]) or "",
field="crs_epsg",
row_number=row_number,
)
if crs_epsg is None:
crs_epsg = current_epsg
elif current_epsg != crs_epsg:
raise ValueError("계획노선 CSV의 crs_epsg는 모든 행에서 같아야 합니다.")
points.append(
tuple(
_parse_route_coordinate(
row.get(header_map[field]) or "",
field=field,
row_number=row_number,
)
for field in ("x", "y", "z")
)
)
if len(points) < 2:
raise ValueError("계획노선 CSV에는 좌표가 2개 이상 있어야 합니다.")
xs, ys, zs = zip(*points, strict=True)
return {
"file": source.name,
"extension": "csv",
"size_bytes": source.stat().st_size,
"purpose": "planned_route",
"route_name": route_name,
"point_count": len(points),
"epsg": crs_epsg,
"columns": list(_PLANNED_ROUTE_COLUMNS),
"bounds": {
"x_min": min(xs),
"x_max": max(xs),
"y_min": min(ys),
"y_max": max(ys),
"z_min": min(zs),
"z_max": max(zs),
},
"start_point": list(points[0]),
"end_point": list(points[-1]),
}
def analyze_dbf_metadata(path: str | Path) -> dict[str, Any]:
"""DBF 머리글만 읽어 레코드 수·속성 수를 낸다 (2026-09-04 카드 표시용).
머리글 32바이트가 전부라 파일 크기와 무관하게 즉시 끝난다.
"""
source = Path(path)
head = source.read_bytes()[:32]
metadata: dict[str, Any] = {
"file": source.name,
"extension": "dbf",
"size_bytes": source.stat().st_size,
}
if len(head) >= 12:
record_count = int.from_bytes(head[4:8], "little")
header_length = int.from_bytes(head[8:10], "little")
metadata["record_count"] = record_count
# 머리글 = 32바이트 고정 + 속성마다 32바이트 + 끝 표시 1바이트.
metadata["field_count"] = max(0, (header_length - 33) // 32)
return metadata
def analyze_shx_metadata(path: str | Path) -> dict[str, Any]:
"""SHX 머리글에서 도형 개수를 센다 (2026-09-04 카드 표시용).
SHX 는 도형마다 8바이트 색인이 한 줄씩이라 파일 길이로 개수가 나온다.
"""
source = Path(path)
head = source.read_bytes()[:100]
metadata: dict[str, Any] = {
"file": source.name,
"extension": "shx",
"size_bytes": source.stat().st_size,
}
if len(head) >= 28:
# 24~27바이트: 파일 길이(16비트 워드 단위, 빅엔디안).
words = int.from_bytes(head[24:28], "big")
metadata["shape_count"] = max(0, (words * 2 - 100) // 8)
return metadata
def analyze_cpg_metadata(path: str | Path) -> dict[str, Any]:
"""CPG 는 인코딩 이름 한 줄이 전부다 (2026-09-04 카드 표시용)."""
source = Path(path)
text = source.read_text(encoding="utf-8", errors="replace").strip()
return {
"file": source.name,
"extension": "cpg",
"size_bytes": source.stat().st_size,
"encoding": text or None,
}
def analyze_input_metadata(path: str | Path) -> dict[str, Any]:
"""입력 파일 확장자에 맞는 B03 메타데이터 분석 함수를 호출한다."""
source = Path(path)
extension = source.suffix.lower()
if extension == ".csv":
return analyze_planned_route_csv(source)
if extension == ".shp":
from B03_FileInput.B03_FileInput_Engine_Shapefile import analyze_shapefile_metadata
return analyze_shapefile_metadata(source)
if extension in {".las", ".laz"}:
return analyze_las_metadata(source)
if extension == ".prj":
return analyze_prj_metadata(source)
if extension == ".tfw":
return analyze_tfw_metadata(source)
# 부속 파일도 카드에 「구분되는 값」을 보여 준다 (2026-09-04 사용자 지시).
if extension == ".dbf":
return analyze_dbf_metadata(source)
if extension == ".shx":
return analyze_shx_metadata(source)
if extension == ".cpg":
return analyze_cpg_metadata(source)
if extension in {".tif", ".tiff"}:
return analyze_tif_metadata(source)
return {
"file": source.name,
"extension": extension.lstrip("."),
"size_bytes": source.stat().st_size,
}