feat(B03): 파일 카드 미리보기 확대 — 전폭·고정 높이, 라이다 점 그림·GeoTIFF 썸네일

- 미리보기를 카드 한 행 전폭·높이 120px 고정으로 키움.
- 라이다: 분류 통계를 훑는 그 길에 XY 를 성기게 주워 탑뷰 점 그림(상한 5천 점,
  파일 재열람 없음). 분류가 없는 파일은 앞 3청크만 봄.
- GeoTIFF: 오버뷰가 있을 때만 128px 흑백 썸네일(2~98 백분위 대비), 없으면 종전 범위 사각형.
- 부속 파일(.shx·.dbf·.cpg·.prj·.tfw)은 그림 대신 구분되는 값 표시 — 머리글만 읽는
  가벼운 분석기 추가.
- 값이 없는 완료 카드는 「보여 줄 값이 없음」 한 줄.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-09-04 19:17:00 +09:00
co-authored by Claude Opus 5
parent 4e4bfa2354
commit 9019ca28ef
4 changed files with 313 additions and 6 deletions
@@ -1,6 +1,8 @@
"""B03 원본 입력 파일 메타데이터 분석."""
import base64
import csv
import io
import logging
import math
import re
@@ -11,6 +13,7 @@ from typing import Any
import laspy
import numpy as np
import rasterio
from PIL import Image
from pyproj import CRS
logger = logging.getLogger(__name__)
@@ -136,6 +139,11 @@ def _prepare_prj_wkt(text: str) -> tuple[str, list[str]]:
return _CUSTOM_VERTICAL_AUTHORITY_PATTERN.sub("", text), codes
# 카드 미리보기 점 그림의 점 수 상한과, 분류가 없는 파일에서 볼 청크 수 (2026-09-04).
_LAS_PREVIEW_MAX_POINTS = 5000
_LAS_PREVIEW_MAX_CHUNKS = 3
def analyze_las_metadata(path: str | Path) -> dict[str, Any]:
"""LAS/LAZ 헤더와 분류 통계를 메모리에 전체 적재하지 않고 분석한다."""
source = Path(path)
@@ -170,6 +178,22 @@ def analyze_las_metadata(path: str | Path) -> dict[str, Any]:
"has_return_number": "return_number" in dimension_names,
}
# 카드 미리보기용 탑뷰 점 그림 (2026-09-04 사용자 지시).
# 분류 통계를 훑는 **그 길에** XY 를 성기게 주워 둔다 — 파일을 다시 읽지 않으므로
# 업로드 시간이 늘지 않는다. 분류가 없어 훑지 않는 파일만 앞 몇 청크를 본다.
preview_points: list[list[int]] = []
stride = max(1, point_count // _LAS_PREVIEW_MAX_POINTS) if point_count else 1
def _collect(chunk: Any) -> None:
if len(preview_points) >= _LAS_PREVIEW_MAX_POINTS:
return
xs = np.asarray(chunk.x, dtype=np.float64)[::stride]
ys = np.asarray(chunk.y, dtype=np.float64)[::stride]
room = _LAS_PREVIEW_MAX_POINTS - len(preview_points)
# 카드 안 작은 그림이라 1m 눈금이면 충분하다 — 정수로 줄여 저장 용량을 아낀다.
for x, y in zip(xs[:room].tolist(), ys[:room].tolist(), strict=True):
preview_points.append([round(x), round(y)])
if metadata["has_classification"] and point_count > 0:
classification_counts: dict[int, int] = {}
for chunk in las_file.chunk_iterator(500_000):
@@ -179,9 +203,22 @@ def analyze_las_metadata(path: str | Path) -> dict[str, Any]:
)
for value, count in zip(values.tolist(), counts.tolist(), strict=True):
classification_counts[value] = classification_counts.get(value, 0) + count
_collect(chunk)
metadata["classification_summary"] = {
str(key): value for key, value in sorted(classification_counts.items())
}
elif point_count > 0:
# 분류가 없으면 전 점을 훑을 이유가 없다 — 앞 몇 청크만 보고 끝낸다.
for index, chunk in enumerate(las_file.chunk_iterator(500_000)):
_collect(chunk)
if (
index + 1 >= _LAS_PREVIEW_MAX_CHUNKS
or len(preview_points) >= _LAS_PREVIEW_MAX_POINTS
):
break
if preview_points:
metadata["preview_points"] = preview_points
return metadata
@@ -247,6 +284,43 @@ def analyze_tfw_metadata(path: str | Path) -> dict[str, Any]:
}
# 카드 미리보기 썸네일 한 변 크기(px) — 128 이면 카드 폭에서 충분히 읽힌다 (2026-09-04).
_TIF_THUMBNAIL_PX = 128
def _geotiff_thumbnail(dataset: Any) -> str | None:
"""저해상 흑백 썸네일을 data URL(PNG) 로 만든다. 만들 수 없으면 None.
오버뷰(피라미드)가 있는 파일만 대상으로 한다 — 오버뷰가 없으면 원본을 훑어 축소해야
해서 업로드가 눈에 띄게 느려진다(사용자 제약 「오래 걸리면 안 됨」).
"""
try:
if not any(dataset.overviews(index) for index in dataset.indexes):
return None
band = dataset.read(
1,
out_shape=(_TIF_THUMBNAIL_PX, _TIF_THUMBNAIL_PX),
masked=True,
)
finite = band.compressed()
if finite.size == 0:
return None
low = float(np.percentile(finite, 2))
high = float(np.percentile(finite, 98))
if high <= low:
return None
# 2~98 백분위로 늘려 대비를 준다 — DEM 은 값 폭이 좁아 그냥 펴면 밋밋하다.
scaled = np.clip((band.filled(low) - low) / (high - low), 0.0, 1.0)
grey = (scaled * 255).astype(np.uint8)
image = Image.fromarray(grey, mode="L")
buffer = io.BytesIO()
image.save(buffer, format="PNG", optimize=True)
return "data:image/png;base64," + base64.b64encode(buffer.getvalue()).decode("ascii")
except Exception as exc: # 썸네일은 곁가지다 — 실패해도 분석을 막지 않는다.
logger.warning("GeoTIFF 썸네일 생성 실패: %s", exc)
return None
def analyze_tif_metadata(path: str | Path) -> dict[str, Any]:
"""TIF/GeoTIFF 데이터셋의 공간 및 밴드 메타데이터를 분석한다."""
source = Path(path)
@@ -261,6 +335,9 @@ def analyze_tif_metadata(path: str | Path) -> dict[str, Any]:
bounds = dataset.bounds
return {
"file": source.name,
# 카드 미리보기용 흑백 썸네일 (2026-09-04 사용자 지시).
# **오버뷰가 있을 때만** 만든다 — 없는 큰 파일을 축소 읽으면 수 초가 걸린다.
"preview_thumbnail": _geotiff_thumbnail(dataset),
"width": int(dataset.width),
"height": int(dataset.height),
"count": int(dataset.count),
@@ -390,6 +467,58 @@ def analyze_planned_route_csv(path: str | Path) -> dict[str, Any]:
}
def analyze_dbf_metadata(path: str | Path) -> dict[str, Any]:
"""DBF 머리글만 읽어 레코드 수·속성 수를 낸다 (2026-09-04 카드 표시용).
머리글 32바이트가 전부라 파일 크기와 무관하게 즉시 끝난다.
"""
source = Path(path)
head = source.read_bytes()[:32]
metadata: dict[str, Any] = {
"file": source.name,
"extension": "dbf",
"size_bytes": source.stat().st_size,
}
if len(head) >= 12:
record_count = int.from_bytes(head[4:8], "little")
header_length = int.from_bytes(head[8:10], "little")
metadata["record_count"] = record_count
# 머리글 = 32바이트 고정 + 속성마다 32바이트 + 끝 표시 1바이트.
metadata["field_count"] = max(0, (header_length - 33) // 32)
return metadata
def analyze_shx_metadata(path: str | Path) -> dict[str, Any]:
"""SHX 머리글에서 도형 개수를 센다 (2026-09-04 카드 표시용).
SHX 는 도형마다 8바이트 색인이 한 줄씩이라 파일 길이로 개수가 나온다.
"""
source = Path(path)
head = source.read_bytes()[:100]
metadata: dict[str, Any] = {
"file": source.name,
"extension": "shx",
"size_bytes": source.stat().st_size,
}
if len(head) >= 28:
# 24~27바이트: 파일 길이(16비트 워드 단위, 빅엔디안).
words = int.from_bytes(head[24:28], "big")
metadata["shape_count"] = max(0, (words * 2 - 100) // 8)
return metadata
def analyze_cpg_metadata(path: str | Path) -> dict[str, Any]:
"""CPG 는 인코딩 이름 한 줄이 전부다 (2026-09-04 카드 표시용)."""
source = Path(path)
text = source.read_text(encoding="utf-8", errors="replace").strip()
return {
"file": source.name,
"extension": "cpg",
"size_bytes": source.stat().st_size,
"encoding": text or None,
}
def analyze_input_metadata(path: str | Path) -> dict[str, Any]:
"""입력 파일 확장자에 맞는 B03 메타데이터 분석 함수를 호출한다."""
source = Path(path)
@@ -406,6 +535,13 @@ def analyze_input_metadata(path: str | Path) -> dict[str, Any]:
return analyze_prj_metadata(source)
if extension == ".tfw":
return analyze_tfw_metadata(source)
# 부속 파일도 카드에 「구분되는 값」을 보여 준다 (2026-09-04 사용자 지시).
if extension == ".dbf":
return analyze_dbf_metadata(source)
if extension == ".shx":
return analyze_shx_metadata(source)
if extension == ".cpg":
return analyze_cpg_metadata(source)
if extension in {".tif", ".tiff"}:
return analyze_tif_metadata(source)
return {