Files
Aislo/B03_FileInput/B03_FileInput_Repository.py
T
eomsangdonandClaude Opus 5 d935013ffa feat(B03): 같은 파일 재업로드는 전송 생략, 다른 내용이면 덮어쓰기 (E2E 결함 2)
사용자 결정(2026-08-08): 복수 파일을 허용하고 중복 기준은 파일명으로 둔다. 같은 이름으로
같은 내용이 다시 들어오면 덮어쓰지 말고 건너뛰고, 내용이 다르면 덮어쓴다.

파일 지문(부분 샘플링)
- B03_FileInput_Fingerprint.ts: 파일 크기 + 앞·중간·끝 8MB를 이어 SHA-256. 24MB만 읽어
  1~2초면 끝난다. 전체 읽기(1.7GB, 10~30초)와 견줘 실용적이고, 자리를 앞·중간·끝으로
  흩어 놓아 머리말만 같은 파일도 갈린다. 한계는 주석에 적었다.
- 화면이 업로드 세션 생성 요청에 지문을 실어 보내고, 서버가 같은 이름의 최신 입력 파일
  메타데이터에 적힌 지문과 견준다. 같으면 already_uploaded=true로 답해 **전송 자체를**
  건너뛴다(1.7GB면 3~5분 절약). 지문이 없거나 다르면 그냥 올린다 — 애매하면 올리는 쪽.
- 완료 요청에도 지문을 실어 input_files.metadata에 남긴다. upload_sessions에 컬럼을
  더하지 않으려는 선택이라 DB 스키마 변경이 없다.

옛 행 정리
- supersede_previous_input_files(): 같은 이름의 이전 행을 SUPERSEDED로 내린다. 조회는
  UPLOADED/PROCESSED만 보므로 목록·분석에서 자동으로 빠지고, 행은 이력으로 남는다.
- 직접 업로드 완료와 보관함 연결 양쪽에 적용.

검증(실서버 f45243b3)
- 같은 파일 재요청 → already_uploaded=true, 세션 미발급.
- 지문이 다르면 → 세션 발급(정상 업로드 경로).
- 옛 행이 SUPERSEDED로 내려가는 것 DB에서 확인.
- 화면 코드(B03_FileInput_Fingerprint.ts)를 그대로 실행해 만든 지문과 서버측 검증
  스크립트의 지문이 20MB 표본에서 완전히 일치(f281fd08…93e4).
typecheck·ruff·prettier 통과, 정적 번들 재빌드.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-08 20:11:23 +09:00

391 lines
12 KiB
Python

"""B03 input_files 테이블의 aiomysql Raw SQL 접근."""
import json
from pathlib import PurePosixPath
from typing import Any
from uuid import UUID
import aiomysql
async def create_input_file(
connection: aiomysql.Connection,
*,
project_id: UUID,
file_type: str,
original_filename: str,
relative_path: str,
file_size_bytes: int,
upload_by: int | None,
crs_epsg: int | None,
metadata: dict[str, Any],
) -> int:
"""업로드 원본 파일 메타데이터를 저장하고 생성된 ID를 반환한다."""
normalized_path = PurePosixPath(relative_path)
if normalized_path.is_absolute() or ".." in normalized_path.parts:
raise ValueError("DB에는 프로젝트 루트 기준 상대 경로만 저장할 수 있습니다.")
if normalized_path.parts[:2] != ("B03_FileInput", "input"):
raise ValueError("입력 파일 경로는 B03_FileInput/input 아래여야 합니다.")
async with connection.cursor() as cursor:
await cursor.execute(
"""
INSERT INTO input_files (
project_id,
file_type,
original_filename,
raw_file_path,
file_size_mb,
upload_by,
crs_epsg,
metadata,
status
)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, 'UPLOADED')
""",
(
str(project_id),
file_type,
original_filename,
normalized_path.as_posix(),
file_size_bytes / (1024 * 1024),
upload_by,
crs_epsg,
json.dumps(metadata, ensure_ascii=False),
),
)
input_file_id = cursor.lastrowid
if not input_file_id:
raise RuntimeError("input_files 레코드 생성 결과에 ID가 없습니다.")
return int(input_file_id)
async def get_project_input_readiness(
connection: aiomysql.Connection,
project_id: UUID,
) -> tuple[set[str], int | None]:
"""현재 업로드 파일 유형과 최신 포인트클라우드 입력 ID를 반환한다."""
async with connection.cursor(aiomysql.DictCursor) as cursor:
await cursor.execute(
"""
SELECT id, LOWER(file_type) AS file_type
FROM input_files
WHERE project_id = %s AND status IN ('UPLOADED', 'PROCESSED')
ORDER BY id DESC
""",
(str(project_id),),
)
rows = await cursor.fetchall()
file_types = {str(row["file_type"]) for row in rows if row.get("file_type")}
point_cloud_id = next(
(int(row["id"]) for row in rows if str(row.get("file_type") or "") in {"las", "laz"}),
None,
)
return file_types, point_cloud_id
async def get_project_storage_relative_path(
connection: aiomysql.Connection,
project_id: UUID,
) -> str:
"""프로젝트의 검증된 저장소 상대 경로를 조회한다."""
async with connection.cursor() as cursor:
await cursor.execute(
"""
SELECT storage_path
FROM projects
WHERE id = %s AND deleted_at IS NULL
""",
(str(project_id),),
)
row = await cursor.fetchone()
if not row or not row[0]:
raise LookupError("프로젝트 또는 프로젝트 저장 경로를 찾을 수 없습니다.")
normalized_path = PurePosixPath(str(row[0]).replace("\\", "/"))
if normalized_path.is_absolute() or ".." in normalized_path.parts:
raise ValueError("프로젝트 저장 경로는 안전한 상대 경로여야 합니다.")
return normalized_path.as_posix()
async def create_upload_session(
connection: aiomysql.Connection,
*,
session_id: str,
project_id: UUID,
original_filename: str,
file_size_bytes: int,
chunk_size_bytes: int,
total_chunks: int,
) -> None:
"""청크 업로드 세션을 생성하거나 동일 세션 ID를 갱신한다."""
async with connection.cursor() as cursor:
await cursor.execute(
"""
INSERT INTO upload_sessions (
id,
project_id,
original_filename,
file_size_bytes,
chunk_size_bytes,
total_chunks,
completed_chunks,
status,
created_at,
updated_at
)
VALUES (%s, %s, %s, %s, %s, %s, 0, 'in_progress', NOW(), NOW())
ON DUPLICATE KEY UPDATE
original_filename = VALUES(original_filename),
file_size_bytes = VALUES(file_size_bytes),
chunk_size_bytes = VALUES(chunk_size_bytes),
total_chunks = VALUES(total_chunks),
status = 'in_progress',
updated_at = NOW()
""",
(
session_id,
str(project_id),
original_filename,
file_size_bytes,
chunk_size_bytes,
total_chunks,
),
)
async def get_upload_session(
connection: aiomysql.Connection,
*,
project_id: UUID,
session_id: str,
) -> dict[str, Any]:
"""청크 업로드 세션을 조회한다."""
async with connection.cursor(aiomysql.DictCursor) as cursor:
await cursor.execute(
"""
SELECT
id,
project_id,
original_filename,
file_size_bytes,
chunk_size_bytes,
total_chunks,
completed_chunks,
status
FROM upload_sessions
WHERE id = %s AND project_id = %s
""",
(session_id, str(project_id)),
)
row = await cursor.fetchone()
if not row:
raise LookupError("업로드 세션을 찾을 수 없습니다.")
return dict(row)
async def upsert_upload_chunk(
connection: aiomysql.Connection,
*,
session_id: str,
chunk_index: int,
chunk_hash: str,
size_bytes: int,
stored_at: str,
) -> int:
"""청크 저장 정보를 기록하고 완료 청크 수를 반환한다."""
async with connection.cursor() as cursor:
await cursor.execute(
"""
INSERT INTO upload_chunks (
session_id,
chunk_index,
chunk_hash,
size_bytes,
stored_at,
completed_at
)
VALUES (%s, %s, %s, %s, %s, NOW())
ON DUPLICATE KEY UPDATE
chunk_hash = VALUES(chunk_hash),
size_bytes = VALUES(size_bytes),
stored_at = VALUES(stored_at),
completed_at = NOW()
""",
(session_id, chunk_index, chunk_hash, size_bytes, stored_at),
)
await cursor.execute(
"""
SELECT COUNT(*)
FROM upload_chunks
WHERE session_id = %s
""",
(session_id,),
)
row = await cursor.fetchone()
completed_chunks = int(row[0]) if row else 0
await cursor.execute(
"""
UPDATE upload_sessions
SET completed_chunks = %s, updated_at = NOW()
WHERE id = %s
""",
(completed_chunks, session_id),
)
return completed_chunks
async def list_completed_chunk_indexes(
connection: aiomysql.Connection,
*,
session_id: str,
) -> list[int]:
"""완료된 청크 인덱스 목록을 반환한다."""
async with connection.cursor() as cursor:
await cursor.execute(
"""
SELECT chunk_index
FROM upload_chunks
WHERE session_id = %s
ORDER BY chunk_index ASC
""",
(session_id,),
)
rows = await cursor.fetchall()
return [int(row[0]) for row in rows]
async def mark_upload_session_completed(
connection: aiomysql.Connection,
*,
session_id: str,
) -> None:
"""업로드 세션을 완료 상태로 표시한다."""
async with connection.cursor() as cursor:
await cursor.execute(
"""
UPDATE upload_sessions
SET status = 'completed', updated_at = NOW()
WHERE id = %s
""",
(session_id,),
)
async def mark_upload_session_failed(
connection: aiomysql.Connection,
*,
session_id: str,
) -> None:
"""업로드 세션을 실패 상태로 표시한다."""
async with connection.cursor() as cursor:
await cursor.execute(
"""
UPDATE upload_sessions
SET status = 'failed', updated_at = NOW()
WHERE id = %s
""",
(session_id,),
)
async def list_project_input_files(
connection: aiomysql.Connection,
project_id: UUID,
) -> list[dict[str, Any]]:
"""업로드 완료된 입력 파일 목록(재접속 현황 표시용) — 서버가 정본이다.
같은 파일명을 다시 올리면 새 레코드가 쌓이므로 파일명별 최신 것만 남긴다.
"""
async with connection.cursor(aiomysql.DictCursor) as cursor:
await cursor.execute(
"""
SELECT f.id, f.file_type, f.original_filename, f.file_size_mb, f.status,
f.upload_at
FROM input_files f
INNER JOIN (
SELECT MAX(id) AS id
FROM input_files
WHERE project_id = %s AND status IN ('UPLOADED', 'PROCESSED')
GROUP BY original_filename
) latest ON latest.id = f.id
ORDER BY f.id ASC
""",
(str(project_id),),
)
rows = await cursor.fetchall()
return [dict(row) for row in rows]
async def list_incomplete_upload_sessions(
connection: aiomysql.Connection,
project_id: UUID,
) -> list[dict[str, Any]]:
"""중단된(미완료) 청크 업로드 세션 목록 — 재접속 시 이어올리기 안내용."""
async with connection.cursor(aiomysql.DictCursor) as cursor:
await cursor.execute(
"""
SELECT id, original_filename, file_size_bytes, chunk_size_bytes,
total_chunks, completed_chunks, updated_at
FROM upload_sessions
WHERE project_id = %s AND status = 'in_progress'
ORDER BY updated_at DESC
""",
(str(project_id),),
)
rows = await cursor.fetchall()
return [dict(row) for row in rows]
async def find_input_file_by_name(
connection: aiomysql.Connection,
project_id: UUID,
original_filename: str,
) -> dict[str, Any] | None:
"""같은 이름으로 등록된 최신 입력 파일 1건. 없으면 None.
같은 파일을 다시 올렸는지 가리는 데 쓴다 — 중복 판정 기준은 파일명이고, 내용이 같은지는
이 행의 메타데이터에 적힌 지문으로 본다(2026-08-08 사용자 결정).
"""
async with connection.cursor(aiomysql.DictCursor) as cursor:
await cursor.execute(
"""
SELECT id, original_filename, file_size_mb, metadata
FROM input_files
WHERE project_id = %s AND original_filename = %s
AND status IN ('UPLOADED', 'PROCESSED')
ORDER BY id DESC
LIMIT 1
""",
(str(project_id), original_filename),
)
row = await cursor.fetchone()
return dict(row) if row else None
async def supersede_previous_input_files(
connection: aiomysql.Connection,
project_id: UUID,
original_filename: str,
keep_input_file_id: int,
) -> int:
"""같은 이름의 옛 행을 `SUPERSEDED`로 내린다. 내린 건수를 돌려준다.
조회 쿼리들이 `UPLOADED`/`PROCESSED`만 보므로, 이렇게만 해도 목록·분석에서 빠진다.
행을 지우지 않는 이유는 언제 무엇이 교체됐는지 추적할 근거를 남기기 위해서다.
"""
async with connection.cursor() as cursor:
await cursor.execute(
"""
UPDATE input_files
SET status = 'SUPERSEDED'
WHERE project_id = %s AND original_filename = %s AND id <> %s
AND status IN ('UPLOADED', 'PROCESSED')
""",
(str(project_id), original_filename, keep_input_file_id),
)
return cursor.rowcount