feat: SAM 3.1 검출 커버리지 튜닝 — 무라벨 16% → 3.6% 2D 검출 단계에서 미검출을 없애는 작업. 지표는 "무라벨 화면%" — 마스크 합집합으로 재서 겹침을 뺀 값이다. 다각형 넓이 단순 합(105%)은 겹침 때문에 미검출을 못 잡아낸다. 원인은 프롬프트가 아니라 통짜 패스였다. wide_v1.txt 의 21개 프롬프트는 8192x5460 을 1x1 로 넣어 대상 하나에 인스턴스가 하나만 살아남았다. 도로는 한쪽 차로만, 논은 한 필지만 잡혔다. BlockYYX 실측 (사진 8장): 기준선 무라벨 평균 21.6% (69장 전체로는 16.1%, 최악 52.1%) 튜닝후 무라벨 평균 3.6% (최대 5.0%) 기여도 (0654 기준): 통짜 -> 타일 패스 -36.7pp conf 0.25 -> 0.10 -5.4pp 타일 4x3 -> 6x4 -4.6pp 프롬프트 추가 9개 -0.5pp <- 거의 기여 없음 비용은 140 -> 548초/장 (3.9배). sam3_multi_prompt.py: --wide-in-tiles 플래그 추가. 통짜 프롬프트를 타일 패스에서도 돌린다. 겹치는 결과는 NMS 가 지운다. 신규 도구: coverage_stats.py 마스크 합집합으로 무라벨% 측정 prompt_stats.py 카테고리-프롬프트별 검출 집계 make_merge_monitor.py 모니터 HTML 생성 (수치는 집계 JSON 에서만) vote_labels.py 에 --rescue-pred 추가 (투표에 졌지만 표가 있고 모델도 동의하는 면을 되돌린다). 3D 작업이 다른 담당으로 넘어가 중단된 상태. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> @
209 lines
9.2 KiB
Python
209 lines
9.2 KiB
Python
#!/usr/bin/env python
|
|
"""여러 사진의 SAM 마스크를 3D 면에 투표해 면별 라벨을 정한다.
|
|
|
|
면 하나가 여러 사진에 보인다 (가림을 뺀 실제 가시 뷰가 중앙 8장). 각 사진에서
|
|
그 면의 중심 픽셀에 어떤 라벨이 있는지 읽어 모으고 다수결한다.
|
|
|
|
득표가 갈리거나 뷰가 모자란 면은 강제로 정하지 않고 0(미결정)으로 남긴다.
|
|
틀린 라벨보다 없는 라벨이 낫다 — 학습 데이터에서 0은 ignore_index로 빠진다.
|
|
|
|
폴리곤을 라벨맵으로 구울 때 순서가 중요하다. 면적 큰 것부터 그려서 작은 것이
|
|
위에 오게 한다. 안 그러면 도로 폴리곤이 그 위의 차량을 덮어버린다.
|
|
|
|
한 면이 보이는데 어느 폴리곤에도 안 걸리는 경우가 있다 (SAM 커버리지가
|
|
장당 87%). 그건 기권으로 세지 어느 클래스 표도 아니다.
|
|
|
|
작은 것은 이 다수결에서 진다. 도로/건물 폴리곤이 크니 차량 위를 스치는 뷰가
|
|
몇 장만 있어도 득표율이 0.5 아래로 내려간다. 그래서 --rescue-pred 를 주면
|
|
"투표에는 졌지만 그 클래스 표가 있고 학습된 모델도 같은 클래스라고 하는 면"을
|
|
되돌린다. 두 증거가 독립이므로 정밀도를 거의 안 깎고 재현을 올린다.
|
|
"""
|
|
import argparse
|
|
import glob
|
|
import json
|
|
import os
|
|
import time
|
|
from collections import OrderedDict
|
|
|
|
import cv2
|
|
import numpy as np
|
|
|
|
|
|
def log(msg):
|
|
print(f"[{time.strftime('%H:%M:%S')}] {msg}", flush=True)
|
|
|
|
|
|
def poly_area(pts):
|
|
x, y = pts[:, 0], pts[:, 1]
|
|
return abs(np.dot(x, np.roll(y, -1)) - np.dot(y, np.roll(x, -1))) / 2
|
|
|
|
|
|
def build_label_map(seg_path, W, H, cls_id):
|
|
"""병합 JSON -> 라벨맵 (int16, 0=없음). 면적 내림차순으로 그린다."""
|
|
with open(seg_path, encoding="utf-8") as fh:
|
|
segs = json.load(fh)["segments"]
|
|
items = []
|
|
for s in segs:
|
|
pts = np.asarray(s["points"], dtype=np.float64).reshape(-1, 2)
|
|
if len(pts) < 3:
|
|
continue
|
|
lab = s["label"]
|
|
if lab not in cls_id:
|
|
continue
|
|
items.append((poly_area(pts), cls_id[lab], pts))
|
|
items.sort(key=lambda t: -t[0])
|
|
lm = np.zeros((H, W), dtype=np.int16)
|
|
for _, cid, pts in items:
|
|
cv2.fillPoly(lm, [np.round(pts).astype(np.int32)], int(cid))
|
|
return lm, len(items)
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser(description=__doc__,
|
|
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
ap.add_argument("--vis-dir", required=True, help="rasterize_visibility.py 출력")
|
|
ap.add_argument("--seg-dir", required=True, help="*_merged.json 폴더")
|
|
ap.add_argument("--out-dir", required=True)
|
|
ap.add_argument("--min-votes", type=int, default=2,
|
|
help="이보다 표가 적으면 미결정")
|
|
ap.add_argument("--min-ratio", type=float, default=0.5,
|
|
help="1위 득표율이 이보다 낮으면 미결정")
|
|
ap.add_argument("--rescue-pred",
|
|
help="PointVector 예측 npy (면 순서 동일). 주면 구제 규칙을 켠다")
|
|
ap.add_argument("--rescue-pred-classes",
|
|
help="--rescue-pred 의 클래스 목록 JSON (classes 키 또는 배열)")
|
|
ap.add_argument("--rescue-class", default="vehicle", help="구제할 클래스 이름")
|
|
ap.add_argument("--rescue-min-votes", type=int, default=1,
|
|
help="구제하려면 그 클래스 표가 최소 몇 개여야 하는지")
|
|
args = ap.parse_args()
|
|
if args.rescue_pred and not args.rescue_pred_classes:
|
|
ap.error("--rescue-pred 를 주면 --rescue-pred-classes 도 필요하다")
|
|
|
|
os.makedirs(args.out_dir, exist_ok=True)
|
|
with open(os.path.join(args.vis_dir, "visibility.json"), encoding="utf-8") as fh:
|
|
vmeta = json.load(fh)
|
|
n_face = vmeta["tile_faces"]
|
|
W, H = vmeta["image_size"]
|
|
names = [r["image"] for r in vmeta["per_camera"]]
|
|
|
|
vis = np.unpackbits(np.load(os.path.join(args.vis_dir, "visible.npy")),
|
|
axis=1, count=n_face).astype(bool)
|
|
pu = np.load(os.path.join(args.vis_dir, "pix_u.npy"))
|
|
pv = np.load(os.path.join(args.vis_dir, "pix_v.npy"))
|
|
log(f"면 {n_face:,} · 카메라 {len(names)} · 영상 {W}x{H}")
|
|
|
|
# 클래스 목록은 실제로 등장한 라벨에서 만든다. 지어내지 않는다.
|
|
labels = OrderedDict()
|
|
seg_files = {}
|
|
for n in names:
|
|
p = os.path.join(args.seg_dir, os.path.splitext(n)[0] + "_multi_merged.json")
|
|
if not os.path.isfile(p):
|
|
raise FileNotFoundError(f"병합 JSON 없음: {p}")
|
|
seg_files[n] = p
|
|
with open(p, encoding="utf-8") as fh:
|
|
for s in json.load(fh)["segments"]:
|
|
labels.setdefault(s["label"], 0)
|
|
labels[s["label"]] += 1
|
|
cls_names = ["미결정"] + sorted(labels, key=lambda k: -labels[k])
|
|
cls_id = {c: i for i, c in enumerate(cls_names) if i > 0}
|
|
n_cls = len(cls_names)
|
|
log(f"클래스 {n_cls} (0=미결정 포함): {', '.join(cls_names[1:])}")
|
|
|
|
votes = np.zeros((n_face, n_cls), dtype=np.int16) # 0열은 안 쓴다
|
|
seen = np.zeros(n_face, dtype=np.int16)
|
|
t0 = time.time()
|
|
for i, n in enumerate(names):
|
|
lm, npoly = build_label_map(seg_files[n], W, H, cls_id)
|
|
m = vis[i]
|
|
idx = np.flatnonzero(m)
|
|
lab = lm[pv[i][idx], pu[i][idx]]
|
|
seen[idx] += 1
|
|
hit = lab > 0
|
|
np.add.at(votes, (idx[hit], lab[hit]), 1)
|
|
log(f" [{i + 1}/{len(names)}] {n} 폴리곤 {npoly} · 가시면 {len(idx):,} "
|
|
f"· 라벨 적중 {int(hit.sum()):,} ({hit.mean():.1%})")
|
|
del lm
|
|
|
|
votes[:, 0] = 0
|
|
total = votes.sum(axis=1)
|
|
best = votes.argmax(axis=1).astype(np.int16)
|
|
top = votes.max(axis=1)
|
|
with np.errstate(divide="ignore", invalid="ignore"):
|
|
ratio = np.where(total > 0, top / np.maximum(total, 1), 0.0)
|
|
|
|
pred = np.where((total >= args.min_votes) & (ratio >= args.min_ratio),
|
|
best, 0).astype(np.int16)
|
|
|
|
rescue = None
|
|
if args.rescue_pred:
|
|
if args.rescue_class not in cls_id:
|
|
raise ValueError(f"투표 클래스에 없다: {args.rescue_class}")
|
|
with open(args.rescue_pred_classes, encoding="utf-8") as fh:
|
|
j = json.load(fh)
|
|
pcls = j["classes"] if isinstance(j, dict) else j
|
|
if args.rescue_class not in pcls:
|
|
raise ValueError(f"예측 클래스에 없다: {args.rescue_class}")
|
|
mp = np.load(args.rescue_pred)
|
|
if len(mp) != n_face:
|
|
raise ValueError(f"예측 {len(mp):,} != 면 {n_face:,}")
|
|
rid = cls_id[args.rescue_class]
|
|
take = ((pred != rid)
|
|
& (votes[:, rid] >= args.rescue_min_votes)
|
|
& (mp == pcls.index(args.rescue_class)))
|
|
was = pred[take].copy()
|
|
pred[take] = rid
|
|
u, c = np.unique(was, return_counts=True)
|
|
rescue = {cls_names[k]: int(v) for k, v in sorted(zip(u, c), key=lambda t: -t[1])}
|
|
log("")
|
|
log(f"구제({args.rescue_class}, 표>={args.rescue_min_votes} & 모델 동의): "
|
|
f"{int(take.sum()):,}면")
|
|
for k, v in rescue.items():
|
|
log(f" {k} 에서 {v:,}")
|
|
|
|
log("")
|
|
log(f"투표 {time.time() - t0:.0f}초")
|
|
log(f"가시 뷰 0인 면 {int((seen == 0).sum()):,} ({(seen == 0).mean():.2%})")
|
|
log(f"보이지만 라벨 표가 0인 면 {int(((seen > 0) & (total == 0)).sum()):,} "
|
|
f"({((seen > 0) & (total == 0)).mean():.2%})")
|
|
log(f"표는 있으나 기준 미달 {int(((total > 0) & (pred == 0)).sum()):,} "
|
|
f"({((total > 0) & (pred == 0)).mean():.2%})")
|
|
log("")
|
|
cnt = np.bincount(pred, minlength=n_cls)
|
|
log(f"{'id':>3} {'클래스':<20} {'면':>10} {'비율':>7} {'평균득표율':>9}")
|
|
for c in np.argsort(-cnt):
|
|
if cnt[c] == 0:
|
|
continue
|
|
r = ratio[pred == c].mean() if c > 0 and cnt[c] else float("nan")
|
|
log(f"{c:>3} {cls_names[c]:<20} {cnt[c]:>10,} {cnt[c] / n_face:>6.2%} "
|
|
f"{'' if c == 0 else f'{r:>8.1%}'}")
|
|
|
|
np.save(os.path.join(args.out_dir, "face_label.npy"), pred)
|
|
np.save(os.path.join(args.out_dir, "face_votes.npy"), votes)
|
|
with open(os.path.join(args.out_dir, "vote.json"), "w", encoding="utf-8") as fh:
|
|
json.dump({
|
|
"vis_dir": os.path.abspath(args.vis_dir),
|
|
"seg_dir": os.path.abspath(args.seg_dir),
|
|
"faces": int(n_face),
|
|
"cameras": len(names),
|
|
"classes": cls_names,
|
|
"min_votes": args.min_votes,
|
|
"min_ratio": args.min_ratio,
|
|
"faces_no_view": int((seen == 0).sum()),
|
|
"faces_no_label_vote": int(((seen > 0) & (total == 0)).sum()),
|
|
"faces_below_threshold": int(((total > 0) & (pred == 0)).sum()),
|
|
"counts": {cls_names[c]: int(cnt[c]) for c in range(n_cls) if cnt[c]},
|
|
"rescue": None if rescue is None else {
|
|
"pred": os.path.abspath(args.rescue_pred),
|
|
"cls": args.rescue_class,
|
|
"min_votes": args.rescue_min_votes,
|
|
"faces": int(sum(rescue.values())),
|
|
"from": rescue,
|
|
},
|
|
"note": "face_label.npy는 타일 면 인덱스 순서. 0 = 미결정 = ignore_index",
|
|
}, fh, indent=2, ensure_ascii=False)
|
|
log(f"기록: {args.out_dir}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|