"""Pipeline Step 8 — V2: V1 Top-K semantic rerank 파이프라인 위치: V1 (pipeline_06_2_mdx_matching.py) → **V2 (이 스크립트)** → V3 → V4 설계 원칙: - V1 결과(mdx_matching_result.yaml) 는 **고정 baseline**. 건드리지 않음. - V1 rank_by_matching_score Top-K 만 rerank 대상 — K=5 (기준 잠금). - 가중합 아님. **순수 rerank** — V1 점수는 저장만, 재정렬 키는 의미 유사도뿐. - AI 판단 없음. ko-sroberta 임베딩은 결정론적 (seed 고정 가정). 입력: - mdx_matching_result.yaml (V1) - MDX_SECTIONS 원문 (pipeline_01 의 MDX_DIR + section config) - Figma frame content (phase_common.load_32_frames() → analysis.md '내용') 처리: 1. 각 MDX 섹션(TARGET 4 + Holdout 3)별로: - V1 Top-5 frame_id 추출 - MDX summary = detect_mdx.build_summary (title + 첫 문단 + slot labels) - Top-5 frame.content ↔ summary 의 ko-sroberta cosine 계산 - cosine 내림차순으로 재정렬 2. lock_snapshot 기록 (V1 yaml + 이 스크립트 + 의존 모듈 sha256) 출력: - v2_semantic_rerank_result.yaml """ import hashlib import sys import datetime from pathlib import Path import yaml HERE = Path(__file__).parent sys.path.insert(0, str(HERE)) from phase_common import load_32_frames, load_frame_index from detect_mdx import detect_mdx_analysis from embeddings import embed_texts, cosine from pipeline_01_extract_nodes import MDX_SECTIONS, MDX_DIR TOP_K = 5 MODEL_ID = 'jhgan/ko-sroberta-multitask' # ============================================================ # 상수 — lock_snapshot 대상 # ============================================================ LOCK_SNAPSHOT_FILES = [ 'pipeline_08_v2_semantic_rerank.py', 'embeddings.py', 'detect_mdx.py', 'phase_common.py', 'pipeline_01_extract_nodes.py', 'synonyms.yaml', 'mdx_matching_result.yaml', ] def sha256_file(path: Path) -> str: return hashlib.sha256(path.read_bytes()).hexdigest() def extract_mdx_raw_section(section_id: str) -> tuple[str, str]: """(title, body_text) 반환. title = 첫 줄 heading 마커 제거, body_text = 원문 그대로.""" cfg = MDX_SECTIONS[section_id] p = MDX_DIR / cfg['file'] lines = p.read_text(encoding='utf-8').split('\n') start_idx = None for i, ln in enumerate(lines): if ln.strip() == cfg['start'].strip(): start_idx = i break if start_idx is None: raise ValueError(f"start_heading 찾지 못함: {cfg['start']!r} in {p}") end_idx = len(lines) if cfg.get('end_prefix'): for i in range(start_idx + 1, len(lines)): if lines[i].strip().startswith(cfg['end_prefix']): end_idx = i break section_lines = lines[start_idx:end_idx] raw = '\n'.join(section_lines) # title = heading 마커(#, ##, ###) 제거한 첫 줄 title_line = section_lines[0].lstrip('#').strip() return title_line, raw def main(): # 1. V1 결과 로드 v1_path = HERE / 'mdx_matching_result.yaml' v1 = yaml.safe_load(v1_path.read_text(encoding='utf-8')) # 2. Figma frame 로드 + content 임베딩 (한 번만) frames = load_32_frames() idx_data, frame_to_short = load_frame_index() fids = list(frames.keys()) frame_contents = [frames[fid].get('content', '') for fid in fids] fid_to_idx = {fid: i for i, fid in enumerate(fids)} print(f"[V2] ko-sroberta 모델 로드 + 32 frame content 임베딩...") frame_vecs = embed_texts(frame_contents) # 3. 섹션별 처리 out_sections = {} for sid, sec in v1['mdx_sections'].items(): top = sec['rank_by_matching_score'][:TOP_K] top_fids = [r['frame_id'] for r in top] # MDX summary 생성 title, raw_text = extract_mdx_raw_section(sid) analysis = detect_mdx_analysis(raw_text, title, anchor_vocab=None) summary = analysis['summary'] # summary 임베딩 mdx_vec = embed_texts([summary])[0] # Top-K 에 대해서만 cosine 계산 rerank = [] for v1_rank_idx, r in enumerate(top, start=1): fid = r['frame_id'] frame_idx = fid_to_idx[fid] sem = cosine(mdx_vec, frame_vecs[frame_idx]) rerank.append({ 'frame_id': fid, 'frame_number': r['frame_number'], 'v1_rank': v1_rank_idx, 'v1_score': r['matching_score'], 'semantic_score': round(float(sem), 4), }) # semantic_score 내림차순 재정렬 rerank.sort(key=lambda x: -x['semantic_score']) for new_rank, item in enumerate(rerank, start=1): item['v2_rank'] = new_rank out_sections[sid] = { 'section_type': sec.get('section_type'), 'answer_frame_number': sec.get('answer_frame_number'), 'mdx_title': title, 'mdx_summary': summary, 'top_k': TOP_K, 'v1_top_k': [ { 'rank': i + 1, 'frame_id': r['frame_id'], 'frame_number': r['frame_number'], 'v1_score': r['matching_score'], } for i, r in enumerate(top) ], 'v2_rerank': rerank, } # 4. lock_snapshot lock = { 'timestamp': datetime.datetime.now().isoformat(timespec='seconds'), 'model': MODEL_ID, 'top_k': TOP_K, 'files': { name: sha256_file(HERE / name) for name in LOCK_SNAPSHOT_FILES }, 'principle': [ 'V1 baseline 고정 — 재정렬 대상만 Top-K 로 한정', '가중합 금지 — 순수 semantic rerank', 'Holdout 성적을 설계 근거로 사용하지 않음', ], } out = { 'meta': { 'pipeline_step': '8.v2', 'description': 'V1 Top-K 후보를 ko-sroberta cosine 으로 재정렬 (캐스케이드 rerank 1단)', 'model': MODEL_ID, 'similarity': 'cosine', 'top_k': TOP_K, 'mdx_summary_spec': 'title + 첫 일반 문단(≤120자) + slot label Top-5 join (detect_mdx.build_summary)', 'frame_source': 'analysis.md "내용" 섹션 (phase_common.load_32_frames)', 'v1_source': 'mdx_matching_result.yaml rank_by_matching_score', 'holdout_sections': v1['meta']['holdout_sections'], 'answer_map': v1['meta']['answer_map'], 'lock_snapshot': lock, }, 'mdx_sections': out_sections, } out_path = HERE / 'v2_semantic_rerank_result.yaml' out_path.write_text( yaml.safe_dump(out, allow_unicode=True, sort_keys=False, width=1000), encoding='utf-8', ) # 콘솔 요약 print() print("=" * 70) print(f"V2 재정렬 완료: {out_path}") print("=" * 70) answer_map = v1['meta']['answer_map'] for sid, s in out_sections.items(): answer_num = answer_map.get(sid) v1_top1 = s['v1_top_k'][0]['frame_number'] v2_top1 = s['v2_rerank'][0]['frame_number'] v2_top1_v1rank = s['v2_rerank'][0]['v1_rank'] mark = '' if answer_num is not None: v1_ok = '✓' if v1_top1 == answer_num else '✗' v2_ok = '✓' if v2_top1 == answer_num else '✗' mark = f" 정답={answer_num} V1{v1_ok} V2{v2_ok}" else: mark = f" (holdout)" print(f" [{sid:8}] V1 top1=Frame {v1_top1:>2} → V2 top1=Frame {v2_top1:>2} (V1 rank {v2_top1_v1rank}){mark}") print() if __name__ == '__main__': main()