Files
KyeongminandClaude Opus 4.8 b836e79ee1 wip: phase_z2 evidence 파이프라인 + matching 실험(phase2~26) + 프론트 trace 패널 진행분 스냅샷
- src: phase_z2 composition/mapper/pipeline/placement_planner/retry, ai_fallback(prompts/schema/validate), mdx_text_atoms 신규
- Front: PipelineTracePanel 신규, FramePanel/SlideCanvas/Home/designAgentApi 등 갱신 + 테스트 4종 추가
- templates/phase_z2: catalog(component_expansion_registry, node_slot_mapping 신규), frames, families, slide_base 갱신
- tests/matching: phase2~26 매칭 실험 스크립트·리포트·온톨로지 전체 (미커밋 진행분)
- tests: b4_v4 evidence, task5~28.5 시리즈, regression(imp95 baseline) 등 신규 테스트 대량 추가
- docs/reference: MDX 구조 인벤토리, MDX→Frame 구조 계약 문서
- scripts: mdx 계약/parity/coverage/viewport 체크, gitea comment, run sync 유틸
- .gitignore: tmp*.json, chromedriver, .orchestrator, *.pkl, Front_test* 등 임시/스냅샷 제외

미완성 작업의 보존용 스냅샷 커밋 (2026-07-02)

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-02 17:03:42 +09:00

146 lines
5.6 KiB
Python

"""Pipeline Step 17 — V4 (template-fit) 를 V3 Top-K 가 아니라 32 프레임 전체에 적용.
기존 V4 r2 는 V3 Top-5 만 평가 → 5 개 중 선택.
이 스크립트는 32 프레임 전체에 template-fit 적용 → confidence 기준 전체 랭킹.
V1/V2/V3 처럼 "32 중 Top-3" 형태 산출.
출력: v4_full32_result.yaml
"""
import datetime
import sys
from pathlib import Path
import yaml
HERE = Path(__file__).parent
sys.path.insert(0, str(HERE))
from phase_common import load_32_frames, load_frame_index
from detect_mdx import detect_mdx_analysis
from template_fit import (
load_templates_v1, collect_anchor_vocab, compute_template_fit, route,
)
from embeddings import embed_texts, cosine
from pipeline_01_extract_nodes import MDX_SECTIONS, MDX_DIR
OUT_PATH = HERE / 'v4_full32_result.yaml'
def extract_mdx_raw(sid):
cfg = MDX_SECTIONS[sid]
p = MDX_DIR / cfg['file']
lines = p.read_text(encoding='utf-8').split('\n')
start = None
for i, ln in enumerate(lines):
if ln.strip() == cfg['start'].strip():
start = i
break
end = len(lines)
if cfg.get('end_prefix'):
for i in range(start + 1, len(lines)):
if lines[i].strip().startswith(cfg['end_prefix']):
end = i
break
section = lines[start:end]
return section[0].lstrip('#').strip(), '\n'.join(section)
def main():
v1 = yaml.safe_load((HERE / 'mdx_matching_result.yaml').read_text(encoding='utf-8'))
answer_map = v1['meta']['answer_map']
holdout = v1['meta']['holdout_sections']
templates = load_templates_v1()
anchor_vocab = collect_anchor_vocab(templates)
frames = load_32_frames()
idx_data, frame_to_short = load_frame_index()
fids = list(frames.keys())
frame_num_map = {fid: int(frame_to_short[fid]) for fid in fids}
frame_contents = [frames[fid].get('content', '') for fid in fids]
print('[V4 full-32] 32 frame content 임베딩 중...')
frame_vecs = embed_texts(frame_contents)
out_sections = {}
for sid in v1['mdx_sections']:
title, raw_text = extract_mdx_raw(sid)
mdx_analysis = detect_mdx_analysis(raw_text, title, anchor_vocab=anchor_vocab)
mdx_summary = mdx_analysis['summary']
mdx_vec = embed_texts([mdx_summary])[0]
judgments = []
for i, fid in enumerate(fids):
if fid not in templates:
continue
template = templates[fid]
content_emb = max(0.0, min(1.0, float(cosine(mdx_vec, frame_vecs[i]))))
fit = compute_template_fit(mdx_analysis, template, content_emb)
label = route(fit['confidence'], fit['axes'], fit['adaptation'], fit['not_suits'])
judgments.append({
'frame_id': fid,
'frame_number': frame_num_map[fid],
'template_id': template.get('template_id'),
'confidence': round(float(fit['confidence']), 4),
'base': round(float(fit['base']), 4),
'penalty': round(float(fit['total_penalty']), 4),
'label': label,
'content_embedding': round(content_emb, 4),
'axes': {
'anchor': round(float(fit['axes']['anchor']['score']), 4),
'cardinality': round(float(fit['axes']['cardinality']), 4),
'relation': round(float(fit['axes']['relation']), 4),
'slot': round(float(fit['axes']['slot']), 4),
'content': round(float(fit['axes']['content']), 4),
},
})
# confidence 내림차순
judgments.sort(key=lambda x: -x['confidence'])
for new_rank, item in enumerate(judgments, start=1):
item['v4_full_rank'] = new_rank
ans = answer_map.get(sid)
out_sections[sid] = {
'mdx_title': title,
'answer_frame_number': ans,
'is_holdout': sid in holdout,
'judgments_full32': judgments,
'usable_count': sum(1 for j in judgments if j['label'] != 'reject'),
'reject_count': sum(1 for j in judgments if j['label'] == 'reject'),
}
out = {
'meta': {
'pipeline_step': '8.v4.full32',
'description': 'V4 template-fit 을 V3 Top-K 가 아니라 32 프레임 전체에 적용',
'note': '기존 v4_template_fit_r2_result.yaml 는 V3 Top-5 만 평가. 이 파일은 32 전체.',
'generated_at': datetime.datetime.now().isoformat(timespec='seconds'),
'answer_map': answer_map,
'holdout_sections': holdout,
},
'mdx_sections': out_sections,
}
OUT_PATH.write_text(
yaml.safe_dump(out, allow_unicode=True, sort_keys=False, width=1000),
encoding='utf-8',
)
print('=' * 70)
print(f'V4 full-32 평가 완료: {OUT_PATH}')
print('=' * 70)
print(f'\n각 섹션별 사용 가능 프레임 (label != reject) 개수 + Top-3:')
for sid, s in out_sections.items():
ans = s.get('answer_frame_number')
ans_str = f'(정답 {ans})' if ans else '(holdout)'
print(f'\n[{sid}] {ans_str} 사용가능 {s["usable_count"]}/{s["usable_count"]+s["reject_count"]}')
usable = [j for j in s['judgments_full32'] if j['label'] != 'reject']
for j in usable[:3]:
mark = ' 🎯' if ans and j['frame_number'] == ans else ''
print(f' Frame {j["frame_number"]:>2}{mark} conf {j["confidence"]:.3f} {j["label"]}')
if not usable:
print(' (사용 가능 후보 없음 — 모두 reject)')
if __name__ == '__main__':
main()