Files
C.E.L_Slide_test2/tests/matching/phase_common.py
T
KyeongminandClaude Opus 4.8 b836e79ee1 wip: phase_z2 evidence 파이프라인 + matching 실험(phase2~26) + 프론트 trace 패널 진행분 스냅샷
- src: phase_z2 composition/mapper/pipeline/placement_planner/retry, ai_fallback(prompts/schema/validate), mdx_text_atoms 신규
- Front: PipelineTracePanel 신규, FramePanel/SlideCanvas/Home/designAgentApi 등 갱신 + 테스트 4종 추가
- templates/phase_z2: catalog(component_expansion_registry, node_slot_mapping 신규), frames, families, slide_base 갱신
- tests/matching: phase2~26 매칭 실험 스크립트·리포트·온톨로지 전체 (미커밋 진행분)
- tests: b4_v4 evidence, task5~28.5 시리즈, regression(imp95 baseline) 등 신규 테스트 대량 추가
- docs/reference: MDX 구조 인벤토리, MDX→Frame 구조 계약 문서
- scripts: mdx 계약/parity/coverage/viewport 체크, gitea comment, run sync 유틸
- .gitignore: tmp*.json, chromedriver, .orchestrator, *.pkl, Front_test* 등 임시/스냅샷 제외

미완성 작업의 보존용 스냅샷 커밋 (2026-07-02)

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-02 17:03:42 +09:00

327 lines
13 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Phase 18~22 공통 모듈
- analysis.md 로드
- 동의어 사전 로드 + normalize
- 키워드 추출 / df 분류
- 점수 계산 (옵션별 축 포함/제외)
"""
import re
import math
import collections
import json
from pathlib import Path
import yaml
from methods import _get_kiwi, _extract_content_tokens, _detect_mdx_layout, _get_cross_encoder
ROOT = Path(r"d:\ad-hoc\kei\design_agent")
BLOCKS_DIR = ROOT / "figma_to_html_agent" / "blocks"
PREVIEW_DIR = ROOT / "data" / "figma_previews"
# ═══ 4개 테스트 유닛 ═══
TARGET_UNITS = [
("MDX01-2-details", "1. (MDX 1) 팝업 — DX와 BIM의 구분", "18", "01.mdx", "2. 용어간 상호관계", None),
("MDX02-2.2-table", "2. (MDX 2) 2.2 DX 시행 주체별 기대효과", "14", "02.mdx", "2. DX 기반 Process 혁신에 따른 주체별 기대효과", "2.2 DX 시행 주체별 기대효과"),
("MDX03-1", "3. (MDX 03) 1. DX 시행을 위한 필수요건", "13", "03.mdx", "1. DX 시행을 위한 필수 요건", None),
("MDX03-2", "4. (MDX 03) 2. Process 혁신과 Product 변화", "29", "03.mdx", "2. Process의 혁신과 Product의 변화", None),
]
def load_synonyms():
p = Path(__file__).parent / "synonyms.yaml"
with open(p, encoding="utf-8") as f:
return yaml.safe_load(f)["synonyms"]
def normalize_with_synonyms(text, synonyms):
"""variants를 canonical 표기로 치환. 긴 variant 먼저."""
replacements = []
for canonical, variants in synonyms.items():
for v in variants:
replacements.append((len(v), v, canonical))
replacements.sort(reverse=True) # 긴 것부터
for _, variant, canonical in replacements:
text = text.replace(variant, canonical)
return text
def load_keyword_base():
"""keyword_base.yaml 로드. normalize=false 는 MDX 치환 대상에서 제외.
반환: {canonical: [variants]} - synonyms.yaml 구조와 호환."""
p = Path(__file__).parent / "keyword_base.yaml"
if not p.exists():
return {}
with open(p, encoding="utf-8") as f:
data = yaml.safe_load(f)
out = {}
for canonical, entry in data.get('keywords', {}).items():
if entry.get('normalize', True) is False:
continue # anchor-only, MDX 치환 비활성
variants = entry.get('variants', [])
if variants:
out[canonical] = variants
# variants 없으면 치환할 게 없으니 dict 에 추가 안 함
return out
def normalize_with_keyword_base(text, kb):
"""keyword_base 로 MDX 정규화. synonyms 와 동일 로직."""
return normalize_with_synonyms(text, kb)
def parse_analysis_md(path):
"""analysis.md 파싱. Legacy ('구조', '내용') + Milestone 2 mirror ('구조 매칭 정보', '내용 설명') 둘 다 지원."""
text = path.read_text(encoding="utf-8")
result = {"keywords": [], "layout": "", "detail": "", "content": ""}
sections = re.split(r"\n## ", text)
for sec in sections[1:]:
lines = sec.split("\n", 1)
heading = lines[0].strip()
body = lines[1].strip() if len(lines) > 1 else ""
if heading in ("구조", "구조 매칭 정보"):
for ln in body.split("\n"):
m = re.match(r"- \*\*layout\*\*:\s*(.+)", ln)
if m: result["layout"] = m.group(1).strip()
m = re.match(r"- \*\*detail\*\*:\s*(.+)", ln)
if m: result["detail"] = m.group(1).strip()
elif heading in ("내용", "내용 설명"):
result["content"] = body.split("\n\n")[0].strip()
elif heading == "후보 키워드":
raw = body.split("\n\n")[0]
result["keywords"] = [k.strip() for k in raw.split(",") if k.strip()]
return result
def load_32_frames():
frames = {}
for d in sorted(BLOCKS_DIR.iterdir()):
if not d.is_dir(): continue
fid = d.name
if not fid.startswith("1171"): continue
a = d / "analysis.md"
if a.exists():
frames[fid] = parse_analysis_md(a)
return frames
def compute_df_idf_tier(frames):
df = collections.Counter()
N = len(frames)
for v in frames.values():
for kw in set(v["keywords"]):
df[kw] += 1
tier = {}
for kw, cnt in df.items():
ratio = cnt / N
if ratio <= 0.10: tier[kw] = "core"
elif ratio >= 0.30: tier[kw] = "general"
else: tier[kw] = "mid"
idf = {kw: math.log(N / cnt) if cnt > 0 else 0 for kw, cnt in df.items()}
return df, idf, tier, N
def extract_mdx_keywords(mdx_text, vocabulary, synonyms=None, keyword_base=None):
"""MDX 텍스트 → 정규화 → direct canonical hit + Kiwi token hit → union.
보강 이유: compound canonical (설계Data, 공사비절감, 3D모델 등) 은
Kiwi 가 분해하여 원형이 tokens 에 없음. vocabulary 에서 직접 substring 매칭
으로 별도 검사 → Kiwi hit 와 union.
Args:
mdx_text: 원본 MDX 텍스트
vocabulary: Figma frame 의 canonical keyword set (비교 대상)
synonyms: legacy synonyms.yaml 사전 (하위호환)
keyword_base: keyword_base.yaml 로드 결과 (우선 적용)
"""
if keyword_base is not None:
mdx_text = normalize_with_keyword_base(mdx_text, keyword_base)
elif synonyms is not None:
mdx_text = normalize_with_synonyms(mdx_text, synonyms)
# ① direct canonical hit (compound 보호)
direct = {t for t in vocabulary if len(t) >= 2 and t in mdx_text}
# ② Kiwi token hit
kiwi = _get_kiwi()
tokens = _extract_content_tokens(mdx_text, kiwi)
kiwi_hit = set(tokens) & vocabulary
return direct | kiwi_hit
def keyword_score(mdx_kws, fig_kws, idf, tier):
"""IDF × tier 가중 Jaccard"""
tier_weight = {"core": 1.0, "mid": 0.5, "general": 0.1}
inter = mdx_kws & fig_kws
union = mdx_kws | fig_kws
num = sum(idf.get(c, 0.5) * tier_weight.get(tier.get(c, "mid"), 0.5) for c in inter)
den = sum(idf.get(c, 0.5) * tier_weight.get(tier.get(c, "mid"), 0.5) for c in union)
return num / den if den > 0 else 0, inter
def content_scores_batch(mdx_content, frames):
"""Cross-encoder 일괄 채점 → {fid: [0,1]}"""
model = _get_cross_encoder()
fids = list(frames.keys())
pairs = [[mdx_content, frames[fid]["content"]] for fid in fids]
raw = model.predict(pairs, show_progress_bar=False)
return {fids[i]: 1 / (1 + math.exp(-float(raw[i]))) for i in range(len(fids))}
# ═══ 구조 매칭 v3 — 우선순위 수정 + 호환도 점수(graded) ═══
def detect_mdx_layout_v2(text):
"""MDX 본문 정밀 구조 감지.
우선순위: ### 서브섹션 > 표 > 블릿 (이전엔 표 먼저였음)"""
lines = text.split("\n")
# 1. ### 서브섹션 우선 (### X.Y 패턴)
subs = [ln for ln in lines if re.match(r"^###\s+\d+\.\d+", ln)]
if len(subs) == 2:
sub_text = " ".join(subs).lower()
if any(kw in sub_text for kw in ["과정", "결과", "process", "product", "as-is", "to-be"]):
return "compare-2banner"
return "compare-2col"
if len(subs) >= 3:
return "multi-section"
# 2. 표 감지 (서브섹션 없을 때)
table_header = None
for ln in lines:
stripped = ln.strip()
if re.match(r"^\|.*\|.*\|", stripped) and not re.match(r"^\|[\s\-:]+\|", stripped):
table_header = stripped
break
if table_header:
cols = [c.strip().replace("*", "").lower() for c in table_header.strip("|").split("|") if c.strip()]
col_text = " ".join(cols)
if any(kw in col_text for kw in ["발주자", "시공자", "설계자"]):
return "persona-3col"
if any(kw in col_text for kw in ["제조업", "건축", "토목"]) and "토목" in col_text:
return "table-3col"
if any(kw in col_text for kw in ["bim", "dx"]) and ("bim" in col_text and "dx" in col_text):
return "compare-rows"
n_cols = len(cols) - 1
if n_cols == 2: return "table-2col"
if n_cols >= 3: return "table-3col"
return "compare-rows"
# 3. 최상위 볼드 블릿
top_bullets = [ln for ln in lines if re.match(r"^[-*]\s+\*\*", ln)]
n = len(top_bullets)
if n == 3: return "3col-parallel"
if n == 2: return "compare-2col"
if n == 4: return "cards-4"
if n >= 5: return "multi-parallel"
return "single-column"
# ═══ 호환도 매트릭스 (compatibility) ═══
# MDX 구조 × Figma layout = 0.0~1.0
# "이 MDX 내용을 이 Figma 레이아웃으로 렌더링 시 얼마나 적합한가"
_COMPAT = {
# 2-column 계열 (서로 호환성 높음)
"compare-2col": {
"compare-2col": 1.0, "compare-2banner-top-2col-bottom": 0.9,
"table-2col": 0.9, "2col-paired": 0.85, "2-boxes": 0.75,
"central-split": 0.7, "paired-rows": 0.6,
"compare-rows": 0.55, "table-3col": 0.3, "persona-3col": 0.3,
"3col-parallel": 0.3, "single-column": 0.5,
},
"compare-2banner": {
"compare-2banner-top-2col-bottom": 1.0, "compare-2col": 0.9,
"table-2col": 0.85, "2col-paired": 0.8, "2-boxes": 0.7,
"paired-rows": 0.7, "central-split": 0.7,
"compare-rows": 0.5, "table-3col": 0.3, "persona-3col": 0.3,
"3col-parallel": 0.3, "single-column": 0.5,
},
"table-2col": {
"table-2col": 1.0, "compare-2col": 0.9,
"compare-2banner-top-2col-bottom": 0.85,
"compare-rows": 0.75, "2col-paired": 0.75,
"table-3col": 0.5, "persona-3col": 0.4, "3col-parallel": 0.3,
"single-column": 0.5,
},
# 3-column 계열
"persona-3col": {
"persona-3col": 1.0, "3col-parallel": 0.75,
"3col-cards": 0.75, "table-3col": 0.7, "3col-compare": 0.65,
"cards-4plus5": 0.4, "compare-rows": 0.35,
"compare-2col": 0.3, "table-2col": 0.4, "single-column": 0.5,
},
"3col-parallel": {
"3col-parallel": 1.0, "3col-cards": 0.9, "3col-compare": 0.9,
"persona-3col": 0.75, "table-3col": 0.6,
"3-emphasis": 0.6, "3-category": 0.65, "3-section": 0.65,
"cards-4": 0.5, "cards-4plus5": 0.4,
"compare-2col": 0.3, "single-column": 0.5,
},
"table-3col": {
"table-3col": 1.0, "3col-parallel": 0.65,
"3col-cards": 0.65, "persona-3col": 0.7,
"compare-rows": 0.7, "3col-compare": 0.75,
"table-2col": 0.5, "single-column": 0.45,
},
"compare-rows": {
"compare-rows": 1.0, "table-2col": 0.75,
"table-3col": 0.7, "paired-rows": 0.75,
"compare-2col": 0.55, "compare-2banner-top-2col-bottom": 0.5,
"2col-paired": 0.5, "single-column": 0.45,
},
# 4+ / 다중
"cards-4": {
"cards-4": 1.0, "cards-4plus5": 0.9,
"policy-4card-plus-list": 0.85, "quadrant-issues": 0.85,
"3col-parallel": 0.5, "table-3col": 0.4, "single-column": 0.4,
},
"multi-parallel": {
"cards-4": 0.85, "cards-4plus5": 0.9,
"policy-4card-plus-list": 0.8, "quadrant-issues": 0.75,
"3col-parallel": 0.55, "persona-3col": 0.45,
"table-3col": 0.4, "single-column": 0.4,
},
"multi-section": {
"3-section": 1.0, "3-category": 0.9, "3-emphasis": 0.85,
"3col-parallel": 0.6, "cards-4": 0.5, "single-column": 0.5,
},
# 단일
"single-column": {
"bullet-cards": 0.85, "list-numbered": 0.85, "list-stacked": 0.85,
"side-card": 0.75, "split-panel-diagram": 0.65,
"split-panel-numbered": 0.65, "central-split": 0.55,
"diagram-labels": 0.55, "diagram-5": 0.55, "central-5-goals": 0.5,
"circular-nodes": 0.5, "cycle-3way": 0.5,
"intro": 0.6, "definition-list": 0.7,
"compare-2col": 0.5, "compare-rows": 0.5, "3col-parallel": 0.5,
},
}
def structural_match_v2(mdx_layout, fig_layout):
"""구조 호환도 매칭 (0~1 graded). 정확일치=1.0, 유사=0.5~0.9, 무관=0.1~0.3."""
if not mdx_layout or not fig_layout: return 0.0
if mdx_layout == fig_layout: return 1.0
row = _COMPAT.get(mdx_layout, {})
return row.get(fig_layout, 0.15) # 명시 안 됐으면 매우 약한 기본 호환도
# ═══ 유닛 로드 ═══
def load_target_units():
"""4개 타겟 유닛의 MDX 본문 + 제목 hierarchy 반환"""
from extract_units import extract_units
from phase10 import extract_titles_only_mdx
units_full = extract_units()
units_title = {}
for uid, _, _, fname, mid, sub in TARGET_UNITS:
units_title[uid] = extract_titles_only_mdx(fname, mid, sub)
return units_full, units_title
def load_frame_index():
with open(PREVIEW_DIR / "index.json", encoding="utf-8") as f:
idx_data = json.load(f)
frame_to_short = {info["frame_id"]: sid for sid, info in idx_data.items()}
return idx_data, frame_to_short