Files
C.E.L_Slide_test2/tests/pipeline/code/detect_mdx.py
T
KyeongminandClaude Opus 4.8 b836e79ee1 wip: phase_z2 evidence 파이프라인 + matching 실험(phase2~26) + 프론트 trace 패널 진행분 스냅샷
- src: phase_z2 composition/mapper/pipeline/placement_planner/retry, ai_fallback(prompts/schema/validate), mdx_text_atoms 신규
- Front: PipelineTracePanel 신규, FramePanel/SlideCanvas/Home/designAgentApi 등 갱신 + 테스트 4종 추가
- templates/phase_z2: catalog(component_expansion_registry, node_slot_mapping 신규), frames, families, slide_base 갱신
- tests/matching: phase2~26 매칭 실험 스크립트·리포트·온톨로지 전체 (미커밋 진행분)
- tests: b4_v4 evidence, task5~28.5 시리즈, regression(imp95 baseline) 등 신규 테스트 대량 추가
- docs/reference: MDX 구조 인벤토리, MDX→Frame 구조 계약 문서
- scripts: mdx 계약/parity/coverage/viewport 체크, gitea comment, run sync 유틸
- .gitignore: tmp*.json, chromedriver, .orchestrator, *.pkl, Front_test* 등 임시/스냅샷 제외

미완성 작업의 보존용 스냅샷 커밋 (2026-07-02)

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-02 17:03:42 +09:00

336 lines
13 KiB
Python

"""MDX 분석 — template-fit-v1 매칭용 코드 기반 추출.
Spec: tests/matching/TEMPLATE_FIT_V1.md §3
출력:
{
title, summary, detected_terms,
item_count: {detected, source},
relation_type: {value, confidence},
content_shape: {has_table, has_subsections, has_bullets},
slot_candidates: [{label, body}, ...]
}
원칙: LLM 호출 없음. 모든 필드는 규칙 기반 코드로 추출.
"""
import re
from phase_common import normalize_with_synonyms, load_synonyms
from methods import _get_kiwi
# ─── 패턴 ────────────────────────────────────────────
RE_TABLE_ROW = re.compile(r'^\|.*\|')
RE_TABLE_SEP = re.compile(r'^\|[\s\-:|]+\|')
RE_SUBSECTION = re.compile(r'^###\s+(\d+\.\d+(?:\.\d+)?)\s+(.+)')
RE_TOP_BULLET = re.compile(r'^[-*]\s+\*\*([^*]+)\*\*')
RE_HTML_TAG = re.compile(r'<[^>]+>')
AXIS_LABEL = '구분' # 표 헤더 중 이 라벨이 있으면 해당 컬럼은 axis 로 제외
NOUN_TAGS = {'NNG', 'NNP', 'NR', 'SL', 'SN', 'SH'}
# relation_type 보정 키워드
COMPARE_HINTS = ['과정', '결과', 'Process', 'Product', 'AS-IS', 'TO-BE',
'Analogue', 'Digital', 'vs', 'VS', '대비']
SEQUENCE_HINTS = ['단계', 'Step', '순서', '차례']
# ═══ 표 파싱 ═══
def parse_first_table(text):
"""첫 markdown 테이블 반환 (header_cells, body_rows). 없으면 None."""
lines = text.split('\n')
tbl_start = None
for i in range(len(lines) - 1):
if RE_TABLE_ROW.match(lines[i].strip()) and RE_TABLE_SEP.match(lines[i+1].strip()):
tbl_start = i
break
if tbl_start is None:
return None
rows = []
for j in range(tbl_start, len(lines)):
s = lines[j].strip()
if not s.startswith('|'):
break
rows.append(s)
if len(rows) < 2:
return None
def cells(line):
return [c.strip().replace('**', '').strip() for c in line.strip('|').split('|')]
header = cells(rows[0])
body = [cells(r) for r in rows[2:]]
return header, body
def find_subsections(text):
"""### X.Y TITLE 라인 목록 → [(number, title), ...]"""
out = []
for ln in text.split('\n'):
m = RE_SUBSECTION.match(ln.strip())
if m:
out.append((m.group(1), m.group(2).strip()))
return out
def find_top_bullets(text):
"""최상위 '* **LABEL**' 블릿 목록 → [full_line, ...]"""
return [ln for ln in text.split('\n') if RE_TOP_BULLET.match(ln)]
# ═══ 슬롯 후보 추출 ═══
def slots_from_table(header, body):
"""축(구분) 제외 컬럼 → label. 첫 데이터 행의 해당 cell → body."""
axis_idx = next((i for i, h in enumerate(header) if AXIS_LABEL in h), None)
first_row = body[0] if body else []
result = []
for i, h in enumerate(header):
if i == axis_idx or not h:
continue
body_cell = first_row[i] if i < len(first_row) else ''
body_clean = RE_HTML_TAG.sub(' ', body_cell).strip()[:120]
result.append({'label': h, 'body': body_clean})
return result
def slots_from_subsections(subs):
"""### TITLE → label (괄호 앞부분), body (전체 타이틀)."""
out = []
for num, title in subs:
main = re.match(r'^([^\(]+)', title)
label = main.group(1).strip() if main else title
out.append({'label': label, 'body': title})
return out
def slots_from_bullets(bullets):
out = []
for b in bullets:
m = RE_TOP_BULLET.match(b)
if not m:
continue
full_label = m.group(1).strip()
main = re.match(r'^([^\(]+)', full_label)
label = main.group(1).strip() if main else full_label
out.append({'label': label, 'body': full_label})
return out
# ═══ detected_terms ═══
def clean_for_kiwi(text):
"""Kiwi 입력 전 HTML/JSX/MDX 노이즈 제거.
구조 파싱에는 원본 사용, Kiwi 명사 추출에만 이 함수 사용."""
# HTML 태그 (open/close/self-closing)
text = re.sub(r'<[^>]+>', ' ', text)
# JSX/MDX 속성 블록 {...} (style, eventHandler 등)
text = re.sub(r'\{[^}]*\}', ' ', text)
# HTML entity
text = re.sub(r'&[a-zA-Z]+;', ' ', text)
# 단위/수치 노이즈 (10px, 0.9rem 등)
text = re.sub(r'\b\d+(?:\.\d+)?(?:px|rem|em|%)\b', ' ', text)
# 공백 정규화
text = re.sub(r'\s+', ' ', text)
return text
def extract_detected_terms(normalized_text, anchor_vocab):
"""Kiwi 명사/외국어/숫자 + anchor_vocab substring.
- HTML/JSX 제거 후 Kiwi noun 추출
- anchor_vocab substring 은 원본(normalized) 에 대해 수행 (compound term 커버)
"""
cleaned = clean_for_kiwi(normalized_text)
kiwi = _get_kiwi()
tokens = kiwi.tokenize(cleaned)
noun_tokens = [t.form for t in tokens
if t.tag in NOUN_TAGS and len(t.form) >= 2]
# anchor_vocab substring (compound terms — Kiwi 가 분해하는 단어 커버)
for term in (anchor_vocab or []):
if len(term) >= 2 and term in normalized_text:
if term not in noun_tokens:
noun_tokens.append(term)
seen = set()
out = []
for t in noun_tokens:
if t not in seen:
seen.add(t)
out.append(t)
return out
# ═══ summary ═══
def build_summary(title, text, slot_candidates):
"""title + 첫 일반 문단(있으면) + 슬롯 라벨 조인."""
parts = [title]
for ln in text.split('\n'):
s = ln.strip()
if not s: continue
if s.startswith(('|', '#', '<', '*', '-', ':', '{', '```', '---')):
continue
if re.match(r'^\d+\.\s', s) or re.match(r'^<br', s):
continue
parts.append(s[:120])
break
if slot_candidates:
parts.append(' / '.join(sc.get('label', '') for sc in slot_candidates[:5]))
return '. '.join(p for p in parts if p)
# ═══ relation_type ═══
def infer_relation_type(item_count, item_source, text, subsections):
# subsection 에서 과정/결과 등 compare hint 있으면 compare
if subsections:
joined = ' '.join(t for _, t in subsections)
if any(kw in joined for kw in COMPARE_HINTS):
return ('compare', 'high')
if any(kw in joined for kw in SEQUENCE_HINTS):
return ('sequence', 'high')
if item_count == 2:
return ('compare', 'high')
if item_count >= 3:
return ('parallel', 'high')
return ('definition', 'low')
# ═══ structure_intent 추론 ═══
def infer_structure_intent(text, title, detected_terms, item_count, relation_type):
"""MDX 의 structure_intent 추론 (list, 여러 개 가능).
설계 원칙:
- Layer 1 (subject-defined): process_product, transformation, persona, industry 등
명확한 subject signal. 이들은 MDX 의 primary intent.
- Layer 2 (mutually exclusive): concept_comparison 은 Layer 1 의 특정 intent
(process_product/transformation) 와 배타적. 둘 다 설정되면 재구성 모호해짐.
- Layer 3 (secondary): multi_attribute / category — 보조 intent.
- Polarity (title 기반): requirement / problem 은 **title** 이나 구체 단어로만
감지 (body noise 방지).
"""
intents = []
terms_set = set(detected_terms)
# Layer 1: subject-defined primary intents
# process_product: terms_set 또는 text 패턴
has_pp_pattern = (
('과정' in text and ('혁신' in text or 'Process' in text)) and
('결과' in text and ('변화' in text or 'Product' in text))
)
if ('과정혁신' in terms_set and '결과혁신' in terms_set) or has_pp_pattern:
intents.append('process_product_split')
if ('AS-IS' in terms_set and 'TO-BE' in terms_set) or \
('Analogue' in terms_set and 'Digital' in terms_set):
if 'transformation_story' not in intents:
intents.append('transformation_story')
if {'발주자', '시공자', '설계자'} <= terms_set:
intents.append('persona_benefit')
if sum(1 for t in ['제조업', '건축', '토목'] if t in terms_set) >= 2:
intents.append('industry_comparison')
# Layer 2: concept_comparison vs category_comparison — exclusive
# (subject 가 개념 대조 vs 유형 분류 — 동시에 primary 될 수 없음)
concept_fired = False
if 'process_product_split' not in intents and 'transformation_story' not in intents:
concept_pairs = [
({'BIM', 'DX'}, ['BIM과 DX', 'DX와 BIM', 'BIM · DX', 'BIM vs DX', 'BIM/DX']),
]
for pair, markers in concept_pairs:
if pair <= terms_set and any(m in text or m in title for m in markers):
intents.append('concept_comparison')
concept_fired = True
break
# Layer 3: multi_attribute_comparison (secondary, co-occurring OK)
multi_markers = ['관점별', '행별', '기준별']
aspect_terms = {'범위', '성과품', '확장성', '수행개념', '수행주체', '프로세스', '활용'}
has_multi_marker = any(m in text for m in multi_markers)
aspect_hits = len(aspect_terms & terms_set)
if (has_multi_marker or aspect_hits >= 3) and relation_type == 'compare':
if 'multi_attribute_comparison' not in intents:
intents.append('multi_attribute_comparison')
# Layer 2 (secondary branch): category_comparison — only if concept didn't fire
# 명시 markers 필요 (title/text 에 분류 주제 언급)
if not concept_fired and 'process_product_split' not in intents and \
'transformation_story' not in intents:
cat_pair_markers = [
({'Package', 'Solution'}, ['Application S/W', 'S/W 구분', 'S/W의 구분', 'Package', 'Solution']),
({'상용', '전용'}, ['상용 Engn', '전용 S/W', '3rd Party', '3rdParty']),
]
for pair, markers in cat_pair_markers:
# pair 전체 + title/text 에 명시 marker 존재
if pair <= terms_set:
# title 에 marker 있으면 강한 신호
has_title_marker = any(m in title for m in markers)
if has_title_marker:
intents.append('category_comparison')
break
# Polarity: requirement_list (title 기반)
if any(s in title for s in ['필수', '요건', '3대', '필수조건']) and relation_type == 'parallel':
intents.append('requirement_list')
# Polarity: problem_diagnosis (title or specific problem-only terms)
problem_title_markers = ['문제', '문제점', '한계', '왜곡', '실정', '부재']
problem_specific = {'개념부재', '전문성부족', '비효율', '전제조건오류', '4대문제'}
has_problem_title = any(s in title for s in problem_title_markers)
has_problem_specific = bool(problem_specific & terms_set)
if has_problem_title or has_problem_specific:
intents.append('problem_diagnosis')
# Fallback: parallel N 이면 일반 requirement_or_pillar
if not intents and relation_type == 'parallel' and item_count in (3, 4, 5):
intents.append('requirement_or_pillar')
return intents
# ═══ 엔트리 포인트 ═══
def detect_mdx_analysis(text, title, anchor_vocab=None):
synonyms = load_synonyms()
normalized = normalize_with_synonyms(text, synonyms)
subs = find_subsections(text)
table = parse_first_table(text)
bullets = find_top_bullets(text)
# item_count 우선순위: subsections > table > bullets
if subs:
item_count = len(subs)
item_source = 'subsections'
slot_candidates = slots_from_subsections(subs)
elif table:
header, body = table
non_axis = [h for h in header if AXIS_LABEL not in h and h]
item_count = len(non_axis)
item_source = 'table_columns'
slot_candidates = slots_from_table(header, body)
elif bullets:
item_count = len(bullets)
item_source = 'bullets'
slot_candidates = slots_from_bullets(bullets)
else:
item_count = 1
item_source = 'none'
slot_candidates = []
rel_value, rel_conf = infer_relation_type(item_count, item_source, text, subs)
detected_terms = extract_detected_terms(normalized, anchor_vocab)
summary = build_summary(title, text, slot_candidates)
# structure_intent 추론 (Phase 24/25 의 구조 축 용)
intents = infer_structure_intent(text, title, detected_terms, item_count, rel_value)
return {
'title': title,
'summary': summary,
'detected_terms': detected_terms,
'item_count': {'detected': item_count, 'source': item_source},
'relation_type': {'value': rel_value, 'confidence': rel_conf},
'content_shape': {
'has_table': table is not None,
'has_subsections': bool(subs),
'has_bullets': bool(bullets),
},
'slot_candidates': slot_candidates,
'structure_intent': intents,
}