"""MDX 분석 — template-fit-v1 매칭용 코드 기반 추출. Spec: tests/matching/TEMPLATE_FIT_V1.md §3 출력: { title, summary, detected_terms, item_count: {detected, source}, relation_type: {value, confidence}, content_shape: {has_table, has_subsections, has_bullets}, slot_candidates: [{label, body}, ...] } 원칙: LLM 호출 없음. 모든 필드는 규칙 기반 코드로 추출. """ import re from phase_common import normalize_with_synonyms, load_synonyms from methods import _get_kiwi # ─── 패턴 ──────────────────────────────────────────── RE_TABLE_ROW = re.compile(r'^\|.*\|') RE_TABLE_SEP = re.compile(r'^\|[\s\-:|]+\|') RE_SUBSECTION = re.compile(r'^###\s+(\d+\.\d+(?:\.\d+)?)\s+(.+)') RE_TOP_BULLET = re.compile(r'^[-*]\s+\*\*([^*]+)\*\*') RE_HTML_TAG = re.compile(r'<[^>]+>') AXIS_LABEL = '구분' # 표 헤더 중 이 라벨이 있으면 해당 컬럼은 axis 로 제외 NOUN_TAGS = {'NNG', 'NNP', 'NR', 'SL', 'SN', 'SH'} # relation_type 보정 키워드 COMPARE_HINTS = ['과정', '결과', 'Process', 'Product', 'AS-IS', 'TO-BE', 'Analogue', 'Digital', 'vs', 'VS', '대비'] SEQUENCE_HINTS = ['단계', 'Step', '순서', '차례'] # ═══ 표 파싱 ═══ def parse_first_table(text): """첫 markdown 테이블 반환 (header_cells, body_rows). 없으면 None.""" lines = text.split('\n') tbl_start = None for i in range(len(lines) - 1): if RE_TABLE_ROW.match(lines[i].strip()) and RE_TABLE_SEP.match(lines[i+1].strip()): tbl_start = i break if tbl_start is None: return None rows = [] for j in range(tbl_start, len(lines)): s = lines[j].strip() if not s.startswith('|'): break rows.append(s) if len(rows) < 2: return None def cells(line): return [c.strip().replace('**', '').strip() for c in line.strip('|').split('|')] header = cells(rows[0]) body = [cells(r) for r in rows[2:]] return header, body def find_subsections(text): """### X.Y TITLE 라인 목록 → [(number, title), ...]""" out = [] for ln in text.split('\n'): m = RE_SUBSECTION.match(ln.strip()) if m: out.append((m.group(1), m.group(2).strip())) return out def find_top_bullets(text): """최상위 '* **LABEL**' 블릿 목록 → [full_line, ...]""" return [ln for ln in text.split('\n') if RE_TOP_BULLET.match(ln)] # ═══ 슬롯 후보 추출 ═══ def slots_from_table(header, body): """축(구분) 제외 컬럼 → label. 첫 데이터 행의 해당 cell → body.""" axis_idx = next((i for i, h in enumerate(header) if AXIS_LABEL in h), None) first_row = body[0] if body else [] result = [] for i, h in enumerate(header): if i == axis_idx or not h: continue body_cell = first_row[i] if i < len(first_row) else '' body_clean = RE_HTML_TAG.sub(' ', body_cell).strip()[:120] result.append({'label': h, 'body': body_clean}) return result def slots_from_subsections(subs): """### TITLE → label (괄호 앞부분), body (전체 타이틀).""" out = [] for num, title in subs: main = re.match(r'^([^\(]+)', title) label = main.group(1).strip() if main else title out.append({'label': label, 'body': title}) return out def slots_from_bullets(bullets): out = [] for b in bullets: m = RE_TOP_BULLET.match(b) if not m: continue full_label = m.group(1).strip() main = re.match(r'^([^\(]+)', full_label) label = main.group(1).strip() if main else full_label out.append({'label': label, 'body': full_label}) return out # ═══ detected_terms ═══ def clean_for_kiwi(text): """Kiwi 입력 전 HTML/JSX/MDX 노이즈 제거. 구조 파싱에는 원본 사용, Kiwi 명사 추출에만 이 함수 사용.""" # HTML 태그 (open/close/self-closing) text = re.sub(r'<[^>]+>', ' ', text) # JSX/MDX 속성 블록 {...} (style, eventHandler 등) text = re.sub(r'\{[^}]*\}', ' ', text) # HTML entity text = re.sub(r'&[a-zA-Z]+;', ' ', text) # 단위/수치 노이즈 (10px, 0.9rem 등) text = re.sub(r'\b\d+(?:\.\d+)?(?:px|rem|em|%)\b', ' ', text) # 공백 정규화 text = re.sub(r'\s+', ' ', text) return text def extract_detected_terms(normalized_text, anchor_vocab): """Kiwi 명사/외국어/숫자 + anchor_vocab substring. - HTML/JSX 제거 후 Kiwi noun 추출 - anchor_vocab substring 은 원본(normalized) 에 대해 수행 (compound term 커버) """ cleaned = clean_for_kiwi(normalized_text) kiwi = _get_kiwi() tokens = kiwi.tokenize(cleaned) noun_tokens = [t.form for t in tokens if t.tag in NOUN_TAGS and len(t.form) >= 2] # anchor_vocab substring (compound terms — Kiwi 가 분해하는 단어 커버) for term in (anchor_vocab or []): if len(term) >= 2 and term in normalized_text: if term not in noun_tokens: noun_tokens.append(term) seen = set() out = [] for t in noun_tokens: if t not in seen: seen.add(t) out.append(t) return out # ═══ summary ═══ def build_summary(title, text, slot_candidates): """title + 첫 일반 문단(있으면) + 슬롯 라벨 조인.""" parts = [title] for ln in text.split('\n'): s = ln.strip() if not s: continue if s.startswith(('|', '#', '<', '*', '-', ':', '{', '```', '---')): continue if re.match(r'^\d+\.\s', s) or re.match(r'^= 3: return ('parallel', 'high') return ('definition', 'low') # ═══ structure_intent 추론 ═══ def infer_structure_intent(text, title, detected_terms, item_count, relation_type): """MDX 의 structure_intent 추론 (list, 여러 개 가능). 설계 원칙: - Layer 1 (subject-defined): process_product, transformation, persona, industry 등 명확한 subject signal. 이들은 MDX 의 primary intent. - Layer 2 (mutually exclusive): concept_comparison 은 Layer 1 의 특정 intent (process_product/transformation) 와 배타적. 둘 다 설정되면 재구성 모호해짐. - Layer 3 (secondary): multi_attribute / category — 보조 intent. - Polarity (title 기반): requirement / problem 은 **title** 이나 구체 단어로만 감지 (body noise 방지). """ intents = [] terms_set = set(detected_terms) # Layer 1: subject-defined primary intents # process_product: terms_set 또는 text 패턴 has_pp_pattern = ( ('과정' in text and ('혁신' in text or 'Process' in text)) and ('결과' in text and ('변화' in text or 'Product' in text)) ) if ('과정혁신' in terms_set and '결과혁신' in terms_set) or has_pp_pattern: intents.append('process_product_split') if ('AS-IS' in terms_set and 'TO-BE' in terms_set) or \ ('Analogue' in terms_set and 'Digital' in terms_set): if 'transformation_story' not in intents: intents.append('transformation_story') if {'발주자', '시공자', '설계자'} <= terms_set: intents.append('persona_benefit') if sum(1 for t in ['제조업', '건축', '토목'] if t in terms_set) >= 2: intents.append('industry_comparison') # Layer 2: concept_comparison vs category_comparison — exclusive # (subject 가 개념 대조 vs 유형 분류 — 동시에 primary 될 수 없음) concept_fired = False if 'process_product_split' not in intents and 'transformation_story' not in intents: concept_pairs = [ ({'BIM', 'DX'}, ['BIM과 DX', 'DX와 BIM', 'BIM · DX', 'BIM vs DX', 'BIM/DX']), ] for pair, markers in concept_pairs: if pair <= terms_set and any(m in text or m in title for m in markers): intents.append('concept_comparison') concept_fired = True break # Layer 3: multi_attribute_comparison (secondary, co-occurring OK) multi_markers = ['관점별', '행별', '기준별'] aspect_terms = {'범위', '성과품', '확장성', '수행개념', '수행주체', '프로세스', '활용'} has_multi_marker = any(m in text for m in multi_markers) aspect_hits = len(aspect_terms & terms_set) if (has_multi_marker or aspect_hits >= 3) and relation_type == 'compare': if 'multi_attribute_comparison' not in intents: intents.append('multi_attribute_comparison') # Layer 2 (secondary branch): category_comparison — only if concept didn't fire # 명시 markers 필요 (title/text 에 분류 주제 언급) if not concept_fired and 'process_product_split' not in intents and \ 'transformation_story' not in intents: cat_pair_markers = [ ({'Package', 'Solution'}, ['Application S/W', 'S/W 구분', 'S/W의 구분', 'Package', 'Solution']), ({'상용', '전용'}, ['상용 Engn', '전용 S/W', '3rd Party', '3rdParty']), ] for pair, markers in cat_pair_markers: # pair 전체 + title/text 에 명시 marker 존재 if pair <= terms_set: # title 에 marker 있으면 강한 신호 has_title_marker = any(m in title for m in markers) if has_title_marker: intents.append('category_comparison') break # Polarity: requirement_list (title 기반) if any(s in title for s in ['필수', '요건', '3대', '필수조건']) and relation_type == 'parallel': intents.append('requirement_list') # Polarity: problem_diagnosis (title or specific problem-only terms) problem_title_markers = ['문제', '문제점', '한계', '왜곡', '실정', '부재'] problem_specific = {'개념부재', '전문성부족', '비효율', '전제조건오류', '4대문제'} has_problem_title = any(s in title for s in problem_title_markers) has_problem_specific = bool(problem_specific & terms_set) if has_problem_title or has_problem_specific: intents.append('problem_diagnosis') # Fallback: parallel N 이면 일반 requirement_or_pillar if not intents and relation_type == 'parallel' and item_count in (3, 4, 5): intents.append('requirement_or_pillar') return intents # ═══ 엔트리 포인트 ═══ def detect_mdx_analysis(text, title, anchor_vocab=None): synonyms = load_synonyms() normalized = normalize_with_synonyms(text, synonyms) subs = find_subsections(text) table = parse_first_table(text) bullets = find_top_bullets(text) # item_count 우선순위: subsections > table > bullets if subs: item_count = len(subs) item_source = 'subsections' slot_candidates = slots_from_subsections(subs) elif table: header, body = table non_axis = [h for h in header if AXIS_LABEL not in h and h] item_count = len(non_axis) item_source = 'table_columns' slot_candidates = slots_from_table(header, body) elif bullets: item_count = len(bullets) item_source = 'bullets' slot_candidates = slots_from_bullets(bullets) else: item_count = 1 item_source = 'none' slot_candidates = [] rel_value, rel_conf = infer_relation_type(item_count, item_source, text, subs) detected_terms = extract_detected_terms(normalized, anchor_vocab) summary = build_summary(title, text, slot_candidates) # structure_intent 추론 (Phase 24/25 의 구조 축 용) intents = infer_structure_intent(text, title, detected_terms, item_count, rel_value) return { 'title': title, 'summary': summary, 'detected_terms': detected_terms, 'item_count': {'detected': item_count, 'source': item_source}, 'relation_type': {'value': rel_value, 'confidence': rel_conf}, 'content_shape': { 'has_table': table is not None, 'has_subsections': bool(subs), 'has_bullets': bool(bullets), }, 'slot_candidates': slot_candidates, 'structure_intent': intents, }