"""Phase Y: 영역 확정 모듈. normalized.sections(Stage 0 산출물)를 기반으로 ## 대목차 구조를 파악하고, Kei 꼭지를 대목차에 매핑하여 영역을 확정한다. source of truth = normalized.sections (Stage 0) raw MDX는 사용하지 않음 (보존용/증거용으로만 존재). 용도: - Kei 꼭지를 대목차에 매핑 - 대목차별 묶음으로 블록 tag 매칭 - 영역 확정 (코드가, Kei가 아님) """ from __future__ import annotations import re import logging from typing import Any logger = logging.getLogger(__name__) def extract_major_sections(normalized_sections: list[dict]) -> list[dict]: """normalized.sections에서 ## 대목차(level=2)를 추출하고, 각 대목차 아래의 소목차(level=3) content를 합쳐서 반환. normalized.sections 구조: [{"level": 2, "title": "DX 시행을 위한 필수 요건", "content": ""}, {"level": 2, "title": "기술(디지털)", "content": "D1: ..."}, {"level": 3, "title": "과정(Process)의 혁신", "content": "D1: ..."}] 반환: [{"title": "DX 시행을 위한 필수 요건", "content": "기술+사람+자연 합침", "sub_titles": ["기술","사람","자연"]}, {"title": "Process의 혁신과 Product의 변화", "content": "과정+결과 합침", "sub_titles": ["과정","결과"]}] """ if not normalized_sections: return [] # level=2 중 content가 비어있는 것 = 대목차 헤더 (아래 level=2/3이 소속) # level=2 중 content가 있는 것 = 대목차 헤더가 없는 독립 섹션 (소목차) # level=3 = 소목차 major_sections = [] current_major = None for sec in normalized_sections: level = sec.get("level", 2) title = sec.get("title", "") content = sec.get("content", "") if level == 2 and not content.strip(): # 대목차 헤더 (빈 content = 아래 섹션들의 그룹 헤더) if current_major: major_sections.append(current_major) current_major = { "title": title, "content": "", "sub_titles": [], } elif level == 2 and content.strip(): # content가 있는 level=2 = 소목차 또는 독립 섹션 if current_major: # 현재 대목차 아래의 소목차 current_major["content"] += f"\n{content}" if current_major["content"] else content current_major["sub_titles"].append(title) else: # 대목차 없이 시작된 독립 섹션 (도입부) current_major = { "title": title, "content": content, "sub_titles": [title], } elif level == 3: # 소목차 → 현재 대목차에 합침 if current_major: current_major["content"] += f"\n{content}" if current_major["content"] else content current_major["sub_titles"].append(title) else: # 대목차 없는 level=3 (비정상이지만 처리) current_major = { "title": title, "content": content, "sub_titles": [title], } if current_major: major_sections.append(current_major) # 빈 섹션 제거 major_sections = [s for s in major_sections if s["content"].strip()] logger.info( f"[section_parser] {len(major_sections)}개 대목차: " + ", ".join(f'"{s["title"]}" (sub: {s["sub_titles"]})' for s in major_sections) ) return major_sections def detect_component_popups(raw_content: str, base_path: str = "") -> list[dict]: """Y-14: MDX에서 import된 Astro 컴포넌트를 감지하고 popup 대상으로 등록. Returns: [{"name": "DxEffect", "source": "components/dx.astro", "resolved_path": "실제 파일 경로", "content_html": "astro HTML 내용"}] """ from pathlib import Path popups = [] # import 문 파싱 imports = re.findall(r'import\s+(\w+)\s+from\s+["\']([^"\']+)["\']', raw_content) # self-closing 태그 사용 여부 used_tags = set(re.findall(r'<(\w+)\s*/>', raw_content)) for name, source in imports: if name not in used_tags: continue # import만 하고 사용 안 한 것은 무시 # astro 파일 경로 해석 resolved = "" content_html = "" if base_path: # MDX 기준 상대경로 → 절대경로 mdx_dir = Path(base_path) candidate = mdx_dir / source if not candidate.exists(): # samples/src/components/ 에서 찾기 candidate = Path(base_path).parent.parent / "src" / "components" / Path(source).name if not candidate.exists(): # 프로젝트 루트에서 찾기 candidate = Path("samples/src/components") / Path(source).name if candidate.exists(): resolved = str(candidate) raw = candidate.read_text(encoding="utf-8") # astro frontmatter 제거 if raw.startswith("---"): end = raw.find("---", 3) if end > 0: content_html = raw[end + 3:].strip() else: content_html = raw else: content_html = raw popups.append({ "name": name, "source": source, "resolved_path": resolved, "content_html": content_html, "tag": f"<{name} />", }) logger.info(f"[Y-14] 컴포넌트 popup 감지: {name} → {resolved or source}") return popups def _classify_sub_types( sub_titles: list[str], full_content: str, normalized_sections: list[dict] | None = None, popup_sub_titles: list[str] | None = None, ) -> list[dict]: """B-1: 각 sub_title의 콘텐츠 유형을 점수 기반 힌트로 판단. 점수 항목: - 병렬 소목차 구조 (sub_titles 수, 대등성) - 각 항목 길이 (D2 본문 길이) - D1/D2 패턴 밀도 - popup/component 존재 여부 (popup_sub_titles) Returns: [{title: str, sub_type: str}] """ results = [] lines = full_content.split("\n") norm_secs = normalized_sections or [] for st in sub_titles: st_key = re.sub(r'\*+', '', st.split("(")[0].strip()).lower() sub_content = "" # 1차: normalized_sections에서 섹션 title로 매칭 for sec in norm_secs: sec_title = sec.get("title", "").lower() if st_key and len(st_key) >= 2 and st_key in sec_title: sub_content = sec.get("content", "") break # 2차: D1: 항목 내 매칭 (sub_title이 D1 항목명인 경우) if not sub_content: capturing = False for line in lines: d1_match = re.match(r'^D1:\s*(.*)', line.strip()) if d1_match: d1_text = re.sub(r'\*+', '', d1_match.group(1)).strip().lower() if capturing: break if st_key and len(st_key) >= 2 and st_key in d1_text: capturing = True sub_content += line.strip() + "\n" elif capturing: stripped = line.strip() if stripped: sub_content += stripped + "\n" # 점수 계산 scores = { "parallel_card_candidate": 0, "text_list_candidate": 0, "visual_detail_candidate": 0, "table_heavy_candidate": 0, } d2_lines = re.findall(r'^D2:', sub_content, re.MULTILINE) d2_total_len = sum(len(l) for l in re.findall(r'^D2:\s*(.*)', sub_content, re.MULTILINE)) has_table = bool(re.search(r'As-is|To-be|\|.*\|.*\|', sub_content)) is_empty = len(sub_content.strip()) < 10 # parallel_card: 짧은 D2, 항목이 대등 if len(d2_lines) >= 1 and d2_total_len < 200: scores["parallel_card_candidate"] += 3 if len(sub_titles) >= 3: scores["parallel_card_candidate"] += 2 # text_list: 긴 D2 본문 if d2_total_len >= 100: scores["text_list_candidate"] += 3 if len(d2_lines) >= 3: scores["text_list_candidate"] += 2 # visual_detail: content 비거나 popup/component if is_empty: scores["visual_detail_candidate"] += 5 if "컴포넌트" in sub_content or "[팝업:" in sub_content: scores["visual_detail_candidate"] += 3 # popup_sub_titles에 포함되면 강하게 visual_detail popup_subs = popup_sub_titles or [] if any(st_key in ps.lower() for ps in popup_subs): scores["visual_detail_candidate"] += 6 # content가 핵심요약/결론 + D1 1줄 이하면 실질적으로 빈 것 — visual_detail # D1이 2개 이상이면 실제 본문 콘텐츠로 봄 d1_lines = re.findall(r'^D1:', sub_content, re.MULTILINE) content_without_markers = re.sub(r'\[핵심요약:[^\]]*\]', '', sub_content).strip() if len(d1_lines) <= 1 and len(content_without_markers) < 50 and sub_content.strip(): scores["visual_detail_candidate"] += 4 # D1이 여러 개면 본문형 content → text_list 가점 if len(d1_lines) >= 2: scores["text_list_candidate"] += 3 # table_heavy if has_table: scores["table_heavy_candidate"] += 5 # 최고 점수 candidate 선택 best_type = max(scores, key=scores.get) best_score = scores[best_type] # 점수가 0이면 content 길이로 fallback if best_score == 0: if sub_content.strip(): best_type = "text_list_candidate" else: best_type = "visual_detail_candidate" results.append({"title": st, "sub_type": best_type}) logger.debug(f"[sub_type] '{st}': {best_type} (scores={scores})") return results def classify_group_relations( major_sections: list[dict], topics: list[dict] | None = None, normalized_sections: list[dict] | None = None, popup_sub_titles: list[str] | None = None, ) -> list[dict]: """Y-13b: 각 대목차의 sub_titles 간 관계를 판단하여 group_schema를 부여. 규칙 기반 판단 (Kei 없이): - sub_titles 3개 + 병렬 → parallel_cluster - sub_titles 2개 + 비대칭 → compare_asymmetric_paired - sub_titles 2개 + 순서/변화 → sequence_list - sub_titles 1개 → single_block - sub_titles 4개+ → card_cluster_N Returns: major_sections에 group_schema 필드 추가하여 반환 """ for sec in major_sections: sub_titles = sec.get("sub_titles", []) content = sec.get("content", "") content_lower = content.lower() n = len(sub_titles) # sub_titles가 1개 이하지만 content에 D1: 항목이 여러 개면 → 실제 병렬 항목 수 if n <= 1: d1_items = re.findall(r'^D1:\s*\*?\*?(.+?)\*?\*?\s*$', content, re.MULTILINE) # 이미지/표 관련 D1 제외 d1_items = [d for d in d1_items if not d.strip().startswith('!') and not d.strip().startswith('As-is')] if len(d1_items) >= 2: n = len(d1_items) sec["sub_titles"] = [re.sub(r'\*+', '', d).strip() for d in d1_items] sub_titles = sec["sub_titles"] if n == 0 or n == 1: sec["group_schema"] = "single_block" elif n == 3: sec["group_schema"] = "parallel_cluster" elif n == 2: has_table = bool(re.search(r'As-is|To-be|\|.*\|.*\|', content)) compare_hints = ["vs", "비교", "차이", "반면"] asymmetric_hints = ["혁신", "변화", "변환", "전환"] process_hints = ["과정", "단계", "수행", "주체"] sub_text = " ".join(sub_titles).lower() effect_hints = ["기대효과", "효과", "성과", "결과물"] all_text = content_lower + " " + sub_text has_compare = any(h in all_text for h in compare_hints) has_asymmetric = any(h in all_text for h in asymmetric_hints) has_process = any(h in all_text for h in process_hints) has_effect = any(h in all_text for h in effect_hints) if has_table and has_asymmetric: sec["group_schema"] = "compare_asymmetric_paired" elif has_process and has_effect: sec["group_schema"] = "sequence_plus_visual" elif has_process: sec["group_schema"] = "sequence_list" elif has_compare: sec["group_schema"] = "compare_paired" else: sec["group_schema"] = "compare_paired" elif n == 4: sec["group_schema"] = "card_cluster_4" else: sec["group_schema"] = f"card_cluster_{n}" # 시각 앵커 포함 여부 (이미지, 차트, 컴포넌트 등) has_visual = "이미지" in content or "![" in content or ".png" in content if has_visual: sec["group_schema"] += "_plus_visual" # B-1: subsection typing — 각 sub_title의 콘텐츠 유형을 점수 기반으로 판단 sec["sub_types"] = _classify_sub_types(sub_titles, content, normalized_sections, popup_sub_titles) logger.info(f"[Y-13b] '{sec['title']}': sub={n}개, schema={sec['group_schema']}, sub_types={[s['sub_type'] for s in sec['sub_types']]}") return major_sections # ══════════════════════════════════════ # schema alias: 회귀 안전을 위해 old → new 매핑 유지 # ══════════════════════════════════════ SCHEMA_ALIASES = { "parallel_3": "parallel_cluster", "parallel_3_with_image": "parallel_cluster_plus_visual", "compare_2": "compare_paired", "compare_asymmetric_2col": "compare_asymmetric_paired", "process_plus_visual": "sequence_plus_visual", "process_list": "sequence_list", "single_section": "single_block", "card_list_4": "card_cluster_4", } def resolve_schema(schema: str) -> str: """old schema 이름 → new 이름으로 해소. 이미 new면 그대로 반환.""" return SCHEMA_ALIASES.get(schema, schema) # ══════════════════════════════════════ # schema → recipe 매핑 (표현 계약) # recipe = 블록 이름이 아닌, 레이아웃 계약 # ══════════════════════════════════════ SCHEMA_RECIPE_MAP = { "parallel_cluster": { "recipe": "single_block", "block_kind": "parallel_cards", "blocks": ["prerequisites-3col", "card-compare-3col", "card-icon-desc"], }, "parallel_cluster_plus_visual": { "recipe": "two_col_text_visual", "left_kind": "parallel_cards", "right_kind": "visual_anchor", "ratio": "7:3", "vertical_align": "center", # direct single-block mapping 금지: p3c는 2층 구조(label+heading)라서 # 1층 구조(목표 제목만)인 plus_visual에서는 부적합. # composition으로 쓸 가능성은 열어둠 (향후 blocks_composition에 추가 가능). "blocks_left": ["card-icon-desc", "card-compare-3col", "card-text-grid"], }, "compare_paired": { "recipe": "single_block", "block_kind": "compare_cards", "blocks": ["compare-detail-gradient", "comparison-2col"], }, "compare_asymmetric_paired": { "recipe": "single_block", "block_kind": "compare_asymmetric", "blocks": ["process-product-2col", "compare-detail-gradient"], }, "sequence_list": { "recipe": "single_block", "block_kind": "sequence_cards", "blocks": ["card-step-vertical", "checklist-dark", "card-numbered"], }, "sequence_plus_visual": { "recipe": "two_col_text_detail", "left_kind": "text_list", "right_kind": "summary_and_popup", "ratio": "6:4", "vertical_align": "top", "blocks_left": ["card-icon-desc", "card-step-vertical", "card-numbered"], }, "single_block": { "recipe": "single_block", "block_kind": "text_list", "blocks": ["dark-bullet-list", "checklist-dark", "card-numbered"], }, "card_cluster_4": { "recipe": "single_block", "block_kind": "card_grid", "blocks": ["card-icon-desc", "card-text-grid", "card-numbered"], }, } def get_recipe_for_schema(schema: str) -> dict: """schema → recipe 표현 계약 반환. alias 자동 해소.""" resolved = resolve_schema(schema) # _plus_visual suffix 분리: base schema에서 recipe 찾고, visual 플래그 추가 base = resolved.replace("_plus_visual", "") has_visual = "_plus_visual" in resolved recipe = SCHEMA_RECIPE_MAP.get(resolved) if recipe: return recipe # base schema로 fallback하되 visual 플래그 추가 recipe = SCHEMA_RECIPE_MAP.get(base) if recipe and has_visual: # base recipe를 복사해서 visual 힌트 추가 r = dict(recipe) r["has_visual"] = True return r # card_cluster_N → card_cluster_4 fallback if base.startswith("card_cluster_"): return SCHEMA_RECIPE_MAP.get("card_cluster_4", {}) return {} # C-1: recipe kind ↔ sub_type 호환 규칙 KIND_SUBTYPE_COMPAT = { "parallel_cards": ["parallel_card_candidate"], "text_list": ["text_list_candidate"], "visual_anchor": ["visual_detail_candidate"], "summary_and_popup": ["visual_detail_candidate"], "compare_cards": ["parallel_card_candidate", "text_list_candidate"], "compare_asymmetric": ["text_list_candidate", "table_heavy_candidate"], "sequence_cards": ["text_list_candidate"], "card_grid": ["parallel_card_candidate"], } def check_kind_compatibility(recipe_kind: str, sub_types: list[dict]) -> bool: """recipe의 left_kind/right_kind가 실제 sub_type과 호환되는지 확인.""" compatible = KIND_SUBTYPE_COMPAT.get(recipe_kind, []) if not compatible: return True # 규칙 없으면 호환 가정 actual_types = [s.get("sub_type", "") for s in sub_types] return any(t in compatible for t in actual_types) def get_candidate_blocks_for_schema(group_schema: str) -> list[str]: """Y-13d: group schema에 맞는 블록 후보 ID 목록 반환. recipe 경유. 주의: *_plus_visual schema는 direct single-block 매칭 금지. 이 함수는 recipe 내부의 블록 후보를 반환할 뿐, 실제 선택은 recipe executor가 담당. """ recipe = get_recipe_for_schema(group_schema) if not recipe: return [] # recipe 유형에 따라 블록 후보 반환 recipe_type = recipe.get("recipe", "") if recipe_type in ("two_col_text_visual", "two_col_text_detail"): return recipe.get("blocks_left", []) else: return recipe.get("blocks", []) def extract_conclusion_text(raw_content: str) -> str: """raw MDX에서 :::note[핵심 요약] 텍스트만 추출. 이것만 raw MDX에서 가져옴 (normalized에 없을 수 있으므로). """ note_match = re.search(r':::note\[([^\]]*)\]\s*([\s\S]*?):::', raw_content) if note_match: text = note_match.group(2).strip() # 마크다운 볼드/불릿 잔여 제거 text = re.sub(r'^\*\s*\*\*', '', text) text = re.sub(r'\*\*$', '', text) text = text.strip("* ") # 선행 불릿 마커(*, •, -) 제거 text = re.sub(r'^[\*•\-]\s*', '', text).strip() return text return "" def map_topics_to_sections( topics: list[dict], sections: list[dict], ) -> dict[str, list[int]]: """Kei 꼭지들을 대목차 섹션에 매핑. 각 꼭지의 title을 보고 어느 섹션의 content에 포함되는지 판단. Returns: {"1. DX 시행을 위한 필수 요건": [1, 2, 3], "2. Process의 혁신과 Product의 변화": [4, 5]} """ section_topics: dict[str, list[int]] = {} for sec in sections: section_topics[sec["title"]] = [] for topic in topics: tid = topic.get("id", 0) t_title = topic.get("title", "").lower() t_hint = topic.get("source_hint", "").lower() best_section = None best_score = 0 for sec in sections: sec_content = sec["content"].lower() sec_title = sec["title"].lower() # sub_titles에서도 매칭 sub_titles_lower = " ".join(s.lower() for s in sec.get("sub_titles", [])) score = 0 # 꼭지 제목이 섹션 content에 포함되는지 key = t_title.split("(")[0].strip() if key and len(key) >= 2: if key in sec_content: score += 10 if key in sec_title: score += 5 if key in sub_titles_lower: score += 8 # sub_title에 직접 매칭 # source_hint에 섹션 제목 키워드가 포함되는지 sec_key = sec_title.split(".")[-1].strip().lower()[:10] if sec_key and len(sec_key) >= 2 and sec_key in t_hint: score += 3 if score > best_score: best_score = score best_section = sec["title"] if best_section and best_score > 0: section_topics[best_section].append(tid) else: # 매칭 안 되면 첫 번째 섹션에 넣음 if sections: section_topics[sections[0]["title"]].append(tid) logger.warning(f"[section_parser] 꼭지 {tid} '{t_title}' 섹션 매핑 실패 → 첫 섹션") # 빈 섹션 제거 section_topics = {k: v for k, v in section_topics.items() if v} logger.info( f"[section_parser] 꼭지-섹션 매핑: " + ", ".join(f'"{k}": {v}' for k, v in section_topics.items()) ) return section_topics