Files
C.E.L_Slide_test2/src/section_parser.py
T

574 lines
22 KiB
Python

"""Phase Y: 영역 확정 모듈.
normalized.sections(Stage 0 산출물)를 기반으로 ## 대목차 구조를 파악하고,
Kei 꼭지를 대목차에 매핑하여 영역을 확정한다.
source of truth = normalized.sections (Stage 0)
raw MDX는 사용하지 않음 (보존용/증거용으로만 존재).
용도:
- Kei 꼭지를 대목차에 매핑
- 대목차별 묶음으로 블록 tag 매칭
- 영역 확정 (코드가, Kei가 아님)
"""
from __future__ import annotations
import re
import logging
from typing import Any
logger = logging.getLogger(__name__)
def extract_major_sections(normalized_sections: list[dict]) -> list[dict]:
"""normalized.sections에서 ## 대목차(level=2)를 추출하고,
각 대목차 아래의 소목차(level=3) content를 합쳐서 반환.
normalized.sections 구조:
[{"level": 2, "title": "DX 시행을 위한 필수 요건", "content": ""},
{"level": 2, "title": "기술(디지털)", "content": "D1: ..."},
{"level": 3, "title": "과정(Process)의 혁신", "content": "D1: ..."}]
반환:
[{"title": "DX 시행을 위한 필수 요건", "content": "기술+사람+자연 합침", "sub_titles": ["기술","사람","자연"]},
{"title": "Process의 혁신과 Product의 변화", "content": "과정+결과 합침", "sub_titles": ["과정","결과"]}]
"""
if not normalized_sections:
return []
# level=2 중 content가 비어있는 것 = 대목차 헤더 (아래 level=2/3이 소속)
# level=2 중 content가 있는 것 = 대목차 헤더가 없는 독립 섹션 (소목차)
# level=3 = 소목차
major_sections = []
current_major = None
for sec in normalized_sections:
level = sec.get("level", 2)
title = sec.get("title", "")
content = sec.get("content", "")
if level == 2 and not content.strip():
# 대목차 헤더 (빈 content = 아래 섹션들의 그룹 헤더)
if current_major:
major_sections.append(current_major)
current_major = {
"title": title,
"content": "",
"sub_titles": [],
}
elif level == 2 and content.strip():
# content가 있는 level=2 = 소목차 또는 독립 섹션
if current_major:
# 현재 대목차 아래의 소목차
current_major["content"] += f"\n{content}" if current_major["content"] else content
current_major["sub_titles"].append(title)
else:
# 대목차 없이 시작된 독립 섹션 (도입부)
current_major = {
"title": title,
"content": content,
"sub_titles": [title],
}
elif level == 3:
# 소목차 → 현재 대목차에 합침
if current_major:
current_major["content"] += f"\n{content}" if current_major["content"] else content
current_major["sub_titles"].append(title)
else:
# 대목차 없는 level=3 (비정상이지만 처리)
current_major = {
"title": title,
"content": content,
"sub_titles": [title],
}
if current_major:
major_sections.append(current_major)
# 빈 섹션 제거
major_sections = [s for s in major_sections if s["content"].strip()]
logger.info(
f"[section_parser] {len(major_sections)}개 대목차: "
+ ", ".join(f'"{s["title"]}" (sub: {s["sub_titles"]})' for s in major_sections)
)
return major_sections
def detect_component_popups(raw_content: str, base_path: str = "") -> list[dict]:
"""Y-14: MDX에서 import된 Astro 컴포넌트를 감지하고 popup 대상으로 등록.
Returns:
[{"name": "DxEffect", "source": "components/dx.astro",
"resolved_path": "실제 파일 경로", "content_html": "astro HTML 내용"}]
"""
from pathlib import Path
popups = []
# import 문 파싱
imports = re.findall(r'import\s+(\w+)\s+from\s+["\']([^"\']+)["\']', raw_content)
# self-closing 태그 사용 여부
used_tags = set(re.findall(r'<(\w+)\s*/>', raw_content))
for name, source in imports:
if name not in used_tags:
continue # import만 하고 사용 안 한 것은 무시
# astro 파일 경로 해석
resolved = ""
content_html = ""
if base_path:
# MDX 기준 상대경로 → 절대경로
mdx_dir = Path(base_path)
candidate = mdx_dir / source
if not candidate.exists():
# samples/src/components/ 에서 찾기
candidate = Path(base_path).parent.parent / "src" / "components" / Path(source).name
if not candidate.exists():
# 프로젝트 루트에서 찾기
candidate = Path("samples/src/components") / Path(source).name
if candidate.exists():
resolved = str(candidate)
raw = candidate.read_text(encoding="utf-8")
# astro frontmatter 제거
if raw.startswith("---"):
end = raw.find("---", 3)
if end > 0:
content_html = raw[end + 3:].strip()
else:
content_html = raw
else:
content_html = raw
popups.append({
"name": name,
"source": source,
"resolved_path": resolved,
"content_html": content_html,
"tag": f"<{name} />",
})
logger.info(f"[Y-14] 컴포넌트 popup 감지: {name}{resolved or source}")
return popups
def _classify_sub_types(
sub_titles: list[str], full_content: str,
normalized_sections: list[dict] | None = None,
popup_sub_titles: list[str] | None = None,
) -> list[dict]:
"""B-1: 각 sub_title의 콘텐츠 유형을 점수 기반 힌트로 판단.
점수 항목:
- 병렬 소목차 구조 (sub_titles 수, 대등성)
- 각 항목 길이 (D2 본문 길이)
- D1/D2 패턴 밀도
- popup/component 존재 여부 (popup_sub_titles)
Returns: [{title: str, sub_type: str}]
"""
results = []
lines = full_content.split("\n")
norm_secs = normalized_sections or []
for st in sub_titles:
st_key = re.sub(r'\*+', '', st.split("(")[0].strip()).lower()
sub_content = ""
# 1차: normalized_sections에서 섹션 title로 매칭
for sec in norm_secs:
sec_title = sec.get("title", "").lower()
if st_key and len(st_key) >= 2 and st_key in sec_title:
sub_content = sec.get("content", "")
break
# 2차: D1: 항목 내 매칭 (sub_title이 D1 항목명인 경우)
if not sub_content:
capturing = False
for line in lines:
d1_match = re.match(r'^D1:\s*(.*)', line.strip())
if d1_match:
d1_text = re.sub(r'\*+', '', d1_match.group(1)).strip().lower()
if capturing:
break
if st_key and len(st_key) >= 2 and st_key in d1_text:
capturing = True
sub_content += line.strip() + "\n"
elif capturing:
stripped = line.strip()
if stripped:
sub_content += stripped + "\n"
# 점수 계산
scores = {
"parallel_card_candidate": 0,
"text_list_candidate": 0,
"visual_detail_candidate": 0,
"table_heavy_candidate": 0,
}
d2_lines = re.findall(r'^D2:', sub_content, re.MULTILINE)
d2_total_len = sum(len(l) for l in re.findall(r'^D2:\s*(.*)', sub_content, re.MULTILINE))
has_table = bool(re.search(r'As-is|To-be|\|.*\|.*\|', sub_content))
is_empty = len(sub_content.strip()) < 10
# parallel_card: 짧은 D2, 항목이 대등
if len(d2_lines) >= 1 and d2_total_len < 200:
scores["parallel_card_candidate"] += 3
if len(sub_titles) >= 3:
scores["parallel_card_candidate"] += 2
# text_list: 긴 D2 본문
if d2_total_len >= 100:
scores["text_list_candidate"] += 3
if len(d2_lines) >= 3:
scores["text_list_candidate"] += 2
# visual_detail: content 비거나 popup/component
if is_empty:
scores["visual_detail_candidate"] += 5
if "컴포넌트" in sub_content or "[팝업:" in sub_content:
scores["visual_detail_candidate"] += 3
# popup_sub_titles에 포함되면 강하게 visual_detail
popup_subs = popup_sub_titles or []
if any(st_key in ps.lower() for ps in popup_subs):
scores["visual_detail_candidate"] += 6
# content가 핵심요약/결론 + D1 1줄 이하면 실질적으로 빈 것 — visual_detail
# D1이 2개 이상이면 실제 본문 콘텐츠로 봄
d1_lines = re.findall(r'^D1:', sub_content, re.MULTILINE)
content_without_markers = re.sub(r'\[핵심요약:[^\]]*\]', '', sub_content).strip()
if len(d1_lines) <= 1 and len(content_without_markers) < 50 and sub_content.strip():
scores["visual_detail_candidate"] += 4
# D1이 여러 개면 본문형 content → text_list 가점
if len(d1_lines) >= 2:
scores["text_list_candidate"] += 3
# table_heavy
if has_table:
scores["table_heavy_candidate"] += 5
# 최고 점수 candidate 선택
best_type = max(scores, key=scores.get)
best_score = scores[best_type]
# 점수가 0이면 content 길이로 fallback
if best_score == 0:
if sub_content.strip():
best_type = "text_list_candidate"
else:
best_type = "visual_detail_candidate"
results.append({"title": st, "sub_type": best_type})
logger.debug(f"[sub_type] '{st}': {best_type} (scores={scores})")
return results
def classify_group_relations(
major_sections: list[dict],
topics: list[dict] | None = None,
normalized_sections: list[dict] | None = None,
popup_sub_titles: list[str] | None = None,
) -> list[dict]:
"""Y-13b: 각 대목차의 sub_titles 간 관계를 판단하여 group_schema를 부여.
규칙 기반 판단 (Kei 없이):
- sub_titles 3개 + 병렬 → parallel_cluster
- sub_titles 2개 + 비대칭 → compare_asymmetric_paired
- sub_titles 2개 + 순서/변화 → sequence_list
- sub_titles 1개 → single_block
- sub_titles 4개+ → card_cluster_N
Returns: major_sections에 group_schema 필드 추가하여 반환
"""
for sec in major_sections:
sub_titles = sec.get("sub_titles", [])
content = sec.get("content", "")
content_lower = content.lower()
n = len(sub_titles)
# sub_titles가 1개 이하지만 content에 D1: 항목이 여러 개면 → 실제 병렬 항목 수
if n <= 1:
d1_items = re.findall(r'^D1:\s*\*?\*?(.+?)\*?\*?\s*$', content, re.MULTILINE)
# 이미지/표 관련 D1 제외
d1_items = [d for d in d1_items if not d.strip().startswith('!') and not d.strip().startswith('As-is')]
if len(d1_items) >= 2:
n = len(d1_items)
sec["sub_titles"] = [re.sub(r'\*+', '', d).strip() for d in d1_items]
sub_titles = sec["sub_titles"]
if n == 0 or n == 1:
sec["group_schema"] = "single_block"
elif n == 3:
sec["group_schema"] = "parallel_cluster"
elif n == 2:
has_table = bool(re.search(r'As-is|To-be|\|.*\|.*\|', content))
compare_hints = ["vs", "비교", "차이", "반면"]
asymmetric_hints = ["혁신", "변화", "변환", "전환"]
process_hints = ["과정", "단계", "수행", "주체"]
sub_text = " ".join(sub_titles).lower()
effect_hints = ["기대효과", "효과", "성과", "결과물"]
all_text = content_lower + " " + sub_text
has_compare = any(h in all_text for h in compare_hints)
has_asymmetric = any(h in all_text for h in asymmetric_hints)
has_process = any(h in all_text for h in process_hints)
has_effect = any(h in all_text for h in effect_hints)
if has_table and has_asymmetric:
sec["group_schema"] = "compare_asymmetric_paired"
elif has_process and has_effect:
sec["group_schema"] = "sequence_plus_visual"
elif has_process:
sec["group_schema"] = "sequence_list"
elif has_compare:
sec["group_schema"] = "compare_paired"
else:
sec["group_schema"] = "compare_paired"
elif n == 4:
sec["group_schema"] = "card_cluster_4"
else:
sec["group_schema"] = f"card_cluster_{n}"
# 시각 앵커 포함 여부 (이미지, 차트, 컴포넌트 등)
has_visual = "이미지" in content or "![" in content or ".png" in content
if has_visual:
sec["group_schema"] += "_plus_visual"
# B-1: subsection typing — 각 sub_title의 콘텐츠 유형을 점수 기반으로 판단
sec["sub_types"] = _classify_sub_types(sub_titles, content, normalized_sections, popup_sub_titles)
logger.info(f"[Y-13b] '{sec['title']}': sub={n}개, schema={sec['group_schema']}, sub_types={[s['sub_type'] for s in sec['sub_types']]}")
return major_sections
# ══════════════════════════════════════
# schema alias: 회귀 안전을 위해 old → new 매핑 유지
# ══════════════════════════════════════
SCHEMA_ALIASES = {
"parallel_3": "parallel_cluster",
"parallel_3_with_image": "parallel_cluster_plus_visual",
"compare_2": "compare_paired",
"compare_asymmetric_2col": "compare_asymmetric_paired",
"process_plus_visual": "sequence_plus_visual",
"process_list": "sequence_list",
"single_section": "single_block",
"card_list_4": "card_cluster_4",
}
def resolve_schema(schema: str) -> str:
"""old schema 이름 → new 이름으로 해소. 이미 new면 그대로 반환."""
return SCHEMA_ALIASES.get(schema, schema)
# ══════════════════════════════════════
# schema → recipe 매핑 (표현 계약)
# recipe = 블록 이름이 아닌, 레이아웃 계약
# ══════════════════════════════════════
SCHEMA_RECIPE_MAP = {
"parallel_cluster": {
"recipe": "single_block",
"block_kind": "parallel_cards",
"blocks": ["prerequisites-3col", "card-compare-3col", "card-icon-desc"],
},
"parallel_cluster_plus_visual": {
"recipe": "two_col_text_visual",
"left_kind": "parallel_cards",
"right_kind": "visual_anchor",
"ratio": "7:3",
"vertical_align": "center",
# direct single-block mapping 금지: p3c는 2층 구조(label+heading)라서
# 1층 구조(목표 제목만)인 plus_visual에서는 부적합.
# composition으로 쓸 가능성은 열어둠 (향후 blocks_composition에 추가 가능).
"blocks_left": ["card-icon-desc", "card-compare-3col", "card-text-grid"],
},
"compare_paired": {
"recipe": "single_block",
"block_kind": "compare_cards",
"blocks": ["compare-detail-gradient", "comparison-2col"],
},
"compare_asymmetric_paired": {
"recipe": "single_block",
"block_kind": "compare_asymmetric",
"blocks": ["process-product-2col", "compare-detail-gradient"],
},
"sequence_list": {
"recipe": "single_block",
"block_kind": "sequence_cards",
"blocks": ["card-step-vertical", "checklist-dark", "card-numbered"],
},
"sequence_plus_visual": {
"recipe": "two_col_text_detail",
"left_kind": "text_list",
"right_kind": "summary_and_popup",
"ratio": "6:4",
"vertical_align": "top",
"blocks_left": ["card-icon-desc", "card-step-vertical", "card-numbered"],
},
"single_block": {
"recipe": "single_block",
"block_kind": "text_list",
"blocks": ["dark-bullet-list", "checklist-dark", "card-numbered"],
},
"card_cluster_4": {
"recipe": "single_block",
"block_kind": "card_grid",
"blocks": ["card-icon-desc", "card-text-grid", "card-numbered"],
},
}
def get_recipe_for_schema(schema: str) -> dict:
"""schema → recipe 표현 계약 반환. alias 자동 해소."""
resolved = resolve_schema(schema)
# _plus_visual suffix 분리: base schema에서 recipe 찾고, visual 플래그 추가
base = resolved.replace("_plus_visual", "")
has_visual = "_plus_visual" in resolved
recipe = SCHEMA_RECIPE_MAP.get(resolved)
if recipe:
return recipe
# base schema로 fallback하되 visual 플래그 추가
recipe = SCHEMA_RECIPE_MAP.get(base)
if recipe and has_visual:
# base recipe를 복사해서 visual 힌트 추가
r = dict(recipe)
r["has_visual"] = True
return r
# card_cluster_N → card_cluster_4 fallback
if base.startswith("card_cluster_"):
return SCHEMA_RECIPE_MAP.get("card_cluster_4", {})
return {}
# C-1: recipe kind ↔ sub_type 호환 규칙
KIND_SUBTYPE_COMPAT = {
"parallel_cards": ["parallel_card_candidate"],
"text_list": ["text_list_candidate"],
"visual_anchor": ["visual_detail_candidate"],
"summary_and_popup": ["visual_detail_candidate"],
"compare_cards": ["parallel_card_candidate", "text_list_candidate"],
"compare_asymmetric": ["text_list_candidate", "table_heavy_candidate"],
"sequence_cards": ["text_list_candidate"],
"card_grid": ["parallel_card_candidate"],
}
def check_kind_compatibility(recipe_kind: str, sub_types: list[dict]) -> bool:
"""recipe의 left_kind/right_kind가 실제 sub_type과 호환되는지 확인."""
compatible = KIND_SUBTYPE_COMPAT.get(recipe_kind, [])
if not compatible:
return True # 규칙 없으면 호환 가정
actual_types = [s.get("sub_type", "") for s in sub_types]
return any(t in compatible for t in actual_types)
def get_candidate_blocks_for_schema(group_schema: str) -> list[str]:
"""Y-13d: group schema에 맞는 블록 후보 ID 목록 반환. recipe 경유.
주의: *_plus_visual schema는 direct single-block 매칭 금지.
이 함수는 recipe 내부의 블록 후보를 반환할 뿐,
실제 선택은 recipe executor가 담당.
"""
recipe = get_recipe_for_schema(group_schema)
if not recipe:
return []
# recipe 유형에 따라 블록 후보 반환
recipe_type = recipe.get("recipe", "")
if recipe_type in ("two_col_text_visual", "two_col_text_detail"):
return recipe.get("blocks_left", [])
else:
return recipe.get("blocks", [])
def extract_conclusion_text(raw_content: str) -> str:
"""raw MDX에서 :::note[핵심 요약] 텍스트만 추출.
이것만 raw MDX에서 가져옴 (normalized에 없을 수 있으므로).
"""
note_match = re.search(r':::note\[([^\]]*)\]\s*([\s\S]*?):::', raw_content)
if note_match:
text = note_match.group(2).strip()
# 마크다운 볼드/불릿 잔여 제거
text = re.sub(r'^\*\s*\*\*', '', text)
text = re.sub(r'\*\*$', '', text)
text = text.strip("* ")
# 선행 불릿 마커(*, •, -) 제거
text = re.sub(r'^[\*•\-]\s*', '', text).strip()
return text
return ""
def map_topics_to_sections(
topics: list[dict],
sections: list[dict],
) -> dict[str, list[int]]:
"""Kei 꼭지들을 대목차 섹션에 매핑.
각 꼭지의 title을 보고 어느 섹션의 content에 포함되는지 판단.
Returns:
{"1. DX 시행을 위한 필수 요건": [1, 2, 3], "2. Process의 혁신과 Product의 변화": [4, 5]}
"""
section_topics: dict[str, list[int]] = {}
for sec in sections:
section_topics[sec["title"]] = []
for topic in topics:
tid = topic.get("id", 0)
t_title = topic.get("title", "").lower()
t_hint = topic.get("source_hint", "").lower()
best_section = None
best_score = 0
for sec in sections:
sec_content = sec["content"].lower()
sec_title = sec["title"].lower()
# sub_titles에서도 매칭
sub_titles_lower = " ".join(s.lower() for s in sec.get("sub_titles", []))
score = 0
# 꼭지 제목이 섹션 content에 포함되는지
key = t_title.split("(")[0].strip()
if key and len(key) >= 2:
if key in sec_content:
score += 10
if key in sec_title:
score += 5
if key in sub_titles_lower:
score += 8 # sub_title에 직접 매칭
# source_hint에 섹션 제목 키워드가 포함되는지
sec_key = sec_title.split(".")[-1].strip().lower()[:10]
if sec_key and len(sec_key) >= 2 and sec_key in t_hint:
score += 3
if score > best_score:
best_score = score
best_section = sec["title"]
if best_section and best_score > 0:
section_topics[best_section].append(tid)
else:
# 매칭 안 되면 첫 번째 섹션에 넣음
if sections:
section_topics[sections[0]["title"]].append(tid)
logger.warning(f"[section_parser] 꼭지 {tid} '{t_title}' 섹션 매핑 실패 → 첫 섹션")
# 빈 섹션 제거
section_topics = {k: v for k, v in section_topics.items() if v}
logger.info(
f"[section_parser] 꼭지-섹션 매핑: "
+ ", ".join(f'"{k}": {v}' for k, v in section_topics.items())
)
return section_topics