"""Step 1: texts.md / MDX 에서 실제 text node 만 추출. 원칙: - Figma texts.md: `- xxx` list item 만 포함 (heading/blockquote 제외) - MDX: heading + 일반 문단 + list item + table cell 포함 (frontmatter / code block / :::note / standalone HTML tag 제외) - HTML tag / markdown bold / markdown italic 제거, HTML entity unescape - 단일 기호 / 1자 / 순수 숫자 제외 (30%, 2D, 3D 등은 보존) - 각 프레임별 + 전체 corpus 카운트 - text_nodes 리스트는 중복 제거 산출: actual_text_nodes.yaml """ import html as html_lib import re import sys from pathlib import Path import yaml sys.path.insert(0, str(Path(__file__).parent)) ROOT = Path(r"d:\ad-hoc\kei\design_agent") BLOCKS_DIR = ROOT / "figma_to_html_agent" / "blocks" MDX_DIR = ROOT / "samples" / "mdx_batch" HERE = Path(__file__).parent BEPS_ID = "1171281171" FRAME_IDS = sorted([ d.name for d in BLOCKS_DIR.iterdir() if d.is_dir() and d.name.startswith("1171") and d.name != BEPS_ID ]) def clean_text(raw): """HTML tag, markdown bold/italic, HTML entity 정리.""" clean = re.sub(r'<[^>]+>', ' ', raw) clean = re.sub(r'\*\*([^*]+)\*\*', r'\1', clean) clean = re.sub(r'(?) — meta 주석 (예: > 패턴, > 원본) - 코드블록 (```...```) - standalone HTML tag (, , ) - markdown table separator (|---|---|) - 빈 줄, 단일 기호, 1자, 순수 숫자 (is_trivial 필터) """ raw_nodes = [] clean_nodes = [] in_codeblock = False for line in path.read_text(encoding='utf-8').split('\n'): stripped = line.strip() # 코드블록 토글 if stripped.startswith('```'): in_codeblock = not in_codeblock continue if in_codeblock: continue if not stripped: continue # heading 제외 (구조 라벨) if stripped.startswith('#'): continue # blockquote 제외 (meta 주석) if stripped.startswith('>'): continue # standalone HTML tag 제외 (opening/closing/self-closing 모두) if re.fullmatch(r']+>', stripped): continue # markdown table separator 제외 (|---|---|) if re.fullmatch(r'\|[\s\-:|]+\|?', stripped): continue # list item → prefix 제거 m_list = re.match(r'^[-*]\s+(.+)$', stripped) if m_list: raw = m_list.group(1).strip() else: # plain line raw = stripped clean = clean_text(raw) if is_trivial(clean): continue raw_nodes.append(raw) clean_nodes.append(clean) return raw_nodes, clean_nodes def extract_mdx_content(path): """MDX 에서 시각 가능한 텍스트 추출. 포함: - heading (#, ##, ### ...) - 일반 문단 (paragraph) - list item (-, *, 숫자. ) - table cell (| a | b | c |) 제외: - frontmatter (--- ... ---) - code block (``` ... ```) - admonition 개행 (:::note, :::) - standalone HTML tag 줄 (예: ) """ raw_nodes = [] clean_nodes = [] lines = path.read_text(encoding='utf-8').split('\n') return _extract_mdx_from_lines(lines, handle_frontmatter=True) def _extract_mdx_from_lines(lines, handle_frontmatter=False): """extract_mdx_content 의 내부 구현 (section 추출 재사용).""" raw_nodes = [] clean_nodes = [] in_codeblock = False in_admonition = False start_idx = 0 if handle_frontmatter and lines and lines[0].strip() == '---': for i in range(1, len(lines)): if lines[i].strip() == '---': start_idx = i + 1 break for line in lines[start_idx:]: stripped = line.strip() # code block 토글 if stripped.startswith('```'): in_codeblock = not in_codeblock continue if in_codeblock: continue # admonition 토글 (:::note ... :::) if stripped.startswith(':::'): in_admonition = not in_admonition if stripped == ':::' else True if stripped == ':::': in_admonition = False else: in_admonition = True continue if not stripped: continue # table separator 배제 (|---|---|) if re.fullmatch(r'\|?[\s\-:|]+\|?', stripped) and '|' in stripped: continue # table row 처리 if stripped.startswith('|') and stripped.endswith('|'): cells = [c.strip() for c in stripped.strip('|').split('|')] for cell in cells: if not cell: continue raw = cell clean = clean_text(raw) if is_trivial(clean): continue raw_nodes.append(raw) clean_nodes.append(clean) continue # list item m_list = re.match(r'^[-*]\s+(.+)$', stripped) if m_list: raw = m_list.group(1).strip() clean = clean_text(raw) if not is_trivial(clean): raw_nodes.append(raw) clean_nodes.append(clean) continue m_ol = re.match(r'^\d+\.\s+(.+)$', stripped) if m_ol: raw = m_ol.group(1).strip() clean = clean_text(raw) if not is_trivial(clean): raw_nodes.append(raw) clean_nodes.append(clean) continue # heading m_h = re.match(r'^#+\s+(.+)$', stripped) if m_h: raw = m_h.group(1).strip() clean = clean_text(raw) if not is_trivial(clean): raw_nodes.append(raw) clean_nodes.append(clean) continue # blockquote 는 일반 문단으로 취급 (단, > 만 있는 줄은 제외) if stripped.startswith('>'): rest = stripped.lstrip('>').strip() if not rest: continue raw = rest clean = clean_text(raw) if not is_trivial(clean): raw_nodes.append(raw) clean_nodes.append(clean) continue # standalone HTML tag 줄 배제 ( 같은) if re.fullmatch(r'<[^>]+>', stripped): continue # HTML 닫는 태그만 있는 줄 if re.fullmatch(r']+>', stripped): continue # 일반 문단 raw = stripped clean = clean_text(raw) if not is_trivial(clean): raw_nodes.append(raw) clean_nodes.append(clean) return raw_nodes, clean_nodes # MDX section 정의 # - TARGET (검증용): 01-2 / 02-2.2 / 03-1 / 03-2 (ANSWER_MAP 매핑 있음) # - 홀드아웃 (블라인드 검증): 01-1 / 02-1 / 02-2.1 (기대 프레임 미지정) MDX_SECTIONS = { # --- TARGET (ANSWER_MAP 매핑 있음) --- '01-2': {'file': '01.mdx', 'start': '## 2. 용어간 상호관계', 'end_prefix': None}, '02-2.2': {'file': '02.mdx', 'start': '### 2.2 DX 시행 주체별 기대효과', 'end_prefix': None}, '03-1': {'file': '03.mdx', 'start': '## 1. DX 시행을 위한 필수 요건', 'end_prefix': '## 2.'}, '03-2': {'file': '03.mdx', 'start': '## 2. Process의 혁신과 Product의 변화', 'end_prefix': None}, # --- 홀드아웃 (블라인드, 기대 프레임 없음) --- '01-1': {'file': '01.mdx', 'start': '## 1. 용어 정의', 'end_prefix': '## 2.'}, '02-1': {'file': '02.mdx', 'start': '## 1. DX의 궁극적 목표', 'end_prefix': '## 2.'}, '02-2.1': {'file': '02.mdx', 'start': '### 2.1 업무 수행 과정(Process)의 변화', 'end_prefix': '### 2.2'}, # --- MDX 04 (DX 지연 요인) — 신규 일반화 테스트, 기대 프레임 미지정 --- '04-1': {'file': '04.mdx', 'start': '## 1. DX에 대한 인식', 'end_prefix': '## 2.'}, '04-2.1': {'file': '04.mdx', 'start': '### 2.1 정책 및 발주 체계', 'end_prefix': '### 2.2'}, '04-2.2': {'file': '04.mdx', 'start': '### 2.2 조직 및 수행 역량', 'end_prefix': None}, # 04-2 통합 (사용자 lock 2026-05-14) — ## 2. DX 추진의 실태 전체 (### 2.1 + ### 2.2 + footer) '04-2': {'file': '04.mdx', 'start': '## 2. DX 추진의 실태', 'end_prefix': None}, # --- MDX 05 (설계 방식의 왜곡) — 신규 추가, 기대 프레임 미지정 (사용자 요청 2026-05-14) --- '05-1': {'file': '05.mdx', 'start': '## 1. 설계의 자동화', 'end_prefix': '## 2.'}, '05-2': {'file': '05.mdx', 'start': '## 2. S/W 중심 설계 방식', 'end_prefix': None}, } def extract_mdx_section(path, start_heading, end_prefix=None): """MDX 파일에서 start_heading 줄부터 end_prefix 직전까지의 section 을 추출.""" lines = path.read_text(encoding='utf-8').split('\n') start_idx = None for i, line in enumerate(lines): if line.strip() == start_heading.strip(): start_idx = i break if start_idx is None: raise ValueError(f"start_heading 찾지 못함: {start_heading!r} in {path}") end_idx = len(lines) if end_prefix: for i in range(start_idx + 1, len(lines)): if lines[i].strip().startswith(end_prefix): end_idx = i break section_lines = lines[start_idx:end_idx] return _extract_mdx_from_lines(section_lines, handle_frontmatter=False) def build_entry(raw, clean): """한 소스의 카운트 + dedup 결과 패키징.""" # dedup: 순서 유지 seen = set() unique = [] for c in clean: if c in seen: continue seen.add(c) unique.append(c) return { 'raw_nodes_count': len(raw), 'clean_nodes_count': len(clean), 'unique_clean_nodes_count': len(unique), 'duplicate_removed_count': len(clean) - len(unique), 'text_nodes': unique, } def main(): output = { 'meta': { 'pipeline_step': 1, 'description': 'texts.md / MDX 에서 실제 text node 추출, HTML/markdown 정리, 중복 제거', 'sources': {}, }, 'frames': {}, 'beps': {}, 'mdx': {}, } # BEPS p = BLOCKS_DIR / BEPS_ID / "texts.md" if p.exists(): raw, clean = extract_figma_nodes(p) entry = build_entry(raw, clean) entry['frame_id'] = BEPS_ID # frame_id를 맨 앞으로 output['beps'] = {'frame_id': BEPS_ID, **{k: v for k, v in entry.items() if k != 'frame_id'}} # 32 frames for fid in FRAME_IDS: p = BLOCKS_DIR / fid / "texts.md" if not p.exists(): continue raw, clean = extract_figma_nodes(p) output['frames'][fid] = build_entry(raw, clean) # MDX section 단위 (TARGET 4개) for section_id, cfg in MDX_SECTIONS.items(): p = MDX_DIR / cfg['file'] if not p.exists(): continue raw, clean = extract_mdx_section(p, cfg['start'], cfg.get('end_prefix')) entry = build_entry(raw, clean) entry['section_id'] = section_id entry['source_file'] = cfg['file'] entry['section_heading'] = cfg['start'] output['mdx'][section_id] = entry # 총계 def _sum(key, grp): return sum(g[key] for g in grp.values()) beps = output['beps'] frames = output['frames'] mdx = output['mdx'] totals = { 'beps_raw': beps.get('raw_nodes_count', 0), 'beps_clean': beps.get('clean_nodes_count', 0), 'beps_unique': beps.get('unique_clean_nodes_count', 0), 'beps_duplicate_removed': beps.get('duplicate_removed_count', 0), 'frames_raw': _sum('raw_nodes_count', frames), 'frames_clean': _sum('clean_nodes_count', frames), 'frames_unique': _sum('unique_clean_nodes_count', frames), 'frames_duplicate_removed': _sum('duplicate_removed_count', frames), 'mdx_raw': _sum('raw_nodes_count', mdx), 'mdx_clean': _sum('clean_nodes_count', mdx), 'mdx_unique': _sum('unique_clean_nodes_count', mdx), 'mdx_duplicate_removed': _sum('duplicate_removed_count', mdx), } totals['total_raw'] = totals['beps_raw'] + totals['frames_raw'] + totals['mdx_raw'] totals['total_clean'] = totals['beps_clean'] + totals['frames_clean'] + totals['mdx_clean'] totals['total_unique'] = totals['beps_unique'] + totals['frames_unique'] + totals['mdx_unique'] totals['total_duplicate_removed'] = ( totals['beps_duplicate_removed'] + totals['frames_duplicate_removed'] + totals['mdx_duplicate_removed'] ) output['meta']['sources'] = { 'beps_count': 1 if beps else 0, 'frame_count': len(frames), 'mdx_count': len(mdx), } output['meta']['totals'] = totals # 저장 out = HERE / "actual_text_nodes.yaml" with open(out, 'w', encoding='utf-8') as f: yaml.safe_dump(output, f, allow_unicode=True, sort_keys=False, width=200) # 화면 요약 print(f"[Step 1] text node 추출 완료") print(f" BEPS: raw={totals['beps_raw']}, clean={totals['beps_clean']}, " f"unique={totals['beps_unique']}, dedup_removed={totals['beps_duplicate_removed']}") print(f" Frames: raw={totals['frames_raw']}, clean={totals['frames_clean']}, " f"unique={totals['frames_unique']}, dedup_removed={totals['frames_duplicate_removed']} " f"({len(frames)}개 frame)") print(f" MDX: raw={totals['mdx_raw']}, clean={totals['mdx_clean']}, " f"unique={totals['mdx_unique']}, dedup_removed={totals['mdx_duplicate_removed']}") print(f" TOTAL: raw={totals['total_raw']}, clean={totals['total_clean']}, " f"unique={totals['total_unique']}, dedup_removed={totals['total_duplicate_removed']}") print() print(f"산출: {out}") # Frame 18/29/14/13 샘플 미리보기 print() print("─" * 60) print("샘플 프레임 (정답 TARGET 4개)") print("─" * 60) for fnum in ['18', '29', '14', '13']: # frame_id 는 BEPS 제외 frame 중에서 frame number 로 접근 어려움 → FRAME_IDS 순 찾기 # 사용자는 Frame 번호(순번)로 표현했으므로 FRAME_IDS[fnum-1] 로 추정 idx = int(fnum) - 1 if idx < 0 or idx >= len(FRAME_IDS): continue fid = FRAME_IDS[idx] info = frames.get(fid) if not info: continue print(f"\n[Frame {fnum}] id={fid}") print(f" raw={info['raw_nodes_count']}, clean={info['clean_nodes_count']}, " f"unique={info['unique_clean_nodes_count']}, dedup_removed={info['duplicate_removed_count']}") print(f" sample text_nodes (최대 10개):") for t in info['text_nodes'][:10]: print(f" - {t}") if __name__ == "__main__": main()