"""Step 4.2 확정 스냅샷 보고서 — NORMALIZED_TOKEN_INVENTORY.md + .html 원칙: - AI 해석/분류 없음. raw 수치만. - 모든 데이터는 yaml 파일에서 읽음. - 검수 메모는 사용자 확정 문구만. 입력: actual_text_nodes.yaml (Step 1) actual_text_tokens.yaml (Step 2) special_forms_inventory.yaml (Step 2.5) normalized_text_tokens.yaml (Step 4.2) replacement_report.yaml (Step 4.2) synonyms.yaml (phrase_variants 룰 정의) 출력: NORMALIZED_TOKEN_INVENTORY.md NORMALIZED_TOKEN_INVENTORY.html """ from pathlib import Path import yaml import markdown HERE = Path(__file__).parent def load_yaml(name): return yaml.safe_load((HERE / name).read_text(encoding='utf-8')) def md_table(headers, rows): out = ['| ' + ' | '.join(str(h) for h in headers) + ' |'] out.append('|' + '|'.join(['---'] * len(headers)) + '|') for r in rows: out.append('| ' + ' | '.join(str(c) for c in r) + ' |') return '\n'.join(out) def section_0_summary(nodes, tokens_step2, normalized): """Executive Summary — 의사결정자용 한 장 요약. 모든 숫자는 yaml 에서 로드.""" nodes_unique = nodes['meta']['totals']['total_unique'] step2_occ = tokens_step2['meta']['totals']['corpus_total_occurrences'] final_occ = normalized['corpus']['total_occurrences'] final_unique = normalized['corpus']['unique_token_count'] return f"""## 0. Executive Summary BEPS / Figma / MDX 에서 실제 텍스트를 수집해 중복 제거 후 **{nodes_unique:,}개** 문장 / 항목을 정리하였다. 형태소 분석 단계에서는 **{step2_occ:,}개** 단어 출현을 확인했고, 정규화 후 최종 **{final_occ:,}개 단어 출현 / {final_unique:,}개 키워드 후보**로 정리하였다. 이 후보 목록을 바탕으로 각 Figma 프레임별 핵심 키워드 (anchor keyword) 를 선별한다. > **1,758개 = 전체 키워드 후보 풀** (anchor keyword 는 이 중에서 프레임별로 선별할 핵심 키워드) """ def section_1_pipeline(): return """## 1. 파이프라인 처리 흐름 | 단계 | 산출 | 역할 | |---|---|---| | Step 1 | `actual_text_nodes.yaml` | Figma texts.md 의 `- ` list item + MDX 본문에서 text_node 추출. HTML tag / markdown / HTML entity 정리. 1글자 / 순수숫자 / 단일기호 제외. | | Step 2 | `actual_text_tokens.yaml` | Kiwi 형태소 분석 (NNG / NNP / SL / SN). 1글자 / 순수숫자 제외. synonym 미적용. | | Step 2.5 | `special_forms_inventory.yaml` | S/W, H/W, 2D, 3D, BIM, DX, AS-IS, TO-BE 실제 표기형 관찰 (occurrences + nodes + source breakdown). | | Step 4.2 | `normalized_text_tokens.yaml`, `replacement_report.yaml` | PRE_COLLAPSE (4 regex) + phrase_variants 치환 → Kiwi user_dict (SL 8 + NNG 5) → 동일 필터. | """ def section_2_totals(normalized): t = normalized['meta']['totals'] total_raw = t['beps_raw'] + t['frames_raw'] + t['mdx_raw'] body = f"""## 2. 전체 카운트 | 소스 | raw tokens | unique tokens | |---|---|---| | BEPS (1 frame) | {t['beps_raw']:,} | {t['beps_unique']:,} | | Frames (32) | {t['frames_raw']:,} | avg {t['frames_unique_avg']:.1f} / frame | | MDX (3) | {t['mdx_raw']:,} | avg {t['mdx_unique_avg']:.1f} / MDX | | **Corpus** | **{total_raw:,}** | **{t['corpus_unique']:,}** | **총 occurrence**: {normalized['corpus']['total_occurrences']:,} """ return body def section_3_special_tokens(normalized, special): tf = normalized['corpus']['token_frequency'] fd = normalized['corpus']['token_frame_df'] md_df = normalized['corpus']['token_mdx_df'] targets = ['S/W', 'H/W', '2D', '3D', 'DX', 'BIM', 'As-is', 'To-Be'] rows = [] for t in targets: rows.append([ t, tf.get(t, 0), f"{fd.get(t, 0)} / 32", f"{md_df.get(t, 0)} / 3", ]) step4_table = md_table(['token', 'occurrence', 'frame_df', 'mdx_df'], rows) # Step 2.5 원본 표기형 관찰 exact = special.get('exact_targets', {}) variant = special.get('variant_targets', {}) s25_rows = [] for t in ['S/W', 'H/W', '2D', '3D', 'DX', 'BIM']: info = exact.get(t, {}) occ = info.get('count', {}).get('occurrences', {}) nodes = info.get('count', {}).get('nodes', {}) s25_rows.append([ t, occ.get('total', 0), f"b{occ.get('beps', 0)} / f{occ.get('frames', 0)} / m{occ.get('mdx', 0)}", nodes.get('total', 0), '—', ]) for t in ['AS-IS', 'TO-BE']: info = variant.get(t, {}) occ = info.get('count', {}).get('occurrences', {}) nodes = info.get('count', {}).get('nodes', {}) observed = info.get('observed_forms', []) obs_str = ', '.join(f"`{f['form']}`×{f['count']}" for f in observed) or '—' s25_rows.append([ t, occ.get('total', 0), f"b{occ.get('beps', 0)} / f{occ.get('frames', 0)} / m{occ.get('mdx', 0)}", nodes.get('total', 0), obs_str, ]) step25_table = md_table( ['target', 'occurrences (total)', 'source breakdown (b/f/m)', 'nodes (total)', 'observed_forms'], s25_rows, ) return f"""## 3. Special tokens 통계 ### 3.1 Step 4.2 정규화 후 (token 기준) {step4_table} ### 3.2 Step 2.5 원본 표기형 관찰 (문자열 기준) {step25_table} """ def section_4_replacements(report, synonyms_yaml): pre_table = md_table( ['rule_id', 'applied'], [[rid, cnt] for rid, cnt in report['pre_collapse_applied'].items()], ) pre_total = report['pre_collapse_total'] phrase_table = md_table( ['canonical', 'applied'], [[c, cnt] for c, cnt in report['applied_replacements'].items()], ) phrase_total = report['total_replacements'] total_changes = report['total_text_changes'] pv = synonyms_yaml.get('phrase_variants', {}) pv_rows = [[canonical, ', '.join(variants)] for canonical, variants in pv.items()] pv_table = md_table(['canonical', 'variants'], pv_rows) return f"""## 4. Applied replacements ### 4.1 PRE_COLLAPSE (regex, 4 rules) {pre_table} **pre_collapse_total**: {pre_total} ### 4.2 phrase_variants 치환 카운트 {phrase_table} **applied_replacements total**: {phrase_total} **total_text_changes** (pre_collapse + phrase): {total_changes} ### 4.3 phrase_variants 룰 정의 (`synonyms.yaml`) {pv_table} """ def section_5_frame_counts(normalized): frame_ids = sorted(normalized['frames'].keys()) rows = [] for i, fid in enumerate(frame_ids, 1): info = normalized['frames'][fid] rows.append([i, fid, info['raw_token_count'], info['unique_token_count']]) table = md_table( ['frame #', 'frame_id', 'raw tokens', 'unique tokens'], rows, ) return f"""## 5. Frame 별 unique token count (32개) {table} """ def section_6_target_frames(normalized): frame_ids = sorted(normalized['frames'].keys()) def by_number(n): return frame_ids[n - 1] targets = [13, 14, 18, 29] out = ["## 6. TARGET frames 상세 (13 / 14 / 18 / 29)", ""] for fnum in targets: fid = by_number(fnum) info = normalized['frames'][fid] out.append(f"### Frame {fnum} / {fid}") out.append(f"- raw_token_count: **{info['raw_token_count']}**") out.append(f"- unique_token_count: **{info['unique_token_count']}**") out.append(f"- unique_tokens:") out.append("") out.append(" " + " / ".join(info['unique_tokens'])) out.append("") return '\n'.join(out) def section_7_top_freq(normalized): rows = [ [r['token'], r['count'], r['frame_count'], r['mdx_count']] for r in normalized['corpus']['top_50_by_frequency'] ] table = md_table(['token', 'count', 'frame_count', 'mdx_count'], rows) return f"""## 7. Top 50 by frequency {table} """ def section_8_top_frame_df(normalized): rows = [ [r['token'], r['frame_count'], r['count'], r['mdx_count']] for r in normalized['corpus']['top_50_by_frame_df'] ] table = md_table(['token', 'frame_count', 'count', 'mdx_count'], rows) return f"""## 8. Top 50 by frame_df {table} """ def section_9_notes(): return """## 9. 검수 메모 - **DX(DX) 제거 완료** — PRE_COLLAPSE 4개 룰 적용. `dx_duplicate` 잔여 0건. - **3D모델 / 2D도면 canonical 은 일부러 만들지 않음** — 3D / 2D 일반 차원 토큰을 살리기 위함. 이후 anchor_set 단계에서 compound 후보로 별도 생성 가능. - **As-is / To-Be 는 frame 에 등장하지 않고 BEPS / MDX 중심** — frame_df=0. BEPS/MDX 에만 각 2건. - **Source of truth 는 `actual_text_nodes.yaml` + `normalized_text_tokens.yaml`** — 이 보고서는 검수용 snapshot 이다. """ def main(): nodes = load_yaml('actual_text_nodes.yaml') tokens_step2 = load_yaml('actual_text_tokens.yaml') normalized = load_yaml('normalized_text_tokens.yaml') report = load_yaml('replacement_report.yaml') special = load_yaml('special_forms_inventory.yaml') synonyms = load_yaml('synonyms.yaml') sections = [ "# NORMALIZED TOKEN INVENTORY — Step 4.2 확정 스냅샷", section_0_summary(nodes, tokens_step2, normalized), section_1_pipeline(), section_2_totals(normalized), section_3_special_tokens(normalized, special), section_4_replacements(report, synonyms), section_5_frame_counts(normalized), section_6_target_frames(normalized), section_7_top_freq(normalized), section_8_top_frame_df(normalized), section_9_notes(), ] md = '\n\n'.join(sections) (HERE / 'NORMALIZED_TOKEN_INVENTORY.md').write_text(md, encoding='utf-8') html_body = markdown.markdown(md, extensions=['tables', 'fenced_code']) html = f""" Normalized Token Inventory — Step 4.2 snapshot {html_body} """ (HERE / 'NORMALIZED_TOKEN_INVENTORY.html').write_text(html, encoding='utf-8') print(f"산출:") print(f" md: {HERE / 'NORMALIZED_TOKEN_INVENTORY.md'}") print(f" html: {HERE / 'NORMALIZED_TOKEN_INVENTORY.html'}") print(f" sections: 9개") print(f" 총 섹션 글자수: {len(md):,}") if __name__ == "__main__": main()