"""Step 2: actual_text_nodes.yaml → Kiwi 형태소 토큰 추출. 원칙: - Kiwi 품사 NNG/NNP/SL/SN 만 (NNB 제외) - 1글자 제거 - 순수 숫자 제거 - synonym/canonical 합침 없음 (Step 4 에서 처리) - frame별 unique tokens - corpus top_50_by_frequency + top_50_by_frame_df - general_candidates 는 삭제 기준 아님, 검토 후보 - special_token_samples: 2D/3D/S/W/H/W/DX/BIM/AS-IS/TO-BE Kiwi 쪼개짐 관찰 산출: actual_text_tokens.yaml """ import re from collections import Counter, defaultdict from pathlib import Path import yaml from kiwipiepy import Kiwi HERE = Path(__file__).parent INPUT = HERE / "actual_text_nodes.yaml" OUTPUT = HERE / "actual_text_tokens.yaml" ALLOWED_TAGS = {'NNG', 'NNP', 'SL', 'SN'} SPECIAL_TOKEN_TARGETS = ['2D', '3D', 'S/W', 'H/W', 'DX', 'BIM', 'AS-IS', 'TO-BE'] kiwi = Kiwi() def extract_tokens(texts): """text 리스트에서 허용 태그 token 만 추출.""" raw = [] for t in texts: for tok in kiwi.tokenize(t): if tok.tag not in ALLOWED_TAGS: continue form = tok.form if len(form) < 2: continue if re.fullmatch(r'\d+', form): continue raw.append(form) return raw def dedup_keep_order(seq): seen = set() out = [] for x in seq: if x in seen: continue seen.add(x) out.append(x) return out def tokenize_to_pairs(text): """tokenize 결과를 [[form, tag], ...] 로 변환.""" return [[tok.form, tok.tag] for tok in kiwi.tokenize(text)] def process_source(source_key, text_nodes, token_freq, token_to_sources): raw = extract_tokens(text_nodes) unique = dedup_keep_order(raw) for tok in raw: token_freq[tok] += 1 for tok in set(raw): token_to_sources[tok].add(source_key) return { 'raw_token_count': len(raw), 'unique_token_count': len(unique), 'unique_tokens': unique, } def build_special_samples(data): """특수 토큰이 코퍼스에서 Kiwi 에 의해 어떻게 분해되는지 + source breakdown.""" # source 별 text_nodes 수집 beps_texts = (data.get('beps') or {}).get('text_nodes', []) frame_texts = [] for info in data.get('frames', {}).values(): frame_texts.extend(info.get('text_nodes', [])) mdx_texts = [] for info in data.get('mdx', {}).values(): mdx_texts.extend(info.get('text_nodes', [])) all_texts = beps_texts + frame_texts + mdx_texts samples = {} for target in SPECIAL_TOKEN_TARGETS: beps_hits = [t for t in beps_texts if target in t] frame_hits = [t for t in frame_texts if target in t] mdx_hits = [t for t in mdx_texts if target in t] all_hits = [t for t in all_texts if target in t] samples[target] = { 'tokenize_alone': tokenize_to_pairs(target), 'found_in': { 'total': len(all_hits), 'beps': len(beps_hits), 'frames': len(frame_hits), 'mdx': len(mdx_hits), }, 'in_context': [ {'text': t, 'tokens': tokenize_to_pairs(t)} for t in all_hits[:5] ], } return samples def main(): data = yaml.safe_load(INPUT.read_text(encoding='utf-8')) output = { 'meta': { 'pipeline_step': 2, 'description': 'Kiwi 형태소 추출 (명사/외국어/숫자). 1글자/순수숫자 제외. synonym 미적용.', 'filters': { 'allowed_tags': sorted(ALLOWED_TAGS), 'exclude_1char': True, 'exclude_pure_number': True, }, }, 'beps': {}, 'frames': {}, 'mdx': {}, 'corpus': {}, } token_freq = Counter() token_to_sources = defaultdict(set) # BEPS beps = data.get('beps') or {} if beps.get('text_nodes'): key = f"beps:{beps['frame_id']}" entry = process_source(key, beps['text_nodes'], token_freq, token_to_sources) output['beps'] = {'frame_id': beps['frame_id'], **entry} # Frames for fid, info in data.get('frames', {}).items(): key = f"frame:{fid}" entry = process_source(key, info.get('text_nodes', []), token_freq, token_to_sources) output['frames'][fid] = entry # MDX for n, info in data.get('mdx', {}).items(): key = f"mdx:{n}" entry = process_source(key, info.get('text_nodes', []), token_freq, token_to_sources) output['mdx'][n] = entry # corpus 통계 total_frames = len(output['frames']) frame_count_of = { t: sum(1 for s in token_to_sources[t] if s.startswith('frame:')) for t in token_freq } mdx_count_of = { t: sum(1 for s in token_to_sources[t] if s.startswith('mdx:')) for t in token_freq } top50_by_freq = [ {'token': t, 'count': c, 'frame_count': frame_count_of[t], 'mdx_count': mdx_count_of[t]} for t, c in token_freq.most_common(50) ] top50_by_frame_df = sorted( [ {'token': t, 'frame_count': frame_count_of[t], 'count': token_freq[t], 'mdx_count': mdx_count_of[t]} for t in token_freq ], key=lambda x: (-x['frame_count'], -x['count'], x['token']), )[:50] threshold = max(3, total_frames // 2) general = sorted( [ {'token': t, 'frame_count': frame_count_of[t], 'total_count': token_freq[t], 'mdx_count': mdx_count_of[t]} for t in token_freq if frame_count_of[t] >= threshold ], key=lambda x: (-x['frame_count'], -x['total_count'], x['token']), ) special_samples = build_special_samples(data) output['corpus'] = { 'unique_token_count': len(token_freq), 'total_occurrences': sum(token_freq.values()), 'top_50_by_frequency': top50_by_freq, 'top_50_by_frame_df': top50_by_frame_df, 'general_candidates_note': ( '제거 기준 아님. 32개 frame 중 threshold 이상에 등장한 토큰 = 검토 후보. ' 'BIM/건설/기술 등 도메인어가 포함될 수 있으므로 삭제하지 말 것.' ), 'general_candidates_threshold': threshold, 'general_candidates_count': len(general), 'general_candidates': general, 'special_token_samples_note': ( 'Step 4 synonym/canonical 룰을 잡기 전에 Kiwi 가 ' '2D/3D/S/W/H/W/DX/BIM/AS-IS/TO-BE 를 어떻게 분해하는지 관찰용. ' 'found_in 에 beps/frames/mdx breakdown 포함.' ), 'special_token_samples': special_samples, } output['meta']['totals'] = { 'beps_raw': output['beps'].get('raw_token_count', 0), 'beps_unique': output['beps'].get('unique_token_count', 0), 'frames_raw': sum(f['raw_token_count'] for f in output['frames'].values()), 'frames_unique_avg': ( sum(f['unique_token_count'] for f in output['frames'].values()) / total_frames if total_frames else 0 ), 'mdx_raw': sum(m['raw_token_count'] for m in output['mdx'].values()), 'mdx_unique_avg': ( sum(m['unique_token_count'] for m in output['mdx'].values()) / len(output['mdx']) if output['mdx'] else 0 ), 'corpus_unique': len(token_freq), 'corpus_total_occurrences': sum(token_freq.values()), } OUTPUT.write_text( yaml.safe_dump(output, allow_unicode=True, sort_keys=False, width=200), encoding='utf-8', ) # 화면 요약 t = output['meta']['totals'] print(f"[Step 2] Kiwi 토큰화 완료") print(f" BEPS: raw={t['beps_raw']}, unique={t['beps_unique']}") print(f" Frames: raw={t['frames_raw']}, unique_avg={t['frames_unique_avg']:.1f} " f"({total_frames}개 frame)") print(f" MDX: raw={t['mdx_raw']}, unique_avg={t['mdx_unique_avg']:.1f}") print(f" Corpus: unique={t['corpus_unique']}, occurrences={t['corpus_total_occurrences']}") print() print(f" 일반 token 후보 (frame_count >= {threshold}, 검토용): {len(general)}개") print(f" 상위 15개:") for g in general[:15]: print(f" {g['token']:12s} frames={g['frame_count']:2d}/{total_frames} " f"count={g['total_count']:3d} mdx={g['mdx_count']}") print() print(f" special_token_samples (Kiwi 쪼개짐 + source breakdown):") for target, info in output['corpus']['special_token_samples'].items(): tokens_alone = ' '.join(f"{f}({tag})" for f, tag in info['tokenize_alone']) fi = info['found_in'] print(f" {target:7s} alone=[{tokens_alone:35s}] " f"total={fi['total']:3d} beps={fi['beps']:3d} " f"frames={fi['frames']:3d} mdx={fi['mdx']:3d}") print() print(f"산출: {OUTPUT}") if __name__ == "__main__": main()