import os import unicodedata import pandas as pd import re dir1 = '/Users/hogyujhang/Library/CloudStorage/Dropbox/Business/Eroom AI Partners/Legal POC portfolio/민사에이전트-사해행위취소및대여금청구/POC - Law-aid Agent/프롬프트/prompt_updates_sequential/요건사실론/사건종류_요건사실론문서자료_done' dir2 = '/Users/hogyujhang/Library/CloudStorage/Dropbox/Business/Eroom AI Partners/Legal POC portfolio/민사에이전트-사해행위취소및대여금청구/POC - Law-aid Agent/프롬프트/prompt_updates_sequential/요건사실론/사건종류_요건사실론문서' output_excel = '/Users/hogyujhang/Library/CloudStorage/Dropbox/Business/Eroom AI Partners/Legal POC portfolio/민사에이전트-사해행위취소및대여금청구/POC - Law-aid Agent/프롬프트/prompt_updates_sequential/요건사실론/요건사실론_추가작업대상리스트.xlsx' # Extract ### from dir1 # File format: 요건사실문서자료_###_1.md set1 = set() for f in os.listdir(dir1): f_norm = unicodedata.normalize('NFC', f) if f_norm.endswith('.md'): # Extract between first '_' and last '_' parts = f_norm.split('_') if len(parts) >= 3: # Reconstruct in case ### has '_' inside name = '_'.join(parts[1:-1]) set1.add(name) # Extract xxx from dir2 # File format: 요건사실론_xxx_v1.md set2 = set() for f in os.listdir(dir2): f_norm = unicodedata.normalize('NFC', f) if f_norm.endswith('.md'): parts = f_norm.split('_') if len(parts) >= 3: name = '_'.join(parts[1:-1]) set2.add(name) diff = sorted(list(set1 - set2)) df = pd.DataFrame(diff, columns=['추가작업대상_사건종류']) df.to_excel(output_excel, index=False) print(f"Excel file created with {len(diff)} items.")