#!/usr/bin/env python3 """Generate information blocks from stage4_claim_packets.json and defense_rebuttal.md. Output block schema: { "claim_id": "C-###", "unit_type": "requirement|defense", "case_type": "", "target": {"collection": "...", "tenant": "..."}, "query_context": { "primary_text": "...", "elements": ["..."] | null, "issue_focus": "..." | null } } """ from __future__ import annotations import argparse import json import re from datetime import datetime from pathlib import Path from typing import Any CASE_TARGET_MAP: dict[str, dict[str, str]] = { "사해행위취소": {"collection": "Analyzed_Cases", "tenant": "Cases_Actio_Pauliana"}, "대여금": {"collection": "Past_Cases", "tenant": "Cases_Loan_Claim"}, "보증금": {"collection": "Past_Cases", "tenant": "Cases_Guarantee_Claim"}, "구상금": {"collection": "Past_Cases", "tenant": "Cases_Indemnity_Claim"}, } def _normalize_case_type(raw_claim_type: str) -> str: raw = (raw_claim_type or "").strip() if not raw: return "" first_phrase = raw.split()[0] normalized = re.sub(r"\([^)]*\)", "", first_phrase).strip() return normalized def _resolve_case_type( raw_claim_type: str, case_group: str | None, strict: bool, ) -> str: normalized = _normalize_case_type(raw_claim_type) if normalized in CASE_TARGET_MAP: return normalized # Fallback 1: keyword scan in full claim_type string. for known in CASE_TARGET_MAP: if known in (raw_claim_type or ""): return known # Fallback 2: minimal group-based rescue for known paulian shape. if (case_group or "").lower() == "paulian": return "사해행위취소" if strict: raise ValueError( f"Unmapped case_type after normalization: raw='{raw_claim_type}', normalized='{normalized}'" ) return normalized def _load_claim_packets(path: Path) -> list[dict[str, Any]]: data = json.loads(path.read_text(encoding="utf-8")) if isinstance(data, dict): packets = data.get("claim_packets") if isinstance(packets, list): return packets if isinstance(data, list): return data raise ValueError("Invalid stage4_claim_packets.json format: expected dict.claim_packets or list") def _is_separator_row(cells: list[str]) -> bool: return all(re.fullmatch(r":?-{3,}:?", c.strip()) for c in cells if c.strip()) def _parse_markdown_table(table_lines: list[str]) -> list[list[str]]: rows: list[list[str]] = [] for line in table_lines: s = line.strip() if not s.startswith("|"): continue cells = [c.strip() for c in s.strip("|").split("|")] if not cells: continue if _is_separator_row(cells): continue rows.append(cells) return rows def _extract_defense_items(defense_md_text: str) -> dict[str, dict[str, list[str]]]: pattern = re.compile(r"^###\s*(C-\d{3})\s*-\s*항변/재반박\s*(\d+)\s*$", re.MULTILINE) matches = list(pattern.finditer(defense_md_text)) by_claim: dict[str, list[tuple[int, str, str]]] = {} for idx, match in enumerate(matches): claim_id = match.group(1) rebuttal_index = int(match.group(2)) start = match.end() end = matches[idx + 1].start() if idx + 1 < len(matches) else len(defense_md_text) block = defense_md_text[start:end] table_lines: list[str] = [] started = False for line in block.splitlines(): if line.strip().startswith("|"): table_lines.append(line) started = True elif started: break rows = _parse_markdown_table(table_lines) row_map: dict[str, str] = {} for row in rows: if len(row) < 2: continue key = row[0].strip() value = row[1].strip() if key == "항목": continue row_map[key] = value expected_defense = row_map.get("상대방 예상 항변/주장", "").strip() core_issue = row_map.get("핵심 쟁점", "").strip() by_claim.setdefault(claim_id, []).append((rebuttal_index, expected_defense, core_issue)) out: dict[str, dict[str, list[str]]] = {} for claim_id, items in by_claim.items(): items.sort(key=lambda x: x[0]) out[claim_id] = { "expected_defenses": [t[1] for t in items if t[1]], "core_issues": [t[2] for t in items if t[2]], } return out def build_information_blocks( claim_packets: list[dict[str, Any]], defense_md_text: str, strict_case_type: bool = True, ) -> list[dict[str, Any]]: defense_map = _extract_defense_items(defense_md_text) blocks: list[dict[str, Any]] = [] for packet in claim_packets: claim_id = str(packet.get("claim_id", "")).strip() if not re.fullmatch(r"C-\d{3}", claim_id): continue raw_claim_type = str(packet.get("claim_type", "")).strip() case_group = str(packet.get("case_group", "")).strip() or None case_type = _resolve_case_type(raw_claim_type, case_group, strict_case_type) if case_type not in CASE_TARGET_MAP: if strict_case_type: raise ValueError( f"case_type '{case_type}' is not in DB matching table for claim_id={claim_id}" ) continue target = CASE_TARGET_MAP[case_type] elements_src = packet.get("elements") or [] element_names: list[str] = [] if isinstance(elements_src, list): for e in elements_src: if not isinstance(e, dict): continue if e.get("element_id") and e.get("element"): element_names.append(str(e["element"]).strip()) purpose_sentence = str(packet.get("purpose_sentence", "")).strip() requirement_block = { "claim_id": claim_id, "unit_type": "requirement", "case_type": case_type, "target": { "collection": target["collection"], "tenant": target["tenant"], }, "query_context": { "primary_text": purpose_sentence, "elements": element_names if element_names else None, "issue_focus": None, }, } blocks.append(requirement_block) defense_info = defense_map.get(claim_id, {"expected_defenses": [], "core_issues": []}) defense_block = { "claim_id": claim_id, "unit_type": "defense", "case_type": case_type, "target": { "collection": target["collection"], "tenant": target["tenant"], }, "query_context": { "primary_text": "\n".join(defense_info["expected_defenses"]).strip(), "elements": None, "issue_focus": "\n".join(defense_info["core_issues"]).strip(), }, } blocks.append(defense_block) return blocks def _parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser( description="Generate information blocks from stage4 claim packets and defense rebuttal." ) parser.add_argument( "--claim-packets", required=True, help="Path to stage4_claim_packets.json", ) parser.add_argument( "--defense-md", required=True, help="Path to defense_rebuttal.md", ) parser.add_argument( "--output", required=True, help="Path to output JSON file", ) parser.add_argument( "--allow-unmapped-case-type", action="store_true", default=False, help="Do not fail when normalized case_type cannot be mapped to DB matching table.", ) parser.add_argument( "--pretty", action="store_true", default=True, help="Write pretty-printed JSON (default: true).", ) return parser.parse_args() def main() -> None: args = _parse_args() claim_packets_path = Path(args.claim_packets) defense_md_path = Path(args.defense_md) output_path = Path(args.output) claim_packets = _load_claim_packets(claim_packets_path) defense_md_text = defense_md_path.read_text(encoding="utf-8") blocks = build_information_blocks( claim_packets=claim_packets, defense_md_text=defense_md_text, strict_case_type=not args.allow_unmapped_case_type, ) payload = { "meta": { "generated_at": datetime.now().isoformat(timespec="seconds"), "input_claim_packets": str(claim_packets_path), "input_defense_md": str(defense_md_path), "block_count": len(blocks), }, "information_blocks": blocks, } output_path.parent.mkdir(parents=True, exist_ok=True) if args.pretty: output_path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8") else: output_path.write_text(json.dumps(payload, ensure_ascii=False), encoding="utf-8") print(f"Generated {len(blocks)} blocks -> {output_path}") if __name__ == "__main__": main()