Files

288 lines
9.1 KiB
Python

#!/usr/bin/env python3
"""Generate information blocks from stage4_claim_packets.json and defense_rebuttal.md.
Output block schema:
{
"claim_id": "C-###",
"unit_type": "requirement|defense",
"case_type": "<normalized>",
"target": {"collection": "...", "tenant": "..."},
"query_context": {
"primary_text": "...",
"elements": ["..."] | null,
"issue_focus": "..." | null
}
}
"""
from __future__ import annotations
import argparse
import json
import re
from datetime import datetime
from pathlib import Path
from typing import Any
CASE_TARGET_MAP: dict[str, dict[str, str]] = {
"사해행위취소": {"collection": "Analyzed_Cases", "tenant": "Cases_Actio_Pauliana"},
"대여금": {"collection": "Past_Cases", "tenant": "Cases_Loan_Claim"},
"보증금": {"collection": "Past_Cases", "tenant": "Cases_Guarantee_Claim"},
"구상금": {"collection": "Past_Cases", "tenant": "Cases_Indemnity_Claim"},
}
def _normalize_case_type(raw_claim_type: str) -> str:
raw = (raw_claim_type or "").strip()
if not raw:
return ""
first_phrase = raw.split()[0]
normalized = re.sub(r"\([^)]*\)", "", first_phrase).strip()
return normalized
def _resolve_case_type(
raw_claim_type: str,
case_group: str | None,
strict: bool,
) -> str:
normalized = _normalize_case_type(raw_claim_type)
if normalized in CASE_TARGET_MAP:
return normalized
# Fallback 1: keyword scan in full claim_type string.
for known in CASE_TARGET_MAP:
if known in (raw_claim_type or ""):
return known
# Fallback 2: minimal group-based rescue for known paulian shape.
if (case_group or "").lower() == "paulian":
return "사해행위취소"
if strict:
raise ValueError(
f"Unmapped case_type after normalization: raw='{raw_claim_type}', normalized='{normalized}'"
)
return normalized
def _load_claim_packets(path: Path) -> list[dict[str, Any]]:
data = json.loads(path.read_text(encoding="utf-8"))
if isinstance(data, dict):
packets = data.get("claim_packets")
if isinstance(packets, list):
return packets
if isinstance(data, list):
return data
raise ValueError("Invalid stage4_claim_packets.json format: expected dict.claim_packets or list")
def _is_separator_row(cells: list[str]) -> bool:
return all(re.fullmatch(r":?-{3,}:?", c.strip()) for c in cells if c.strip())
def _parse_markdown_table(table_lines: list[str]) -> list[list[str]]:
rows: list[list[str]] = []
for line in table_lines:
s = line.strip()
if not s.startswith("|"):
continue
cells = [c.strip() for c in s.strip("|").split("|")]
if not cells:
continue
if _is_separator_row(cells):
continue
rows.append(cells)
return rows
def _extract_defense_items(defense_md_text: str) -> dict[str, dict[str, list[str]]]:
pattern = re.compile(r"^###\s*(C-\d{3})\s*-\s*항변/재반박\s*(\d+)\s*$", re.MULTILINE)
matches = list(pattern.finditer(defense_md_text))
by_claim: dict[str, list[tuple[int, str, str]]] = {}
for idx, match in enumerate(matches):
claim_id = match.group(1)
rebuttal_index = int(match.group(2))
start = match.end()
end = matches[idx + 1].start() if idx + 1 < len(matches) else len(defense_md_text)
block = defense_md_text[start:end]
table_lines: list[str] = []
started = False
for line in block.splitlines():
if line.strip().startswith("|"):
table_lines.append(line)
started = True
elif started:
break
rows = _parse_markdown_table(table_lines)
row_map: dict[str, str] = {}
for row in rows:
if len(row) < 2:
continue
key = row[0].strip()
value = row[1].strip()
if key == "항목":
continue
row_map[key] = value
expected_defense = row_map.get("상대방 예상 항변/주장", "").strip()
core_issue = row_map.get("핵심 쟁점", "").strip()
by_claim.setdefault(claim_id, []).append((rebuttal_index, expected_defense, core_issue))
out: dict[str, dict[str, list[str]]] = {}
for claim_id, items in by_claim.items():
items.sort(key=lambda x: x[0])
out[claim_id] = {
"expected_defenses": [t[1] for t in items if t[1]],
"core_issues": [t[2] for t in items if t[2]],
}
return out
def build_information_blocks(
claim_packets: list[dict[str, Any]],
defense_md_text: str,
strict_case_type: bool = True,
) -> list[dict[str, Any]]:
defense_map = _extract_defense_items(defense_md_text)
blocks: list[dict[str, Any]] = []
for packet in claim_packets:
claim_id = str(packet.get("claim_id", "")).strip()
if not re.fullmatch(r"C-\d{3}", claim_id):
continue
raw_claim_type = str(packet.get("claim_type", "")).strip()
case_group = str(packet.get("case_group", "")).strip() or None
case_type = _resolve_case_type(raw_claim_type, case_group, strict_case_type)
if case_type not in CASE_TARGET_MAP:
if strict_case_type:
raise ValueError(
f"case_type '{case_type}' is not in DB matching table for claim_id={claim_id}"
)
continue
target = CASE_TARGET_MAP[case_type]
elements_src = packet.get("elements") or []
element_names: list[str] = []
if isinstance(elements_src, list):
for e in elements_src:
if not isinstance(e, dict):
continue
if e.get("element_id") and e.get("element"):
element_names.append(str(e["element"]).strip())
purpose_sentence = str(packet.get("purpose_sentence", "")).strip()
requirement_block = {
"claim_id": claim_id,
"unit_type": "requirement",
"case_type": case_type,
"target": {
"collection": target["collection"],
"tenant": target["tenant"],
},
"query_context": {
"primary_text": purpose_sentence,
"elements": element_names if element_names else None,
"issue_focus": None,
},
}
blocks.append(requirement_block)
defense_info = defense_map.get(claim_id, {"expected_defenses": [], "core_issues": []})
defense_block = {
"claim_id": claim_id,
"unit_type": "defense",
"case_type": case_type,
"target": {
"collection": target["collection"],
"tenant": target["tenant"],
},
"query_context": {
"primary_text": "\n".join(defense_info["expected_defenses"]).strip(),
"elements": None,
"issue_focus": "\n".join(defense_info["core_issues"]).strip(),
},
}
blocks.append(defense_block)
return blocks
def _parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Generate information blocks from stage4 claim packets and defense rebuttal."
)
parser.add_argument(
"--claim-packets",
required=True,
help="Path to stage4_claim_packets.json",
)
parser.add_argument(
"--defense-md",
required=True,
help="Path to defense_rebuttal.md",
)
parser.add_argument(
"--output",
required=True,
help="Path to output JSON file",
)
parser.add_argument(
"--allow-unmapped-case-type",
action="store_true",
default=False,
help="Do not fail when normalized case_type cannot be mapped to DB matching table.",
)
parser.add_argument(
"--pretty",
action="store_true",
default=True,
help="Write pretty-printed JSON (default: true).",
)
return parser.parse_args()
def main() -> None:
args = _parse_args()
claim_packets_path = Path(args.claim_packets)
defense_md_path = Path(args.defense_md)
output_path = Path(args.output)
claim_packets = _load_claim_packets(claim_packets_path)
defense_md_text = defense_md_path.read_text(encoding="utf-8")
blocks = build_information_blocks(
claim_packets=claim_packets,
defense_md_text=defense_md_text,
strict_case_type=not args.allow_unmapped_case_type,
)
payload = {
"meta": {
"generated_at": datetime.now().isoformat(timespec="seconds"),
"input_claim_packets": str(claim_packets_path),
"input_defense_md": str(defense_md_path),
"block_count": len(blocks),
},
"information_blocks": blocks,
}
output_path.parent.mkdir(parents=True, exist_ok=True)
if args.pretty:
output_path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
else:
output_path.write_text(json.dumps(payload, ensure_ascii=False), encoding="utf-8")
print(f"Generated {len(blocks)} blocks -> {output_path}")
if __name__ == "__main__":
main()