288 lines
9.1 KiB
Python
288 lines
9.1 KiB
Python
#!/usr/bin/env python3
|
|
"""Generate information blocks from stage4_claim_packets.json and defense_rebuttal.md.
|
|
|
|
Output block schema:
|
|
{
|
|
"claim_id": "C-###",
|
|
"unit_type": "requirement|defense",
|
|
"case_type": "<normalized>",
|
|
"target": {"collection": "...", "tenant": "..."},
|
|
"query_context": {
|
|
"primary_text": "...",
|
|
"elements": ["..."] | null,
|
|
"issue_focus": "..." | null
|
|
}
|
|
}
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import re
|
|
from datetime import datetime
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
|
|
CASE_TARGET_MAP: dict[str, dict[str, str]] = {
|
|
"사해행위취소": {"collection": "Analyzed_Cases", "tenant": "Cases_Actio_Pauliana"},
|
|
"대여금": {"collection": "Past_Cases", "tenant": "Cases_Loan_Claim"},
|
|
"보증금": {"collection": "Past_Cases", "tenant": "Cases_Guarantee_Claim"},
|
|
"구상금": {"collection": "Past_Cases", "tenant": "Cases_Indemnity_Claim"},
|
|
}
|
|
|
|
|
|
def _normalize_case_type(raw_claim_type: str) -> str:
|
|
raw = (raw_claim_type or "").strip()
|
|
if not raw:
|
|
return ""
|
|
first_phrase = raw.split()[0]
|
|
normalized = re.sub(r"\([^)]*\)", "", first_phrase).strip()
|
|
return normalized
|
|
|
|
|
|
def _resolve_case_type(
|
|
raw_claim_type: str,
|
|
case_group: str | None,
|
|
strict: bool,
|
|
) -> str:
|
|
normalized = _normalize_case_type(raw_claim_type)
|
|
if normalized in CASE_TARGET_MAP:
|
|
return normalized
|
|
|
|
# Fallback 1: keyword scan in full claim_type string.
|
|
for known in CASE_TARGET_MAP:
|
|
if known in (raw_claim_type or ""):
|
|
return known
|
|
|
|
# Fallback 2: minimal group-based rescue for known paulian shape.
|
|
if (case_group or "").lower() == "paulian":
|
|
return "사해행위취소"
|
|
|
|
if strict:
|
|
raise ValueError(
|
|
f"Unmapped case_type after normalization: raw='{raw_claim_type}', normalized='{normalized}'"
|
|
)
|
|
return normalized
|
|
|
|
|
|
def _load_claim_packets(path: Path) -> list[dict[str, Any]]:
|
|
data = json.loads(path.read_text(encoding="utf-8"))
|
|
if isinstance(data, dict):
|
|
packets = data.get("claim_packets")
|
|
if isinstance(packets, list):
|
|
return packets
|
|
if isinstance(data, list):
|
|
return data
|
|
raise ValueError("Invalid stage4_claim_packets.json format: expected dict.claim_packets or list")
|
|
|
|
|
|
def _is_separator_row(cells: list[str]) -> bool:
|
|
return all(re.fullmatch(r":?-{3,}:?", c.strip()) for c in cells if c.strip())
|
|
|
|
|
|
def _parse_markdown_table(table_lines: list[str]) -> list[list[str]]:
|
|
rows: list[list[str]] = []
|
|
for line in table_lines:
|
|
s = line.strip()
|
|
if not s.startswith("|"):
|
|
continue
|
|
cells = [c.strip() for c in s.strip("|").split("|")]
|
|
if not cells:
|
|
continue
|
|
if _is_separator_row(cells):
|
|
continue
|
|
rows.append(cells)
|
|
return rows
|
|
|
|
|
|
def _extract_defense_items(defense_md_text: str) -> dict[str, dict[str, list[str]]]:
|
|
pattern = re.compile(r"^###\s*(C-\d{3})\s*-\s*항변/재반박\s*(\d+)\s*$", re.MULTILINE)
|
|
matches = list(pattern.finditer(defense_md_text))
|
|
by_claim: dict[str, list[tuple[int, str, str]]] = {}
|
|
|
|
for idx, match in enumerate(matches):
|
|
claim_id = match.group(1)
|
|
rebuttal_index = int(match.group(2))
|
|
start = match.end()
|
|
end = matches[idx + 1].start() if idx + 1 < len(matches) else len(defense_md_text)
|
|
block = defense_md_text[start:end]
|
|
|
|
table_lines: list[str] = []
|
|
started = False
|
|
for line in block.splitlines():
|
|
if line.strip().startswith("|"):
|
|
table_lines.append(line)
|
|
started = True
|
|
elif started:
|
|
break
|
|
|
|
rows = _parse_markdown_table(table_lines)
|
|
row_map: dict[str, str] = {}
|
|
for row in rows:
|
|
if len(row) < 2:
|
|
continue
|
|
key = row[0].strip()
|
|
value = row[1].strip()
|
|
if key == "항목":
|
|
continue
|
|
row_map[key] = value
|
|
|
|
expected_defense = row_map.get("상대방 예상 항변/주장", "").strip()
|
|
core_issue = row_map.get("핵심 쟁점", "").strip()
|
|
|
|
by_claim.setdefault(claim_id, []).append((rebuttal_index, expected_defense, core_issue))
|
|
|
|
out: dict[str, dict[str, list[str]]] = {}
|
|
for claim_id, items in by_claim.items():
|
|
items.sort(key=lambda x: x[0])
|
|
out[claim_id] = {
|
|
"expected_defenses": [t[1] for t in items if t[1]],
|
|
"core_issues": [t[2] for t in items if t[2]],
|
|
}
|
|
return out
|
|
|
|
|
|
def build_information_blocks(
|
|
claim_packets: list[dict[str, Any]],
|
|
defense_md_text: str,
|
|
strict_case_type: bool = True,
|
|
) -> list[dict[str, Any]]:
|
|
defense_map = _extract_defense_items(defense_md_text)
|
|
blocks: list[dict[str, Any]] = []
|
|
|
|
for packet in claim_packets:
|
|
claim_id = str(packet.get("claim_id", "")).strip()
|
|
if not re.fullmatch(r"C-\d{3}", claim_id):
|
|
continue
|
|
|
|
raw_claim_type = str(packet.get("claim_type", "")).strip()
|
|
case_group = str(packet.get("case_group", "")).strip() or None
|
|
case_type = _resolve_case_type(raw_claim_type, case_group, strict_case_type)
|
|
if case_type not in CASE_TARGET_MAP:
|
|
if strict_case_type:
|
|
raise ValueError(
|
|
f"case_type '{case_type}' is not in DB matching table for claim_id={claim_id}"
|
|
)
|
|
continue
|
|
target = CASE_TARGET_MAP[case_type]
|
|
|
|
elements_src = packet.get("elements") or []
|
|
element_names: list[str] = []
|
|
if isinstance(elements_src, list):
|
|
for e in elements_src:
|
|
if not isinstance(e, dict):
|
|
continue
|
|
if e.get("element_id") and e.get("element"):
|
|
element_names.append(str(e["element"]).strip())
|
|
|
|
purpose_sentence = str(packet.get("purpose_sentence", "")).strip()
|
|
|
|
requirement_block = {
|
|
"claim_id": claim_id,
|
|
"unit_type": "requirement",
|
|
"case_type": case_type,
|
|
"target": {
|
|
"collection": target["collection"],
|
|
"tenant": target["tenant"],
|
|
},
|
|
"query_context": {
|
|
"primary_text": purpose_sentence,
|
|
"elements": element_names if element_names else None,
|
|
"issue_focus": None,
|
|
},
|
|
}
|
|
blocks.append(requirement_block)
|
|
|
|
defense_info = defense_map.get(claim_id, {"expected_defenses": [], "core_issues": []})
|
|
defense_block = {
|
|
"claim_id": claim_id,
|
|
"unit_type": "defense",
|
|
"case_type": case_type,
|
|
"target": {
|
|
"collection": target["collection"],
|
|
"tenant": target["tenant"],
|
|
},
|
|
"query_context": {
|
|
"primary_text": "\n".join(defense_info["expected_defenses"]).strip(),
|
|
"elements": None,
|
|
"issue_focus": "\n".join(defense_info["core_issues"]).strip(),
|
|
},
|
|
}
|
|
blocks.append(defense_block)
|
|
|
|
return blocks
|
|
|
|
|
|
def _parse_args() -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(
|
|
description="Generate information blocks from stage4 claim packets and defense rebuttal."
|
|
)
|
|
parser.add_argument(
|
|
"--claim-packets",
|
|
required=True,
|
|
help="Path to stage4_claim_packets.json",
|
|
)
|
|
parser.add_argument(
|
|
"--defense-md",
|
|
required=True,
|
|
help="Path to defense_rebuttal.md",
|
|
)
|
|
parser.add_argument(
|
|
"--output",
|
|
required=True,
|
|
help="Path to output JSON file",
|
|
)
|
|
parser.add_argument(
|
|
"--allow-unmapped-case-type",
|
|
action="store_true",
|
|
default=False,
|
|
help="Do not fail when normalized case_type cannot be mapped to DB matching table.",
|
|
)
|
|
parser.add_argument(
|
|
"--pretty",
|
|
action="store_true",
|
|
default=True,
|
|
help="Write pretty-printed JSON (default: true).",
|
|
)
|
|
return parser.parse_args()
|
|
|
|
|
|
def main() -> None:
|
|
args = _parse_args()
|
|
claim_packets_path = Path(args.claim_packets)
|
|
defense_md_path = Path(args.defense_md)
|
|
output_path = Path(args.output)
|
|
|
|
claim_packets = _load_claim_packets(claim_packets_path)
|
|
defense_md_text = defense_md_path.read_text(encoding="utf-8")
|
|
|
|
blocks = build_information_blocks(
|
|
claim_packets=claim_packets,
|
|
defense_md_text=defense_md_text,
|
|
strict_case_type=not args.allow_unmapped_case_type,
|
|
)
|
|
|
|
payload = {
|
|
"meta": {
|
|
"generated_at": datetime.now().isoformat(timespec="seconds"),
|
|
"input_claim_packets": str(claim_packets_path),
|
|
"input_defense_md": str(defense_md_path),
|
|
"block_count": len(blocks),
|
|
},
|
|
"information_blocks": blocks,
|
|
}
|
|
|
|
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
if args.pretty:
|
|
output_path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
|
|
else:
|
|
output_path.write_text(json.dumps(payload, ensure_ascii=False), encoding="utf-8")
|
|
|
|
print(f"Generated {len(blocks)} blocks -> {output_path}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|