    - name: stage4_0_parallel-executor
      description: 'Stage 4 Phase A: 4개의 Python 스크립트를 DAG 기반으로 병렬 실행 (Direct MCP)'
      llm_provider: openai
      llm_model: gpt-4o
      tools:
        mcpServers:
          code-executor:
            type: streamable-http
            url: https://code-executor.mcp.eroomai.com/mcp
            description: Run scripts of programming languages
            headers:
              Authorization: Bearer rR8OXqWrVZA1gFEo8oWfBkw2XgpWoGrrspw5ObsxTCM=
      tasks:
      - task_name: run_index
        mcp: code-executor
        tool_name: run_code
        parameters:
          language: python
          code: |
            #!/usr/bin/env python3
            """
            stage4_index.py — Stage 4, Phase A-1: Deterministic Index Generator

            Reads 5 REQUIRED inputs via MCP localdocs and produces stage4_index.json.
            Runs inside code-executor Docker container.
            """

            import json
            import re
            import sys
            from datetime import datetime
            from typing import Any
            import httpx


            # ──────────────────────────────────────────────────────────────────────
            # 0-1) MCP localdocs helpers
            # ──────────────────────────────────────────────────────────────────────
            LOCALDOCS_URL = "http://mcp-localdocs:8012/mcp"
            HEADERS = {
                "Content-Type": "application/json",
                "Accept": "application/json, text/event-stream",
            }


            def parse_sse(text):
                for line in text.strip().split("\n"):
                    if line.startswith("data: "):
                        return json.loads(line[6:])
                try:
                    return json.loads(text)
                except Exception:
                    return None


            def call_tool(c, name, arguments, msg_id=10):
                r = c.post(LOCALDOCS_URL, json={
                    "jsonrpc": "2.0", "id": msg_id,
                    "method": "tools/call",
                    "params": {"name": name, "arguments": arguments}
                }, headers=HEADERS)
                result = parse_sse(r.text)
                if result and "result" in result:
                    return result
                print(f"Tool {name} error: {json.dumps(result)[:300]}",
                      file=sys.stderr)
                return result


            def read_doc(c, doc_name, msg_id=10):
                """Read a JSON document via MCP localdocs."""
                result = call_tool(c, "read_doc", {"doc_name": doc_name}, msg_id)
                if result and "result" in result:
                    text = result["result"]["content"][0]["text"]
                    if not text or not text.strip():
                        return None
                    try:
                        return json.loads(text)
                    except json.JSONDecodeError:
                        print(f"read_doc({doc_name}): JSON parse failed",
                              file=sys.stderr)
                        return None
                return None


            def read_doc_text(c, doc_name, msg_id=10):
                """Read a text/markdown document via MCP localdocs (no JSON parse)."""
                result = call_tool(c, "read_doc", {"doc_name": doc_name}, msg_id)
                if result and "result" in result:
                    text = result["result"]["content"][0]["text"]
                    return text if text and text.strip() else None
                return None


            # ──────────────────────────────────────────────────────────────────────
            # 1) evidence_index builder
            # ──────────────────────────────────────────────────────────────────────
            def build_evidence_index(evidence_data: list[dict],
                                    claim_evidence_map: dict[str, list[str]]) -> dict:
                """
                evidence_indexed.json →
                { "E-###": { title, doc_type, key_facts, related_claims } }
                """
                ev_to_claims: dict[str, list[str]] = {}
                for cid, ev_list in claim_evidence_map.items():
                    for eid in ev_list:
                        ev_to_claims.setdefault(eid, [])
                        if cid not in ev_to_claims[eid]:
                            ev_to_claims[eid].append(cid)

                index = {}
                for item in evidence_data:
                    eid = item.get("evidence_index", "")
                    if not eid:
                        continue
                    doc_type = _classify_doc_type(item.get("document_type", ""))
                    key_facts = _split_key_info(item.get("key_info", ""))
                    index[eid] = {
                        "title": item.get("title", ""),
                        "doc_type": doc_type,
                        "key_facts": key_facts,
                        "related_claims": sorted(ev_to_claims.get(eid, []))
                    }
                return index


            def _classify_doc_type(raw: str) -> str:
                if not raw:
                    return "기타"
                mapping = {
                    "등기부등본": "공문서", "등기사항": "공문서",
                    "법인등기부등본": "공문서", "주민등록": "공문서",
                    "법원문서": "판결/결정", "판결": "판결/결정",
                    "결정": "판결/결정", "배당표": "판결/결정",
                    "계약서": "처분문서", "약정서": "처분문서",
                    "감정평가서": "기타",
                    "영수증": "거래기록", "금융기록": "거래기록",
                    "확인서": "거래기록", "명세표": "거래기록",
                }
                for key, val in mapping.items():
                    if key in raw:
                        return val
                return "기타"


            def _split_key_info(key_info: str) -> list[str]:
                if not key_info:
                    return []
                parts = re.split(r",\s*(?![^()]*\))", key_info)
                return [p.strip() for p in parts if p.strip()]


            # ──────────────────────────────────────────────────────────────────────
            # 3) fact_index builder
            # ──────────────────────────────────────────────────────────────────────
            def build_fact_index(fact_ledger: list[dict],
                                claim_fact_map: dict[str, list[str]]) -> dict:
                """
                Fact_Ledger.json →
                { "F-###": { source_bo_id, summary, credibility,
                            evidence_refs, related_claims } }

                ★ 핵심 원칙: Fact_Ledger의 fact_id ↔ source_bo_id 매핑을 원본
                  그대로 보존한다. 재정렬·재할당하지 않는다.
                  summary는 Fact_Ledger의 action 필드로부터 결정론적으로 생성한다.
                """
                fact_to_claims: dict[str, list[str]] = {}
                for cid, fact_list in claim_fact_map.items():
                    for fid in fact_list:
                        fact_to_claims.setdefault(fid, [])
                        if cid not in fact_to_claims[fid]:
                            fact_to_claims[fid].append(cid)

                index = {}
                for item in fact_ledger:
                    fid = item.get("fact_id", "")
                    if not fid:
                        continue
                    bo_id = item.get("source_bo_id", "")
                    ev_refs = _extract_evidence_ids(item.get("evidence_refs", []))
                    summary = _build_fact_summary(item)

                    claims = sorted(set(
                        fact_to_claims.get(fid, []) +
                        fact_to_claims.get(bo_id, [])
                    ))

                    index[fid] = {
                        "source_bo_id": bo_id,
                        "summary": summary,
                        "credibility": item.get("credibility", "unknown"),
                        "evidence_refs": ev_refs,
                        "related_claims": claims
                    }
                return index


            def _extract_evidence_ids(refs: list) -> list[str]:
                result = []
                for ref in refs:
                    if not isinstance(ref, str):
                        continue
                    for m in re.findall(r"E-\d+", ref):
                        if m not in result:
                            result.append(m)
                return result


            def _build_fact_summary(item: dict) -> str:
                """Build concise summary from Fact_Ledger entry."""
                parties = item.get("parties", [])
                action = item.get("action", "")
                date = item.get("date", "")

                summary = ""
                if parties and action:
                    subject = parties[0]
                    if subject in action:
                        summary = action
                    else:
                        particle = "이" if _ends_with_consonant(subject) else "가"
                        summary = f"{subject}{particle} {action}"
                elif action:
                    summary = action

                if date:
                    summary = f"{summary}({date})"
                return summary


            def _ends_with_consonant(text: str) -> bool:
                if not text:
                    return False
                last = text[-1]
                if '가' <= last <= '힣':
                    return (ord(last) - 0xAC00) % 28 != 0
                return False


            # ──────────────────────────────────────────────────────────────────────
            # 4) goal_index builder
            # ──────────────────────────────────────────────────────────────────────
            def build_goal_index(client_goal: dict) -> dict:
                goals = {}
                idx = 1

                primary = client_goal.get("primary_goal", "")
                if primary:
                    goals[f"G-{idx:03d}"] = {"summary": primary}
                    idx += 1

                constraints = client_goal.get("constraints", [])
                if constraints:
                    goals[f"G-{idx:03d}"] = {"summary": ", ".join(constraints)}
                    idx += 1

                defendants = client_goal.get("parties", {}).get("defendants", [])
                has_pauliana = any(
                    "사해행위" in d.get("role", "") or "수익자" in d.get("role", "")
                    for d in defendants
                )
                if has_pauliana:
                    goals[f"G-{idx:03d}"] = {
                        "summary": "사해행위취소를 통한 책임재산 원상회복"
                    }
                    idx += 1

                return goals


            # ──────────────────────────────────────────────────────────────────────
            # 5) legal_elements_index builder
            # ──────────────────────────────────────────────────────────────────────
            def build_legal_elements_index(lrf_text: str) -> dict:
                index = {}
                q_sections = re.split(r"(?=^## Q-\d+)", lrf_text, flags=re.MULTILINE)

                for section in q_sections:
                    q_match = re.match(r"## (Q-\d+)\s*[—\-]\s*(.*)", section)
                    if not q_match:
                        continue
                    query_id = q_match.group(1)
                    claim_ids = _extract_applicable_claims(section)

                    points_match = re.search(
                        r"###\s*요건사실 핵심 포인트\s*\n(.*?)"
                        r"(?=\n###|\n---|\n## |\Z)",
                        section, re.DOTALL
                    )
                    if not points_match:
                        continue

                    point_pattern = re.compile(
                        r"^\s*(\d+)\.\s+\*\*(.+?)\*\*\s*[:：]\s*(.*?)"
                        r"(?=\n\s*\d+\.\s+\*\*|\Z)",
                        re.MULTILINE | re.DOTALL
                    )

                    for m in point_pattern.finditer(points_match.group(1)):
                        point_num = int(m.group(1))
                        element_label = m.group(2).strip()
                        description = re.sub(r"\s+", " ", m.group(3)).strip()

                        q_num = re.search(r"\d+", query_id).group()
                        element_id = f"LF-Q{q_num.zfill(3)}-P{point_num}"

                        index[element_id] = {
                            "query_id": query_id,
                            "element": element_label,
                            "description": description,
                            "applicable_claims": claim_ids or ["UNKNOWN"]
                        }

                return index


            def _extract_applicable_claims(section_text: str) -> list[str]:
                match = re.search(
                    r"적용\s*claim_id\s*\*?\*?\s*[:：]\s*(.*)", section_text)
                if not match:
                    return []
                raw = match.group(1).strip()
                claims: list[str] = []
                for rm in re.finditer(r"(C-\d+)\s*[~～]\s*(C-\d+)", raw):
                    s = int(re.search(r"\d+", rm.group(1)).group())
                    e = int(re.search(r"\d+", rm.group(2)).group())
                    for i in range(s, e + 1):
                        cid = f"C-{i:03d}"
                        if cid not in claims:
                            claims.append(cid)
                for cm in re.finditer(r"C-\d+", raw):
                    if cm.group() not in claims:
                        claims.append(cm.group())
                return sorted(claims)


            # ──────────────────────────────────────────────────────────────────────
            # 6) claim_index builder
            # ──────────────────────────────────────────────────────────────────────
            def build_claim_index(pre_claim_text: str,
                                  min_score: int = 5) -> tuple[dict, list[str]]:
                claims = {}
                ordered = []

                for row in _parse_ssot_table(pre_claim_text):
                    cid = row["claim_id"]
                    score = row["total_score"]
                    claims[cid] = {
                        "claim_type": row["claim_type"],
                        "total_score": score,
                        "eligible": score >= min_score,
                        "rank": row["rank"],
                        "plaintiff": "",
                        "defendant": "",
                        "summary": ""
                    }
                    if score >= min_score:
                        ordered.append(cid)

                for cid, detail in _parse_claim_details(pre_claim_text).items():
                    if cid in claims:
                        claims[cid]["summary"] = detail.get("summary", "")

                for cid, pmap in _parse_party_mappings(pre_claim_text).items():
                    if cid in claims:
                        claims[cid]["plaintiff"] = pmap.get("plaintiff", "")
                        claims[cid]["defendant"] = pmap.get("defendant", "")

                return claims, ordered


            def _parse_ssot_table(text: str) -> list[dict]:
                rows = []
                section_match = re.search(
                    r"##\s*2\.\s*청구권\s*우선순위\s*요약\s*\n(.*?)(?=\n---|\n##)",
                    text, re.DOTALL
                )
                if not section_match:
                    section_match = re.search(
                        r"(\|.*claim_id.*\|.*\n(?:\|.*\n)+)", text, re.DOTALL)
                if not section_match:
                    return rows

                header_line = None
                col_indices: dict[str, int] = {}

                for line in section_match.group(1).strip().split("\n"):
                    line = line.strip()
                    if not line.startswith("|"):
                        continue
                    cells = [c.strip() for c in line.split("|") if c.strip()]

                    if cells and all(re.match(r"^[-:]+$", c) for c in cells):
                        continue

                    if header_line is None and any(
                        "claim_id" in c.lower() for c in cells
                    ):
                        header_line = cells
                        for i, h in enumerate(cells):
                            hl = h.strip().lower()
                            if "순위" in hl or "rank" in hl:
                                col_indices["rank"] = i
                            elif "claim_id" in hl:
                                col_indices["claim_id"] = i
                            elif "청구권" in hl or "claim" in hl:
                                col_indices["claim_type"] = i
                            elif "총점" in hl or "total" in hl:
                                col_indices["total_score"] = i
                        continue

                    if header_line and len(cells) >= len(col_indices):
                        try:
                            cid = cells[col_indices.get("claim_id", 1)].strip()
                            if not re.match(r"C-\d+", cid):
                                continue
                            rr = cells[col_indices.get("rank", 0)].strip()
                            rank = (int(re.search(r"\d+", rr).group())
                                    if re.search(r"\d+", rr) else 0)
                            ct = cells[col_indices.get("claim_type", 2)].strip()
                            sr = cells[col_indices.get("total_score", -1)].strip()
                            score = (int(re.search(r"\d+", sr).group())
                                    if re.search(r"\d+", sr) else 0)
                            rows.append({"claim_id": cid, "rank": rank,
                                        "claim_type": ct, "total_score": score})
                        except (IndexError, ValueError, AttributeError):
                            continue
                return rows


            def _parse_claim_details(text: str) -> dict[str, dict]:
                details = {}

                for m in re.finditer(
                    r"###\s*\(\d+\)\s*(C-\d+)\s*[:：]\s*(.*?)\n"
                    r"(.*?)(?=\n###|\n---|\n##|\Z)", text, re.DOTALL
                ):
                    cid = m.group(1)
                    dt = m.group(3)

                    pm = re.search(
                        r"[-\*]\s*\*?\*?청구취지\*?\*?\s*[:：]\s*(.*?)"
                        r"(?=\n[-\*]|\n\n|\Z)", dt)
                    purport = pm.group(1).strip() if pm else ""

                    cm = re.search(
                        r"[-\*]\s*\*?\*?청구원인\*?\*?\s*[:：]\s*(.*?)"
                        r"(?=\n[-\*]|\n\n|\Z)", dt)
                    cause = cm.group(1).strip() if cm else ""

                    tags = []
                    ft = list(dict.fromkeys(re.findall(r"F-\d+", dt)))
                    et = list(dict.fromkeys(re.findall(r"E-\d+", dt)))
                    if ft:
                        tags.append(f"({', '.join(ft)})")
                    if et:
                        tags.append(f"({', '.join(et)})")

                    base = cause or purport
                    details[cid] = {
                        "summary": f"{base} {''.join(tags)}".strip() if base else ""
                    }

                s4 = re.search(
                    r"##\s*4\.\s*기타\s*청구권.*?\n(.*?)(?=\n---|\n##|\Z)",
                    text, re.DOTALL)
                if s4:
                    for line in s4.group(1).strip().split("\n"):
                        if not line.strip().startswith("|"):
                            continue
                        cells = [c.strip() for c in line.split("|") if c.strip()]
                        if len(cells) < 3:
                            continue
                        cm = re.match(r"C-\d+", cells[0])
                        if cm and cm.group() not in details:
                            cid = cm.group()
                            memo = cells[-1] if len(cells) > 2 else ""
                            ct = cells[1] if len(cells) > 1 else ""
                            tags = []
                            ft = list(dict.fromkeys(re.findall(r"F-\d+", line)))
                            et = list(dict.fromkeys(re.findall(r"E-\d+", line)))
                            if ft:
                                tags.append(f"({', '.join(ft)})")
                            if et:
                                tags.append(f"({', '.join(et)})")
                            details[cid] = {
                                "summary": f"{ct}: {memo} {''.join(tags)}".strip()
                            }
                return details


            def _parse_party_mappings(text: str) -> dict[str, dict]:
                mappings = {}
                section = re.search(
                    r"(?:###\s*5\.3|청구권별\s*매핑).*?\n(.*?)(?=\n---|\n##|\Z)",
                    text, re.DOTALL)
                if not section:
                    return mappings

                header_found = False
                p_col = d_col = cid_col = None

                for line in section.group(1).strip().split("\n"):
                    if not line.strip().startswith("|"):
                        continue
                    cells = [c.strip() for c in line.split("|") if c.strip()]

                    if cells and all(re.match(r"^[-:]+$", c) for c in cells):
                        continue

                    if not header_found:
                        for i, h in enumerate(cells):
                            if "claim_id" in h.lower():
                                cid_col = i
                            elif "원고" in h:
                                p_col = i
                            elif "피고" in h:
                                d_col = i
                        if cid_col is not None:
                            header_found = True
                        continue

                    if header_found and len(cells) > max(
                        filter(None, [cid_col, p_col, d_col]), default=0
                    ):
                        cm = re.match(
                            r"C-\d+", cells[cid_col] if cid_col is not None else "")
                        if cm:
                            mappings[cm.group()] = {
                                "plaintiff": (cells[p_col].strip()
                                              if p_col and p_col < len(cells) else ""),
                                "defendant": (cells[d_col].strip()
                                              if d_col and d_col < len(cells) else ""),
                            }
                return mappings


            # ──────────────────────────────────────────────────────────────────────
            # 7) Cross-reference maps: claim → facts, claim → evidence
            # ──────────────────────────────────────────────────────────────────────
            def build_claim_fact_evidence_maps(
                pre_claim_text: str,
                fact_ledger: list[dict]
            ) -> tuple[dict[str, list[str]], dict[str, list[str]]]:
                """
                Build claim_id → [fact_ids] and claim_id → [evidence_ids].
                Sources:
                  1) 청구전작업.md §3/§4 explicit references
                  2) evidence propagation from Fact_Ledger evidence_refs
                  3) Party-based heuristic for unassigned facts
                """
                claim_facts: dict[str, list[str]] = {}
                claim_evidence: dict[str, list[str]] = {}

                # ── Source 1: Explicit references ──
                for m in re.finditer(
                    r"###\s*\(\d+\)\s*(C-\d+)\s*[:：].*?\n"
                    r"(.*?)(?=\n###|\n---|\n##|\Z)",
                    pre_claim_text, re.DOTALL
                ):
                    cid = m.group(1)
                    body = m.group(2)
                    claim_facts[cid] = (
                        list(dict.fromkeys(re.findall(r"F-\d+", body))) +
                        list(dict.fromkeys(re.findall(r"bh\d+", body)))
                    )
                    claim_evidence[cid] = list(dict.fromkeys(
                        re.findall(r"E-\d+", body)))

                s4 = re.search(r"##\s*4\..*?\n(.*?)(?=\n---|\n##|\Z)",
                              pre_claim_text, re.DOTALL)
                if s4:
                    for line in s4.group(1).split("\n"):
                        cm = re.search(r"C-\d+", line)
                        if cm and cm.group() not in claim_facts:
                            cid = cm.group()
                            claim_facts[cid] = (
                                list(dict.fromkeys(re.findall(r"F-\d+", line))) +
                                list(dict.fromkeys(re.findall(r"bh\d+", line)))
                            )
                            claim_evidence[cid] = list(dict.fromkeys(
                                re.findall(r"E-\d+", line)))

                # ── Fact_Ledger lookups ──
                fact_to_evidence: dict[str, list[str]] = {}
                bh_to_fid: dict[str, str] = {}
                fid_to_item: dict[str, dict] = {}
                for item in fact_ledger:
                    fid = item.get("fact_id", "")
                    bo_id = item.get("source_bo_id", "")
                    fact_to_evidence[fid] = _extract_evidence_ids(
                        item.get("evidence_refs", []))
                    fid_to_item[fid] = item
                    if bo_id:
                        bh_to_fid[bo_id] = fid

                # ── Source 3: Party-based expansion ──
                claim_seed_parties: dict[str, set[str]] = {}
                for cid, fids in claim_facts.items():
                    pset: set[str] = set()
                    for fid_or_bh in fids:
                        actual = bh_to_fid.get(fid_or_bh, fid_or_bh)
                        if actual in fid_to_item:
                            pset.update(fid_to_item[actual].get("parties", []))
                    claim_seed_parties[cid] = pset

                assigned: set[str] = set()
                for fids in claim_facts.values():
                    for fob in fids:
                        assigned.add(bh_to_fid.get(fob, fob))

                # Generic parties: appearing in >50% of claims
                generic: set[str] = set()
                if claim_seed_parties:
                    pcc: dict[str, int] = {}
                    for pset in claim_seed_parties.values():
                        for p in pset:
                            pcc[p] = pcc.get(p, 0) + 1
                    thr = len(claim_seed_parties) * 0.5
                    generic = {p for p, c in pcc.items() if c > thr}

                for fid, item in fid_to_item.items():
                    if fid in assigned:
                        continue
                    fp = set(item.get("parties", []))
                    if not fp:
                        continue

                    best_claims: list[str] = []
                    best_score = 0
                    for cid, sp in claim_seed_parties.items():
                        overlap = fp & sp
                        ngo = overlap - generic
                        score = len(ngo) * 2 + len(overlap)
                        if score > best_score:
                            best_score = score
                            best_claims = [cid]
                        elif score == best_score and score > 0:
                            best_claims.append(cid)

                    if best_score >= 2:
                        for cid in best_claims:
                            claim_facts.setdefault(cid, [])
                            if fid not in claim_facts[cid]:
                                claim_facts[cid].append(fid)
                                assigned.add(fid)

                # ── Propagate evidence ──
                for cid, fids in claim_facts.items():
                    for fob in fids:
                        actual = bh_to_fid.get(fob, fob) if fob.startswith("bh") else fob
                        for eid in fact_to_evidence.get(actual, []):
                            if eid not in claim_evidence.get(cid, []):
                                claim_evidence.setdefault(cid, []).append(eid)

                return claim_facts, claim_evidence


            # ──────────────────────────────────────────────────────────────────────
            # 8) parties & procedural_structures
            # ──────────────────────────────────────────────────────────────────────
            def build_parties(pre_claim_text: str, client_goal: dict) -> dict:
                parties: dict[str, list[str]] = {
                    "plaintiffs": [], "defendants": [], "excluded": []
                }

                ps = re.search(
                    r"###\s*5\.1\s*원고\s*\n(.*?)(?=\n###|\n---|\n##|\Z)",
                    pre_claim_text, re.DOTALL)
                if ps:
                    for line in ps.group(1).split("\n"):
                        if "|" in line and "확정" in line:
                            cells = [c.strip() for c in line.split("|") if c.strip()]
                            if cells and not re.match(r"^[-:]+$", cells[0]):
                                n = cells[0].strip()
                                if n and n not in parties["plaintiffs"] and "원고" not in n:
                                    parties["plaintiffs"].append(n)

                ds = re.search(
                    r"###\s*5\.2\s*피고\s*\n(.*?)(?=\n###|\n---|\n##|\Z)",
                    pre_claim_text, re.DOTALL)
                if ds:
                    for line in ds.group(1).split("\n"):
                        if "|" not in line:
                            continue
                        cells = [c.strip() for c in line.split("|") if c.strip()]
                        if len(cells) < 2 or re.match(r"^[-:]+$", cells[0]):
                            continue
                        n = cells[0].strip()
                        if "피고" in n or not n:
                            continue
                        st = cells[1].strip() if len(cells) > 1 else ""
                        if "제외" in st:
                            if n not in parties["excluded"]:
                                parties["excluded"].append(n)
                        elif "확정" in st or "후보" in st:
                            if n not in parties["defendants"]:
                                parties["defendants"].append(n)

                if not parties["plaintiffs"]:
                    for p in client_goal.get("parties", {}).get("plaintiffs", []):
                        parties["plaintiffs"].append(p.get("name", ""))
                if not parties["defendants"]:
                    for d in client_goal.get("parties", {}).get("defendants", []):
                        n = d.get("name", "")
                        st = d.get("asset_status", "")
                        if "재산 전무" in st or "폐업" in d.get("status", ""):
                            parties["excluded"].append(n)
                        else:
                            parties["defendants"].append(n)

                return parties


            def extract_procedural_structures(pre_claim_text: str) -> list[str]:
                structures: list[str] = []
                keywords = [
                    "단순병합", "예비적병합", "선택적병합",
                    "공동소송", "필수적공동소송", "통상공동소송",
                    "반소", "반소가능성", "소송고지", "보조참가", "채권자대위"
                ]

                s7 = re.search(
                    r"##\s*7\.\s*청구방식.*?\n(.*?)(?=\n---|\n##|\Z)",
                    pre_claim_text, re.DOTALL)
                scan = s7.group(1) if s7 else ""

                s9 = re.search(
                    r"##\s*9\.\s*후속\s*단계.*?\n(.*?)(?=\n---|\n##|\Z)",
                    pre_claim_text, re.DOTALL)
                if s9:
                    scan += "\n" + s9.group(1)

                for kw in keywords:
                    if kw in scan:
                        structures.append(kw)

                if s7:
                    sm = re.search(
                        r"[-\*]\s*\*?\*?구조\*?\*?\s*[:：]\s*(.*?)(?:\n|$)",
                        s7.group(1))
                    if sm:
                        for kw in keywords:
                            if kw in sm.group(1) and kw not in structures:
                                structures.append(kw)

                return structures


            # ══════════════════════════════════════════════════════════════════════
            # 9) POST-GENERATION INTEGRITY VALIDATION
            # ══════════════════════════════════════════════════════════════════════

            class ValidationReport:
                """Collects and reports validation findings."""

                def __init__(self):
                    self.errors: list[str] = []
                    self.warnings: list[str] = []

                def error(self, msg: str):
                    self.errors.append(msg)

                def warn(self, msg: str):
                    self.warnings.append(msg)

                @property
                def ok(self) -> bool:
                    return len(self.errors) == 0

                def print_report(self):
                    if self.ok and not self.warnings:
                        print("[VALIDATE] ✓ All integrity checks passed.",
                              file=sys.stderr)
                        return
                    if self.errors:
                        print(f"[VALIDATE] ✗ {len(self.errors)} error(s):",
                              file=sys.stderr)
                        for e in self.errors:
                            print(f"  ERROR: {e}", file=sys.stderr)
                    if self.warnings:
                        print(f"[VALIDATE] △ {len(self.warnings)} warning(s):",
                              file=sys.stderr)
                        for w in self.warnings:
                            print(f"  WARN:  {w}", file=sys.stderr)


            def validate_index(result: dict, fact_ledger: list[dict],
                              bo_data: list[dict] | None = None) -> ValidationReport:
                """
                Post-generation integrity checks:

                V1. fact_index fact_id/source_bo_id ↔ Fact_Ledger alignment
                V2. fact_index content ↔ BO.json content alignment (if available)
                V3. fact_index evidence_refs → evidence_index referential integrity
                V4. No duplicate source_bo_ids in fact_index
                V5. legal_elements → claim_index referential integrity
                V6. All Fact_Ledger entries present in fact_index (completeness)
                V7. Evidence completeness (all referenced E-### exist in index)
                """
                rpt = ValidationReport()

                fi = result.get("fact_index", {})
                ei = result.get("evidence_index", {})
                ci = result.get("claim_index", {})
                le = result.get("legal_elements_index", {})

                fl_by_id = {it["fact_id"]: it for it in fact_ledger if "fact_id" in it}

                # ── V1: fact_index ↔ Fact_Ledger alignment ──
                for fid, entry in fi.items():
                    bo_id = entry.get("source_bo_id", "")
                    if fid not in fl_by_id:
                        rpt.error(f"V1: fact_index[{fid}] not in Fact_Ledger")
                        continue
                    fl = fl_by_id[fid]
                    expected_bo = fl.get("source_bo_id", "")
                    if bo_id != expected_bo:
                        rpt.error(
                            f"V1: fact_index[{fid}].source_bo_id='{bo_id}' "
                            f"≠ Fact_Ledger='{expected_bo}'")

                    # Content consistency: summary must derive from FL action
                    fl_action = fl.get("action", "")
                    summary = entry.get("summary", "")
                    if fl_action and fl_action not in summary:
                        core = fl_action[:15]
                        if core not in summary:
                            rpt.warn(
                                f"V1: fact_index[{fid}].summary mismatch "
                                f"('{summary[:35]}…' vs FL.action='{fl_action[:35]}…')")

                # ── V2: BO.json content alignment (optional) ──
                if bo_data is not None:
                    bo_by_id = {it["id"]: it for it in bo_data if "id" in it}

                    for fid, entry in fi.items():
                        bo_id = entry.get("source_bo_id", "")
                        if not bo_id:
                            continue
                        if bo_id not in bo_by_id:
                            rpt.error(
                                f"V2: source_bo_id='{bo_id}' (in {fid}) "
                                f"not found in BO.json")
                            continue

                        bo_action = bo_by_id[bo_id].get("Action", "")
                        summary = entry.get("summary", "")

                        if bo_action:
                            # Character-set overlap ratio for content alignment
                            bo_chars = {c for c in bo_action
                                        if c.strip() and c not in "을를에의이가"}
                            sm_chars = {c for c in summary if c.strip()}
                            ratio = (len(bo_chars & sm_chars) /
                                    max(len(bo_chars), 1))

                            if ratio < 0.3:
                                rpt.error(
                                    f"V2: {fid}(bo={bo_id}) content mismatch "
                                    f"(overlap {ratio:.0%})\n"
                                    f"      BO : '{bo_action[:50]}'\n"
                                    f"      IDX: '{summary[:50]}'")

                # ── V3: evidence_refs → evidence_index ──
                for fid, entry in fi.items():
                    for eref in entry.get("evidence_refs", []):
                        if eref not in ei:
                            rpt.warn(
                                f"V3: {fid}.evidence_refs → '{eref}' "
                                f"not in evidence_index")

                # ── V4: No duplicate source_bo_ids ──
                seen_bo: dict[str, str] = {}
                for fid, entry in fi.items():
                    bo = entry.get("source_bo_id", "")
                    if not bo:
                        continue
                    if bo in seen_bo:
                        rpt.error(
                            f"V4: Duplicate source_bo_id '{bo}' "
                            f"in {fid} and {seen_bo[bo]}")
                    seen_bo[bo] = fid

                # ── V5: legal_elements → claim_index ──
                for le_id, le_entry in le.items():
                    for cid in le_entry.get("applicable_claims", []):
                        if cid != "UNKNOWN" and cid not in ci:
                            rpt.warn(
                                f"V5: legal_elements[{le_id}] → '{cid}' "
                                f"not in claim_index")

                # ── V6: Fact_Ledger completeness ──
                for item in fact_ledger:
                    fid = item.get("fact_id", "")
                    if fid and fid not in fi:
                        rpt.warn(f"V6: Fact_Ledger[{fid}] missing from fact_index")

                # ── V7: Evidence completeness ──
                all_erefs: set[str] = set()
                for fe in fi.values():
                    all_erefs.update(fe.get("evidence_refs", []))
                for eid in all_erefs:
                    if eid not in ei:
                        rpt.warn(f"V7: '{eid}' referenced but not in evidence_index")

                return rpt


            # ──────────────────────────────────────────────────────────────────────
            # 10) Main assembly (data-only, no file I/O)
            # ──────────────────────────────────────────────────────────────────────
            def build_stage4_index_from_data(
                evidence_data: list[dict],
                fact_ledger: list[dict],
                client_goal: dict,
                lrf_text: str,
                pre_claim_text: str,
                bo_data: list[dict] | None = None,
                top_n: int = 8,
                min_score: int = 5,
                do_validate: bool = False,
                strict: bool = False,
            ) -> dict:
                """Assemble stage4_index.json from pre-loaded data (no file I/O)."""
                cfm, cem = build_claim_fact_evidence_maps(pre_claim_text, fact_ledger)

                evidence_index = build_evidence_index(evidence_data, cem)
                fact_index = build_fact_index(fact_ledger, cfm)
                goal_index = build_goal_index(client_goal)
                legal_elements_index = build_legal_elements_index(lrf_text)
                claim_index, ordered = build_claim_index(pre_claim_text, min_score)

                result = {
                    "meta": {
                        "stage": "4",
                        "generated_at": datetime.now().strftime("%Y-%m-%d"),
                        "source_files": [
                            "legally_required_facts_information.md",
                            "청구전작업.md", "Fact_Ledger.json",
                            "evidence_indexed.json", "client_goal.json"
                        ]
                    },
                    "evidence_index": evidence_index,
                    "fact_index": fact_index,
                    "goal_index": goal_index,
                    "legal_elements_index": legal_elements_index,
                    "claim_index": claim_index,
                    "top_n_claims": ordered[:top_n],
                    "parties": build_parties(pre_claim_text, client_goal),
                    "procedural_structures": extract_procedural_structures(
                        pre_claim_text)
                }

                if do_validate:
                    rpt = validate_index(result, fact_ledger, bo_data)
                    rpt.print_report()
                    if strict and not rpt.ok:
                        raise RuntimeError(
                            f"Strict validation failed: {len(rpt.errors)} error(s)")

                return result


            def _log(msg: str):
                """진행/디버그 메시지 → stderr (stdout은 결과 JSON 전용)."""
                print(msg, file=sys.stderr)


            # ──────────────────────────────────────────────────────────────────────
            # 11) Entry point — MCP localdocs mode
            # ──────────────────────────────────────────────────────────────────────
            def main():
                top_n = 8
                min_score = 5
                do_validate = True
                strict = False
                output_name = "stage4_index.json"

                with httpx.Client(timeout=60) as c:
                    # ===== 1) localdocs 연결 =====
                    r = c.post(LOCALDOCS_URL, json={
                        "jsonrpc": "2.0", "id": 1, "method": "initialize",
                        "params": {
                            "protocolVersion": "2025-03-26",
                            "capabilities": {},
                            "clientInfo": {"name": "stage4-index-gen", "version": "1.0"}
                        }
                    }, headers=HEADERS)
                    sid = r.headers.get("mcp-session-id")
                    if sid:
                        HEADERS["mcp-session-id"] = sid
                    c.post(LOCALDOCS_URL, json={
                        "jsonrpc": "2.0", "method": "notifications/initialized"
                    }, headers=HEADERS)
                    _log(f"1) Connected to localdocs (session: {sid})")

                    # 도구 목록 확인
                    r = c.post(LOCALDOCS_URL, json={
                        "jsonrpc": "2.0", "id": 2, "method": "tools/list"
                    }, headers=HEADERS)
                    tools_result = parse_sse(r.text)
                    tools = (tools_result.get("result", {}).get("tools", [])
                            if tools_result else [])
                    _log(f"   Tools: {[t['name'] for t in tools]}")

                    # write 도구 찾기
                    write_tool = next(
                        (t for t in tools if "write" in t["name"]), None)
                    if write_tool:
                        props = write_tool.get("inputSchema", {}).get("properties", {})
                        required = write_tool.get("inputSchema", {}).get("required", [])
                        _log(f"   Write tool: {write_tool['name']}, "
                            f"params: {list(props.keys())}, required: {required}")

                    # 문서 목록
                    docs = call_tool(c, "list_docs", {}, 3)
                    if docs and "result" in docs:
                        _log(f"   Docs: {docs['result']['content'][0]['text'][:500]}")

                    # ===== 2) 파일 로딩 (MCP read_doc) =====
                    _log("\n2) Loading input files via MCP...")

                    evidence_data = read_doc(c, "evidence_indexed.json", 10)
                    fact_ledger = read_doc(c, "Fact_Ledger.json", 11)
                    if fact_ledger is None:
                        fact_ledger = read_doc(c, "fact_ledger.json", 12)
                    client_goal = read_doc(c, "client_goal.json", 13)
                    lrf_text = read_doc_text(
                        c, "legally_required_facts_information.md", 14)
                    pre_claim_text = read_doc_text(c, "청구전작업.md", 15)

                    # Optional
                    bo_data = read_doc(c, "BO.json", 16)

                    # 로딩 검증
                    required_files = {
                        "evidence_indexed.json": evidence_data,
                        "Fact_Ledger.json": fact_ledger,
                        "client_goal.json": client_goal,
                        "legally_required_facts_information.md": lrf_text,
                        "청구전작업.md": pre_claim_text,
                    }
                    for name, data in required_files.items():
                        if data is None:
                            raise RuntimeError(
                                f"Failed to load required file: {name}")
                        size = len(data) if hasattr(data, '__len__') else '?'
                        _log(f"   {name}: loaded "
                            f"({type(data).__name__}, len={size})")

                    if bo_data:
                        _log(f"   BO.json: loaded ({len(bo_data)} entries)")
                    else:
                        _log("   BO.json: not found (optional, skipping)")

                    # ===== 3) 인덱스 생성 =====
                    _log("\n3) Building stage4_index...")
                    result = build_stage4_index_from_data(
                        evidence_data=evidence_data,
                        fact_ledger=fact_ledger,
                        client_goal=client_goal,
                        lrf_text=lrf_text,
                        pre_claim_text=pre_claim_text,
                        bo_data=bo_data,
                        top_n=top_n,
                        min_score=min_score,
                        do_validate=do_validate,
                        strict=strict,
                    )

                    # ===== 4) 결과 저장 (MCP write_doc + stdout) =====
                    output_json = json.dumps(result, ensure_ascii=False, indent=2)

                    if write_tool:
                        write_result = call_tool(c, write_tool["name"], {
                            "path": output_name,
                            "content": output_json,
                        }, 20)
                        if write_result and "result" in write_result:
                            _log(f"\n4) Written to localdocs: {output_name}")
                        else:
                            _log("\n4) Write to localdocs failed")
                    else:
                        _log("\n4) No write tool available")

                    # 항상 stdout으로 출력 (code-executor가 캡처)
                    print(output_json)


            if __name__ == "__main__":
                main()

          requirements: "httpx"
          network: "agent-network"
          timeout: 120

      - task_name: run_compute_inputs
        mcp: code-executor
        tool_name: run_code
        parameters:
          language: python
          code: |
            #!/usr/bin/env python3
            """
            stage4_compute_inputs.py — Stage 4, Phase A-3: Deterministic Compute Input Extractor

            Reads inputs via MCP localdocs and produces stage4_compute_inputs.json.
            Runs inside code-executor Docker container.
            """

            import json
            import re
            import sys
            from datetime import datetime, date
            from typing import Any, Optional
            import httpx


            # ══════════════════════════════════════════════════════════════════════════════
            # 0) MCP localdocs helpers
            # ══════════════════════════════════════════════════════════════════════════════
            LOCALDOCS_URL = "http://mcp-localdocs:8012/mcp"
            HEADERS = {
                "Content-Type": "application/json",
                "Accept": "application/json, text/event-stream",
            }


            def parse_sse(text):
                for line in text.strip().split("\n"):
                    if line.startswith("data: "):
                        return json.loads(line[6:])
                try:
                    return json.loads(text)
                except Exception:
                    return None


            def call_tool(c, name, arguments, msg_id=10):
                r = c.post(LOCALDOCS_URL, json={
                    "jsonrpc": "2.0", "id": msg_id,
                    "method": "tools/call",
                    "params": {"name": name, "arguments": arguments}
                }, headers=HEADERS)
                result = parse_sse(r.text)
                if result and "result" in result:
                    return result
                print(f"Tool {name} error: {json.dumps(result)[:300]}",
                      file=sys.stderr)
                return result


            def read_doc(c, doc_name, msg_id=10):
                """Read a JSON document via MCP localdocs."""
                result = call_tool(c, "read_doc", {"doc_name": doc_name}, msg_id)
                if result and "result" in result:
                    text = result["result"]["content"][0]["text"]
                    if not text or not text.strip():
                        return None
                    try:
                        return json.loads(text)
                    except json.JSONDecodeError:
                        print(f"read_doc({doc_name}): JSON parse failed",
                              file=sys.stderr)
                        return None
                return None


            def read_doc_text(c, doc_name, msg_id=10):
                """Read a text/markdown document via MCP localdocs (no JSON parse)."""
                result = call_tool(c, "read_doc", {"doc_name": doc_name}, msg_id)
                if result and "result" in result:
                    text = result["result"]["content"][0]["text"]
                    return text if text and text.strip() else None
                return None


            # ══════════════════════════════════════════════════════════════════════════════
            # 2) Amount Parsing Utilities
            # ══════════════════════════════════════════════════════════════════════════════
            # Korean number unit map
            _KO_UNITS = {
                "원": 1,
                "만원": 10_000,
                "십만원": 100_000,
                "백만원": 1_000_000,
                "천만원": 10_000_000,
                "억원": 100_000_000,
                "억": 100_000_000,
                "만": 10_000,
                "천": 1_000,
                "백": 100,
            }

            def parse_korean_amount(text: str) -> Optional[int]:
                """
                Parse a Korean currency string into an integer value.
                Handles patterns like: "10억원", "3억원", "2억 2천만원", "4억 3천만원",
                "1억원", "3천만원", "5천만원", "2천만원", "3천만원", "약 3억원"
                Also handles: "채권최고액 15억원", "보증금 1억원"
                Returns None if unparseable.
                """
                if not text:
                    return None

                # Remove common prefixes
                text = re.sub(r'(약|채권최고액|보증금|보증원금|보증한도|대금|매매대금)\s*', '', text.strip())
                # Remove parenthetical content
                text = re.sub(r'\(.*?\)', '', text).strip()

                # Try direct numeric (e.g. "300000000")
                m = re.match(r'^(\d[\d,]+)\s*원?$', text)
                if m:
                    return int(m.group(1).replace(",", ""))

                # Pattern: "X억 Y천만원", "X억원", "X천만원", etc.
                total = 0
                remaining = text

                # Extract 억
                m_eok = re.search(r'(\d+(?:\.\d+)?)\s*억', remaining)
                if m_eok:
                    total += int(float(m_eok.group(1)) * 100_000_000)
                    remaining = remaining[m_eok.end():]

                # Extract 천만
                m_cheonman = re.search(r'(\d+(?:\.\d+)?)\s*천만', remaining)
                if m_cheonman:
                    total += int(float(m_cheonman.group(1)) * 10_000_000)
                    remaining = remaining[m_cheonman.end():]

                # Extract 백만
                m_baekman = re.search(r'(\d+(?:\.\d+)?)\s*백만', remaining)
                if m_baekman:
                    total += int(float(m_baekman.group(1)) * 1_000_000)
                    remaining = remaining[m_baekman.end():]

                # Extract 만
                m_man = re.search(r'(\d+(?:\.\d+)?)\s*만', remaining)
                if m_man:
                    total += int(float(m_man.group(1)) * 10_000)
                    remaining = remaining[m_man.end():]

                # Extract 천 (standalone, not 천만)
                m_cheon = re.search(r'(\d+(?:\.\d+)?)\s*천(?!만)', remaining)
                if m_cheon:
                    total += int(float(m_cheon.group(1)) * 1_000)
                    remaining = remaining[m_cheon.end():]

                if total > 0:
                    return total

                # Fallback: try extracting a decimal with unit
                m_decimal = re.search(r'(\d+(?:\.\d+)?)\s*억', text)
                if m_decimal:
                    return int(float(m_decimal.group(1)) * 100_000_000)

                return None


            def parse_rate_from_text(text: str) -> list[dict]:
                """
                Extract interest rate candidates from text.
                Returns list of {"type": "약정이율"|"지연손해금율", "annual_pct": float, "raw": str}

                Handles patterns:
                  - "이자 월0.5%"  → annual 6%
                  - "지연손해금 월1%" → annual 12%
                  - "연 12%", "연이율 6%", "이자율 5%"
                  - "연체이율 연 24%"
                """
                rates = []

                # Monthly rate patterns
                for m in re.finditer(r'(이자|지연손해금|연체이자|연체이율|이율)\s*월\s*(\d+(?:\.\d+)?)\s*%', text):
                    label = m.group(1)
                    monthly = float(m.group(2))
                    annual = monthly * 12
                    rtype = "지연손해금율" if "지연" in label or "연체" in label else "약정이율"
                    rates.append({
                        "type": rtype,
                        "annual_pct": min(annual, 24.0),  # 24% cap
                        "raw": m.group(0)
                    })

                # Annual rate patterns
                for m in re.finditer(
                    r'(약정이율|약정이자|이자율?|지연손해금율?|연체이율?|법정이율|연이율)\s*'
                    r'(?:연\s*)?(\d+(?:\.\d+)?)\s*%', text
                ):
                    label = m.group(1)
                    annual = float(m.group(2))
                    rtype = "지연손해금율" if "지연" in label or "연체" in label else "약정이율"
                    # Avoid duplicates already captured as monthly
                    if not any(r["annual_pct"] == annual for r in rates):
                        rates.append({
                            "type": rtype,
                            "annual_pct": min(annual, 24.0),
                            "raw": m.group(0)
                        })

                return rates


            def parse_date_str(date_str: str) -> Optional[str]:
                """Normalise various date formats to ISO YYYY-MM-DD. Returns None on failure."""
                if not date_str:
                    return None
                # Already ISO
                m = re.match(r'^(\d{4})-(\d{1,2})-(\d{1,2})$', str(date_str).strip())
                if m:
                    return f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
                # Korean: 2017.9.25 or 2017. 9. 25
                m = re.match(r'(\d{4})\s*[.\-/]\s*(\d{1,2})\s*[.\-/]\s*(\d{1,2})', str(date_str))
                if m:
                    return f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
                return None


            def extract_dates_from_text(text: str) -> list[str]:
                """Extract all date strings from a text, return as ISO dates."""
                dates = []
                for m in re.finditer(r'(\d{4})\s*[.\-/]\s*(\d{1,2})\s*[.\-/]\s*(\d{1,2})', text):
                    iso = f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
                    if iso not in dates:
                        dates.append(iso)
                return dates


            # ══════════════════════════════════════════════════════════════════════════════
            # 3) Claim-Type Classification for Rate Determination
            # ══════════════════════════════════════════════════════════════════════════════
            _PAULIAN_KEYWORDS = ["사해행위", "채권자취소", "취소", "원상회복"]
            _COMMERCIAL_KEYWORDS = ["상사", "회사", "주식회사", "상법"]
            _MONETARY_KEYWORDS = ["대여금", "구상금", "대출", "보증채무", "이행"]

            def _is_paulian_claim(claim_type: str) -> bool:
                return any(kw in claim_type for kw in _PAULIAN_KEYWORDS)

            def _is_monetary_claim(claim_type: str) -> bool:
                return any(kw in claim_type for kw in _MONETARY_KEYWORDS)

            def _classify_rate_basis(claim_type: str, plaintiff: str) -> dict:
                """
                Determine the statutory rate basis for a claim.
                Returns {"basis": str, "annual_pct": float, "note": str}
                """
                # 사해행위취소 → no delay damages on the cancellation itself
                if _is_paulian_claim(claim_type):
                    return {
                        "basis": "사해행위취소(형성의소)",
                        "annual_pct": None,
                        "note": "사해행위취소 자체는 금전청구가 아니므로 지연손해금 산정 불요. "
                                "원상회복으로 가액배상 시에만 법정이율 적용 가능."
                    }

                # Check if commercial (상사) → 6%
                is_commercial = any(kw in plaintiff for kw in ["주식회사", "회사", "캐피탈", "보증"])
                if is_commercial:
                    return {
                        "basis": "상사법정이율(상법 제54조)",
                        "annual_pct": 6.0,
                        "note": "상인 간 금전채무 → 연 6%"
                    }

                # Default civil → 5%
                return {
                    "basis": "민사법정이율(민법 제379조)",
                    "annual_pct": 5.0,
                    "note": "민사 금전채무 → 연 5%"
                }


            # ══════════════════════════════════════════════════════════════════════════════
            # 4) Principal Extraction per Claim
            # ══════════════════════════════════════════════════════════════════════════════
            def extract_principal_candidates(
                claim_id: str,
                claim_data: dict,
                fact_index: dict,
                evidence_index: dict,
                fact_ledger: list[dict],
                preclaim_text: str
            ) -> dict:
                """
                Extract principal (원금) candidates for a claim.
                Returns: {
                    "value": int | None,
                    "src": [tag list],
                    "status": "OK" | "CALCULATION_PENDING",
                    "missing_reason": str | None,
                    "candidates": [list of alternatives]
                }
                """
                candidates = []

                # Strategy 1: Parse from 청구전작업.md claim detail sections
                # Look for patterns like "약 3억원", "2.2억원", "4억원" in the claim section
                claim_section = _extract_claim_section(claim_id, preclaim_text)
                if claim_section:
                    # Extract amount from 청구취지 line
                    for line in claim_section.split("\n"):
                        if "청구취지" in line:
                            amounts_in_line = re.findall(
                                r'(\d+(?:\.\d+)?(?:\s*억\s*)?(?:\d+\s*천만)?(?:\d+\s*만)?\s*원)',
                                line)
                            for amt_str in amounts_in_line:
                                val = parse_korean_amount(amt_str)
                                if val and val > 0:
                                    candidates.append({
                                        "value": val,
                                        "src": [claim_id],
                                        "origin": "청구전작업_청구취지",
                                        "raw": amt_str.strip()
                                    })

                # Strategy 2: From Fact_Ledger amounts linked to this claim
                for fid, fdata in fact_index.items():
                    if claim_id not in fdata.get("related_claims", []):
                        continue
                    # Find original fact_ledger entry for amount
                    bo_id = fdata.get("source_bo_id", "")
                    for fl_entry in fact_ledger:
                        if fl_entry.get("fact_id") == fid and fl_entry.get("amount"):
                            val = parse_korean_amount(fl_entry["amount"])
                            if val and val > 0:
                                src_tags = [fid]
                                if bo_id:
                                    src_tags.append(bo_id)
                                # Add evidence refs
                                for eref in fl_entry.get("evidence_refs", []):
                                    eid_match = re.match(r'(E-\d+)', eref)
                                    if eid_match:
                                        src_tags.append(eid_match.group(1))
                                candidates.append({
                                    "value": val,
                                    "src": src_tags,
                                    "origin": f"Fact_Ledger[{fid}].amount",
                                    "raw": fl_entry["amount"]
                                })

                # Strategy 3: From evidence key_info
                for eid, edata in evidence_index.items():
                    if claim_id not in edata.get("related_claims", []):
                        continue
                    key_info = edata.get("key_info", "")
                    if not key_info:
                        for kf_text in edata.get("key_facts", []):
                            key_info += " " + kf_text
                    # Extract amounts from key_info
                    amount_matches = re.findall(
                        r'(\d+(?:\.\d+)?(?:\s*억\s*)?(?:\d+\s*천만)?(?:\d+\s*만)?\s*원)',
                        key_info)
                    for amt_str in amount_matches:
                        val = parse_korean_amount(amt_str)
                        if val and val > 0:
                            candidates.append({
                                "value": val,
                                "src": [eid],
                                "origin": f"evidence[{eid}].key_info",
                                "raw": amt_str.strip()
                            })

                # Strategy 4: From claim summary
                summary = claim_data.get("summary", "")
                summary_amounts = re.findall(
                    r'(\d+(?:\.\d+)?(?:\s*억\s*)?(?:\d+\s*천만)?(?:\d+\s*만)?\s*원)',
                    summary)
                for amt_str in summary_amounts:
                    val = parse_korean_amount(amt_str)
                    if val and val > 0:
                        # Extract tags from summary
                        tags_in_summary = re.findall(r'([FE]-\d+|bh\d+)', summary)
                        if tags_in_summary:
                            candidates.append({
                                "value": val,
                                "src": tags_in_summary[:5],  # cap at 5 tags
                                "origin": f"claim_index[{claim_id}].summary",
                                "raw": amt_str.strip()
                            })

                # Deduplicate and rank candidates
                candidates = _deduplicate_candidates(candidates)

                if not candidates:
                    return {
                        "value": None,
                        "src": [],
                        "status": "CALCULATION_PENDING",
                        "missing_reason": "C2_FAIL: 원금을 문서 태그로 추적할 수 없음",
                        "candidates": []
                    }

                # Select best candidate: prefer more source tags, then largest value
                best = _select_best_candidate(candidates)
                return {
                    "value": best["value"],
                    "src": best["src"],
                    "status": "OK",
                    "missing_reason": None,
                    "candidates": candidates
                }


            # ══════════════════════════════════════════════════════════════════════════════
            # 5) Rate Extraction per Claim
            # ══════════════════════════════════════════════════════════════════════════════
            def extract_rate_candidates(
                claim_id: str,
                claim_data: dict,
                evidence_index: dict,
                evidence_raw: list[dict],
                fact_ledger: list[dict],
                fact_index: dict,
                preclaim_text: str
            ) -> dict:
                """
                Extract interest rate candidates for a claim.
                Returns: {
                    "contractual_rate": { ... } | None,
                    "statutory_rate": { ... },
                    "recommended_rate": { ... },
                    "status": "OK" | "CALCULATION_PENDING",
                    "missing_reason": str | None
                }
                """
                claim_type = claim_data.get("claim_type", "")
                plaintiff = claim_data.get("plaintiff", "")

                # Get statutory rate basis
                statutory = _classify_rate_basis(claim_type, plaintiff)

                # For 사해행위취소, no direct rate needed
                if _is_paulian_claim(claim_type):
                    return {
                        "contractual_rate": None,
                        "statutory_rate": statutory,
                        "recommended_rate": statutory,
                        "status": "OK",
                        "missing_reason": None
                    }

                # Try to find contractual rate from evidence linked to this claim
                contractual_candidates = []
                for eid, edata in evidence_index.items():
                    if claim_id not in edata.get("related_claims", []):
                        continue
                    # Search in raw evidence data
                    for ev_raw in evidence_raw:
                        if ev_raw.get("evidence_index") == eid:
                            key_info = ev_raw.get("key_info", "")
                            rates = parse_rate_from_text(key_info)
                            for r in rates:
                                r["src"] = [eid]
                                contractual_candidates.append(r)

                # Also check facts linked to this claim for rate mentions
                for fid, fdata in fact_index.items():
                    if claim_id not in fdata.get("related_claims", []):
                        continue
                    for fl_entry in fact_ledger:
                        if fl_entry.get("fact_id") == fid:
                            action = fl_entry.get("action", "")
                            rates = parse_rate_from_text(action)
                            for r in rates:
                                r["src"] = [fid, fdata.get("source_bo_id", "")]
                                contractual_candidates.append(r)

                # Check 청구전작업 for rate info
                claim_section = _extract_claim_section(claim_id, preclaim_text)
                if claim_section:
                    rates = parse_rate_from_text(claim_section)
                    for r in rates:
                        tags = re.findall(r'([FE]-\d+|bh\d+)', claim_section)
                        r["src"] = tags[:3] if tags else []
                        contractual_candidates.append(r)

                contractual = None
                if contractual_candidates:
                    # Prefer 약정이율, then highest with most tags
                    약정 = [c for c in contractual_candidates if c["type"] == "약정이율"]
                    지연 = [c for c in contractual_candidates if c["type"] == "지연손해금율"]

                    if 약정:
                        best_약정 = max(약정, key=lambda x: (len(x.get("src", [])), x["annual_pct"]))
                        contractual = {
                            "type": best_약정["type"],
                            "annual_pct": best_약정["annual_pct"],
                            "src": best_약정.get("src", []),
                            "raw": best_약정.get("raw", ""),
                            "cap_applied": best_약정["annual_pct"] >= 24.0
                        }
                    elif 지연:
                        best_지연 = max(지연, key=lambda x: (len(x.get("src", [])), x["annual_pct"]))
                        contractual = {
                            "type": best_지연["type"],
                            "annual_pct": best_지연["annual_pct"],
                            "src": best_지연.get("src", []),
                            "raw": best_지연.get("raw", ""),
                            "cap_applied": best_지연["annual_pct"] >= 24.0
                        }

                # Determine recommended rate
                if contractual and contractual.get("src"):
                    recommended = contractual
                elif statutory["annual_pct"] is not None:
                    recommended = {
                        "type": "법정이율",
                        "annual_pct": statutory["annual_pct"],
                        "src": [],
                        "raw": statutory["basis"],
                        "cap_applied": False
                    }
                else:
                    recommended = None

                if recommended and (recommended.get("src") or statutory["annual_pct"] is not None):
                    status = "OK"
                    missing = None
                else:
                    status = "CALCULATION_PENDING"
                    missing = "C3_FAIL: 이율 근거 확정 불가 (약정이율 E-### 명시 없음, 법정이율 적용 조건 미확인)"

                return {
                    "contractual_rate": contractual,
                    "statutory_rate": statutory,
                    "recommended_rate": recommended,
                    "status": status,
                    "missing_reason": missing
                }


            # ══════════════════════════════════════════════════════════════════════════════
            # 6) Date Extraction per Claim (start/end for delay damages)
            # ══════════════════════════════════════════════════════════════════════════════
            def extract_date_candidates(
                claim_id: str,
                claim_data: dict,
                fact_index: dict,
                fact_ledger: list[dict],
                evidence_index: dict,
                evidence_raw: list[dict],
                preclaim_text: str
            ) -> dict:
                """
                Extract start_date (기산일) and end_date (종기) for delay damages.

                start_date heuristics:
                  - For 대여금/구상금: 변제기 다음날 or 대위변제일 다음날
                  - For 사해행위취소: not applicable (형성의소)

                end_date:
                  - Typically "소장부본 송달일" (unknown at filing) or "완제일"
                  - We record this as unknown with reason
                """
                claim_type = claim_data.get("claim_type", "")

                # Paulian claims: no delay damages on the cancellation itself
                if _is_paulian_claim(claim_type):
                    return {
                        "start_date": {
                            "value": None, "src": [],
                            "status": "NOT_APPLICABLE",
                            "missing_reason": "사해행위취소(형성의소)는 지연손해금 기산일 불요",
                            "candidates": []
                        },
                        "end_date": {
                            "value": None, "src": [],
                            "status": "NOT_APPLICABLE",
                            "missing_reason": "사해행위취소(형성의소)는 지연손해금 종기 불요",
                            "candidates": []
                        }
                    }

                start_candidates = []
                end_candidates = []

                # Gather relevant dates from facts
                related_facts = []
                for fid, fdata in fact_index.items():
                    if claim_id in fdata.get("related_claims", []):
                        related_facts.append(fid)

                for fl_entry in fact_ledger:
                    fid = fl_entry.get("fact_id", "")
                    if fid not in related_facts:
                        continue
                    d = parse_date_str(fl_entry.get("date"))
                    if not d:
                        continue
                    bo_id = fl_entry.get("source_bo_id", "")
                    action = fl_entry.get("action", "")
                    src = [fid]
                    if bo_id:
                        src.append(bo_id)
                    # Add evidence refs
                    for eref in fl_entry.get("evidence_refs", []):
                        eid_m = re.match(r'(E-\d+)', eref)
                        if eid_m:
                            src.append(eid_m.group(1))

                    ftype = fl_entry.get("type", "")

                    # Start date heuristics
                    if ftype in ("채무불이행", "기한도래"):
                        start_candidates.append({
                            "value": d, "src": src,
                            "origin": f"FL[{fid}] 채무불이행/기한도래",
                            "note": "변제기 또는 부도일"
                        })
                    elif "만기" in action or "기한" in action:
                        start_candidates.append({
                            "value": d, "src": src,
                            "origin": f"FL[{fid}] 만기/기한",
                            "note": "만기일(기산일 후보)"
                        })
                    elif "대위변제" in action:
                        start_candidates.append({
                            "value": d, "src": src,
                            "origin": f"FL[{fid}] 대위변제",
                            "note": "대위변제일(구상금 기산일 후보)"
                        })

                # Also extract maturity from evidence key_info
                for eid, edata in evidence_index.items():
                    if claim_id not in edata.get("related_claims", []):
                        continue
                    for ev_raw in evidence_raw:
                        if ev_raw.get("evidence_index") == eid:
                            key_info = ev_raw.get("key_info", "")
                            if "만기" in key_info:
                                dates = extract_dates_from_text(key_info)
                                for d in dates:
                                    start_candidates.append({
                                        "value": d, "src": [eid],
                                        "origin": f"evidence[{eid}] 만기",
                                        "note": "증거 key_info 내 만기일"
                                    })

                # Scan preclaim text for rate/date info
                claim_section = _extract_claim_section(claim_id, preclaim_text)
                if claim_section:
                    dates_in_section = extract_dates_from_text(claim_section)
                    tags_in_section = re.findall(r'([FE]-\d+|bh\d+)', claim_section)
                    for d in dates_in_section:
                        start_candidates.append({
                            "value": d, "src": tags_in_section[:3],
                            "origin": f"청구전작업[{claim_id}]",
                            "note": "청구전작업 내 날짜"
                        })

                # end_date: typically unknown (소장부본 송달일)
                end_result = {
                    "value": None,
                    "src": [],
                    "status": "CALCULATION_PENDING",
                    "missing_reason": "C2_FAIL: 종기(소장부본 송달일)는 소제기 후 확정되므로 현재 추출 불가",
                    "candidates": []
                }

                # Deduplicate start candidates
                start_candidates = _deduplicate_date_candidates(start_candidates)

                if not start_candidates:
                    start_result = {
                        "value": None,
                        "src": [],
                        "status": "CALCULATION_PENDING",
                        "missing_reason": "C2_FAIL: 기산일을 문서 태그로 추적할 수 없음",
                        "candidates": []
                    }
                else:
                    best = start_candidates[0]  # Already sorted
                    start_result = {
                        "value": best["value"],
                        "src": best["src"],
                        "status": "OK",
                        "missing_reason": None,
                        "candidates": start_candidates
                    }

                return {
                    "start_date": start_result,
                    "end_date": end_result
                }


            # ══════════════════════════════════════════════════════════════════════════════
            # 7) Fee Calculations (Stamp Fee + Service Fee)
            # ══════════════════════════════════════════════════════════════════════════════
            def calc_stamp_fee(value: int) -> int:
                """2025 Korean court stamp fee schedule."""
                if value < 10_000_000:
                    return int(value * 0.0050)
                elif value < 100_000_000:
                    return int(value * 0.0045 + 5_000)
                elif value < 1_000_000_000:
                    return int(value * 0.0040 + 55_000)
                else:
                    return int(value * 0.0035 + 555_000)


            def calc_service_fee(party_count: int) -> int:
                """Korean court service fee: 당사자수 × 15회분 × 5,200원"""
                return party_count * 15 * 5_200


            def compute_fees(
                principal_value: Optional[int],
                claim_data: dict,
                parties: dict
            ) -> dict:
                """Compute stamp fee and service fee if principal is available."""
                claim_type = claim_data.get("claim_type", "")

                # For 사해행위취소: claim value is the property value
                if _is_paulian_claim(claim_type):
                    # Extract property value from summary or return pending
                    return {
                        "claim_value": {
                            "value": principal_value,
                            "status": "OK" if principal_value else "CALCULATION_PENDING",
                            "missing_reason": None if principal_value else "사해행위취소 소가 산정은 목적물 가액 기준이나 현재 미확정"
                        },
                        "stamp_fee": {
                            "value": calc_stamp_fee(principal_value) if principal_value else None,
                            "status": "OK" if principal_value else "CALCULATION_PENDING",
                            "missing_reason": None if principal_value else "소가 미확정으로 인지대 산정 불가"
                        },
                        "service_fee": _compute_service_fee(claim_data, parties)
                    }

                if not principal_value:
                    return {
                        "claim_value": {
                            "value": None, "status": "CALCULATION_PENDING",
                            "missing_reason": "원금 미확정으로 소가 산정 불가"
                        },
                        "stamp_fee": {
                            "value": None, "status": "CALCULATION_PENDING",
                            "missing_reason": "소가 미확정으로 인지대 산정 불가"
                        },
                        "service_fee": _compute_service_fee(claim_data, parties)
                    }

                stamp = calc_stamp_fee(principal_value)
                return {
                    "claim_value": {"value": principal_value, "status": "OK", "missing_reason": None},
                    "stamp_fee": {"value": stamp, "status": "OK", "missing_reason": None},
                    "service_fee": _compute_service_fee(claim_data, parties)
                }


            def _compute_service_fee(claim_data: dict, parties: dict) -> dict:
                """Compute service fee from party count."""
                # Count unique parties for this claim
                plaintiff_str = claim_data.get("plaintiff", "")
                defendant_str = claim_data.get("defendant", "")

                p_count = len([p.strip() for p in re.split(r'[,，]', plaintiff_str) if p.strip()])
                d_count = len([d.strip() for d in re.split(r'[,，]', defendant_str) if d.strip()])
                total_parties = max(p_count + d_count, 2)  # At least 2

                fee = calc_service_fee(total_parties)
                return {
                    "value": fee,
                    "party_count": total_parties,
                    "status": "OK",
                    "missing_reason": None
                }


            # ══════════════════════════════════════════════════════════════════════════════
            # 8) Helper Functions
            # ══════════════════════════════════════════════════════════════════════════════
            def _extract_claim_section(claim_id: str, preclaim_text: str) -> Optional[str]:
                """Extract the section for a specific claim from 청구전작업.md"""
                # Match patterns like "### (1) C-001:" or "| C-001 |"
                lines = preclaim_text.split("\n")
                in_section = False
                section_lines = []

                for line in lines:
                    if claim_id in line and (
                        re.match(r'###\s*\(?\d+\)?', line) or
                        line.strip().startswith(f"| {claim_id}")
                    ):
                        in_section = True
                        section_lines.append(line)
                        continue
                    if in_section:
                        # End when next claim section starts
                        if re.match(r'###\s*\(?\d+\)', line) and claim_id not in line:
                            break
                        if re.match(r'^---', line):
                            break
                        if re.match(r'^##\s+\d+\.', line):
                            break
                        section_lines.append(line)

                return "\n".join(section_lines) if section_lines else None


            def _deduplicate_candidates(candidates: list[dict]) -> list[dict]:
                """Deduplicate amount candidates by value, keeping the one with most src tags."""
                if not candidates:
                    return []
                seen = {}
                for c in candidates:
                    val = c["value"]
                    if val not in seen or len(c.get("src", [])) > len(seen[val].get("src", [])):
                        seen[val] = c
                # Sort: most src tags first, then highest value
                result = sorted(seen.values(),
                                key=lambda x: (-len(x.get("src", [])), -x["value"]))
                return result


            def _deduplicate_date_candidates(candidates: list[dict]) -> list[dict]:
                """Deduplicate date candidates by value, keeping most tagged."""
                if not candidates:
                    return []
                seen = {}
                for c in candidates:
                    val = c["value"]
                    key = val
                    if key not in seen or len(c.get("src", [])) > len(seen[key].get("src", [])):
                        seen[key] = c
                return sorted(seen.values(),
                              key=lambda x: (-len(x.get("src", [])), x["value"]))


            def _select_best_candidate(candidates: list[dict]) -> dict:
                """Select the best amount candidate: most tags, then largest value."""
                if not candidates:
                    return {"value": None, "src": []}
                return candidates[0]  # Already sorted by _deduplicate_candidates


            # ══════════════════════════════════════════════════════════════════════════════
            # 9) Gate Assessment
            # ══════════════════════════════════════════════════════════════════════════════
            def assess_compute_gate(claim_entry: dict) -> dict:
                """
                Assess the 3-condition gate for deterministic computation:
                  C1: python code execution → always True
                  C2: principal/rate/start/end all traceable to tags
                  C3: normative rate basis confirmed
                """
                principal = claim_entry.get("principal", {})
                rate = claim_entry.get("rate", {})
                dates = claim_entry.get("dates", {})
                start = dates.get("start_date", {})
                end = dates.get("end_date", {})

                c1 = True  # Always true (we're running python)

                c2_principal = principal.get("status") == "OK" and bool(principal.get("src"))
                c2_rate = (rate.get("status") == "OK")
                c2_start = (start.get("status") in ("OK", "NOT_APPLICABLE"))
                c2_end = (end.get("status") in ("OK", "NOT_APPLICABLE"))
                c2 = c2_principal and c2_rate and c2_start and c2_end

                # C3: rate basis is confirmed
                rec_rate = rate.get("recommended_rate")
                statutory = rate.get("statutory_rate", {})
                is_paulian = (statutory.get("basis", "").startswith("사해행위취소") or
                              start.get("status") == "NOT_APPLICABLE")
                c3 = (is_paulian or  # 사해행위취소는 지연손해금 산정 불요
                      (rec_rate is not None and rec_rate.get("annual_pct") is not None))

                gate_pass = c1 and c2 and c3

                missing_conditions = []
                if not c2_principal:
                    missing_conditions.append("C2: principal 추적 불가")
                if not c2_rate:
                    missing_conditions.append("C2: rate 추적 불가")
                if not c2_start:
                    missing_conditions.append("C2: start_date 추적 불가")
                if not c2_end:
                    missing_conditions.append("C2: end_date 추적 불가")
                if not c3:
                    missing_conditions.append("C3: 이율 규범값 근거 미확정")

                return {
                    "gate_pass": gate_pass,
                    "C1": c1,
                    "C2": c2,
                    "C3": c3,
                    "missing_conditions": missing_conditions
                }


            # ══════════════════════════════════════════════════════════════════════════════
            # 10) Main Builder
            # ══════════════════════════════════════════════════════════════════════════════
            def build_compute_inputs(
                index: dict,
                fact_ledger: list[dict],
                evidence_raw: list[dict],
                preclaim_text: str
            ) -> dict:
                """Build the complete stage4_compute_inputs.json structure."""

                claim_index = index["claim_index"]
                fact_index = index["fact_index"]
                evidence_index = index["evidence_index"]
                top_n = index["top_n_claims"]
                parties = index.get("parties", {})

                claims = []

                for claim_id in top_n:
                    claim_data = claim_index.get(claim_id, {})

                    # Extract principal
                    principal = extract_principal_candidates(
                        claim_id, claim_data, fact_index, evidence_index,
                        fact_ledger, preclaim_text
                    )

                    # Extract rate
                    rate = extract_rate_candidates(
                        claim_id, claim_data, evidence_index, evidence_raw,
                        fact_ledger, fact_index, preclaim_text
                    )

                    # Extract dates
                    dates = extract_date_candidates(
                        claim_id, claim_data, fact_index, fact_ledger,
                        evidence_index, evidence_raw, preclaim_text
                    )

                    # Compute fees
                    fees = compute_fees(principal.get("value"), claim_data, parties)

                    # Build claim entry
                    entry = {
                        "claim_id": claim_id,
                        "claim_type": claim_data.get("claim_type", ""),
                        "principal": {
                            "value": principal["value"],
                            "src": principal["src"],
                            "status": principal["status"],
                            "missing_reason": principal["missing_reason"],
                        },
                        "rate": {
                            "contractual_rate": rate["contractual_rate"],
                            "statutory_rate": rate["statutory_rate"],
                            "recommended_rate": rate["recommended_rate"],
                            "status": rate["status"],
                            "missing_reason": rate["missing_reason"],
                        },
                        "dates": {
                            "start_date": {
                                "value": dates["start_date"]["value"],
                                "src": dates["start_date"]["src"],
                                "status": dates["start_date"]["status"],
                                "missing_reason": dates["start_date"]["missing_reason"],
                            },
                            "end_date": {
                                "value": dates["end_date"]["value"],
                                "src": dates["end_date"]["src"],
                                "status": dates["end_date"]["status"],
                                "missing_reason": dates["end_date"]["missing_reason"],
                            }
                        },
                        "fees": fees,
                    }

                    # Assess gate
                    entry["gate"] = assess_compute_gate(entry)

                    # Include candidate details for transparency
                    if principal.get("candidates"):
                        entry["principal"]["candidates"] = principal["candidates"]
                    if dates["start_date"].get("candidates"):
                        entry["dates"]["start_date"]["candidates"] = dates["start_date"]["candidates"]

                    claims.append(entry)

                # Formulas reference (for downstream use)
                formulas = {
                    "delay_damages": "principal * (annual_rate / 100) * (days / 365)",
                    "stamp_fee": "2025 schedule: <10M→0.50%, <100M→0.45%+5k, <1B→0.40%+55k, ≥1B→0.35%+555k",
                    "service_fee": "party_count × 15 × 5,200원"
                }

                return {
                    "meta": {
                        "stage": "4",
                        "phase": "A-3",
                        "generated_at": datetime.now().strftime("%Y-%m-%d"),
                        "description": "Deterministic compute inputs per claim (§4 GATED COMPUTE)",
                        "gate_conditions": {
                            "C1": "python code execution (always true)",
                            "C2": "principal/rate/start/end traceable to document tags",
                            "C3": "normative rate basis confirmed (contractual ≤24% or statutory)"
                        }
                    },
                    "formulas": formulas,
                    "claims": claims
                }


            # ══════════════════════════════════════════════════════════════════════════════
            # 11) Validation
            # ══════════════════════════════════════════════════════════════════════════════
            class ValidationReport:
                def __init__(self):
                    self.errors: list[str] = []
                    self.warnings: list[str] = []
                    self.info: list[str] = []
                    self.ok = True

                def error(self, msg: str):
                    self.errors.append(msg)
                    self.ok = False

                def warn(self, msg: str):
                    self.warnings.append(msg)

                def add_info(self, msg: str):
                    self.info.append(msg)

                def print_report(self):
                    print("\n=== stage4_compute_inputs Validation ===",
                          file=sys.stderr)
                    for e in self.errors:
                        print(f"  ERROR: {e}", file=sys.stderr)
                    for w in self.warnings:
                        print(f"  WARN:  {w}", file=sys.stderr)
                    for i in self.info:
                        print(f"  INFO:  {i}", file=sys.stderr)
                    status = "PASS" if self.ok else "FAIL"
                    print(f"  Status: {status} "
                          f"({len(self.errors)} errors, {len(self.warnings)} warnings)",
                          file=sys.stderr)


            def validate_compute_inputs(result: dict, index: dict) -> ValidationReport:
                """Validate the generated compute inputs."""
                rpt = ValidationReport()

                claims = result.get("claims", [])
                top_n = index.get("top_n_claims", [])

                # V1: All top_n claims present
                claim_ids_present = {c["claim_id"] for c in claims}
                for cid in top_n:
                    if cid not in claim_ids_present:
                        rpt.error(f"V1: top_n claim {cid} missing from compute_inputs")

                # V2: No extra claims
                for cid in claim_ids_present:
                    if cid not in top_n:
                        rpt.warn(f"V2: claim {cid} in compute_inputs but not in top_n")

                for claim in claims:
                    cid = claim["claim_id"]

                    # V3: Principal sources are valid tags
                    for src in claim.get("principal", {}).get("src", []):
                        if not re.match(r'^(F-\d+|E-\d+|bh\d+|C-\d+)$', src):
                            rpt.warn(f"V3: {cid} principal src '{src}' is not a standard tag")

                    # V4: Rate has valid structure
                    rate = claim.get("rate", {})
                    if rate.get("status") == "OK":
                        rec = rate.get("recommended_rate")
                        if rec and rec.get("annual_pct") is not None:
                            if rec["annual_pct"] > 24.0:
                                rpt.error(f"V4: {cid} rate {rec['annual_pct']}% exceeds 24% cap")
                            if rec["annual_pct"] < 0:
                                rpt.error(f"V4: {cid} rate {rec['annual_pct']}% is negative")

                    # V5: Date format
                    for dk in ("start_date", "end_date"):
                        d = claim.get("dates", {}).get(dk, {})
                        val = d.get("value")
                        if val and not re.match(r'^\d{4}-\d{2}-\d{2}$', val):
                            rpt.error(f"V5: {cid} {dk} '{val}' is not ISO date format")

                    # V6: Gate assessment consistency
                    gate = claim.get("gate", {})
                    if gate.get("gate_pass"):
                        if claim["principal"]["status"] != "OK":
                            rpt.error(f"V6: {cid} gate_pass=true but principal status != OK")

                    # V7: Fee calculations
                    fees = claim.get("fees", {})
                    stamp = fees.get("stamp_fee", {})
                    if stamp.get("value") is not None and stamp["value"] < 0:
                        rpt.error(f"V7: {cid} stamp_fee is negative")

                    svc = fees.get("service_fee", {})
                    if svc.get("value") is not None:
                        if svc["value"] <= 0:
                            rpt.warn(f"V7: {cid} service_fee is zero or negative")

                    # V8: missing_reason present when status != OK
                    for field_name in ("principal",):
                        field = claim.get(field_name, {})
                        if field.get("status") == "CALCULATION_PENDING" and not field.get("missing_reason"):
                            rpt.warn(f"V8: {cid} {field_name} is PENDING but no missing_reason")

                # Summary
                gate_pass_count = sum(1 for c in claims if c.get("gate", {}).get("gate_pass"))
                rpt.add_info(f"Total claims: {len(claims)}")
                rpt.add_info(f"Gate PASS: {gate_pass_count}/{len(claims)}")
                rpt.add_info(f"Gate FAIL: {len(claims) - gate_pass_count}/{len(claims)}")

                return rpt


            def _log(msg: str):
                """진행/디버그 메시지 → stderr (stdout은 결과 JSON 전용)."""
                print(msg, file=sys.stderr)


            # ══════════════════════════════════════════════════════════════════════════════
            # 12) Entry Point
            # ══════════════════════════════════════════════════════════════════════════════
            def main():
                output_name = "stage4_compute_inputs.json"
                do_validate = True

                with httpx.Client(timeout=60) as c:
                    # ===== 1) localdocs 연결 =====
                    r = c.post(LOCALDOCS_URL, json={
                        "jsonrpc": "2.0", "id": 1, "method": "initialize",
                        "params": {
                            "protocolVersion": "2025-03-26",
                            "capabilities": {},
                            "clientInfo": {"name": "stage4-compute-inputs-gen", "version": "1.0"}
                        }
                    }, headers=HEADERS)
                    sid = r.headers.get("mcp-session-id")
                    if sid:
                        HEADERS["mcp-session-id"] = sid
                    c.post(LOCALDOCS_URL, json={
                        "jsonrpc": "2.0", "method": "notifications/initialized"
                    }, headers=HEADERS)
                    _log(f"1) Connected to localdocs (session: {sid})")

                    # 도구 목록 확인
                    r = c.post(LOCALDOCS_URL, json={
                        "jsonrpc": "2.0", "id": 2, "method": "tools/list"
                    }, headers=HEADERS)
                    tools_result = parse_sse(r.text)
                    tools = (tools_result.get("result", {}).get("tools", [])
                            if tools_result else [])
                    _log(f"   Tools: {[t['name'] for t in tools]}")

                    write_tool = next(
                        (t for t in tools if "write" in t["name"]), None)

                    # 문서 목록
                    docs = call_tool(c, "list_docs", {}, 3)
                    if docs and "result" in docs:
                        _log(f"   Docs: {docs['result']['content'][0]['text'][:500]}")

                    # ===== 2) 파일 로딩 (MCP read_doc) =====
                    _log("\n2) Loading input files via MCP...")

                    index = read_doc(c, "stage4_index.json", 10)
                    if index is None:
                        raise RuntimeError("Failed to load stage4_index.json")

                    fact_ledger = read_doc(c, "Fact_Ledger.json", 11)
                    if fact_ledger is None:
                        fact_ledger = read_doc(c, "fact_ledger.json", 12)
                    if fact_ledger is None:
                        raise RuntimeError("Failed to load Fact_Ledger.json")

                    evidence_raw = read_doc(c, "evidence_indexed.json", 13)
                    if evidence_raw is None:
                        raise RuntimeError("Failed to load evidence_indexed.json")

                    preclaim_text = read_doc_text(c, "청구전작업.md", 14) or ""

                    for name, data in [("stage4_index.json", index),
                                      ("Fact_Ledger.json", fact_ledger),
                                      ("evidence_indexed.json", evidence_raw)]:
                        size = len(data) if hasattr(data, '__len__') else '?'
                        _log(f"   {name}: loaded ({type(data).__name__}, len={size})")
                    _log(f"   청구전작업.md: {'loaded' if preclaim_text else 'not found'}")

                    # ===== 3) 빌드 =====
                    _log("\n3) Building compute inputs...")
                    result = build_compute_inputs(index, fact_ledger, evidence_raw, preclaim_text)

                    # ===== 4) 검증 =====
                    if do_validate:
                        rpt = validate_compute_inputs(result, index)
                        rpt.print_report()

                    # ===== 5) 결과 저장 (MCP write_doc + stdout) =====
                    output_json = json.dumps(result, ensure_ascii=False, indent=2)

                    if write_tool:
                        write_result = call_tool(c, write_tool["name"], {
                            "path": output_name,
                            "content": output_json,
                        }, 20)
                        if write_result and "result" in write_result:
                            _log(f"\n4) Written to localdocs: {output_name}")
                        else:
                            _log("\n4) Write to localdocs failed")
                    else:
                        _log("\n4) No write tool available")

                    # 항상 stdout으로 출력 (code-executor가 캡처)
                    print(output_json)


            if __name__ == "__main__":
                main()

          requirements: "httpx"
          network: "agent-network"
          timeout: 120

      - task_name: run_claim_packets
        mcp: code-executor
        tool_name: run_code
        parameters:
          language: python
          code: |
            #!/usr/bin/env python3
            """
            stage4_claim_packets.py — Stage 4, Phase A-2 (Generalized)

            Reads inputs via MCP localdocs and produces stage4_claim_packets.json.
            Runs inside code-executor Docker container.
            """

            import json
            import re
            import sys
            from datetime import datetime
            from typing import Any
            import httpx


            # -----------------------------------------------------------------------------
            # 0) MCP localdocs helpers
            # -----------------------------------------------------------------------------
            LOCALDOCS_URL = "http://mcp-localdocs:8012/mcp"
            HEADERS = {
                "Content-Type": "application/json",
                "Accept": "application/json, text/event-stream",
            }


            def parse_sse(text):
                for line in text.strip().split("\n"):
                    if line.startswith("data: "):
                        return json.loads(line[6:])
                try:
                    return json.loads(text)
                except Exception:
                    return None


            def call_tool(c, name, arguments, msg_id=10):
                r = c.post(LOCALDOCS_URL, json={
                    "jsonrpc": "2.0", "id": msg_id,
                    "method": "tools/call",
                    "params": {"name": name, "arguments": arguments}
                }, headers=HEADERS)
                result = parse_sse(r.text)
                if result and "result" in result:
                    return result
                print(f"Tool {name} error: {json.dumps(result)[:300]}",
                      file=sys.stderr)
                return result


            def read_doc(c, doc_name, msg_id=10):
                """Read a JSON document via MCP localdocs."""
                result = call_tool(c, "read_doc", {"doc_name": doc_name}, msg_id)
                if result and "result" in result:
                    text = result["result"]["content"][0]["text"]
                    if not text or not text.strip():
                        return None
                    try:
                        return json.loads(text)
                    except json.JSONDecodeError:
                        print(f"read_doc({doc_name}): JSON parse failed",
                              file=sys.stderr)
                        return None
                return None


            def read_doc_text(c, doc_name, msg_id=10):
                """Read a text/markdown document via MCP localdocs (no JSON parse)."""
                result = call_tool(c, "read_doc", {"doc_name": doc_name}, msg_id)
                if result and "result" in result:
                    text = result["result"]["content"][0]["text"]
                    return text if text and text.strip() else None
                return None


            # -----------------------------------------------------------------------------
            # 2) Lightweight JSON Schema Validator (subset)
            # -----------------------------------------------------------------------------

            JSON_TYPE_MAP = {
                "object": dict,
                "array": list,
                "string": str,
                "number": (int, float),
                "integer": int,
                "boolean": bool,
                "null": type(None),
            }


            def _type_ok(value: Any, schema_type: str) -> bool:
                py_t = JSON_TYPE_MAP.get(schema_type)
                if py_t is None:
                    return True
                if schema_type == "integer":
                    return isinstance(value, int) and not isinstance(value, bool)
                if schema_type == "number":
                    return (isinstance(value, (int, float)) and
                            not isinstance(value, bool))
                return isinstance(value, py_t)


            def validate_schema(instance: Any, schema: dict,
                                path: str = "$") -> list[str]:
                errors: list[str] = []

                expected = schema.get("type")
                if expected and not _type_ok(instance, expected):
                    errors.append(f"{path}: expected type '{expected}', got '{type(instance).__name__}'")
                    return errors

                enum = schema.get("enum")
                if enum is not None and instance not in enum:
                    errors.append(f"{path}: value '{instance}' is not in enum {enum}")

                if isinstance(instance, dict):
                    required = schema.get("required", [])
                    for key in required:
                        if key not in instance:
                            errors.append(f"{path}: missing required key '{key}'")

                    props = schema.get("properties", {})
                    additional = schema.get("additionalProperties", True)

                    for k, v in instance.items():
                        if k in props:
                            errors.extend(validate_schema(v, props[k], f"{path}.{k}"))
                        else:
                            if isinstance(additional, dict):
                                errors.extend(validate_schema(v, additional, f"{path}.{k}"))
                            elif additional is False:
                                errors.append(f"{path}: additional property '{k}' not allowed")

                if isinstance(instance, list):
                    min_items = schema.get("minItems")
                    max_items = schema.get("maxItems")
                    if min_items is not None and len(instance) < min_items:
                        errors.append(f"{path}: expected minItems={min_items}, got {len(instance)}")
                    if max_items is not None and len(instance) > max_items:
                        errors.append(f"{path}: expected maxItems={max_items}, got {len(instance)}")

                    item_schema = schema.get("items")
                    if item_schema:
                        for i, item in enumerate(instance):
                            errors.extend(validate_schema(item, item_schema, f"{path}[{i}]"))

                pattern = schema.get("pattern")
                if pattern and isinstance(instance, str):
                    if re.match(pattern, instance) is None:
                        errors.append(f"{path}: string '{instance}' does not match /{pattern}/")

                return errors


            # -----------------------------------------------------------------------------
            # 3) Rules Engine Utilities
            # -----------------------------------------------------------------------------

            def build_synonym_map(rules: dict) -> dict[str, set[str]]:
                syn_map: dict[str, set[str]] = {}
                for group in rules.get("synonym_groups", []):
                    gset = set(group)
                    for token in gset:
                        syn_map.setdefault(token, set()).update(gset)
                return syn_map


            def tokenize_ko(text: str, rules: dict) -> set[str]:
                tokenizer = rules.get("tokenizer", {})
                stopwords = set(tokenizer.get("stopwords", []))
                suffixes = tokenizer.get("suffixes", [])

                raw_tokens = re.findall(r"[가-힣A-Za-z0-9]+", text or "")
                out: set[str] = set()
                for tok in raw_tokens:
                    if len(tok) < 2 or tok in stopwords:
                        continue
                    out.add(tok)
                    for suf in suffixes:
                        if tok.endswith(suf) and len(tok) > len(suf) + 1:
                            stripped = tok[:-len(suf)]
                            if len(stripped) >= 2:
                                out.add(stripped)
                            break
                return out


            def overlap_score(a: str, b: str, rules: dict,
                              syn_map: dict[str, set[str]]) -> int:
                ta = tokenize_ko(a, rules)
                tb = tokenize_ko(b, rules)
                direct = len(ta & tb)

                exp_a = set(ta)
                exp_b = set(tb)
                for t in ta:
                    if t in syn_map:
                        exp_a.update(syn_map[t])
                for t in tb:
                    if t in syn_map:
                        exp_b.update(syn_map[t])

                extra = len((exp_a & tb) | (ta & exp_b)) - direct
                return direct * 2 + max(extra, 0)


            def claim_text(cdata: dict) -> str:
                return " ".join([
                    cdata.get("claim_type", ""),
                    cdata.get("summary", ""),
                    cdata.get("plaintiff", ""),
                    cdata.get("defendant", ""),
                ]).strip()


            def _matches_any(patterns: list[str], text: str) -> bool:
                return any(re.search(p, text, flags=re.IGNORECASE) for p in patterns)


            def detect_case_group(claim_id: str, claim_data: dict,
                                  legal_elements: dict, rules: dict) -> str:
                """
                Strategy 2: case-group detection + fallback to general.
                Score groups by claim_type and element text matches.
                """
                groups = rules.get("case_groups", [])
                ctext = claim_data.get("claim_type", "")

                elem_texts = []
                for _, ed in legal_elements.items():
                    if claim_id in ed.get("applicable_claims", []):
                        elem_texts.append(
                            (ed.get("element", "") + " " + ed.get("description", "")).strip())
                elem_blob = "\n".join(elem_texts)

                best_group = "general"
                best_score = 0

                for g in groups:
                    gid = g.get("id", "")
                    if not gid:
                        continue
                    c_weight = int(g.get("claim_type_weight", 3))
                    e_weight = int(g.get("element_weight", 1))

                    score = 0
                    c_pats = g.get("claim_type_patterns", [])
                    e_pats = g.get("element_patterns", [])

                    if c_pats and _matches_any(c_pats, ctext):
                        score += c_weight

                    if e_pats and elem_blob:
                        em = 0
                        for p in e_pats:
                            if re.search(p, elem_blob, flags=re.IGNORECASE):
                                em += 1
                        score += em * e_weight

                    if score > best_score:
                        best_score = score
                        best_group = gid

                return best_group if best_score > 0 else "general"


            # -----------------------------------------------------------------------------
            # 4) Plugin System
            # -----------------------------------------------------------------------------

            class PluginContext:
                def __init__(self, claim_id: str, claim_data: dict, claim_index: dict,
                            fact_index: dict, legal_elements: dict,
                            rules: dict, syn_map: dict[str, set[str]]):
                    self.claim_id = claim_id
                    self.claim_data = claim_data
                    self.claim_index = claim_index
                    self.fact_index = fact_index
                    self.legal_elements = legal_elements
                    self.rules = rules
                    self.syn_map = syn_map


            class BasePlugin:
                """Generic/unbiased baseline plugin."""

                def __init__(self, group_id: str, plugin_rules: dict):
                    self.group_id = group_id
                    self.plugin_rules = plugin_rules

                def expand_candidate_claims(self, ctx: PluginContext) -> set[str]:
                    # Generic: no expansion. Keep unbiased default.
                    return {ctx.claim_id}

                def fact_bonus(self, element_text: str, fact_text: str) -> int:
                    # Apply boost rules from external config only.
                    bonus = 0
                    for br in self.plugin_rules.get("fact_boost_rules", []):
                        ep = br.get("element_any", [])
                        fp = br.get("fact_any", [])
                        sc = int(br.get("score", 0))
                        if ep and not _matches_any(ep, element_text):
                            continue
                        if fp and not _matches_any(fp, fact_text):
                            continue
                        bonus += sc
                    return bonus

                def special_modules(self) -> list[str]:
                    mods = self.plugin_rules.get("special_modules", [])
                    return sorted(set([m for m in mods if isinstance(m, str)]))

                def build_special_actions(self, claim_data: dict, claim_index: dict,
                                          fact_index: dict, fact_ledger: list[dict],
                                          legal_elements: dict) -> list[dict]:
                    return []


            class LoanPlugin(BasePlugin):
                pass


            class GuaranteePlugin(BasePlugin):
                pass


            class GeneralPlugin(BasePlugin):
                pass


            class PaulianPlugin(BasePlugin):
                def expand_candidate_claims(self, ctx: PluginContext) -> set[str]:
                    # include own claim + preserved candidates discovered by shared facts/similar text
                    cands = {ctx.claim_id}

                    own_text = claim_text(ctx.claim_data)

                    # text-similar siblings
                    for cid2, cdata2 in ctx.claim_index.items():
                        if cid2 == ctx.claim_id:
                            continue
                        s = overlap_score(own_text, claim_text(cdata2), ctx.rules, ctx.syn_map)
                        if s >= 4:
                            cands.add(cid2)

                    # fact-linked siblings
                    own_related_facts = [
                        fid for fid, fdata in ctx.fact_index.items()
                        if ctx.claim_id in fdata.get("related_claims", [])
                    ]
                    for fid in own_related_facts:
                        for rcid in ctx.fact_index.get(fid, {}).get("related_claims", []):
                            cands.add(rcid)

                    return cands

                def _build_fact_ledger_maps(self, fact_ledger: list[dict]) -> tuple[dict, dict]:
                    by_fact = {}
                    by_bo = {}
                    for row in fact_ledger:
                        fid = row.get("fact_id", "")
                        bo = row.get("source_bo_id", "")
                        if fid:
                            by_fact[fid] = row
                        if bo:
                            by_bo[bo] = row
                    return by_fact, by_bo

                def _pick_preserved_claim_link(self, claim_id: str, claim_index: dict,
                                              fact_index: dict) -> str:
                    # pick non-paulian claim most connected via shared fact relations
                    candidates = []
                    for cid, cdata in claim_index.items():
                        if cid == claim_id:
                            continue
                        ct = cdata.get("claim_type", "")
                        if _matches_any(self.plugin_rules.get("claim_type_patterns", []), ct):
                            continue
                        candidates.append(cid)

                    if not candidates:
                        return ""

                    best = ""
                    best_score = -1
                    for cid in candidates:
                        score = 0
                        for _, fdata in fact_index.items():
                            rc = set(fdata.get("related_claims", []))
                            if claim_id in rc and cid in rc:
                                score += 1
                        if score > best_score:
                            best_score = score
                            best = cid

                    return best

                def build_special_actions(self, claim_data: dict, claim_index: dict,
                                          fact_index: dict, fact_ledger: list[dict],
                                          legal_elements: dict) -> list[dict]:
                    transfer_patterns = self.plugin_rules.get("transfer_patterns", [])
                    act_patterns = self.plugin_rules.get("act_type_patterns", [])

                    claim_id = claim_data.get("claim_id", "")
                    if not claim_id:
                        return []

                    by_fact, by_bo = self._build_fact_ledger_maps(fact_ledger)
                    rows = []
                    for fid, fdata in fact_index.items():
                        if not str(fid).startswith("F-"):
                            continue
                        if claim_id not in fdata.get("related_claims", []):
                            continue

                        fl = by_fact.get(fid)
                        if not fl:
                            fl = by_bo.get(fdata.get("source_bo_id", ""), {})

                        summary = fdata.get("summary", "")
                        action = fl.get("action", "")
                        combined = f"{summary} {action}".strip()
                        if _matches_any(transfer_patterns, combined):
                            rows.append((fid, fdata, fl, combined))

                    if not rows:
                        return []

                    preserved_link = self._pick_preserved_claim_link(claim_id, claim_index, fact_index)

                    defendants = [d.strip() for d in claim_data.get("defendant", "").split(",") if d.strip()]
                    remedy = "주위(원물반환)"
                    ctype = claim_data.get("claim_type", "")
                    if _matches_any([r"전득자", r"가액배상", r"예비"], ctype):
                        remedy = "주위(원물반환)|예비(가액배상)"

                    out = []
                    for i, (fid, fdata, fl, combined) in enumerate(sorted(rows, key=lambda x: x[0]), start=1):
                        act_type = "기타"
                        for pat, name in act_patterns:
                            if re.search(pat, combined):
                                act_type = name
                                break

                        beneficiary = ""
                        parties = fl.get("parties", []) if isinstance(fl.get("parties", []), list) else []
                        for p in parties:
                            if p in defendants:
                                beneficiary = p
                                break
                        if not beneficiary:
                            beneficiary = defendants[0] if defendants else "불명"

                        obj = combined[:40] if combined else claim_data.get("summary", "")[:40]
                        loc = re.search(
                            r"((?:[\w]+(?:시|구|군|동|리)\s*){0,4}[\w\s]*?(?:아파트|토지|건물|부동산|대지|주택))",
                            combined
                        )
                        if loc:
                            obj = loc.group(1).strip()
                        obj = f"{obj} ({fid})"

                        date = fl.get("date", "")
                        time = f"{date} ({fid})" if date else "불명"

                        ev_ids = []
                        for eref in fdata.get("evidence_refs", []):
                            m = re.match(r"(E-\d+)", str(eref))
                            if m:
                                ev_ids.append(m.group(1))

                        src = [fid] + ev_ids
                        src = list(dict.fromkeys(src))

                        out.append({
                            "paul_id": f"PAUL-{i}",
                            "act_type": act_type,
                            "object": obj,
                            "time": time,
                            "beneficiary_or_transferee": f"{beneficiary} ({fid})",
                            "preserved_claim_link": preserved_link,
                            "remedy_structure": remedy,
                            "src": src,
                        })

                    return out


            def build_plugin_registry(rules: dict) -> dict[str, BasePlugin]:
                plugin_cfg = rules.get("plugins", {})
                reg: dict[str, BasePlugin] = {
                    "general": GeneralPlugin("general", plugin_cfg.get("general", {})),
                    "loan": LoanPlugin("loan", plugin_cfg.get("loan", {})),
                    "guarantee": GuaranteePlugin("guarantee", plugin_cfg.get("guarantee", {})),
                    "paulian": PaulianPlugin("paulian", plugin_cfg.get("paulian", {})),
                }
                return reg


            # -----------------------------------------------------------------------------
            # 5) Core Selection Logic (A-2 contract)
            # -----------------------------------------------------------------------------

            def select_elements_for_claim(claim_id: str, legal_elements: dict,
                                          max_elements: int) -> list[dict]:
                matched: list[tuple[int, str, dict]] = []
                for eid, edata in legal_elements.items():
                    if claim_id in edata.get("applicable_claims", []):
                        m = re.search(r"P(\d+)$", eid)
                        pnum = int(m.group(1)) if m else 999
                        matched.append((pnum, eid, edata))

                matched.sort(key=lambda x: (x[0], x[1]))
                return [{
                    "element_id": eid,
                    "element": edata.get("element", ""),
                    "description": edata.get("description", ""),
                } for _, eid, edata in matched[:max_elements]]


            def _credibility_bonus(cred: str, rules: dict, case_group: str) -> int:
                cfg = rules.get("plugins", {}).get(case_group, {})
                table = cfg.get("credibility_bonus", {})
                return int(table.get((cred or "").lower(), 0))


            def gather_candidate_facts(
                claim_id: str,
                claim_data: dict,
                case_group: str,
                plugin: BasePlugin,
                element: dict,
                claim_index: dict,
                fact_index: dict,
                legal_elements: dict,
                rules: dict,
                syn_map: dict[str, set[str]],
                max_facts: int,
            ) -> list[str]:
                elem_text = (element.get("element", "") + " " +
                            element.get("description", "")).strip()

                ctx = PluginContext(
                    claim_id=claim_id,
                    claim_data=claim_data,
                    claim_index=claim_index,
                    fact_index=fact_index,
                    legal_elements=legal_elements,
                    rules=rules,
                    syn_map=syn_map,
                )

                candidate_claims = plugin.expand_candidate_claims(ctx)

                # General fallback if plugin returns empty unexpectedly
                if not candidate_claims:
                    candidate_claims = {claim_id}

                has_direct = any(
                    claim_id in fdata.get("related_claims", [])
                    for _, fdata in fact_index.items()
                )
                if not has_direct:
                    # deterministic sibling augmentation
                    base = claim_text(claim_data)
                    for cid2, cdata2 in claim_index.items():
                        if cid2 == claim_id:
                            continue
                        if overlap_score(base, claim_text(cdata2), rules, syn_map) >= 4:
                            candidate_claims.add(cid2)

                scored: list[tuple[int, str]] = []
                for fid, fdata in fact_index.items():
                    if not str(fid).startswith("F-"):
                        continue
                    related = set(fdata.get("related_claims", []))
                    if not related.intersection(candidate_claims):
                        continue

                    fact_text = fdata.get("summary", "")
                    score = overlap_score(elem_text, fact_text, rules, syn_map)

                    # Strategy 3 baseline: overlap + traceability + deterministic tie-break.
                    if fdata.get("evidence_refs"):
                        score += 1

                    # plugin/rule bonus
                    score += _credibility_bonus(fdata.get("credibility", ""), rules, case_group)
                    score += plugin.fact_bonus(elem_text, fact_text)

                    scored.append((score, fid))

                scored.sort(key=lambda x: (-x[0], x[1]))
                return [fid for _, fid in scored[:max_facts]]


            def gather_candidate_evidence(
                claim_id: str,
                element: dict,
                fact_candidates: list[str],
                fact_index: dict,
                evidence_index: dict,
                rules: dict,
                syn_map: dict[str, set[str]],
                max_evidence: int,
            ) -> list[str]:
                elem_text = (element.get("element", "") + " " +
                            element.get("description", "")).strip()

                scores: dict[str, int] = {}

                # traceability first: from selected facts
                for fid in fact_candidates:
                    fdata = fact_index.get(fid, {})
                    for eref in fdata.get("evidence_refs", []):
                        m = re.match(r"(E-\d+)", str(eref))
                        if not m:
                            continue
                        eid = m.group(1)
                        if eid in evidence_index:
                            scores[eid] = scores.get(eid, 0) + 2

                # claim-level evidence
                for eid, edata in evidence_index.items():
                    if claim_id in edata.get("related_claims", []):
                        scores[eid] = scores.get(eid, 0) + 1

                # add overlap for determinism against element content
                for eid in list(scores.keys()):
                    ed = evidence_index.get(eid, {})
                    ev_text = " ".join([
                        ed.get("title", ""),
                        ed.get("doc_type", ""),
                        " ".join(ed.get("key_facts", [])),
                    ]).strip()
                    scores[eid] += overlap_score(elem_text, ev_text, rules, syn_map)

                ranked = sorted(scores.items(), key=lambda x: (-x[1], x[0]))
                return [eid for eid, _ in ranked[:max_evidence]]


            # -----------------------------------------------------------------------------
            # 6) Optional fields builders
            # -----------------------------------------------------------------------------

            def _build_fact_ledger_maps(fact_ledger: list[dict]) -> tuple[dict, dict]:
                by_fact = {}
                by_bo = {}
                for row in fact_ledger:
                    fid = row.get("fact_id", "")
                    bo = row.get("source_bo_id", "")
                    if fid:
                        by_fact[fid] = row
                    if bo:
                        by_bo[bo] = row
                return by_fact, by_bo


            def _derive_fact_parties(fid: str, fdata: dict,
                                    by_fact: dict, by_bo: dict) -> list[str]:
                row = by_fact.get(fid)
                if not row:
                    row = by_bo.get(fdata.get("source_bo_id", ""), {})
                parts = row.get("parties", []) if isinstance(row.get("parties", []), list) else []
                return [p for p in parts if isinstance(p, str) and p.strip()]


            def build_claim_parties(claim_id: str, claim_data: dict,
                                    fact_index: dict, fact_ledger: list[dict],
                                    global_parties: dict) -> dict:
                plaintiff = claim_data.get("plaintiff", "")
                defendant_str = claim_data.get("defendant", "")
                defendants = [d.strip() for d in defendant_str.split(",") if d.strip()]

                by_fact, by_bo = _build_fact_ledger_maps(fact_ledger)

                known_pl = set(global_parties.get("plaintiffs", []))
                known_def = set(global_parties.get("defendants", []))
                excluded = set(global_parties.get("excluded", []))

                third = set()
                for fid, fdata in fact_index.items():
                    if not str(fid).startswith("F-"):
                        continue
                    if claim_id not in fdata.get("related_claims", []):
                        continue
                    for p in _derive_fact_parties(fid, fdata, by_fact, by_bo):
                        if (p not in known_pl and p not in known_def and
                                p not in excluded and p not in defendants):
                            third.add(p)

                return {
                    "plaintiff": plaintiff,
                    "defendants": defendants,
                    "third_parties": sorted(third),
                }


            def detect_procedural_for_claim(claim_id: str, preclaim_text: str,
                                            all_proc: list[str], rules: dict) -> list[str]:
                kws = rules.get("procedural_keywords", [])
                found = set()
                scan = preclaim_text or ""
                for kw in kws:
                    if kw in scan and (claim_id in scan or kw in all_proc):
                        found.add(kw)
                for kw in all_proc:
                    if kw in kws:
                        found.add(kw)
                return sorted(found)


            def extract_warnings(claim_id: str, preclaim_text: str) -> list[str]:
                out = []
                in_warn = False
                for line in (preclaim_text or "").split("\n"):
                    if re.match(r"^##\s+8\.\s+VALIDATION", line, flags=re.IGNORECASE):
                        in_warn = True
                        continue
                    if in_warn and re.match(r"^##\s+\d+\.", line):
                        break
                    if not in_warn:
                        continue

                    m = re.match(r"\s*[-*]\s*\*\*(WARNING-\d+)\*\*:\s*(.*)", line)
                    if not m:
                        continue

                    wid = m.group(1)
                    wtxt = m.group(2).strip()
                    if claim_id in wtxt or _warning_applies(claim_id, wtxt):
                        out.append(f"{wid}: {wtxt}")

                return out


            def _warning_applies(claim_id: str, warn_text: str) -> bool:
                m = re.search(r"C-(\d+)", claim_id or "")
                if not m:
                    return False
                num = int(m.group(1))

                for s, e in re.findall(r"C-(\d+)\s*[~～]\s*C-(\d+)", warn_text):
                    if int(s) <= num <= int(e):
                        return True

                return claim_id in re.findall(r"C-\d+", warn_text)


            # -----------------------------------------------------------------------------
            # 7) Builder
            # -----------------------------------------------------------------------------

            def build_claim_packets(index: dict, fact_ledger: list[dict],
                                    preclaim_text: str, rules: dict,
                                    max_elements: int = 6, max_facts: int = 3,
                                    max_evidence: int = 3,
                                    rules_name: str = "claim_scoring_rules.json") -> dict:
                claim_index = index.get("claim_index", {})
                fact_index = index.get("fact_index", {})
                evidence_index = index.get("evidence_index", {})
                legal_elements = index.get("legal_elements_index", {})
                top_n = index.get("top_n_claims", [])
                parties = index.get("parties", {})
                all_proc = index.get("procedural_structures", [])

                syn_map = build_synonym_map(rules)
                plugins = build_plugin_registry(rules)

                packets = []
                global_paul_counter = 0

                for claim_id in top_n:
                    cdata = claim_index.get(claim_id, {})
                    ctype = cdata.get("claim_type", "")

                    case_group = detect_case_group(claim_id, cdata, legal_elements, rules)
                    plugin = plugins.get(case_group, plugins["general"])

                    elements_src = select_elements_for_claim(claim_id, legal_elements, max_elements)
                    elements = []
                    for elem in elements_src:
                        facts = gather_candidate_facts(
                            claim_id=claim_id,
                            claim_data=cdata,
                            case_group=case_group,
                            plugin=plugin,
                            element=elem,
                            claim_index=claim_index,
                            fact_index=fact_index,
                            legal_elements=legal_elements,
                            rules=rules,
                            syn_map=syn_map,
                            max_facts=max_facts,
                        )
                        evidences = gather_candidate_evidence(
                            claim_id=claim_id,
                            element=elem,
                            fact_candidates=facts,
                            fact_index=fact_index,
                            evidence_index=evidence_index,
                            rules=rules,
                            syn_map=syn_map,
                            max_evidence=max_evidence,
                        )
                        elements.append({
                            "element_id": elem["element_id"],
                            "element": elem["element"],
                            "fact_candidates": facts,
                            "evidence_candidates": evidences,
                        })

                    claim_parties = build_claim_parties(
                        claim_id, cdata, fact_index, fact_ledger, parties)
                    proc = detect_procedural_for_claim(
                        claim_id, preclaim_text, all_proc, rules)

                    modules = plugin.special_modules()

                    # paulian action expansion via plugin only
                    cdata_with_id = dict(cdata)
                    cdata_with_id["claim_id"] = claim_id
                    actions = plugin.build_special_actions(
                        claim_data=cdata_with_id,
                        claim_index=claim_index,
                        fact_index=fact_index,
                        fact_ledger=fact_ledger,
                        legal_elements=legal_elements,
                    )

                    for act in actions:
                        global_paul_counter += 1
                        act["paul_id"] = f"PAUL-{global_paul_counter}"

                    warns = extract_warnings(claim_id, preclaim_text)
                    summary = cdata.get("summary", "")
                    summary_clean = re.sub(r"\s*\((?:[FE]-\d+(?:,\s*)?)+\)", "", summary).strip()

                    packet = {
                        "claim_id": claim_id,
                        "claim_type": ctype,
                        "case_group": case_group,
                        "rank": cdata.get("rank", 0),
                        "total_score": cdata.get("total_score", 0),
                        "plaintiff": cdata.get("plaintiff", ""),
                        "defendant": cdata.get("defendant", ""),
                        "summary": summary_clean,
                        "elements": elements,
                        "parties": claim_parties,
                        "procedural_structures": proc,
                        "special_claim_modules": modules,
                        "paulian_actions": actions,
                    }
                    if warns:
                        packet["warnings"] = warns

                    packets.append(packet)

                return {
                    "meta": {
                        "stage": "4",
                        "phase": "A-2",
                        "top_n": len(top_n),
                        "generated_at": datetime.now().strftime("%Y-%m-%d"),
                        "rules_file": rules_name,
                    },
                    "claim_packets": packets,
                }


            # -----------------------------------------------------------------------------
            # 8) Validation
            # -----------------------------------------------------------------------------

            class ValidationReport:
                def __init__(self):
                    self.errors: list[str] = []
                    self.warnings: list[str] = []

                def error(self, msg: str):
                    self.errors.append(msg)

                def warn(self, msg: str):
                    self.warnings.append(msg)

                @property
                def ok(self) -> bool:
                    return len(self.errors) == 0

                def print_report(self):
                    if self.ok and not self.warnings:
                        print("[VALIDATE] ✓ All checks passed.", file=sys.stderr)
                        return
                    if self.errors:
                        print(f"[VALIDATE] ✗ {len(self.errors)} error(s):",
                              file=sys.stderr)
                        for e in self.errors:
                            print(f"  ERROR: {e}", file=sys.stderr)
                    if self.warnings:
                        print(f"[VALIDATE] △ {len(self.warnings)} warning(s):",
                              file=sys.stderr)
                        for w in self.warnings:
                            print(f"  WARN:  {w}", file=sys.stderr)


            def validate_contracts(index: dict, result: dict,
                                  index_schema: dict, output_schema: dict,
                                  max_elements: int = 6, max_facts: int = 3,
                                  max_evidence: int = 3) -> ValidationReport:
                rpt = ValidationReport()

                i_errors = validate_schema(index, index_schema)
                for e in i_errors:
                    rpt.error(f"INDEX_SCHEMA: {e}")

                o_errors = validate_schema(result, output_schema)
                for e in o_errors:
                    rpt.error(f"OUTPUT_SCHEMA: {e}")

                # Additional strict count checks
                top_n = index.get("top_n_claims", [])
                packets = result.get("claim_packets", [])

                if len(top_n) != len(packets):
                    rpt.error(
                        f"A-2 cardinality mismatch: top_n={len(top_n)} packets={len(packets)}")

                packet_map = {p.get("claim_id", ""): p for p in packets}
                for cid in top_n:
                    if cid not in packet_map:
                        rpt.error(f"Missing packet for top_n claim: {cid}")

                for p in packets:
                    cid = p.get("claim_id", "")
                    elements = p.get("elements", [])
                    if len(elements) > max_elements:
                        rpt.error(f"{cid}: elements>{max_elements}")

                    for el in elements:
                        if len(el.get("fact_candidates", [])) > max_facts:
                            rpt.error(f"{cid}/{el.get('element_id')}: fact_candidates>{max_facts}")
                        if len(el.get("evidence_candidates", [])) > max_evidence:
                            rpt.error(f"{cid}/{el.get('element_id')}: evidence_candidates>{max_evidence}")

                return rpt


            def _log(msg: str):
                """진행/디버그 메시지 → stderr (stdout은 결과 JSON 전용)."""
                print(msg, file=sys.stderr)


            # -----------------------------------------------------------------------------
            # 9) Main
            # -----------------------------------------------------------------------------

            def main():
                output_name = "stage4_claim_packets.json"
                max_elements = 6
                max_facts = 3
                max_evidence = 3
                do_validate = True

                with httpx.Client(timeout=60) as c:
                    # ===== 1) localdocs 연결 =====
                    r = c.post(LOCALDOCS_URL, json={
                        "jsonrpc": "2.0", "id": 1, "method": "initialize",
                        "params": {
                            "protocolVersion": "2025-03-26",
                            "capabilities": {},
                            "clientInfo": {"name": "stage4-claim-packets-gen", "version": "1.0"}
                        }
                    }, headers=HEADERS)
                    sid = r.headers.get("mcp-session-id")
                    if sid:
                        HEADERS["mcp-session-id"] = sid
                    c.post(LOCALDOCS_URL, json={
                        "jsonrpc": "2.0", "method": "notifications/initialized"
                    }, headers=HEADERS)
                    _log(f"1) Connected to localdocs (session: {sid})")

                    # 도구 목록 확인
                    r = c.post(LOCALDOCS_URL, json={
                        "jsonrpc": "2.0", "id": 2, "method": "tools/list"
                    }, headers=HEADERS)
                    tools_result = parse_sse(r.text)
                    tools = (tools_result.get("result", {}).get("tools", [])
                            if tools_result else [])
                    _log(f"   Tools: {[t['name'] for t in tools]}")

                    write_tool = next(
                        (t for t in tools if "write" in t["name"]), None)

                    # 문서 목록
                    docs = call_tool(c, "list_docs", {}, 3)
                    if docs and "result" in docs:
                        _log(f"   Docs: {docs['result']['content'][0]['text'][:500]}")

                    # ===== 2) 파일 로딩 (MCP read_doc) =====
                    _log("\n2) Loading input files via MCP...")

                    index = read_doc(c, "stage4_index.json", 10)
                    if index is None:
                        raise RuntimeError("Failed to load required: stage4_index.json")

                    rules = read_doc(c, "Default_Agent/claim_scoring_rules.json", 11)
                    if rules is None:
                        raise RuntimeError(
                            "Failed to load required: Default_Agent/claim_scoring_rules.json"
                        )
                    _log(f"   Default_Agent/claim_scoring_rules.json: loaded")

                    # 스키마 (optional — 없으면 검증 스킵)
                    index_schema = read_doc(c, "stage4_index.schema.json", 12)
                    output_schema = read_doc(c, "stage4_claim_packets.schema.json", 13)
                    if index_schema:
                        _log(f"   stage4_index.schema.json: loaded")
                    else:
                        _log(f"   stage4_index.schema.json: not found (skip validation)")
                    if output_schema:
                        _log(f"   stage4_claim_packets.schema.json: loaded")
                    else:
                        _log(f"   stage4_claim_packets.schema.json: not found (skip validation)")

                    fact_ledger = read_doc(c, "Fact_Ledger.json", 14)
                    if fact_ledger is None:
                        fact_ledger = read_doc(c, "fact_ledger.json", 15)
                    fact_ledger = fact_ledger or []

                    preclaim_text = read_doc_text(c, "청구전작업.md", 16) or ""

                    _log(f"   stage4_index.json: loaded")
                    _log(f"   Fact_Ledger: {len(fact_ledger)} entries")
                    _log(f"   청구전작업.md: {'loaded' if preclaim_text else 'not found'}")

                    # ===== 3) 빌드 =====
                    _log("\n3) Building claim packets...")
                    result = build_claim_packets(
                        index, fact_ledger, preclaim_text, rules,
                        max_elements=max_elements,
                        max_facts=max_facts,
                        max_evidence=max_evidence,
                    )

                    # ===== 4) 검증 =====
                    if do_validate and index_schema and output_schema:
                        rpt = validate_contracts(
                            index, result, index_schema, output_schema,
                            max_elements=max_elements,
                            max_facts=max_facts,
                            max_evidence=max_evidence,
                        )
                        rpt.print_report()
                    elif do_validate:
                        _log("[VALIDATE] Skipped: schema files not available")

                    # ===== 5) 결과 저장 (MCP write_doc + stdout) =====
                    output_json = json.dumps(result, ensure_ascii=False, indent=2)

                    if write_tool:
                        write_result = call_tool(c, write_tool["name"], {
                            "path": output_name,
                            "content": output_json,
                        }, 20)
                        if write_result and "result" in write_result:
                            _log(f"\n4) Written to localdocs: {output_name}")
                        else:
                            _log("\n4) Write to localdocs failed")
                    else:
                        _log("\n4) No write tool available")

                    # 항상 stdout으로 출력 (code-executor가 캡처)
                    print(output_json)


            if __name__ == "__main__":
                main()

          requirements: "httpx"
          network: "agent-network"
          timeout: 120

      task_procedure:
        IN:
          nexts: ["run_index"]
          wait_until: []
        run_index:
          nexts: ["run_compute_inputs", "run_claim_packets"]
          wait_until: []
        run_compute_inputs:
          nexts: ["OUT"]
          wait_until: ["run_index"]
        run_claim_packets:
          nexts: ["OUT"]
          wait_until: ["run_index"]