Files
Liti-agent-Development/Caching_설계/Claude_Opus_4_6_캐싱설계.md

6.8 KiB

질문

API를 사용해서 Opus 4.6을 불러서 추론 작업을 시키면, Opus 4.6은 읽은 파일을 다시 읽고, 다시 쓰고 하는 등 반복적으로 토큰을 사용한다. Opus 4.6을 API로 호출하여 사용할 때, input file을 한 번만 읽고 추론하며, 반복적으로 추론할 때도, caching된 자료만 사용하고, 파일을 생성할 때도, 추론을 통해 최종적으로 1회만 파일을 생성하도록 제약을 걸고 싶다. API로 Opus 4.6을 호출할 때 어떤 작업을 해야하는가?

답변

Opus 4.6 API 호출 시

토큰 효율화 방법

핵심은 Prompt Caching과 단일 턴 설계입니다.


1. Prompt Caching으로 파일을 1회만 읽기

파일 내용을 cache_control 블록에 넣으면, 첫 호출 후 캐시되어 이후 호출에서 재읽기 비용이 사라집니다.

import anthropic

client = anthropic.Anthropic()

# 파일 내용을 한 번만 로드
with open("input_document.md", "r") as f:
    file_content = f.read()

response = client.messages.create(
    model="claude-opus-4-6",
    max_tokens=8000,
    system=[
        {
            "type": "text",
            "text": "당신은 전문 분석가입니다. 주어진 문서를 분석하여 최종 결과물을 1회 생성하세요.",
        },
        {
            "type": "text",
            "text": file_content,          # ← 파일 내용
            "cache_control": {"type": "ephemeral"}  # ← 캐시 지정
        }
    ],
    messages=[
        {
            "role": "user",
            "content": "위 문서를 분석하고 최종 보고서를 작성하세요."
        }
    ]
)

캐시 효과:

  • 첫 호출: 파일 토큰 전체 과금 (캐시 저장)
  • 이후 호출: 캐시 토큰 비용만 과금 (약 90% 절감)
  • 캐시 유효 시간: 5분 (ephemeral 기준)

2. 반복 추론 시 캐시 재사용 구조

멀티턴 없이 누적 대화 히스토리에 캐시를 붙여 반복 추론을 최소화합니다.

import anthropic

client = anthropic.Anthropic()

with open("large_case_document.md", "r") as f:
    file_content = f.read()

# 캐시할 시스템 프롬프트 + 문서 (변하지 않는 부분)
SYSTEM_PROMPT = [
    {
        "type": "text",
        "text": "당신은 한국 민사소송 전문 AI입니다.",
    },
    {
        "type": "text", 
        "text": file_content,
        "cache_control": {"type": "ephemeral"}  # 문서는 캐시
    }
]

def single_reasoning_call(task_instruction: str) -> str:
    """파일 재읽기 없이 캐시된 문서로 단일 추론"""
    response = client.messages.create(
        model="claude-opus-4-6",
        max_tokens=8000,
        system=SYSTEM_PROMPT,           # 캐시된 시스템 사용
        messages=[
            {
                "role": "user",
                "content": task_instruction
            }
        ]
    )
    
    # 캐시 사용 현황 확인
    usage = response.usage
    print(f"캐시 읽기 토큰: {usage.cache_read_input_tokens}")
    print(f"캐시 생성 토큰: {usage.cache_creation_input_tokens}")
    print(f"일반 입력 토큰: {usage.input_tokens}")
    
    return response.content[0].text

# 동일 문서에 대해 여러 작업 → 파일은 1회만 읽힘
result_analysis  = single_reasoning_call("사실관계를 분석하세요.")
result_legal     = single_reasoning_call("법적 쟁점을 추출하세요.")
result_final     = single_reasoning_call("최종 소장을 작성하세요.")

3. 파일 생성 1회 제한 - 프롬프트 제약

모델이 중간에 파일을 생성하지 않도록 시스템 프롬프트에 명시적 제약을 걸고, 구조화된 출력을 요구합니다.

STRICT_SYSTEM = [
    {
        "type": "text",
        "text": """
## 출력 규칙 (엄수)
1. 중간 초안, 임시 결과물을 절대 출력하지 마세요.
2. 내부 추론(reasoning)은 <thinking> 태그 안에서만 수행하세요.
3. 최종 결과물만 <output> 태그로 1회 출력하세요.
4. <output> 태그는 응답 전체에서 정확히 1번만 사용하세요.

출력 형식:
<thinking>
[내부 추론 과정 - 토큰 절약을 위해 간결하게]
</thinking>
<output>
[최종 결과물만 작성]
</output>
        """
    },
    {
        "type": "text",
        "text": file_content,
        "cache_control": {"type": "ephemeral"}
    }
]

4. 파일을 1회만 쓰는 완성 코드

import anthropic
import re

client = anthropic.Anthropic()

def process_document_once(
    input_file: str, 
    output_file: str, 
    task: str
) -> dict:
    """
    - 파일 1회 읽기 (캐시)
    - 추론 1회 실행
    - 파일 1회 쓰기
    """
    
    # ① 파일 1회 읽기
    with open(input_file, "r", encoding="utf-8") as f:
        content = f.read()
    
    # ② API 단일 호출 (캐시 적용)
    response = client.messages.create(
        model="claude-opus-4-6",
        max_tokens=8000,
        system=[
            {
                "type": "text",
                "text": """최종 결과물만 <output></output> 태그로 1회 출력하세요.
중간 결과물, 반복 출력 금지."""
            },
            {
                "type": "text",
                "text": content,
                "cache_control": {"type": "ephemeral"}
            }
        ],
        messages=[{"role": "user", "content": task}]
    )
    
    raw_output = response.content[0].text
    
    # ③ <output> 태그에서 최종 결과만 추출
    match = re.search(r"<output>(.*?)</output>", raw_output, re.DOTALL)
    final_output = match.group(1).strip() if match else raw_output
    
    # ④ 파일 1회 쓰기
    with open(output_file, "w", encoding="utf-8") as f:
        f.write(final_output)
    
    return {
        "cache_read_tokens": response.usage.cache_read_input_tokens,
        "cache_creation_tokens": response.usage.cache_creation_input_tokens,
        "output_tokens": response.usage.output_tokens
    }

# 실행
stats = process_document_once(
    input_file="case_document.md",
    output_file="final_report.md",
    task="이 사건의 소장을 작성하세요."
)
print(stats)

요약

목표 방법
파일 1회 읽기 cache_control: ephemeral을 system에 적용
반복 추론 시 캐시 사용 동일 system 구조를 재사용 (5분 내 호출)
중간 출력 방지 <output> 태그 제약 + 단일 API 호출 설계
파일 1회 쓰기 응답 파싱 후 단일 write() 호출

캐시 TTL이 5분이므로, 동일 문서로 반복 작업 시 5분 이내 호출을 유지하거나, 더 긴 파이프라인에서는 각 스테이지마다 캐시를 재설정하는 구조가 필요합니다.