From 48fd0435a8a34b2b70c055d34a5065a7e78d4afe Mon Sep 17 00:00:00 2001 From: jhogyu Date: Wed, 30 Sep 2026 20:07:39 +0900 Subject: [PATCH] docs: record S2_00 IO analysis and request preparation plans --- .../Stage_2_S2_00_outdated_9_09.yml | 6248 +++++++++++++++++ .../YAML_Prompts/2. Stage_2/MEMORY.md | 15 + .../2. Stage_2/Stage_2_00_IO_info.md | 332 + .../plans/s2-00-request-preparation.md | 262 + .../plans/s2-00-request-preparation_v1.md | 206 + 5 files changed, 7063 insertions(+) create mode 100644 Case_02_Comparison_Research/YAML_Prompts/2. Stage_2/Default_Agent/Stage_2_Clean/agent_scripts/Stage_2_S2_00_outdated_9_09.yml create mode 100644 Case_02_Comparison_Research/YAML_Prompts/2. Stage_2/Stage_2_00_IO_info.md create mode 100644 Case_02_Comparison_Research/plans/s2-00-request-preparation.md create mode 100644 Case_02_Comparison_Research/plans/s2-00-request-preparation_v1.md diff --git a/Case_02_Comparison_Research/YAML_Prompts/2. Stage_2/Default_Agent/Stage_2_Clean/agent_scripts/Stage_2_S2_00_outdated_9_09.yml b/Case_02_Comparison_Research/YAML_Prompts/2. Stage_2/Default_Agent/Stage_2_Clean/agent_scripts/Stage_2_S2_00_outdated_9_09.yml new file mode 100644 index 00000000..b8c468d2 --- /dev/null +++ b/Case_02_Comparison_Research/YAML_Prompts/2. Stage_2/Default_Agent/Stage_2_Clean/agent_scripts/Stage_2_S2_00_outdated_9_09.yml @@ -0,0 +1,6248 @@ +Agent: + name: Stage_2_S2_00_v2 + version: "1.2.0" + description: >- + Stage 1 Part 1-4의 고정 입력을 검증·보존·정규화하고 claim-neutral + cluster slice와 S2_10 structured-context handoff plan을 생성하는 + S2_00 deterministic ingress Agent. + metadata: + workflow_id: S2_00 + execution_class: NON-LLM-DETERMINISTIC + execution_authority: MCP_CODE_EXECUTOR_INLINE + release_ref: Default_Agent/Stage_2_Clean/manifest/stage2_release.json + workflow_contract_ref: Default_Agent/Stage_2_Clean/workflows/S2_00_stage1_ingress_normalize_and_bundle_compile.yml + deployment_binding_ref: Default_Agent/Stage_2_Clean/deployment/stage2_code_executor_binding.yml + implementation_status: S2_00_CODE_AND_FIXTURE_OFFLINE_VERIFIED_S2_10_HYBRID_REIMPLEMENTATION_AND_RESEAL_PENDING_LIVE_ADMISSION_PENDING + Stages: + - name: S2_00 + description: >- + 단일 Code Executor 호출 안에서 C00, C05, C10, C15를 순차 실행하고 + localdocs binary IO와 status-last 논리 배리어로 결과를 발행한다. + prevs: [] + nexts: [] + tools: + mcpServers: + localdocs: + type: streamable-http + url: http://mcp-localdocs:8012/mcp + code-executor: + type: streamable-http + url: https://code-executor.mcp.eroomai.com/mcp + tasks: + - task_name: Task_S2_00_deterministic_ingress + description: >- + 고정 request와 release를 hydration하고 S2_00 pure core를 실행한 뒤 + binding-derived root에 결과를 status-last 방식으로 발행한다. + mcp: code-executor + tool_name: run_code + parameters: + language: python + requirements: "httpx==0.28.1" + network: agent-network + timeout: 300 + code: |- + #!/usr/bin/env python3 + """Deterministic Stage 2 ingress for the S2_00 contract. + + The pure ingress core uses the Python standard library; the inline MCP adapter + adds only pinned ``httpx`` transport to the fixed localdocs endpoint. It does + not perform legal reasoning, model calls, external-network access, dynamic + imports, or runtime installation. The release, source, conservation, context, + cluster, bundle, hydration, and logical-publish contracts are re-checked here. + """ + + from __future__ import annotations + + import argparse + import base64 + import binascii + from collections import Counter, defaultdict + import contextlib + from dataclasses import dataclass + import hashlib + import io + import itertools + import json + import math + import os + from pathlib import Path, PurePosixPath + import re + import shutil + import stat + import sys + import tempfile + import unicodedata + from typing import Any, Callable, Iterable, Mapping, MutableMapping, Sequence + + + ALGORITHM_VERSION = "s2_00_ingress/1.2.0" + ALGORITHM_SEMANTIC_DIGEST = hashlib.sha256( + b"LITI-S2_00-INLINE-ALGORITHM\x00" + ALGORITHM_VERSION.encode("ascii") + ).hexdigest() + MAX_FILE_BYTES = 32 * 1024 * 1024 + MAX_RUN_BYTES = 256 * 1024 * 1024 + MAX_JSON_DEPTH = 96 + MAX_JSON_ITEMS = 1_000_000 + ROUTES = frozenset({"TO_S2_10", "TO_S2_10_WITH_ISSUES", "TO_S2_40_STATUS_ONLY"}) + RELEASE_MODE = { + "STRUCTURAL_FIXTURE": "DEV_FIXTURE_RELEASE", + "SUBSET_CANARY": "SUBSET_CANARY_RELEASE", + "PRODUCTION": "PRODUCTION_RELEASE", + } + SEMANTIC_SIGNAL_KINDS = frozenset({"canonical", "domain_signal"}) + HARD_RELATION_KINDS = frozenset( + { + "SAME_BO_ID", + "SOURCE_BO_ATTACHMENT", + "SAME_EVIDENCE_REF", + "SAME_EVENT_REF", + "EXPLICIT_CASE_RELATION", + } + ) + CANDIDATE_RELATION_KINDS = frozenset( + {"claim_precondition", "accessory_of", "incompatible_with", "EXPLICIT_DEPENDENCY"} + ) + P1_DIGEST_KEYS = { + "evidence_indexed_sha256": "evidence_indexed", + "evidence_event_candidates_sha256": "evidence_event_candidates", + "b1_gate_sha256": "b1_evidence_indexed_gate", + "b2_gate_sha256": "b2_event_candidates_gate", + "screening_sha256": "domain_screening", + "activation_manifest_sha256": "domain_activation_manifest", + "registry_index_sha256": "stage1_domain_registry_index", + } + V2_DOMAIN_IDS = frozenset({"E-01", "E-06", "E-12", "E-16", "E-18", "E-19", "E-20", "E-21"}) + CONTEXT_SCHEMA_ID = "https://schemas.liti-agent.local/stage2/s2_00/context.schema.v2.json" + INGRESS_SCHEMA_ID = "https://schemas.liti-agent.local/stage2/s2_00/ingress.schema.v1.json" + REVIEW_SCHEMA_ID = "https://schemas.liti-agent.local/stage2/shared/review_status.schema.v1.json" + REQUIREMENT_CLASS_ENUM = { + "identity_backbone": "IDENTITY_BACKBONE", + "routing_profile_backbone": "ROUTING_PROFILE_BACKBONE", + "evidence_scope": "EVIDENCE_EVENT_SCOPE", + "event_scope": "EVIDENCE_EVENT_SCOPE", + "integrity_corroborator": "INTEGRITY_CORROBORATOR", + "optimization_context": "OPTIMIZATION_CONTEXT", + } + ADAPTER_IDS = { + "evidence_indexed": "S2A-EVIDENCE-V3-ENVELOPE-V1", + "evidence_event_candidates": "S2A-EVENTS-V1-ENVELOPE-V1", + "client_goal": "S2A-CLIENT-GOAL-V8-V1", + "domain_screening": "S2A-DOMAIN-SCREENING-V1", + "domain_activation_manifest": "S2A-DUAL-SG01-V1", + "b1_evidence_indexed_gate": "S2A-B1-GATE-V1", + "b2_event_candidates_gate": "S2A-B2-GATE-V1", + "stage1_part1_soft_gate_handoff": "S2A-P1-HANDOFF-FLAT-V1", + "bo": "S2A-BO-V8-LIST-V1", + "signal_manifest": "S2A-SIGNAL-ALL-V1", + "stage1_part2_review_handoff": "S2A-P2-HANDOFF-FLAT-V1", + "legal_effect_structures": "S2A-LES-CURRENT-V8-V1", + "stage1_part3_review_handoff": "S2A-P3-HANDOFF-WRAPPED-V1", + "fact_ledger_base": "S2A-FACT-LEDGER-CURRENT-V8-V1", + "fact_ledger_writer_report": "S2A-FACT-LEDGER-WRITER-REPORT-V1", + "stage1_part4_review_handoff": "S2A-P4-HANDOFF-WRAPPED-V1", + } + SG01_PROJECTION_FIELDS: tuple[str, ...] = ( + "schema_version", + "signal_id", + "status", + "registry_version", + "registry_index_sha256", + "screening_sha256", + "domain_entries", + "active_domain_ids", + "supporting_domain_ids", + "monitor_domain_ids", + "expected_runnable_domain_ids", + "required_calculation_domains", + "unrouted_material", + "conservation_gate", + "fail_open_policy", + "review_items", + "contract_guards", + ) + SG01_SET_FIELDS = frozenset( + { + "active_domain_ids", + "supporting_domain_ids", + "monitor_domain_ids", + "expected_runnable_domain_ids", + "required_calculation_domains", + } + ) + OUTPUT_SCHEMA_TARGETS: tuple[tuple[re.Pattern[str], str, str], ...] = ( + (re.compile(r"^ingress/stage1_input_manifest\.json$"), "ingress.schema.json", "#/$defs/stage1_input_manifest"), + (re.compile(r"^ingress/intake_report\.json$"), "ingress.schema.json", "#/$defs/intake_report"), + (re.compile(r"^ingress/ingress_status\.json$"), "ingress.schema.json", "#/$defs/ingress_status"), + (re.compile(r"^ingress/technical_diagnostic\.json$"), "ingress.schema.json", "#/$defs/technical_diagnostic"), + (re.compile(r"^review/issue_ledger\.base\.json$"), "review_status.schema.json", "#/$defs/issue_ledger_base"), + (re.compile(r"^context/case_context\.json$"), "context.schema.json", "#/$defs/case_context"), + (re.compile(r"^context/evidence_inventory\.json$"), "context.schema.json", "#/$defs/evidence_inventory"), + (re.compile(r"^context/object_registry\.json$"), "context.schema.json", "#/$defs/object_registry"), + (re.compile(r"^context/party_and_title_context\.json$"), "context.schema.json", "#/$defs/party_and_title_context"), + (re.compile(r"^context/slot_crosswalk\.json$"), "context.schema.json", "#/$defs/slot_crosswalk"), + (re.compile(r"^context/cluster_plan\.json$"), "context.schema.json", "#/$defs/cluster_plan"), + (re.compile(r"^context/cluster_slices/[^/]+\.json$"), "context.schema.json", "#/$defs/cluster_slice"), + (re.compile(r"^context/bundle_plan\.json$"), "context.schema.json", "#/$defs/bundle_plan"), + ) + _RAW_VALUE_UNSET = object() + + + # AgentBackend substitutes these two values in the deployed Agent YAML before + # Code Executor runs the byte-identical source. They intentionally remain + # literal placeholders in the offline parity mirror and its unit tests. + INLINE_USER_HASH = "{{__user_hash__}}" + INLINE_WORKSPACE_HASH = "{{__workspace_hash__}}" + EXPECTED_STAGE2_RELEASE_SHA256 = "9fa85bb94c5f14f675dcdaa0cf94d06745b21675d97797539ffc14015dcc9f53" + INLINE_REQUEST_PATH = "stage2_control/s2_00_request.json" + INLINE_STAGE2_ASSET_ROOT = "Default_Agent/Stage_2_Clean" + INLINE_STAGE2_RELEASE_PATH = ( + f"{INLINE_STAGE2_ASSET_ROOT}/manifest/stage2_release.json" + ) + LOCALDOCS_URL = "http://mcp-localdocs:8012/mcp" + MCP_PROTOCOL_VERSION = "2025-03-26" + INLINE_CLIENT_NAME = "liti-stage2-s2-00-inline" + INLINE_CLIENT_VERSION = "1.2.0" + INLINE_SCHEMA_MODULE_IDS = frozenset( + {"SCHEMA-INGRESS", "SCHEMA-CONTEXT", "SCHEMA-REVIEW-STATUS"} + ) + + + DEFAULT_SOURCE_CONTRACTS: tuple[dict[str, Any], ...] = ( + {"logical_input_id": "evidence_indexed", "path": "evidence_indexed.json", "criticality": "evidence_scope"}, + {"logical_input_id": "evidence_event_candidates", "path": "evidence_event_candidates.json", "criticality": "event_scope"}, + {"logical_input_id": "client_goal", "path": "client_goal.json", "criticality": "optimization_context"}, + {"logical_input_id": "domain_screening", "path": "routing/domain_screening.json", "criticality": "routing_profile_backbone"}, + {"logical_input_id": "domain_activation_manifest", "path": "routing/domain_activation_manifest.json", "criticality": "routing_profile_backbone"}, + {"logical_input_id": "b1_evidence_indexed_gate", "path": "quality_gates/B1_evidence_indexed_gate.json", "criticality": "integrity_corroborator"}, + {"logical_input_id": "b2_event_candidates_gate", "path": "quality_gates/B2_event_candidates_gate.json", "criticality": "integrity_corroborator"}, + {"logical_input_id": "stage1_part1_soft_gate_handoff", "path": "quality_gates/stage1_part1_soft_gate_handoff.json", "criticality": "integrity_corroborator"}, + {"logical_input_id": "bo", "path": "BO.json", "criticality": "identity_backbone"}, + {"logical_input_id": "signal_manifest", "path": "signals/signal_manifest.json", "criticality": "routing_profile_backbone"}, + {"logical_input_id": "stage1_part2_review_handoff", "path": "quality_gates/stage1_part2_review_handoff.json", "criticality": "integrity_corroborator"}, + {"logical_input_id": "legal_effect_structures", "path": "legal_effect_structures.json", "criticality": "routing_profile_backbone"}, + {"logical_input_id": "stage1_part3_review_handoff", "path": "quality_gates/stage1_part3_review_handoff.json", "criticality": "integrity_corroborator"}, + {"logical_input_id": "fact_ledger_base", "path": "Fact_Ledger_base.json", "criticality": "identity_backbone"}, + {"logical_input_id": "fact_ledger_writer_report", "path": "stage1_tmp/fact_ledger/fact_ledger_writer_report.json", "criticality": "integrity_corroborator"}, + {"logical_input_id": "stage1_part4_review_handoff", "path": "quality_gates/stage1_part4_review_handoff.json", "criticality": "integrity_corroborator"}, + ) + + + class IngressError(RuntimeError): + """A machine-readable deterministic ingress failure.""" + + def __init__( + self, + code: str, + message: str, + *, + logical_input_id: str | None = None, + details: Mapping[str, Any] | None = None, + ) -> None: + super().__init__(message) + self.code = code + self.logical_input_id = logical_input_id + self.details = dict(details or {}) + + def as_dict(self) -> dict[str, Any]: + result: dict[str, Any] = {"code": self.code, "message": str(self)} + if self.logical_input_id is not None: + result["logical_input_id"] = self.logical_input_id + if self.details: + result["details"] = self.details + return result + + + @dataclass(frozen=True, slots=True) + class Snapshot: + logical_input_id: str + relative_path: str + resolved_path: str + raw: bytes + raw_sha256: str + byte_length: int + device: int + inode: int + mtime_ns: int + + + def _reject_constant(value: str) -> None: + raise ValueError(f"non-finite JSON number is forbidden: {value}") + + + def _pairs_without_duplicates(pairs: Sequence[tuple[str, Any]]) -> dict[str, Any]: + result: dict[str, Any] = {} + for key, value in pairs: + if key in result: + raise ValueError(f"duplicate JSON key: {key}") + result[key] = value + return result + + + def _walk_json_limits(value: Any, *, max_depth: int, max_items: int) -> int: + count = 0 + stack: list[tuple[Any, int]] = [(value, 1)] + while stack: + current, depth = stack.pop() + if depth > max_depth: + raise IngressError("JSON_DEPTH_LIMIT", "JSON nesting depth exceeded") + if isinstance(current, dict): + count += len(current) + stack.extend((item, depth + 1) for item in current.values()) + elif isinstance(current, list): + count += len(current) + stack.extend((item, depth + 1) for item in current) + if count > max_items: + raise IngressError("JSON_ITEM_LIMIT", "JSON aggregate item limit exceeded") + return count + + + def load_json_strict( + source: Snapshot | bytes | bytearray | memoryview | str, + *, + max_depth: int = MAX_JSON_DEPTH, + max_items: int = MAX_JSON_ITEMS, + ) -> Any: + """Parse one UTF-8 JSON value, rejecting duplicate keys and non-finite numbers.""" + + if isinstance(source, Snapshot): + raw = source.raw + elif isinstance(source, str): + raw = source.encode("utf-8") + else: + raw = bytes(source) + try: + text = raw.decode("utf-8", errors="strict") + except UnicodeDecodeError as exc: + raise IngressError("INVALID_UTF8", "JSON source is not strict UTF-8") from exc + try: + value = json.loads( + text, + object_pairs_hook=_pairs_without_duplicates, + parse_constant=_reject_constant, + ) + except (json.JSONDecodeError, ValueError) as exc: + message = str(exc) + code = "DUPLICATE_JSON_KEY" if "duplicate JSON key" in message else "STRICT_JSON_PARSE_FAILED" + raise IngressError(code, message) from exc + _walk_json_limits(value, max_depth=max_depth, max_items=max_items) + return value + + + def canonical_json_bytes(value: Any) -> bytes: + """Return the project canonical parsed representation without normalizing strings.""" + + def reject_nonfinite(item: Any) -> None: + if isinstance(item, float) and not math.isfinite(item): + raise IngressError("NON_FINITE_NUMBER", "NaN and Infinity are forbidden") + if isinstance(item, dict): + for nested in item.values(): + reject_nonfinite(nested) + elif isinstance(item, (list, tuple)): + for nested in item: + reject_nonfinite(nested) + + reject_nonfinite(value) + try: + rendered = json.dumps( + value, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ) + except (TypeError, ValueError) as exc: + raise IngressError("CANONICAL_SERIALIZATION_FAILED", str(exc)) from exc + return (rendered + "\n").encode("utf-8") + + + def canonical_digest(value: Any) -> str: + return hashlib.sha256(canonical_json_bytes(value)).hexdigest() + + + class _SchemaViolation(ValueError): + """Internal deterministic JSON Schema validation failure.""" + + + def _json_equal(left: Any, right: Any) -> bool: + try: + return canonical_json_bytes(left) == canonical_json_bytes(right) + except IngressError: + return False + + + def _schema_pointer(document: Mapping[str, Any], fragment: str) -> Mapping[str, Any]: + if fragment in {"", "#"}: + return document + pointer = fragment[1:] if fragment.startswith("#") else fragment + if not pointer.startswith("/"): + raise _SchemaViolation(f"unsupported schema fragment: {fragment}") + current: Any = document + for token in pointer[1:].split("/"): + key = token.replace("~1", "/").replace("~0", "~") + if not isinstance(current, dict) or key not in current: + raise _SchemaViolation(f"unresolved schema pointer: {fragment}") + current = current[key] + if not isinstance(current, dict): + raise _SchemaViolation(f"schema pointer is not an object: {fragment}") + return current + + + def _schema_type_matches(value: Any, expected: str) -> bool: + return { + "object": isinstance(value, dict), + "array": isinstance(value, list), + "string": isinstance(value, str), + "integer": isinstance(value, int) and not isinstance(value, bool), + "number": isinstance(value, (int, float)) and not isinstance(value, bool), + "boolean": isinstance(value, bool), + "null": value is None, + }.get(expected, False) + + + def _validate_schema_node( + value: Any, + schema: Mapping[str, Any], + *, + root_schema: Mapping[str, Any], + schema_documents: Mapping[str, Mapping[str, Any]], + instance_path: str, + ) -> None: + reference = schema.get("$ref") + if isinstance(reference, str): + if reference.startswith("#"): + target_root = root_schema + fragment = reference + else: + name, separator, tail = reference.partition("#") + target_root = schema_documents.get(name) + if target_root is None: + raise _SchemaViolation(f"{instance_path}: external schema ref is not release-local: {reference}") + fragment = f"#{tail}" if separator else "#" + _validate_schema_node( + value, + _schema_pointer(target_root, fragment), + root_schema=target_root, + schema_documents=schema_documents, + instance_path=instance_path, + ) + return + if "const" in schema and not _json_equal(value, schema["const"]): + raise _SchemaViolation(f"{instance_path}: const mismatch") + if "enum" in schema and not any(_json_equal(value, candidate) for candidate in schema["enum"]): + raise _SchemaViolation(f"{instance_path}: enum mismatch") + forbidden = schema.get("not") + if isinstance(forbidden, dict) and _schema_branch_matches( + value, + forbidden, + root_schema=root_schema, + schema_documents=schema_documents, + instance_path=instance_path, + ): + raise _SchemaViolation(f"{instance_path}: forbidden schema branch matched") + expected_type = schema.get("type") + if expected_type is not None: + alternatives = [expected_type] if isinstance(expected_type, str) else list(expected_type) + if not any(_schema_type_matches(value, item) for item in alternatives): + raise _SchemaViolation(f"{instance_path}: expected type {alternatives}") + for keyword in ("oneOf", "anyOf"): + branches = schema.get(keyword) + if isinstance(branches, list): + matches = 0 + for branch in branches: + try: + _validate_schema_node( + value, + branch, + root_schema=root_schema, + schema_documents=schema_documents, + instance_path=instance_path, + ) + matches += 1 + except _SchemaViolation: + continue + required_matches = 1 if keyword == "oneOf" else None + if (required_matches is not None and matches != required_matches) or (keyword == "anyOf" and matches == 0): + raise _SchemaViolation(f"{instance_path}: {keyword} matched {matches} branches") + all_of = schema.get("allOf") + if isinstance(all_of, list): + for branch in all_of: + _validate_schema_node( + value, + branch, + root_schema=root_schema, + schema_documents=schema_documents, + instance_path=instance_path, + ) + condition = schema.get("if") + if isinstance(condition, dict): + condition_matches = True + try: + _validate_schema_node( + value, + condition, + root_schema=root_schema, + schema_documents=schema_documents, + instance_path=instance_path, + ) + except _SchemaViolation: + condition_matches = False + selected = schema.get("then" if condition_matches else "else") + if isinstance(selected, dict): + _validate_schema_node( + value, + selected, + root_schema=root_schema, + schema_documents=schema_documents, + instance_path=instance_path, + ) + if isinstance(value, dict): + minimum_properties = schema.get("minProperties") + maximum_properties = schema.get("maxProperties") + if isinstance(minimum_properties, int) and len(value) < minimum_properties: + raise _SchemaViolation(f"{instance_path}: minProperties {minimum_properties}") + if isinstance(maximum_properties, int) and len(value) > maximum_properties: + raise _SchemaViolation(f"{instance_path}: maxProperties {maximum_properties}") + required = schema.get("required", []) + if isinstance(required, list): + missing = [key for key in required if key not in value] + if missing: + raise _SchemaViolation(f"{instance_path}: missing required keys {missing}") + properties = schema.get("properties", {}) + if isinstance(properties, dict): + pattern_properties = schema.get("patternProperties", {}) + matched_by_pattern: set[str] = set() + if isinstance(pattern_properties, dict): + for key, child_value in value.items(): + for pattern_text, child_schema in pattern_properties.items(): + if re.search(pattern_text, key) is not None and isinstance(child_schema, dict): + matched_by_pattern.add(key) + _validate_schema_node( + child_value, + child_schema, + root_schema=root_schema, + schema_documents=schema_documents, + instance_path=f"{instance_path}/{key}", + ) + extras = sorted(set(value) - set(properties) - matched_by_pattern) + additional = schema.get("additionalProperties") + if additional is False: + if extras: + raise _SchemaViolation(f"{instance_path}: additional properties {extras}") + elif isinstance(additional, dict): + for key in extras: + _validate_schema_node( + value[key], + additional, + root_schema=root_schema, + schema_documents=schema_documents, + instance_path=f"{instance_path}/{key}", + ) + for key, child_schema in properties.items(): + if key in value and isinstance(child_schema, dict): + _validate_schema_node( + value[key], + child_schema, + root_schema=root_schema, + schema_documents=schema_documents, + instance_path=f"{instance_path}/{key}", + ) + if isinstance(value, list): + minimum = schema.get("minItems") + maximum = schema.get("maxItems") + if isinstance(minimum, int) and len(value) < minimum: + raise _SchemaViolation(f"{instance_path}: minItems {minimum}") + if isinstance(maximum, int) and len(value) > maximum: + raise _SchemaViolation(f"{instance_path}: maxItems {maximum}") + if schema.get("uniqueItems") is True: + digests = [canonical_digest(item) for item in value] + if len(digests) != len(set(digests)): + raise _SchemaViolation(f"{instance_path}: duplicate array items") + prefix_items = schema.get("prefixItems") + if isinstance(prefix_items, list): + for index, child_schema in enumerate(prefix_items): + if index < len(value) and isinstance(child_schema, dict): + _validate_schema_node( + value[index], + child_schema, + root_schema=root_schema, + schema_documents=schema_documents, + instance_path=f"{instance_path}/{index}", + ) + item_schema = schema.get("items") + if item_schema is False and isinstance(prefix_items, list) and len(value) > len(prefix_items): + raise _SchemaViolation(f"{instance_path}: additional array items are forbidden") + if isinstance(item_schema, dict): + start = len(prefix_items) if isinstance(prefix_items, list) else 0 + for index, item in enumerate(value[start:], start=start): + _validate_schema_node( + item, + item_schema, + root_schema=root_schema, + schema_documents=schema_documents, + instance_path=f"{instance_path}/{index}", + ) + contains = schema.get("contains") + if isinstance(contains, dict): + if not any( + _schema_branch_matches( + item, + contains, + root_schema=root_schema, + schema_documents=schema_documents, + instance_path=f"{instance_path}/{index}", + ) + for index, item in enumerate(value) + ): + raise _SchemaViolation(f"{instance_path}: contains did not match") + if isinstance(value, str): + min_length = schema.get("minLength") + if isinstance(min_length, int) and len(value) < min_length: + raise _SchemaViolation(f"{instance_path}: minLength {min_length}") + max_length = schema.get("maxLength") + if isinstance(max_length, int) and len(value) > max_length: + raise _SchemaViolation(f"{instance_path}: maxLength {max_length}") + pattern = schema.get("pattern") + if isinstance(pattern, str) and re.search(pattern, value) is None: + raise _SchemaViolation(f"{instance_path}: pattern mismatch") + if isinstance(value, (int, float)) and not isinstance(value, bool): + minimum = schema.get("minimum") + if isinstance(minimum, (int, float)) and value < minimum: + raise _SchemaViolation(f"{instance_path}: minimum {minimum}") + maximum = schema.get("maximum") + if isinstance(maximum, (int, float)) and value > maximum: + raise _SchemaViolation(f"{instance_path}: maximum {maximum}") + + + def _schema_branch_matches( + value: Any, + schema: Mapping[str, Any], + *, + root_schema: Mapping[str, Any], + schema_documents: Mapping[str, Mapping[str, Any]], + instance_path: str, + ) -> bool: + try: + _validate_schema_node( + value, + schema, + root_schema=root_schema, + schema_documents=schema_documents, + instance_path=instance_path, + ) + return True + except _SchemaViolation: + return False + + + def _load_output_schemas(asset_root: Path) -> dict[str, Mapping[str, Any]]: + documents: dict[str, Mapping[str, Any]] = {} + for name in ("ingress.schema.json", "context.schema.json", "review_status.schema.json"): + snapshot = open_bounded_snapshot(asset_root, f"schemas/{name}", logical_input_id=f"schema:{name}") + value = load_json_strict(snapshot) + if not isinstance(value, dict): + raise IngressError("OUTPUT_SCHEMA_SHAPE", f"schema is not an object: {name}") + documents[name] = value + return documents + + + def _validate_output_artifact( + relative_path: str, + value: Any, + schema_documents: Mapping[str, Mapping[str, Any]], + ) -> None: + target = next( + ( + (schema_name, schema_pointer) + for path_pattern, schema_name, schema_pointer in OUTPUT_SCHEMA_TARGETS + if path_pattern.fullmatch(relative_path) + ), + None, + ) + if target is None: + raise IngressError( + "OUTPUT_ARTIFACT_PATH_UNDECLARED", + f"no output schema target is declared for {relative_path}", + details={"path": relative_path}, + ) + schema_name, schema_pointer = target + root_schema = schema_documents[schema_name] + try: + _validate_schema_node( + value, + _schema_pointer(root_schema, schema_pointer), + root_schema=root_schema, + schema_documents=schema_documents, + instance_path=relative_path, + ) + except _SchemaViolation as exc: + raise IngressError( + "OUTPUT_SCHEMA_VALIDATION_FAILED", + str(exc), + details={"path": relative_path, "schema": schema_name, "schema_pointer": schema_pointer}, + ) from exc + + + def _safe_relative_path(relative_path: str) -> PurePosixPath: + if not isinstance(relative_path, str) or not relative_path: + raise IngressError("INVALID_SOURCE_PATH", "source path must be a non-empty string") + if "\x00" in relative_path or "\\" in relative_path: + raise IngressError("INVALID_SOURCE_PATH", "NUL and backslash are forbidden in logical paths") + logical = PurePosixPath(relative_path) + if logical.is_absolute() or any(part in {"", ".", ".."} for part in logical.parts): + raise IngressError("PATH_TRAVERSAL", f"unsafe relative path: {relative_path}") + return logical + + + def _assert_no_symlink_components(root: Path, logical: PurePosixPath) -> None: + current = root + for part in logical.parts: + current = current / part + try: + current_stat = current.lstat() + except FileNotFoundError: + return + if stat.S_ISLNK(current_stat.st_mode): + raise IngressError("SYMLINK_ESCAPE", f"symlink component rejected: {logical}") + + + def open_bounded_snapshot( + approved_root: str | os.PathLike[str], + relative_path: str, + *, + logical_input_id: str = "anonymous", + max_bytes: int = MAX_FILE_BYTES, + require_single_link: bool = True, + ) -> Snapshot: + """Read one regular file once from one descriptor and verify post-read identity.""" + + root_arg = Path(approved_root) + if root_arg.is_symlink(): + raise IngressError("SYMLINK_ROOT_REJECTED", "approved root itself may not be a symlink") + try: + root = root_arg.resolve(strict=True) + except FileNotFoundError as exc: + raise IngressError("APPROVED_ROOT_MISSING", "approved root does not exist") from exc + if not root.is_dir(): + raise IngressError("APPROVED_ROOT_NOT_DIRECTORY", "approved root must be a directory") + logical = _safe_relative_path(relative_path) + _assert_no_symlink_components(root, logical) + candidate = root.joinpath(*logical.parts) + try: + resolved = candidate.resolve(strict=True) + except FileNotFoundError as exc: + raise IngressError("SOURCE_MISSING", f"source is missing: {relative_path}", logical_input_id=logical_input_id) from exc + try: + resolved.relative_to(root) + except ValueError as exc: + raise IngressError("PATH_ESCAPE", f"resolved source escaped approved root: {relative_path}") from exc + flags = os.O_RDONLY + if hasattr(os, "O_CLOEXEC"): + flags |= os.O_CLOEXEC + if hasattr(os, "O_NOFOLLOW"): + flags |= os.O_NOFOLLOW + try: + descriptor = os.open(candidate, flags) + except OSError as exc: + raise IngressError("SOURCE_OPEN_FAILED", f"unable to open source: {relative_path}") from exc + try: + before = os.fstat(descriptor) + if not stat.S_ISREG(before.st_mode): + raise IngressError("NON_REGULAR_SOURCE", f"source is not a regular file: {relative_path}") + if require_single_link and before.st_nlink != 1: + raise IngressError("HARDLINK_POLICY_VIOLATION", f"source link count is {before.st_nlink}") + if before.st_size > max_bytes: + raise IngressError("SOURCE_SIZE_LIMIT", f"source exceeds {max_bytes} bytes") + chunks: list[bytes] = [] + total = 0 + while True: + chunk = os.read(descriptor, min(1024 * 1024, max_bytes + 1 - total)) + if not chunk: + break + chunks.append(chunk) + total += len(chunk) + if total > max_bytes: + raise IngressError("SOURCE_SIZE_LIMIT", f"source exceeds {max_bytes} bytes") + after = os.fstat(descriptor) + finally: + os.close(descriptor) + try: + path_after = candidate.stat(follow_symlinks=False) + except FileNotFoundError as exc: + raise IngressError("SOURCE_SNAPSHOT_CHANGED", "source disappeared after snapshot") from exc + identity_before = (before.st_dev, before.st_ino, before.st_size, before.st_mtime_ns) + identity_after = (after.st_dev, after.st_ino, after.st_size, after.st_mtime_ns) + path_identity = (path_after.st_dev, path_after.st_ino, path_after.st_size, path_after.st_mtime_ns) + if identity_before != identity_after or identity_after != path_identity: + raise IngressError("SOURCE_SNAPSHOT_CHANGED", f"source changed during snapshot: {relative_path}") + raw = b"".join(chunks) + return Snapshot( + logical_input_id=logical_input_id, + relative_path=logical.as_posix(), + resolved_path=str(resolved), + raw=raw, + raw_sha256=hashlib.sha256(raw).hexdigest(), + byte_length=len(raw), + device=after.st_dev, + inode=after.st_ino, + mtime_ns=after.st_mtime_ns, + ) + + + def resolve_stage1_sources( + stage1_run_root: str | os.PathLike[str], + contract_manifest: Mapping[str, Any] | None = None, + ) -> list[dict[str, Any]]: + """Resolve only approved logical kinds; a relocation manifest cannot invent kinds.""" + + root = Path(stage1_run_root).resolve(strict=True) + if not root.is_dir(): + raise IngressError("STAGE1_ROOT_NOT_DIRECTORY", "Stage 1 run root must be a directory") + contracts = [dict(row) for row in DEFAULT_SOURCE_CONTRACTS] + overrides = dict((contract_manifest or {}).get("path_overrides", {})) + approved_ids = {row["logical_input_id"] for row in contracts} + invented = sorted(set(overrides) - approved_ids) + if invented: + raise IngressError("UNAPPROVED_LOGICAL_KIND", "relocation manifest invented logical kinds", details={"ids": invented}) + seen_paths: set[str] = set() + for row in contracts: + path = overrides.get(row["logical_input_id"], row["path"]) + safe = _safe_relative_path(path).as_posix() + if safe in seen_paths: + raise IngressError("DUPLICATE_LOGICAL_MAPPING", f"duplicate physical mapping: {safe}") + seen_paths.add(safe) + row["expected_path"] = row.pop("path") + row["observed_path"] = safe + row["resolution_source"] = ( + "RELEASE_BOUND_CONTRACT_MANIFEST" + if row["logical_input_id"] in overrides + else "DEFAULT_EXACT_PATH" + ) + return contracts + + + def load_release_lock(release_path: str | os.PathLike[str]) -> dict[str, Any]: + """Load a strict release lock without interpreting a historical status label.""" + + path = Path(release_path) + snapshot = open_bounded_snapshot(path.parent, path.name, logical_input_id="stage2_release_lock") + value = load_json_strict(snapshot) + if not isinstance(value, dict): + raise IngressError("RELEASE_LOCK_SHAPE", "release lock must be an object") + release_class = value.get("release_class") + if release_class not in set(RELEASE_MODE.values()): + raise IngressError("RELEASE_CLASS_INVALID", f"unsupported release class: {release_class}") + value = dict(value) + value["_release_raw_sha256"] = snapshot.raw_sha256 + return value + + + def _issue( + code: str, + *, + impact_scope: str = "GLOBAL", + source_refs: Sequence[str] = (), + severity: str = "ERROR", + message: str | None = None, + ) -> dict[str, Any]: + return { + "issue_code": code, + "severity": severity, + "impact_scope": impact_scope, + "scope_refs": sorted(set(source_refs)), + "source_contract_row_refs": sorted(set(source_refs)), + "reason_codes": [code], + "downstream_allowed_actions": [], + "message": message or code, + } + + + def _shape_required(value: Any, keys: Sequence[str]) -> list[str]: + if not isinstance(value, dict): + return list(keys) + return [key for key in keys if key not in value] + + + def _json_pointer_value(document: Any, pointer: str | None) -> tuple[bool, Any]: + if pointer in {None, ""}: + return (pointer == "", document) + if not isinstance(pointer, str) or not pointer.startswith("/"): + return False, None + current = document + for raw_token in pointer[1:].split("/"): + token = raw_token.replace("~1", "/").replace("~0", "~") + if isinstance(current, dict) and token in current: + current = current[token] + elif isinstance(current, list) and token.isdigit() and int(token) < len(current): + current = current[int(token)] + else: + return False, None + return True, current + + + def _release_stage1_source_rows(release_lock: Mapping[str, Any]) -> list[Mapping[str, Any]]: + rows = release_lock.get("stage1_sources") + if not isinstance(rows, list): + dependency = release_lock.get("dependency_locks", {}).get("stage1", {}) + rows = dependency.get("stage1_sources") if isinstance(dependency, dict) else None + return [row for row in rows if isinstance(row, dict)] if isinstance(rows, list) else [] + + + def _adapter_decision(release_lock: Mapping[str, Any], adapter_id: str) -> Mapping[str, Any] | None: + for row in release_lock.get("adapter_decisions", []): + if isinstance(row, dict) and row.get("adapter_id") == adapter_id and isinstance(row.get("decision"), dict): + return row["decision"] + return None + + + def _closed_adapter_shape_errors( + document: Any, + *, + logical_id: str, + adapter_id: str, + required_keys: Sequence[str], + release_lock: Mapping[str, Any], + ) -> list[str]: + errors: list[str] = [] + if required_keys: + errors.extend(f"missing root key {key}" for key in _shape_required(document, required_keys)) + decision = _adapter_decision(release_lock, adapter_id) + if decision is not None: + root_shape = decision.get("root_shape") + if root_shape == "ARRAY" and not isinstance(document, list): + errors.append("root must be an array") + elif root_shape == "OBJECT_ENVELOPE" and not isinstance(document, dict): + errors.append("root must be an object envelope") + if isinstance(document, dict): + errors.extend( + f"missing root key {key}" + for key in _shape_required(document, decision.get("required_root_fields", [])) + ) + if isinstance(document, list): + required_item_fields = decision.get("required_item_fields", decision.get("required_row_fields", [])) + if isinstance(required_item_fields, list): + for index, item in enumerate(document): + for key in _shape_required(item, required_item_fields): + errors.append(f"row {index} missing {key}") + if decision is None: + fallback_required: dict[str, tuple[str, ...]] = { + "evidence_indexed": ("schema_contract_version", "items"), + "evidence_event_candidates": ("schema_version", "items"), + "domain_activation_manifest": SG01_PROJECTION_FIELDS, + "signal_manifest": ("downstream_read_sets", "files"), + "legal_effect_structures": ("schema_version", "structure_records"), + "fact_ledger_writer_report": ( + "schema_version", + "row_count", + "gate_firings", + "domain_effect_coverage", + "calculation_readiness", + "blocked_review_items", + "conservation", + "final_sha256", + ), + } + fallback = fallback_required.get(logical_id, ()) + if fallback: + errors.extend(f"missing root key {key}" for key in _shape_required(document, fallback)) + if logical_id in {"bo", "fact_ledger_base"} and not isinstance(document, list): + errors.append("root must be an array") + return sorted(set(errors)) + + + def _schema_document_index(deployment_documents: Mapping[str, Any]) -> dict[str, Mapping[str, Any]]: + result: dict[str, Mapping[str, Any]] = {} + for path, document in deployment_documents.items(): + if not isinstance(document, dict): + continue + result[path] = document + result[PurePosixPath(path).name] = document + schema_id = document.get("$id") + if isinstance(schema_id, str): + result[schema_id] = document + return result + + + def _source_hash_index(document: Mapping[str, Any] | None) -> dict[str, str]: + result: dict[str, str] = {} + if not isinstance(document, dict): + return result + candidate_arrays: list[Any] = [] + for key in ("source_rows", "sources", "artifacts", "files", "entries"): + if isinstance(document.get(key), list): + candidate_arrays.append(document[key]) + for wrapper in ("completion_seal", "manifest", "payload", "data"): + nested = document.get(wrapper) + if isinstance(nested, dict): + for key in ("source_rows", "sources", "artifacts", "files", "entries"): + if isinstance(nested.get(key), list): + candidate_arrays.append(nested[key]) + for rows in candidate_arrays: + for row in rows: + if not isinstance(row, dict): + continue + digest = row.get("raw_sha256", row.get("sha256")) + if not isinstance(digest, str) or re.fullmatch(r"[A-Fa-f0-9]{64}", digest) is None: + continue + for key in ("logical_input_id", "path", "observed_path", "logical_id"): + identifier = row.get(key) + if isinstance(identifier, str) and identifier: + result[identifier] = digest.lower() + return result + + + def _source_producer_index(document: Mapping[str, Any] | None) -> dict[str, str]: + """Index producer evidence carried by a bounded completion/manifest row.""" + + result: dict[str, str] = {} + if not isinstance(document, dict): + return result + candidate_arrays: list[Any] = [] + for key in ("source_rows", "sources", "artifacts", "files", "entries"): + if isinstance(document.get(key), list): + candidate_arrays.append(document[key]) + for wrapper in ("completion_seal", "manifest", "payload", "data"): + nested = document.get(wrapper) + if isinstance(nested, dict): + for key in ("source_rows", "sources", "artifacts", "files", "entries"): + if isinstance(nested.get(key), list): + candidate_arrays.append(nested[key]) + for rows in candidate_arrays: + for row in rows: + if not isinstance(row, dict): + continue + producer = next( + ( + row.get(key) + for key in ("producer_id", "created_by", "writer_id", "writer", "finalized_by") + if isinstance(row.get(key), str) and row.get(key) + ), + None, + ) + if not isinstance(producer, str): + continue + for key in ("logical_input_id", "path", "observed_path", "logical_id"): + identifier = row.get(key) + if isinstance(identifier, str) and identifier: + result[identifier] = producer + return result + + + def _producer_value(document: Any) -> str | None: + if not isinstance(document, dict): + return None + for key in ("producer_id", "created_by", "writer_id", "writer", "finalized_by"): + value = document.get(key) + if isinstance(value, str) and value: + return value + for wrapper in ("metadata", "meta", "handoff", "payload"): + nested = document.get(wrapper) + if isinstance(nested, dict): + for key in ("producer_id", "created_by", "writer_id", "writer", "finalized_by"): + value = nested.get(key) + if isinstance(value, str) and value: + return value + # P3/P4 are closed one-key wrappers in the Stage 1 v8 handoff contract. + for wrapper in ( + "stage1_part3_review_handoff", + "stage1_part4_review_handoff", + ): + nested = document.get(wrapper) + if isinstance(nested, dict): + for key in ("created_by", "finalized_by"): + value = nested.get(key) + if isinstance(value, str) and value: + return value + return None + + + def _producer_matches( + observed: str, + expected: str, + alias_id: str | None, + release_lock: Mapping[str, Any], + ) -> bool: + if observed == expected: + return True + if alias_id is None: + return False + decision = _adapter_decision(release_lock, alias_id) + if decision is None or decision.get("bidirectional_match_allowed") is not True: + return False + pair = {decision.get("schema_writer_id"), decision.get("orchestration_producer_id")} + return {observed, expected} == pair + + + def _identity_ref(document: Any, pointer: str | None, logical_id: str) -> dict[str, Any]: + if pointer is None: + return {"value": None, "disposition": "NOT_APPLICABLE", "source_ref": logical_id} + found, value = _json_pointer_value(document, pointer) + if not found or value is None: + return {"value": None, "disposition": "MISSING", "source_ref": f"{logical_id}#{pointer}"} + return {"value": str(value), "disposition": "OBSERVED", "source_ref": f"{logical_id}#{pointer}"} + + + def validate_ingress_contracts( + snapshots: Mapping[str, Snapshot], + contracts: Sequence[Mapping[str, Any]], + release_lock: Mapping[str, Any], + *, + deployment_snapshots: Mapping[str, Snapshot] | None = None, + deployment_documents: Mapping[str, Any] | None = None, + completion_seal: Mapping[str, Any] | None = None, + contract_manifest: Mapping[str, Any] | None = None, + ) -> dict[str, Any]: + """Strictly parse sources and verify release-bound schema, producer, identity, and seal rows.""" + + documents: dict[str, Any] = {} + rows: list[dict[str, Any]] = [] + issues: list[dict[str, Any]] = [] + deployment_snapshots = deployment_snapshots or {} + deployment_documents = deployment_documents or {} + deployment_by_path = {snapshot.relative_path: snapshot for snapshot in deployment_snapshots.values()} + schema_documents = _schema_document_index(deployment_documents) + release_source_rows = _release_stage1_source_rows(release_lock) + release_ids = [str(row.get("logical_input_id")) for row in release_source_rows] + duplicate_release_ids = sorted(key for key, count in Counter(release_ids).items() if count > 1) + if duplicate_release_ids: + raise IngressError( + "RELEASE_SOURCE_CONTRACT_DUPLICATE", + "release stage1_sources contains duplicate logical_input_id rows", + details={"logical_input_ids": duplicate_release_ids}, + ) + expected_fixed = { + str(row["logical_input_id"]): str(row["path"]) + for row in DEFAULT_SOURCE_CONTRACTS + } + expected_release_ids = set(expected_fixed) | {"signal_payload_family"} + observed_release_ids = set(release_ids) + if observed_release_ids != expected_release_ids: + raise IngressError( + "RELEASE_SOURCE_CONTRACT_SET_MISMATCH", + "release stage1_sources must be the exact 16 fixed inputs plus signal_payload_family", + details={ + "missing": sorted(expected_release_ids - observed_release_ids), + "extra": sorted(observed_release_ids - expected_release_ids), + }, + ) + release_rows = {str(row.get("logical_input_id")): row for row in release_source_rows} + for logical_id, expected_path in expected_fixed.items(): + release_row = release_rows[logical_id] + if release_row.get("path") != expected_path or release_row.get("path_rule") not in {None, ""}: + raise IngressError( + "RELEASE_SOURCE_FIXED_PATH_MISMATCH", + f"fixed source path contract mismatch: {logical_id}", + ) + signal_family = release_rows["signal_payload_family"] + if ( + signal_family.get("path") is not None + or signal_family.get("path_rule") != "signals/" + or signal_family.get("adapter_id") != "S2A-SIGNAL-PAYLOAD-FAMILY-V1" + or signal_family.get("raw_hash_source") != "MANIFEST_ROW" + ): + raise IngressError( + "SIGNAL_PAYLOAD_FAMILY_CONTRACT_MISMATCH", + "signal_payload_family must use the approved manifest-expanded path contract", + ) + completion_hashes = _source_hash_index(completion_seal) + manifest_hashes = _source_hash_index(contract_manifest) + completion_producers = _source_producer_index(completion_seal) + manifest_producers = _source_producer_index(contract_manifest) + for contract in contracts: + logical_id = str(contract["logical_input_id"]) + snapshot = snapshots.get(logical_id) + release_row = release_rows.get(logical_id) + contract_missing = release_row is None + release_row = release_row or {} + alias_value = release_row.get("producer_alias", release_row.get("producer_alias_id")) + alias_id = str(alias_value) if isinstance(alias_value, str) else None + schema_ref = release_row.get("schema_ref") if isinstance(release_row.get("schema_ref"), dict) else None + row = { + "logical_input_id": logical_id, + "requirement_class": REQUIREMENT_CLASS_ENUM.get( + str(contract.get("criticality")), + "INTEGRITY_CORROBORATOR", + ), + "expected_path": contract.get("expected_path"), + "observed_path": contract.get("observed_path"), + "resolution_source": contract.get("resolution_source"), + "schema_id": schema_ref.get("$id") if schema_ref else release_row.get("schema_id"), + "schema_sha256": schema_ref.get("sha256") if schema_ref else release_row.get("schema_sha256"), + "producer_id": release_row.get("producer_id"), + "producer_alias_id": alias_id, + "adapter_id": release_row.get("adapter_id", ADAPTER_IDS.get(logical_id, "S2A-UNBOUND-V1")), + "run_identity_ref": release_row.get("run_identity_ref", {"value": None, "disposition": "MISSING", "source_ref": logical_id}), + "transaction_identity_ref": release_row.get("transaction_identity_ref", {"value": None, "disposition": "MISSING", "source_ref": logical_id}), + "scope_refs": [logical_id], + "source_contract_row_refs": [logical_id], + "reason_codes": [], + "downstream_allowed_actions": [], + "issue_codes": [], + } + if contract_missing: + code = "RELEASE_SOURCE_CONTRACT_MISSING" + row["reason_codes"].append(code) + row["issue_codes"].append(code) + issues.append(_issue(code, impact_scope="GLOBAL", source_refs=[logical_id])) + declared_path = release_row.get("path") + if isinstance(declared_path, str) and declared_path != contract.get("expected_path"): + code = "RELEASE_SOURCE_PATH_MISMATCH" + row["reason_codes"].append(code) + row["issue_codes"].append(code) + issues.append(_issue(code, impact_scope="GLOBAL", source_refs=[logical_id])) + if snapshot is None: + row.update( + { + "raw_sha256": None, + "byte_length": 0, + "parse_status": "NOT_OBSERVED", + "schema_status": "UNEVALUABLE", + "seal_status": "UNEVALUABLE", + "scope_technical_disposition": "UNAVAILABLE", + "impact_scope": "GLOBAL" if contract.get("criticality") == "identity_backbone" else "CLUSTER", + } + ) + code = "SOURCE_MISSING" + row["reason_codes"].append(code) + row["issue_codes"].append(code) + issues.append(_issue(code, impact_scope=row["impact_scope"], source_refs=[logical_id])) + rows.append(row) + continue + row["raw_sha256"] = snapshot.raw_sha256 + row["byte_length"] = snapshot.byte_length + try: + document = load_json_strict( + snapshot, + max_depth=int(release_lock.get("limits", {}).get("max_json_depth", MAX_JSON_DEPTH)), + max_items=int(release_lock.get("limits", {}).get("max_json_items", MAX_JSON_ITEMS)), + ) + documents[logical_id] = document + row["parse_status"] = "PASS" + except IngressError as exc: + row["parse_status"] = "FAIL" + row["schema_status"] = "UNEVALUABLE" + row["seal_status"] = "UNEVALUABLE" + row["scope_technical_disposition"] = "UNAVAILABLE" + row["impact_scope"] = "GLOBAL" if contract.get("criticality") == "identity_backbone" else "CLUSTER" + row["reason_codes"].append(exc.code) + row["issue_codes"].append(exc.code) + issues.append(_issue(exc.code, impact_scope=row["impact_scope"], source_refs=[logical_id], message=str(exc))) + rows.append(row) + continue + expected_adapter = ADAPTER_IDS.get(logical_id) + if expected_adapter is not None and release_row.get("adapter_id") not in {None, expected_adapter}: + code = "ADAPTER_ID_MISMATCH" + row["schema_status"] = "FAIL" + row["reason_codes"].append(code) + row["issue_codes"].append(code) + issues.append(_issue(code, impact_scope="GLOBAL", source_refs=[logical_id])) + if schema_ref is not None: + schema_path = schema_ref.get("path") + schema_snapshot = deployment_by_path.get(schema_path) if isinstance(schema_path, str) else None + schema_document = deployment_documents.get(schema_path) if isinstance(schema_path, str) else None + expected_schema_hash = schema_ref.get("sha256") + expected_schema_id = schema_ref.get("$id") + if schema_snapshot is None or not isinstance(schema_document, dict): + schema_error = "SCHEMA_REF_NOT_IN_BOUNDED_DEPLOYMENT" + elif not isinstance(expected_schema_hash, str) or schema_snapshot.raw_sha256 != expected_schema_hash.lower(): + schema_error = "SCHEMA_HASH_MISMATCH" + elif expected_schema_id is not None and schema_document.get("$id") != expected_schema_id: + schema_error = "SCHEMA_ID_MISMATCH" + else: + schema_error = None + try: + _validate_schema_node( + document, + schema_document, + root_schema=schema_document, + schema_documents=schema_documents, + instance_path=logical_id, + ) + except _SchemaViolation as exc: + schema_error = "SOURCE_SCHEMA_VALIDATION_FAILED" + issues.append( + _issue( + schema_error, + impact_scope="CLUSTER", + source_refs=[logical_id], + message=str(exc), + ) + ) + if schema_error is not None: + row["schema_status"] = "FAIL" + row["reason_codes"].append(schema_error) + row["issue_codes"].append(schema_error) + if schema_error != "SOURCE_SCHEMA_VALIDATION_FAILED": + issues.append(_issue(schema_error, impact_scope="GLOBAL", source_refs=[logical_id])) + else: + row["schema_status"] = "PASS" + else: + adapter_errors = _closed_adapter_shape_errors( + document, + logical_id=logical_id, + adapter_id=str(row["adapter_id"]), + required_keys=release_row.get("required_keys", []), + release_lock=release_lock, + ) + if contract_missing: + row["schema_status"] = "UNEVALUABLE" + elif adapter_errors: + code = "ADAPTER_REQUIRED_KEY_MISSING" + row["schema_status"] = "FAIL" + row["reason_codes"].append(code) + row["issue_codes"].append(code) + issues.append( + _issue( + code, + impact_scope="CLUSTER", + source_refs=[logical_id], + message="; ".join(adapter_errors), + ) + ) + else: + row["schema_status"] = "PASS" + expected_producer = release_row.get("producer_id") + document_producer = _producer_value(document) + sealed_producer = ( + completion_producers.get(logical_id) + or completion_producers.get(str(contract.get("observed_path"))) + or manifest_producers.get(logical_id) + or manifest_producers.get(str(contract.get("observed_path"))) + ) + if ( + document_producer is not None + and sealed_producer is not None + and document_producer != sealed_producer + ): + code = "PRODUCER_EVIDENCE_CONFLICT" + row["reason_codes"].append(code) + row["issue_codes"].append(code) + issues.append(_issue(code, impact_scope="GLOBAL", source_refs=[logical_id])) + observed_producer = document_producer or sealed_producer + if isinstance(expected_producer, str): + if observed_producer is None: + code = "PRODUCER_ID_UNEVALUABLE" + row["reason_codes"].append(code) + row["issue_codes"].append(code) + issues.append(_issue(code, impact_scope="CLUSTER", source_refs=[logical_id])) + elif not _producer_matches(observed_producer, expected_producer, alias_id, release_lock): + code = "PRODUCER_ID_MISMATCH" + row["reason_codes"].append(code) + row["issue_codes"].append(code) + issues.append(_issue(code, impact_scope="GLOBAL", source_refs=[logical_id])) + row["run_identity_ref"] = _identity_ref(document, release_row.get("run_identity_pointer"), logical_id) + row["transaction_identity_ref"] = _identity_ref( + document, + release_row.get("transaction_identity_pointer"), + logical_id, + ) + raw_hash_source = str(release_row.get("raw_hash_source", "NONE")) + if raw_hash_source in {"CASE_RUN_COMPLETION_SEAL", "COMPLETION_SEAL", "COMPLETION_SEAL_ROW"}: + expected_hash = completion_hashes.get(logical_id) or completion_hashes.get(str(contract.get("observed_path"))) + elif raw_hash_source in {"CONTRACT_MANIFEST", "CONTRACT_MANIFEST_ROW", "MANIFEST_ROW"}: + expected_hash = manifest_hashes.get(logical_id) or manifest_hashes.get(str(contract.get("observed_path"))) + elif raw_hash_source in {"COMPLETION_SEAL_OR_CONTRACT_MANIFEST", "SEALED_ROW"}: + expected_hash = ( + completion_hashes.get(logical_id) + or completion_hashes.get(str(contract.get("observed_path"))) + or manifest_hashes.get(logical_id) + or manifest_hashes.get(str(contract.get("observed_path"))) + ) + elif raw_hash_source in {"UNAVAILABLE_DEV", "NONE"}: + expected_hash = None + else: + expected_hash = None + code = "RAW_HASH_SOURCE_UNAPPROVED" + row["reason_codes"].append(code) + row["issue_codes"].append(code) + issues.append(_issue(code, impact_scope="GLOBAL", source_refs=[logical_id])) + if expected_hash is not None and expected_hash != snapshot.raw_sha256: + code = "RAW_HASH_MISMATCH" + row["seal_status"] = "FAIL" + row["reason_codes"].append(code) + row["issue_codes"].append(code) + issues.append(_issue(code, impact_scope="GLOBAL", source_refs=[logical_id])) + else: + row["seal_status"] = "PASS" if expected_hash else "UNEVALUABLE" + row["scope_technical_disposition"] = ( + "UNAVAILABLE" + if contract_missing or any(code in row["issue_codes"] for code in {"RAW_HASH_MISMATCH", "SCHEMA_HASH_MISMATCH", "SCHEMA_ID_MISMATCH"}) + else "AVAILABLE" + if not row["issue_codes"] + else "AVAILABLE_WITH_ISSUES" + ) + row["impact_scope"] = "GLOBAL" if contract.get("criticality") == "identity_backbone" else "CLUSTER" + rows.append(row) + for identity_kind, field in ( + ("RUN", "run_identity_ref"), + ("TRANSACTION", "transaction_identity_ref"), + ): + observed_values = { + str(row[field]["value"]) + for row in rows + if row[field].get("disposition") == "OBSERVED" and row[field].get("value") is not None + } + if len(observed_values) > 1: + code = f"{identity_kind}_IDENTITY_CONFLICT" + issues.append(_issue(code, impact_scope="GLOBAL", source_refs=sorted(observed_values))) + for row in rows: + if row[field].get("disposition") == "OBSERVED": + row["issue_codes"] = sorted(set(row["issue_codes"] + [code])) + row["reason_codes"] = sorted(set(row["reason_codes"] + [code])) + row["scope_technical_disposition"] = "UNAVAILABLE" + return {"documents": documents, "source_contract_rows": rows, "issues": issues} + + + def _records_from_signal_document(document: Any) -> list[Any]: + if isinstance(document, list): + return list(document) + if isinstance(document, dict): + for key in ("signals", "records", "items"): + value = document.get(key) + if isinstance(value, list): + return list(value) + return [document] + return [document] + + + def _record_signal_id(record: Any) -> str | None: + if not isinstance(record, dict): + return None + value = record.get("signal_id") + if isinstance(value, str) and value: + return value + for wrapper in ("domain_activation_manifest", "payload", "data"): + nested = record.get(wrapper) + if isinstance(nested, dict) and isinstance(nested.get("signal_id"), str): + return nested["signal_id"] + return None + + + def expand_stage2_signal_all( + stage1_run_root: str | os.PathLike[str], + signal_manifest: Mapping[str, Any], + *, + max_file_bytes: int = MAX_FILE_BYTES, + max_total_bytes: int = MAX_RUN_BYTES, + signal_registry: Mapping[str, Any] | None = None, + ) -> dict[str, Any]: + """Expand Stage 2 ALL while separating semantic and integrity-only universes.""" + + downstream = signal_manifest.get("downstream_read_sets", {}) + stage2 = downstream.get("stage2", []) if isinstance(downstream, dict) else [] + if stage2 != ["ALL"]: + raise IngressError("SIGNAL_ALL_CONTRACT", "downstream_read_sets.stage2 must equal ['ALL']") + files = signal_manifest.get("files") + if not isinstance(files, list): + raise IngressError("SIGNAL_FILES_SHAPE", "signal manifest files must be an array") + transaction_id = str(signal_manifest.get("manifest_transaction_id", signal_manifest.get("transaction_id", "MISSING"))) + file_rows: list[dict[str, Any]] = [] + semantic_rows: list[dict[str, Any]] = [] + integrity_rows: list[dict[str, Any]] = [] + occurrences: list[dict[str, Any]] = [] + payload_snapshots: list[Snapshot] = [] + issues: list[dict[str, Any]] = [] + path_counter: Counter[str] = Counter() + parsed_documents: dict[str, Any] = {} + aggregate_bytes = 0 + registry_entries = { + str(row.get("file")): row + for row in (signal_registry or {}).get("entries", []) + if isinstance(row, dict) and isinstance(row.get("file"), str) + } + compatibility_files = { + str(path) + for path in (signal_registry or {}).get("compatibility_views", []) + if isinstance(path, str) + } + domain_envelope_schema = (signal_registry or {}).get("domain_envelope") + observed_registry_files: set[str] = set() + for index, entry in enumerate(files): + if not isinstance(entry, dict) or not isinstance(entry.get("path"), str): + raise IngressError("SIGNAL_FILE_ROW_SHAPE", f"invalid signal file row at index {index}") + relative_payload = _safe_relative_path(entry["path"]).as_posix() + if relative_payload.startswith("signals/"): + raise IngressError("SIGNAL_PATH_PREFIX_FORBIDDEN", "manifest file path must not include signals/ prefix") + physical = f"signals/{relative_payload}" + snapshot = open_bounded_snapshot( + stage1_run_root, + physical, + logical_input_id=f"signal_file:{index}", + max_bytes=max_file_bytes, + ) + document = load_json_strict(snapshot) + payload_snapshots.append(snapshot) + aggregate_bytes += snapshot.byte_length + if aggregate_bytes > max_total_bytes: + raise IngressError("AGGREGATE_RUN_SIZE_LIMIT", "signal ALL payloads exceed remaining run byte budget") + parsed_documents[relative_payload] = document + kind = entry.get("kind", "canonical") + if kind not in SEMANTIC_SIGNAL_KINDS | {"compatibility_view"}: + raise IngressError("SIGNAL_KIND_UNAPPROVED", f"unapproved signal file kind: {kind}") + semantic = kind in SEMANTIC_SIGNAL_KINDS + expected_hash = entry.get( + "file_sha256", entry.get("sha256", entry.get("raw_sha256")) + ) + row = { + "manifest_index": index, + "file_path": relative_payload, + "physical_path": physical, + "kind": kind, + "raw_sha256": snapshot.raw_sha256, + "byte_length": snapshot.byte_length, + "semantic": semantic, + "manifest_declared_record_count": entry.get("record_count"), + } + if expected_hash is not None and expected_hash != snapshot.raw_sha256: + row["hash_status"] = "FAIL" + issues.append(_issue("SIGNAL_FILE_HASH_MISMATCH", impact_scope="SIGNAL", source_refs=[physical])) + else: + row["hash_status"] = "PASS" if expected_hash else "UNEVALUABLE" + records = _records_from_signal_document(document) + row["observed_record_count"] = len(records) + declared_count = entry.get("record_count") + if isinstance(declared_count, int) and declared_count != len(records): + row["record_count_status"] = "FAIL" + issues.append(_issue("SIGNAL_RECORD_COUNT_MISMATCH", impact_scope="SIGNAL", source_refs=[physical])) + else: + row["record_count_status"] = "PASS" if isinstance(declared_count, int) else "UNEVALUABLE" + registry_row = registry_entries.get(relative_payload) + if kind == "canonical": + if signal_registry is not None and registry_row is None: + issues.append(_issue("SIGNAL_REGISTRY_COVERAGE_MISMATCH", impact_scope="SIGNAL", source_refs=[physical])) + elif registry_row is not None: + observed_registry_files.add(relative_payload) + declared_schema = entry.get("schema", entry.get("schema_path")) + if declared_schema is not None and declared_schema != registry_row.get("schema"): + issues.append(_issue("SIGNAL_SCHEMA_LINEAGE_MISMATCH", impact_scope="SIGNAL", source_refs=[physical])) + elif kind == "compatibility_view": + if relative_payload in registry_entries: + issues.append(_issue("SIGNAL_COMPATIBILITY_SUBSTITUTION", impact_scope="SIGNAL", source_refs=[physical])) + if signal_registry is not None and relative_payload not in compatibility_files: + issues.append(_issue("SIGNAL_REGISTRY_COVERAGE_MISMATCH", impact_scope="SIGNAL", source_refs=[physical])) + elif kind == "domain_signal": + declared_schema = entry.get("schema", entry.get("schema_path")) + if signal_registry is not None and declared_schema not in {None, domain_envelope_schema}: + issues.append(_issue("SIGNAL_SCHEMA_LINEAGE_MISMATCH", impact_scope="SIGNAL", source_refs=[physical])) + file_rows.append(row) + path_counter[relative_payload] += 1 + if semantic: + semantic_rows.append(row) + for record_ordinal, record in enumerate(records): + signal_id = _record_signal_id(record) + occurrence_key = [transaction_id, relative_payload, record_ordinal, signal_id] + occurrences.append( + { + "occurrence_key": occurrence_key, + "occurrence_ref": f"SIGO-{canonical_digest(occurrence_key)[:24]}", + "manifest_transaction_id": transaction_id, + "file_path": relative_payload, + "record_ordinal": record_ordinal, + "signal_id": signal_id, + "disposition": "UNMAPPED" if signal_id is None else "UNUSED", + "binding_refs": [], + "raw_record_sha256": canonical_digest(record), + "record": record, + } + ) + else: + integrity_rows.append(row) + duplicates = sorted(path for path, count in path_counter.items() if count > 1) + if duplicates: + issues.append(_issue("SIGNAL_ALL_DUPLICATE_FILE_ROW", impact_scope="SIGNAL", source_refs=duplicates)) + manifest_counter = Counter((i, row["file_path"], row["kind"]) for i, row in enumerate(file_rows)) + partition_counter = Counter((row["manifest_index"], row["file_path"], row["kind"]) for row in semantic_rows + integrity_rows) + missing_registry_files = sorted(set(registry_entries) - observed_registry_files) if signal_registry is not None else [] + if missing_registry_files: + issues.append( + _issue( + "SIGNAL_REGISTRY_COVERAGE_MISMATCH", + impact_scope="SIGNAL", + source_refs=[f"signals/{path}" for path in missing_registry_files], + ) + ) + file_conservation = ( + manifest_counter == partition_counter + and not duplicates + and not missing_registry_files + and not any(row["hash_status"] == "FAIL" or row["record_count_status"] == "FAIL" for row in file_rows) + ) + record_counter = Counter(tuple(row["occurrence_key"]) for row in occurrences) + partitioned_record_counter = Counter( + tuple(row["occurrence_key"]) + for row in occurrences + if row["disposition"] in {"USED", "UNUSED", "UNMAPPED"} + ) + record_conservation = record_counter == partitioned_record_counter + return { + "manifest_transaction_id": transaction_id, + "ordered_file_rows": file_rows, + "semantic_file_rows": semantic_rows, + "integrity_only_file_rows": integrity_rows, + "record_occurrences": occurrences, + "used_record_occurrences": [], + "unused_record_occurrences": [row for row in occurrences if row["disposition"] == "UNUSED"], + "unmapped_record_occurrences": [row for row in occurrences if row["disposition"] == "UNMAPPED"], + "_parsed_documents_by_path": parsed_documents, + "_payload_snapshots": payload_snapshots, + "file_conservation_pass": file_conservation, + "record_conservation_pass": record_conservation, + "aggregate_payload_bytes": aggregate_bytes, + "issues": issues, + } + + + def _collect_values_for_keys(value: Any, keys: frozenset[str]) -> set[str]: + result: set[str] = set() + stack = [value] + while stack: + current = stack.pop() + if isinstance(current, dict): + for key, child in current.items(): + if key in keys: + if isinstance(child, list): + result.update(str(item) for item in child if item is not None) + elif child is not None: + result.add(str(child)) + stack.append(child) + elif isinstance(current, list): + stack.extend(current) + return result + + + def bind_signal_occurrences(signal_all: MutableMapping[str, Any], documents: Mapping[str, Any]) -> dict[str, Any]: + """Bind each semantic signal occurrence to explicit Stage 1 references without deduplication.""" + + explicit_signal_ids = _collect_values_for_keys( + documents, + frozenset({"signal_id", "signal_ids", "signal_refs", "emitted_signal_ids", "required_signal_ids"}), + ) + known_refs = { + "fact_id": _collect_values_for_keys(documents.get("fact_ledger_base"), frozenset({"fact_id"})), + "source_bo_id": _collect_values_for_keys(documents, frozenset({"BO_ID", "source_bo_id", "source_bo_ids"})), + "bo_id": _collect_values_for_keys(documents, frozenset({"BO_ID", "bo_id"})), + "structure_id": _collect_values_for_keys(documents.get("legal_effect_structures"), frozenset({"structure_id"})), + "domain_id": _collect_values_for_keys(documents, frozenset({"domain_id", "domain_ids", "active_domain_ids"})), + "evidence_id": _collect_values_for_keys(documents.get("evidence_indexed"), frozenset({"evidence_id", "id"})), + "event_id": _collect_values_for_keys(documents.get("evidence_event_candidates"), frozenset({"event_id", "id"})), + } + link_keys = { + "fact_id": ("fact_id", "fact_ids"), + "source_bo_id": ("source_bo_id", "source_bo_ids"), + "bo_id": ("bo_id", "bo_ids"), + "structure_id": ("structure_id", "structure_ids"), + "domain_id": ("domain_id", "domain_ids"), + "evidence_id": ("evidence_id", "evidence_ids"), + "event_id": ("event_id", "event_ids"), + } + for occurrence in signal_all.get("record_occurrences", []): + signal_id = occurrence.get("signal_id") + record = occurrence.get("record") + bindings: set[str] = set() + if isinstance(signal_id, str) and signal_id in explicit_signal_ids: + bindings.add(f"signal_id:{signal_id}") + for ref_kind, candidate_keys in link_keys.items(): + observed = _collect_values_for_keys(record, frozenset(candidate_keys)) + for ref in sorted(observed & known_refs[ref_kind]): + bindings.add(f"{ref_kind}:{ref}") + if not isinstance(signal_id, str) or not signal_id: + occurrence["disposition"] = "UNMAPPED" + elif bindings: + occurrence["disposition"] = "USED" + else: + occurrence["disposition"] = "UNUSED" + occurrence["binding_refs"] = sorted(bindings) + for disposition, key in ( + ("USED", "used_record_occurrences"), + ("UNUSED", "unused_record_occurrences"), + ("UNMAPPED", "unmapped_record_occurrences"), + ): + signal_all[key] = [ + row for row in signal_all.get("record_occurrences", []) if row.get("disposition") == disposition + ] + source_counter = Counter(tuple(row["occurrence_key"]) for row in signal_all.get("record_occurrences", [])) + partition_counter = Counter( + tuple(row["occurrence_key"]) + for key in ("used_record_occurrences", "unused_record_occurrences", "unmapped_record_occurrences") + for row in signal_all[key] + ) + signal_all["record_conservation_pass"] = source_counter == partition_counter + return dict(signal_all) + + + def _activation_payload(value: Mapping[str, Any]) -> Mapping[str, Any]: + for key in ("domain_activation_manifest", "activation", "payload", "data"): + nested = value.get(key) + if isinstance(nested, dict) and any(field in nested for field in SG01_PROJECTION_FIELDS): + return nested + return value + + + def verify_activation_projection( + routing_activation: Mapping[str, Any], + signal_activation: Mapping[str, Any], + *, + routing_raw_sha256: str | None = None, + signal_raw_sha256: str | None = None, + ) -> dict[str, Any]: + """Compare approved semantic SG-01 projection while retaining both raw hashes.""" + + left = _activation_payload(routing_activation) + right = _activation_payload(signal_activation) + missing_left = [field for field in SG01_PROJECTION_FIELDS if field not in left] + missing_right = [field for field in SG01_PROJECTION_FIELDS if field not in right] + if missing_left or missing_right: + raise IngressError( + "SG01_PROJECTION_SHAPE", + "both activation artifacts must expose the complete approved 17-field projection", + details={"routing_missing": missing_left, "signal_missing": missing_right}, + ) + + def project(value: Mapping[str, Any]) -> dict[str, Any]: + result: dict[str, Any] = {} + for field in SG01_PROJECTION_FIELDS: + child = value[field] + if field in SG01_SET_FIELDS: + if not isinstance(child, list): + raise IngressError("SG01_PROJECTION_SHAPE", f"{field} must be an array") + child = sorted({canonical_json_bytes(item): item for item in child}.values(), key=canonical_json_bytes) + result[field] = child + return result + + left_projection = project(left) + right_projection = project(right) + if left_projection != right_projection: + raise IngressError( + "SG01_SEMANTIC_DRIFT", + "routing activation and signal SG-01 semantic projections differ", + details={"routing_projection": left_projection, "signal_projection": right_projection}, + ) + return { + "status": "PASS", + "projection": left_projection, + "projection_sha256": canonical_digest(left_projection), + "routing_raw_sha256": routing_raw_sha256, + "signal_raw_sha256": signal_raw_sha256, + "compared_keys": list(SG01_PROJECTION_FIELDS), + } + + + def _array_rows(value: Any, preferred_keys: Sequence[str]) -> list[Any]: + if isinstance(value, list): + return list(value) + if isinstance(value, dict): + for key in preferred_keys: + candidate = value.get(key) + if isinstance(candidate, list): + return list(candidate) + return [] + + + def verify_cross_artifact_seals( + documents: Mapping[str, Any], + snapshots: Mapping[str, Snapshot], + deployment_snapshots: Mapping[str, Snapshot] | None = None, + ) -> dict[str, Any]: + """Recompute the P1 guard and current-v8 producer invariants.""" + + checks: list[dict[str, Any]] = [] + issues: list[dict[str, Any]] = [] + deployment_snapshots = deployment_snapshots or {} + p1 = documents.get("stage1_part1_soft_gate_handoff") + if isinstance(p1, dict): + digest_guard = p1.get("digest_guard") + if not isinstance(digest_guard, dict): + issues.append(_issue("P1_SEVEN_KEY_MISSING", source_refs=["stage1_part1_soft_gate_handoff"])) + digest_guard = {} + elif any(key not in digest_guard for key in P1_DIGEST_KEYS): + issues.append(_issue("P1_SEVEN_KEY_MISSING", source_refs=["stage1_part1_soft_gate_handoff#digest_guard"])) + for digest_key, logical_id in P1_DIGEST_KEYS.items(): + source = snapshots.get(logical_id) or deployment_snapshots.get(logical_id) + observed = source.raw_sha256 if source else None + expected = digest_guard.get(digest_key) + passed = expected is not None and observed is not None and expected == observed + checks.append({"check_id": f"P1:{digest_key}", "status": "PASS" if passed else "UNEVALUABLE" if source is None else "FAIL"}) + if expected is not None and observed is not None and not passed: + issues.append(_issue("P1_DIGEST_MISMATCH", source_refs=[logical_id])) + else: + issues.append(_issue("P1_HANDOFF_NOT_FLAT_OBJECT", source_refs=["stage1_part1_soft_gate_handoff"])) + p2 = documents.get("stage1_part2_review_handoff") + if p2 is not None and not isinstance(p2, dict): + issues.append(_issue("P2_HANDOFF_NOT_FLAT_OBJECT", source_refs=["stage1_part2_review_handoff"])) + for stage in (3, 4): + logical = f"stage1_part{stage}_review_handoff" + value = documents.get(logical) + if value is not None: + wrapper_present = isinstance(value, dict) and isinstance(value.get(logical), dict) + if not wrapper_present: + issues.append(_issue(f"P{stage}_WRAPPER_MISSING", source_refs=[logical])) + ledger_rows = _array_rows(documents.get("fact_ledger_base"), ("facts", "fact_ledger", "rows", "items")) + for index, row in enumerate(ledger_rows): + if not isinstance(row, dict) or "domain_effects" not in row or "calculation_requests" not in row: + issues.append(_issue("CURRENT_V8_LEDGER_EXTENSION_MISSING", impact_scope="FACT", source_refs=[f"fact_ledger_base#/{index}"])) + return {"checks": checks, "issues": issues, "passed": not any(item["severity"] == "ERROR" for item in issues)} + + + def mint_stage2_id( + namespace: str, + canonical_tuple: Any, + *, + prefix: str | None = None, + algorithm_version: str = ALGORITHM_VERSION, + collision_registry: MutableMapping[str, str] | None = None, + ) -> dict[str, str]: + """Mint a domain-separated ID and reject short-ID collisions.""" + + if not re.fullmatch(r"[A-Za-z][A-Za-z0-9_.:-]{0,127}", namespace): + raise IngressError("INVALID_ID_NAMESPACE", f"invalid namespace: {namespace}") + mint_input = { + "namespace": namespace, + "algorithm_version": algorithm_version, + "canonical_tuple": canonical_tuple, + } + full_digest = canonical_digest(mint_input) + external_prefix = prefix or namespace.upper().replace("_", "-") + external_id = f"{external_prefix}-{full_digest[:24]}" + if collision_registry is not None: + prior = collision_registry.get(external_id) + if prior is not None and prior != full_digest: + raise IngressError("MINTED_ID_COLLISION", f"collision for {external_id}") + collision_registry[external_id] = full_digest + return { + "id": external_id, + "mint_input_sha256": full_digest, + "algorithm_version": algorithm_version, + } + + + def mint_context_occurrence_id( + kind: str, + logical_artifact_id: str, + json_pointer: str, + raw_value: Any, + explicit_lineage_refs: Sequence[str], + *, + stage1_canonical_id: str | None = None, + collision_registry: MutableMapping[str, str] | None = None, + ) -> dict[str, Any]: + """Mint an occurrence/lineage context ID without fuzzy entity resolution.""" + + if kind not in {"party", "object"}: + raise IngressError("CONTEXT_KIND_INVALID", "context kind must be party or object") + raw_value_sha256 = canonical_digest(raw_value) + canonical_tuple = [ + logical_artifact_id, + json_pointer, + raw_value_sha256, + sorted(set(explicit_lineage_refs)), + ] + minted = mint_stage2_id( + f"{kind}_context_occurrence", + canonical_tuple, + prefix="PC" if kind == "party" else "OC", + collision_registry=collision_registry, + ) + source_ref = _source_ref( + logical_artifact_id, + json_pointer, + raw_value, + stage1_id=stage1_canonical_id, + ) + identity_kind = "PARTY_CONTEXT" if kind == "party" else "OBJECT_CONTEXT" + result: dict[str, Any] = { + "identity_kind": identity_kind, + f"{kind}_context_id": stage1_canonical_id or minted["id"], + f"stage1_{kind}_id": stage1_canonical_id, + "identity_disposition": ( + "PRESERVED_STAGE1_CANONICAL_ID" + if stage1_canonical_id is not None + else "IDENTITY_UNRESOLVED" + ), + "source_refs": [source_ref], + "explicit_lineage_refs": sorted(set(explicit_lineage_refs)), + "display_label_nfc": unicodedata.normalize("NFC", str(raw_value)), + "raw_display_value_sha256": raw_value_sha256, + } + if stage1_canonical_id is None: + result["derivation"] = { + "derivation_id": minted["id"], + "algorithm_version": ALGORITHM_VERSION, + "sorted_input_refs": sorted(set(explicit_lineage_refs)) or [f"{logical_artifact_id}:{json_pointer}"], + "mint_input_sha256": minted["mint_input_sha256"], + } + return result + + + def _extract_review_arrays(document: Any) -> list[Any]: + if not isinstance(document, dict): + return [] + result: list[Any] = [] + for key in ("review_items", "review_queue", "blocked_review_items", "unresolved_review_items"): + value = document.get(key) + if isinstance(value, list): + result.extend(value) + for wrapper in ( + "handoff", + "payload", + "review", + "data", + "stage1_part3_review_handoff", + "stage1_part4_review_handoff", + ): + nested = document.get(wrapper) + if isinstance(nested, dict): + result.extend(_extract_review_arrays(nested)) + return result + + + def normalize_review_items( + review_documents: Mapping[str, Any], + release_lock: Mapping[str, Any] | None = None, + ) -> dict[str, Any]: + """Preserve every raw review occurrence in exactly one normalized partition.""" + + stage_map = { + "stage1_part1_soft_gate_handoff": ("P1", "S2A-P1-HANDOFF-FLAT-V1"), + "stage1_part2_review_handoff": ("P2", "S2A-P2-HANDOFF-FLAT-V1"), + "stage1_part3_review_handoff": ("P3", "S2A-P3-HANDOFF-WRAPPED-V1"), + "stage1_part4_review_handoff": ("P4", "S2A-P4-HANDOFF-WRAPPED-V1"), + } + release_lock = release_lock or {} + mapping_decision = _adapter_decision(release_lock, "S2-REVIEW-MAP-V1") + mapping_rows = mapping_decision.get("mappings", []) if isinstance(mapping_decision, dict) else [] + closed_mappings: dict[tuple[str, str, str], Mapping[str, Any]] = {} + for mapping in mapping_rows: + if not isinstance(mapping, dict): + continue + key = ( + str(mapping.get("source_stage", "ANY")), + str(mapping.get("source_field_kind")), + str(mapping.get("source_value")), + ) + if key in closed_mappings: + raise IngressError("REVIEW_MAPPING_DUPLICATE", f"duplicate release review mapping: {key}") + closed_mappings[key] = mapping + adapter_issues: list[dict[str, Any]] = [] + if not closed_mappings: + adapter_issues.append(_issue("REVIEW_MAPPING_CONTRACT_MISSING", impact_scope="REVIEW_ITEM")) + collected: list[tuple[str, Any, str]] = [] + adapter_ids_seen: set[str] = set() + for logical_id in sorted(review_documents): + source_stage, adapter_id = stage_map.get(logical_id, (None, None)) + if source_stage is None or adapter_id is None: + continue + adapter_ids_seen.add(adapter_id) + decision = _adapter_decision(release_lock, adapter_id) + if decision is None: + adapter_issues.append(_issue("HANDOFF_ADAPTER_CONTRACT_MISSING", impact_scope="REVIEW_ITEM", source_refs=[logical_id])) + continue + document = review_documents[logical_id] + wrapper_pointer = decision.get("wrapper_json_pointer") + wrapper_found, wrapper = _json_pointer_value(document, wrapper_pointer) + if not wrapper_found or not isinstance(wrapper, dict): + code = "P2_HANDOFF_NOT_FLAT_OBJECT" if source_stage == "P2" else f"{source_stage}_WRAPPER_MISSING" + adapter_issues.append(_issue(code, impact_scope="REVIEW_ITEM", source_refs=[logical_id])) + continue + expected_version = decision.get("schema_version") + if wrapper.get("schema_version") != expected_version: + adapter_issues.append( + _issue( + f"{source_stage}_HANDOFF_SCHEMA_VERSION_MISMATCH", + impact_scope="REVIEW_ITEM", + source_refs=[logical_id], + ) + ) + items_found, raw_items = _json_pointer_value(wrapper, decision.get("review_items_json_pointer")) + if not items_found or not isinstance(raw_items, list): + adapter_issues.append( + _issue( + f"{source_stage}_REVIEW_ITEMS_SHAPE", + impact_scope="REVIEW_ITEM", + source_refs=[logical_id], + ) + ) + raw_items = [] + if decision.get("count_field_required") is True: + declared_count = wrapper.get("review_item_count") + if not isinstance(declared_count, int) or declared_count != len(raw_items): + adapter_issues.append( + _issue("REVIEW_CONSERVATION_FAILED", impact_scope="REVIEW_ITEM", source_refs=[logical_id]) + ) + for raw_item in raw_items: + raw_hash = canonical_digest(raw_item) + collected.append((logical_id, raw_item, raw_hash)) + grouped: dict[tuple[str, str], list[Any]] = defaultdict(list) + for logical_id, raw_item, raw_hash in collected: + grouped[(logical_id, raw_hash)].append(raw_item) + raw_occurrences: list[dict[str, Any]] = [] + normalized: list[dict[str, Any]] = [] + partition_counts = {key: 0 for key in ("SUPPORTED", "CONDITIONAL", "UNRESOLVED", "EXCLUDED", "UNMAPPED")} + for (logical_id, raw_hash), items in sorted(grouped.items()): + for duplicate_index, raw_item in enumerate(items): + raw_status = raw_item.get("status") if isinstance(raw_item, dict) else None + raw_severity = raw_item.get("severity") if isinstance(raw_item, dict) else None + source_stage, adapter_id = stage_map.get(logical_id, ("P1", "S2A-P1-HANDOFF-FLAT-V1")) + field_kind = "REVIEW_ITEM_STATUS" if raw_status is not None else "REVIEW_ITEM_SEVERITY" + source_value = str(raw_status if raw_status is not None else raw_severity) + mapping = closed_mappings.get((source_stage, field_kind, source_value)) or closed_mappings.get( + ("ANY", field_kind, source_value) + ) + partition = str(mapping.get("normalized_partition")) if mapping is not None else "UNMAPPED" + if partition not in partition_counts: + raise IngressError("REVIEW_MAPPING_PARTITION_INVALID", f"release mapping produced {partition}") + upstream_id = raw_item.get("review_id") if isinstance(raw_item, dict) else None + if upstream_id: + review_id = str(upstream_id) + minted_flag = False + else: + minted = mint_stage2_id( + "review_occurrence", + [logical_id, raw_hash, duplicate_index], + prefix="REV", + ) + review_id = minted["id"] + minted_flag = True + raw_occurrence = { + "source_stage": source_stage, + "logical_input_id": logical_id, + "canonical_raw_item_sha256": raw_hash, + "duplicate_occurrence_index": duplicate_index, + } + raw_occurrences.append(raw_occurrence) + reason_codes = ["UNMAPPED_REVIEW_STATUS"] if partition == "UNMAPPED" else [] + row = { + "review_key": { + "value": review_id, + "minted": minted_flag, + "source_stage": source_stage, + "logical_input_id": logical_id, + "canonical_raw_item_sha256": raw_hash, + "duplicate_occurrence_index": duplicate_index, + "upstream_review_id": str(upstream_id) if upstream_id is not None else None, + }, + "partition": partition, + "source_status_raw": str(raw_status) if raw_status is not None else None, + "source_severity_raw": str(raw_severity) if raw_severity is not None else None, + "source_item_raw_sha256": raw_hash, + "mapping_id": str(mapping.get("mapping_id")) if mapping is not None else f"S2-REVIEW-MAP-V1:{source_stage}:UNMAPPED", + "impact_scope": "REVIEW_ITEM", + "scope_refs": [review_id], + "source_contract_row_refs": [logical_id], + "reason_codes": reason_codes, + "downstream_allowed_actions": ["CARRY_FORWARD_TO_LAWYER_REVIEW"], + "_adapter_id": adapter_id, + } + normalized.append(row) + partition_counts[partition] += 1 + raw_counter = Counter((logical_id, raw_hash) for logical_id, _, raw_hash in collected) + normalized_counter = Counter( + (row["review_key"]["logical_input_id"], row["review_key"]["canonical_raw_item_sha256"]) + for row in normalized + ) + adapter_ids = sorted({row.pop("_adapter_id") for row in normalized} | adapter_ids_seen) + adapter_conservation_pass = not any( + item["issue_code"] == "REVIEW_CONSERVATION_FAILED" for item in adapter_issues + ) + return { + "schema_version": "stage2_review_normalization_receipt.v1", + "adapter_ids": adapter_ids or ["S2A-P1-HANDOFF-FLAT-V1"], + "mapping_table_version": "S2-REVIEW-MAP-V1", + "raw_occurrences": sorted( + raw_occurrences, + key=lambda row: ( + row["source_stage"], + row["logical_input_id"], + row["canonical_raw_item_sha256"], + row["duplicate_occurrence_index"], + ), + ), + "normalized_occurrences": sorted( + normalized, + key=lambda row: row["review_key"]["value"], + ), + "partition_counts": partition_counts, + "conservation_status": "PASS" if raw_counter == normalized_counter and adapter_conservation_pass else "FAIL", + "_issues": adapter_issues, + } + + + def check_conservation( + documents: Mapping[str, Any], + *, + signal_all: Mapping[str, Any] | None = None, + normalized_reviews: Mapping[str, Any] | None = None, + source_snapshots: Mapping[str, Snapshot] | None = None, + ) -> dict[str, Any]: + """Independently compute core set, cardinality, and multiset invariants.""" + + checks: list[dict[str, Any]] = [] + issues: list[dict[str, Any]] = [] + source_snapshots = source_snapshots or {} + + def add_check( + check_id: str, + passed: bool | None, + left: Sequence[Any] | Counter[Any] | None, + right: Sequence[Any] | Counter[Any] | None, + *, + issue_code: str, + impact_scope: str, + source_refs: Sequence[str], + details: Mapping[str, Any] | None = None, + ) -> None: + left_counter = left if isinstance(left, Counter) else Counter(left or []) + right_counter = right if isinstance(right, Counter) else Counter(right or []) + row: dict[str, Any] = { + "check_id": check_id, + "status": "UNEVALUABLE" if passed is None else "PASS" if passed else "FAIL", + "left_count": sum(left_counter.values()) if left is not None else None, + "right_count": sum(right_counter.values()) if right is not None else None, + "left_counter_digest": canonical_digest(sorted((canonical_digest(key), count) for key, count in left_counter.items())) if left is not None else None, + "right_counter_digest": canonical_digest(sorted((canonical_digest(key), count) for key, count in right_counter.items())) if right is not None else None, + } + if details: + row.update(details) + checks.append(row) + if passed is False: + issues.append(_issue(issue_code, impact_scope=impact_scope, source_refs=source_refs)) + + bo_rows = _array_rows(documents.get("bo"), ("business_objects", "BO", "rows", "items")) + ledger_rows = _array_rows(documents.get("fact_ledger_base"), ("facts", "fact_ledger", "rows", "items")) + bo_ids = [str(row["BO_ID"]) for row in bo_rows if isinstance(row, dict) and row.get("BO_ID") is not None] + source_bo_ids = [ + str(row["source_bo_id"]) + for row in ledger_rows + if isinstance(row, dict) and row.get("source_bo_id") is not None + ] + missing_bo_id_rows = [index for index, row in enumerate(bo_rows) if not isinstance(row, dict) or row.get("BO_ID") is None] + missing_source_bo_rows = [ + index for index, row in enumerate(ledger_rows) if not isinstance(row, dict) or row.get("source_bo_id") is None + ] + bo_pass = ( + not missing_bo_id_rows + and not missing_source_bo_rows + and Counter(bo_ids) == Counter(source_bo_ids) + ) + add_check( + "BO_FACT_MULTISET", + bo_pass, + bo_ids, + source_bo_ids, + issue_code="BO_FACT_CONSERVATION_FAILED", + impact_scope="FACT", + source_refs=["bo", "fact_ledger_base"], + details={ + "missing_bo_id_rows": missing_bo_id_rows, + "missing_source_bo_id_rows": missing_source_bo_rows, + "duplicate_bo_ids": sorted(key for key, count in Counter(bo_ids).items() if count > 1), + "dangling_source_bo_ids": sorted(set(source_bo_ids) - set(bo_ids)), + }, + ) + missing_fact_id_rows = [ + index for index, row in enumerate(ledger_rows) if not isinstance(row, dict) or row.get("fact_id") is None + ] + fact_ids = [str(row["fact_id"]) for row in ledger_rows if isinstance(row, dict) and row.get("fact_id") is not None] + expected_fact_ids = [f"F-{index:03d}" for index in range(1, len(ledger_rows) + 1)] + fact_pass = not missing_fact_id_rows and fact_ids == expected_fact_ids and len(fact_ids) == len(set(fact_ids)) + add_check( + "FACT_ID_SEQUENCE", + fact_pass, + fact_ids, + expected_fact_ids, + issue_code="FACT_ID_CONSERVATION_FAILED", + impact_scope="FACT", + source_refs=["fact_ledger_base"], + details={"missing_fact_id_rows": missing_fact_id_rows, "observed": fact_ids}, + ) + extension_missing = [ + index + for index, row in enumerate(ledger_rows) + if not isinstance(row, dict) + or not isinstance(row.get("domain_effects"), dict) + or not isinstance(row.get("calculation_requests"), list) + ] + add_check( + "CURRENT_V8_LEDGER_EXTENSIONS", + not extension_missing, + list(range(len(ledger_rows))), + [index for index in range(len(ledger_rows)) if index not in extension_missing], + issue_code="CURRENT_V8_LEDGER_EXTENSION_MISSING", + impact_scope="FACT", + source_refs=["fact_ledger_base"], + details={"missing_row_indices": extension_missing}, + ) + les_rows = _array_rows( + documents.get("legal_effect_structures"), + ("structures", "structure_records", "legal_effect_structures", "rows", "items"), + ) + dangling_les: list[str] = [] + les_ids: list[str] = [] + for row in les_rows: + if not isinstance(row, dict): + continue + structure_id = row.get("structure_id", row.get("legal_effect_structure_id")) + if structure_id is not None: + les_ids.append(str(structure_id)) + refs = row.get("source_bo_ids", []) + if isinstance(refs, list): + dangling_les.extend(str(ref) for ref in refs if ref not in set(bo_ids)) + duplicate_les_ids = sorted(key for key, count in Counter(les_ids).items() if count > 1) + les_pass = not dangling_les and not duplicate_les_ids and len(les_ids) == len(les_rows) + add_check( + "LES_BO_JOIN", + les_pass, + [str(row.get("structure_id", row.get("legal_effect_structure_id"))) for row in les_rows if isinstance(row, dict)], + les_ids, + issue_code="LES_BO_JOIN_FAILED", + impact_scope="CLUSTER", + source_refs=["legal_effect_structures", "bo"], + details={"dangling_refs": sorted(dangling_les), "duplicate_structure_ids": duplicate_les_ids}, + ) + declared_les_count = None + les_document = documents.get("legal_effect_structures") + if isinstance(les_document, dict): + for key in ("declared_structure_count", "structure_count", "record_count"): + if isinstance(les_document.get(key), int): + declared_les_count = int(les_document[key]) + break + declared_les_pass = None if declared_les_count is None else declared_les_count == len(les_rows) + add_check( + "LES_DECLARED_ACTUAL_COUNT", + declared_les_pass, + [None] * declared_les_count if declared_les_count is not None else None, + [None] * len(les_rows), + issue_code="LES_DECLARED_COUNT_MISMATCH", + impact_scope="CLUSTER", + source_refs=["legal_effect_structures"], + ) + actual_domain_index: dict[str, list[str]] = defaultdict(list) + actual_bo_index: dict[str, list[str]] = defaultdict(list) + ledger_structure_refs: list[tuple[str, str, str]] = [] + ledger_type_refs: list[tuple[str, str, str]] = [] + actual_structure_refs: list[tuple[str, str, str]] = [] + actual_type_refs: list[tuple[str, str, str]] = [] + route_count_errors: list[str] = [] + for row in les_rows: + if not isinstance(row, dict): + continue + structure_id = str(row.get("structure_id", row.get("legal_effect_structure_id", "MISSING"))) + domain_id = str(row.get("domain_id", "MISSING")) + type_id = str(row.get("type_id", row.get("type", "MISSING"))) + actual_domain_index[domain_id].append(structure_id) + source_ids = row.get("source_bo_ids", []) + if isinstance(source_ids, list): + for bo_id in source_ids: + actual_bo_index[str(bo_id)].append(structure_id) + actual_structure_refs.append((str(bo_id), domain_id, structure_id)) + actual_type_refs.append((str(bo_id), domain_id, type_id)) + routes = row.get("routes", []) + if isinstance(routes, list) and row.get("route_count", len(routes)) != len(routes): + route_count_errors.append(structure_id) + for row in ledger_rows: + if not isinstance(row, dict): + continue + bo_id = str(row.get("source_bo_id", "MISSING")) + effects = row.get("domain_effects", {}) + if not isinstance(effects, dict): + continue + for domain_id, effect in effects.items(): + if not isinstance(effect, dict): + continue + for structure_id in effect.get("structure_ids", []) if isinstance(effect.get("structure_ids"), list) else []: + ledger_structure_refs.append((bo_id, str(domain_id), str(structure_id))) + for type_id in effect.get("type_ids", []) if isinstance(effect.get("type_ids"), list) else []: + ledger_type_refs.append((bo_id, str(domain_id), str(type_id))) + structure_index = les_document.get("structure_index", {}) if isinstance(les_document, dict) else {} + index_present = isinstance(structure_index, dict) and bool(structure_index) + index_ok = True + if index_present: + declared_by_domain = structure_index.get("by_domain_id", {}) + declared_by_bo = structure_index.get("by_bo_id", {}) + index_ok = ( + isinstance(declared_by_domain, dict) + and isinstance(declared_by_bo, dict) + and {str(key): Counter(map(str, value)) for key, value in declared_by_domain.items() if isinstance(value, list)} + == {key: Counter(value) for key, value in actual_domain_index.items()} + and {str(key): Counter(map(str, value)) for key, value in declared_by_bo.items() if isinstance(value, list)} + == {key: Counter(value) for key, value in actual_bo_index.items()} + ) + reverse_ok = ( + (not ledger_structure_refs or Counter(ledger_structure_refs) == Counter(actual_structure_refs)) + and (not ledger_type_refs or Counter(ledger_type_refs) == Counter(actual_type_refs)) + and not route_count_errors + and index_ok + ) + add_check( + "LES_REVERSE_INDEX", + reverse_ok, + ledger_structure_refs + ledger_type_refs, + actual_structure_refs + actual_type_refs, + issue_code="LES_REVERSE_INDEX_MISMATCH", + impact_scope="CLUSTER", + source_refs=["legal_effect_structures", "fact_ledger_base"], + details={"index_present": index_present, "route_count_errors": route_count_errors}, + ) + evidence_rows = _array_rows(documents.get("evidence_indexed"), ("evidence", "evidence_items", "rows", "items")) + event_rows = _array_rows(documents.get("evidence_event_candidates"), ("events", "event_candidates", "rows", "items")) + evidence_ids = [ + str(row.get("evidence_id", row.get("id"))) + for row in evidence_rows + if isinstance(row, dict) and (row.get("evidence_id") is not None or row.get("id") is not None) + ] + event_ids = [ + str(row.get("event_id", row.get("id"))) + for row in event_rows + if isinstance(row, dict) and (row.get("event_id") is not None or row.get("id") is not None) + ] + fact_evidence_refs: list[str] = [] + fact_event_refs: list[str] = [] + event_evidence_refs: list[str] = [] + for row in ledger_rows: + if not isinstance(row, dict): + continue + evidence_values = row.get("evidence_refs", row.get("evidence_ids", [])) + event_values = row.get("event_refs", row.get("event_ids", [])) + if isinstance(evidence_values, list): + fact_evidence_refs.extend(str(ref) for ref in evidence_values) + if isinstance(event_values, list): + fact_event_refs.extend(str(ref) for ref in event_values) + for row in event_rows: + if not isinstance(row, dict): + continue + evidence_values = row.get("evidence_refs", row.get("evidence_ids", [])) + if isinstance(evidence_values, list): + event_evidence_refs.extend(str(ref) for ref in evidence_values) + evidence_failures = sorted( + set(fact_evidence_refs + event_evidence_refs) - set(evidence_ids) + ) + duplicate_evidence_ids = sorted(key for key, count in Counter(evidence_ids).items() if count > 1) + evidence_pass = not evidence_failures and not duplicate_evidence_ids + add_check( + "EVIDENCE_REFERENCE_CONSERVATION", + evidence_pass, + fact_evidence_refs + event_evidence_refs, + evidence_ids, + issue_code="EVIDENCE_REFERENCE_CONSERVATION_FAILED", + impact_scope="EVIDENCE", + source_refs=["evidence_indexed", "fact_ledger_base", "evidence_event_candidates"], + details={"dangling_refs": evidence_failures, "duplicate_evidence_ids": duplicate_evidence_ids}, + ) + event_failures = sorted(set(fact_event_refs) - set(event_ids)) + duplicate_event_ids = sorted(key for key, count in Counter(event_ids).items() if count > 1) + event_pass = not event_failures and not duplicate_event_ids + add_check( + "EVENT_REFERENCE_CONSERVATION", + event_pass, + fact_event_refs, + event_ids, + issue_code="EVENT_REFERENCE_CONSERVATION_FAILED", + impact_scope="EVIDENCE", + source_refs=["evidence_event_candidates", "fact_ledger_base"], + details={"dangling_refs": event_failures, "duplicate_event_ids": duplicate_event_ids}, + ) + disposition_rows = [row.get("disposition") for row in event_rows if isinstance(row, dict) and "disposition" in row] + b2_gate = documents.get("b2_event_candidates_gate") + declared_dispositions = None + if isinstance(b2_gate, dict): + declared_dispositions = b2_gate.get("event_disposition_counts") + if declared_dispositions is None and isinstance(b2_gate.get("summary"), dict): + declared_dispositions = b2_gate["summary"].get("event_disposition_counts") + if isinstance(declared_dispositions, dict): + disposition_expected = Counter( + {str(key): int(value) for key, value in declared_dispositions.items() if isinstance(value, int)} + ) + disposition_actual = Counter(str(value) for value in disposition_rows) + disposition_pass: bool | None = disposition_actual == disposition_expected + elif disposition_rows: + disposition_expected = Counter(str(value) for value in disposition_rows) + disposition_actual = Counter(str(value) for value in disposition_rows) + disposition_pass = all(isinstance(value, str) and value for value in disposition_rows) + else: + disposition_expected = Counter() + disposition_actual = Counter() + disposition_pass = None + add_check( + "EVENT_DISPOSITION_CONSERVATION", + disposition_pass, + disposition_actual, + disposition_expected, + issue_code="EVENT_DISPOSITION_CONSERVATION_FAILED", + impact_scope="EVIDENCE", + source_refs=["evidence_event_candidates", "b2_event_candidates_gate"], + ) + writer_report = documents.get("fact_ledger_writer_report") + if isinstance(writer_report, dict): + observed_domain_coverage = Counter( + str(domain_id) + for row in ledger_rows + if isinstance(row, dict) and isinstance(row.get("domain_effects"), dict) + for domain_id in row["domain_effects"] + ) + declared_domain_coverage = Counter( + {str(key): int(value) for key, value in writer_report.get("domain_effect_coverage", {}).items() if isinstance(value, int)} + ) + observed_readiness = Counter( + str(request.get("operand_state")) + for row in ledger_rows + if isinstance(row, dict) and isinstance(row.get("calculation_requests"), list) + for request in row["calculation_requests"] + if isinstance(request, dict) + ) + declared_readiness = Counter( + {str(key): int(value) for key, value in writer_report.get("calculation_readiness", {}).items() if isinstance(value, int)} + ) + ledger_snapshot = source_snapshots.get("fact_ledger_base") + final_hash = writer_report.get("final_sha256") + writer_pass = ( + writer_report.get("row_count") == len(ledger_rows) + and declared_domain_coverage == observed_domain_coverage + and declared_readiness == observed_readiness + and (ledger_snapshot is None or final_hash == ledger_snapshot.raw_sha256) + ) + add_check( + "FACT_LEDGER_WRITER_REPORT_CONNECTION", + writer_pass, + [len(ledger_rows), observed_domain_coverage, observed_readiness, ledger_snapshot.raw_sha256 if ledger_snapshot else None], + [writer_report.get("row_count"), declared_domain_coverage, declared_readiness, final_hash], + issue_code="FACT_LEDGER_WRITER_REPORT_MISMATCH", + impact_scope="FACT", + source_refs=["fact_ledger_base", "fact_ledger_writer_report"], + ) + else: + add_check( + "FACT_LEDGER_WRITER_REPORT_CONNECTION", + None, + None, + None, + issue_code="FACT_LEDGER_WRITER_REPORT_MISMATCH", + impact_scope="FACT", + source_refs=["fact_ledger_base", "fact_ledger_writer_report"], + ) + if signal_all is not None: + file_pass = bool(signal_all.get("file_conservation_pass")) + record_pass = bool(signal_all.get("record_conservation_pass")) + checks.append({"check_id": "SIGNAL_FILE_ROW_CONSERVATION", "status": "PASS" if file_pass else "FAIL"}) + checks.append({"check_id": "SIGNAL_RECORD_OCCURRENCE_CONSERVATION", "status": "PASS" if record_pass else "FAIL"}) + issues.extend(signal_all.get("issues", [])) + if not file_pass: + issues.append(_issue("SIGNAL_FILE_CONSERVATION_FAILED", impact_scope="SIGNAL")) + if not record_pass: + issues.append(_issue("SIGNAL_RECORD_CONSERVATION_FAILED", impact_scope="SIGNAL")) + if normalized_reviews is not None: + review_pass = normalized_reviews.get("conservation_status") == "PASS" + checks.append({"check_id": "REVIEW_OCCURRENCE_CONSERVATION", "status": "PASS" if review_pass else "FAIL"}) + if not review_pass: + issues.append(_issue("REVIEW_CONSERVATION_FAILED", impact_scope="REVIEW_ITEM")) + issues.extend(normalized_reviews.get("_issues", [])) + return {"checks": checks, "issues": issues, "passed": not any(check["status"] == "FAIL" for check in checks)} + + + def _source_ref( + logical_id: str, + pointer: str, + raw_value: Any = _RAW_VALUE_UNSET, + *, + stage1_id: str | None = None, + ) -> dict[str, Any]: + """Build a truthful RFC 6901 provenance row without pointer narrowing.""" + + row: dict[str, Any] = { + "logical_artifact_id": logical_id, + "json_pointer": pointer, + "raw_value_sha256": canonical_digest( + [logical_id, pointer] + if raw_value is _RAW_VALUE_UNSET + else raw_value + ), + "source_contract_row_ref": logical_id, + } + if stage1_id is not None: + row["stage1_id"] = stage1_id + return row + + + def _dedupe_source_refs(rows: Iterable[Mapping[str, Any]]) -> list[dict[str, Any]]: + by_digest = {canonical_digest(dict(row)): dict(row) for row in rows} + return [by_digest[key] for key in sorted(by_digest)] + + + def _coerce_source_ref(value: Mapping[str, Any] | str) -> dict[str, Any]: + if isinstance(value, dict) and { + "logical_artifact_id", + "json_pointer", + "raw_value_sha256", + }.issubset(value): + return dict(value) + text = str(value) + if "#" in text: + logical_id, pointer = text.split("#", 1) + else: + logical_id, pointer = text, "" + if pointer and not pointer.startswith("/"): + pointer = "/" + pointer + return _source_ref(logical_id or "unknown", pointer, text) + + + def _artifact_header( + schema_id: str = CONTEXT_SCHEMA_ID, + *, + schema_sha256: str = "0" * 64, + run_id: str = "STRUCTURAL-FIXTURE", + input_set_digest: str = "0" * 64, + stage2_release_digest: str = "0" * 64, + algorithm_digest: str = "0" * 64, + release_class: str = "DEV_FIXTURE_RELEASE", + ) -> dict[str, Any]: + return { + "schema_id": schema_id, + "schema_sha256": schema_sha256, + "schema_version": "stage2_s2_00_context.v2", + "producer_id": "S2_00", + "run_id": run_id, + "input_set_digest": input_set_digest, + "stage2_release_digest": stage2_release_digest, + "algorithm_digest": algorithm_digest, + "release_class": release_class, + } + + + def _derivation( + namespace: str, + input_refs: Sequence[str], + payload: Any, + ) -> dict[str, Any]: + sorted_refs = sorted(set(str(item) for item in input_refs)) or ["S2_00:EMPTY_INPUT_SET"] + minted = mint_stage2_id(namespace, [sorted_refs, payload], prefix="DRV") + return { + "derivation_id": minted["id"], + "algorithm_version": ALGORITHM_VERSION, + "sorted_input_refs": sorted_refs, + "mint_input_sha256": minted["mint_input_sha256"], + } + + + def build_case_context( + documents: Mapping[str, Any], + *, + run_binding_digest: str, + issues: Sequence[Mapping[str, Any]] = (), + artifact_header: Mapping[str, Any] | None = None, + signal_all: Mapping[str, Any] | None = None, + ) -> dict[str, Any]: + routing = documents.get("domain_activation_manifest") + routing_payload = _activation_payload(routing) if isinstance(routing, dict) else {} + active = routing_payload.get("active_domain_ids", []) + expected = routing_payload.get("expected_runnable_domain_ids", []) + bo_rows = _array_rows(documents.get("bo"), ("business_objects", "BO", "rows", "items")) + ledger_rows = _array_rows(documents.get("fact_ledger_base"), ("facts", "fact_ledger", "rows", "items")) + les_rows = _array_rows(documents.get("legal_effect_structures"), ("structures", "structure_records", "legal_effect_structures", "rows", "items")) + bo_ids = sorted( + str(row["BO_ID"]) + for row in bo_rows + if isinstance(row, dict) and row.get("BO_ID") is not None + ) + structure_ids = sorted( + str(row.get("structure_id", row.get("legal_effect_structure_id"))) + for row in les_rows + if isinstance(row, dict) and (row.get("structure_id") is not None or row.get("legal_effect_structure_id") is not None) + ) + structures_by_bo: dict[str, list[str]] = defaultdict(list) + for row in les_rows: + if not isinstance(row, dict): + continue + structure_id = row.get("structure_id", row.get("legal_effect_structure_id")) + for bo_id in row.get("source_bo_ids", []) if isinstance(row.get("source_bo_ids"), list) else []: + if structure_id is not None: + structures_by_bo[str(bo_id)].append(str(structure_id)) + fact_contexts: list[dict[str, Any]] = [] + for index, row in enumerate(ledger_rows): + if not isinstance(row, dict): + continue + if row.get("fact_id") is None or row.get("source_bo_id") is None: + continue + fact_id = str(row["fact_id"]) + source_bo_id = str(row.get("source_bo_id", "MISSING")) + evidence_refs = row.get("evidence_refs", row.get("evidence_ids", [])) + event_refs = row.get("event_refs", row.get("event_ids", [])) + calculation_requests = row.get("calculation_requests", []) + law_version_refs = row.get("law_version_refs", []) + signal_occurrence_refs = sorted( + str(occurrence.get("occurrence_ref")) + for occurrence in (signal_all or {}).get("record_occurrences", []) + if occurrence.get("disposition") == "USED" + and ( + f"fact_id:{fact_id}" in occurrence.get("binding_refs", []) + or f"source_bo_id:{source_bo_id}" in occurrence.get("binding_refs", []) + or f"bo_id:{source_bo_id}" in occurrence.get("binding_refs", []) + ) + ) + fact_contexts.append( + { + "fact_id": fact_id, + "source_bo_id": source_bo_id, + "party_context_ids": [], + "object_context_ids": [], + "structure_refs": sorted(set(structures_by_bo.get(source_bo_id, []))), + "signal_occurrence_refs": signal_occurrence_refs, + "calculation_request_refs": sorted( + canonical_digest(value) + for value in (calculation_requests if isinstance(calculation_requests, list) else []) + ), + "law_version_refs": sorted(str(value) for value in law_version_refs) if isinstance(law_version_refs, list) else [], + "evidence_refs": sorted(str(value) for value in evidence_refs) if isinstance(evidence_refs, list) else [], + "event_refs": sorted(str(value) for value in event_refs) if isinstance(event_refs, list) else [], + "review_keys": [], + "source_refs": [_source_ref("fact_ledger_base", f"/rows/{index}", row, stage1_id=fact_id)], + } + ) + source_refs = _dedupe_source_refs( + [ + _source_ref("bo", "", documents.get("bo")), + _source_ref("fact_ledger_base", "", documents.get("fact_ledger_base")), + _source_ref("legal_effect_structures", "", documents.get("legal_effect_structures")), + _source_ref("domain_activation_manifest", "", documents.get("domain_activation_manifest")), + _source_ref("client_goal", "", documents.get("client_goal")), + ] + ) + minted = mint_stage2_id( + "case_context", + [run_binding_digest, [row["fact_id"] for row in fact_contexts], bo_ids, structure_ids], + prefix="CC", + ) + unavailable_scope_refs = sorted( + { + str(scope_ref) + for item in issues + if item.get("issue_code") + for scope_ref in item.get("scope_refs", []) + } + ) + return { + "artifact_header": dict(artifact_header or _artifact_header()), + "case_context_id": minted["id"], + "facts": sorted(fact_contexts, key=lambda row: row["fact_id"]), + "bo_ids": bo_ids, + "structure_ids": structure_ids, + "active_domain_ids": sorted(set(str(value) for value in active)) if isinstance(active, list) else [], + "expected_runnable_domain_ids": sorted(set(str(value) for value in expected)) if isinstance(expected, list) else [], + "unavailable_scope_refs": unavailable_scope_refs, + "source_refs": source_refs, + "derivation": _derivation("case_context_derivation", [f"fact:{row['fact_id']}" for row in fact_contexts], minted["mint_input_sha256"]), + } + + + def build_evidence_inventory( + documents: Mapping[str, Any], + *, + run_binding_digest: str, + artifact_header: Mapping[str, Any] | None = None, + ) -> dict[str, Any]: + evidence_rows = _array_rows(documents.get("evidence_indexed"), ("evidence", "evidence_items", "rows", "items")) + event_rows = _array_rows(documents.get("evidence_event_candidates"), ("events", "event_candidates", "rows", "items")) + ledger_rows = _array_rows(documents.get("fact_ledger_base"), ("facts", "fact_ledger", "rows", "items")) + event_by_evidence: dict[str, set[str]] = defaultdict(set) + for index, row in enumerate(event_rows): + if not isinstance(row, dict): + continue + event_id = str(row.get("event_id", f"EVENT-OCCURRENCE-{index}")) + refs = row.get("evidence_refs", row.get("evidence_ids", [])) + if isinstance(refs, list): + for ref in refs: + event_by_evidence[str(ref)].add(event_id) + facts_by_evidence: dict[str, set[str]] = defaultdict(set) + for row in ledger_rows: + if not isinstance(row, dict) or row.get("fact_id") is None: + continue + refs = row.get("evidence_refs", row.get("evidence_ids", [])) + if isinstance(refs, list): + for ref in refs: + facts_by_evidence[str(ref)].add(str(row["fact_id"])) + items: list[dict[str, Any]] = [] + for index, row in enumerate(evidence_rows): + if not isinstance(row, dict): + continue + evidence_id = str(row.get("evidence_id", row.get("id", f"EVIDENCE-OCCURRENCE-{index}"))) + items.append( + { + "evidence_id": evidence_id, + "event_ids": sorted(event_by_evidence.get(evidence_id, set())), + "linked_fact_ids": sorted(facts_by_evidence.get(evidence_id, set())), + "technical_disposition": "AVAILABLE", + "source_refs": [_source_ref("evidence_indexed", f"/items/{index}", row, stage1_id=evidence_id)], + } + ) + source_refs = _dedupe_source_refs( + [ + _source_ref("evidence_indexed", "", documents.get("evidence_indexed")), + _source_ref("evidence_event_candidates", "", documents.get("evidence_event_candidates")), + _source_ref("fact_ledger_base", "", documents.get("fact_ledger_base")), + ] + ) + return { + "artifact_header": dict(artifact_header or _artifact_header()), + "items": sorted(items, key=lambda row: row["evidence_id"]), + "source_refs": source_refs, + "derivation": _derivation("evidence_inventory_derivation", [f"evidence:{row['evidence_id']}" for row in items], run_binding_digest), + } + + + def _candidate_values(row: Mapping[str, Any], keys: Sequence[str]) -> list[tuple[str, Any]]: + result: list[tuple[str, Any]] = [] + for key in keys: + value = row.get(key) + if value is None: + continue + if isinstance(value, list): + result.extend((f"/{key}/{index}", item) for index, item in enumerate(value)) + else: + result.append((f"/{key}", value)) + return result + + + def build_object_registry( + documents: Mapping[str, Any], + *, + run_binding_digest: str, + artifact_header: Mapping[str, Any] | None = None, + ) -> dict[str, Any]: + bo_rows = _array_rows(documents.get("bo"), ("business_objects", "BO", "rows", "items")) + objects: list[dict[str, Any]] = [] + collision_registry: dict[str, str] = {} + for index, row in enumerate(bo_rows): + if not isinstance(row, dict): + continue + explicit_bo = row.get("BO_ID") + for pointer, value in _candidate_values(row, ("object", "objects", "asset", "subject_matter")): + stage1_object_id = value.get("object_id") if isinstance(value, dict) else None + context = mint_context_occurrence_id( + "object", + "bo", + f"/{index}{pointer}", + value, + [str(explicit_bo)] if explicit_bo is not None else [], + stage1_canonical_id=str(stage1_object_id) if stage1_object_id is not None else None, + collision_registry=collision_registry, + ) + objects.append(context) + return { + "artifact_header": dict(artifact_header or _artifact_header()), + "objects": sorted(objects, key=lambda row: row["object_context_id"]), + "source_refs": [_source_ref("bo", "", documents.get("bo"))], + "derivation": _derivation("object_registry_derivation", [row["object_context_id"] for row in objects], run_binding_digest), + } + + + def build_party_and_title_context( + documents: Mapping[str, Any], + *, + run_binding_digest: str, + artifact_header: Mapping[str, Any] | None = None, + ) -> dict[str, Any]: + bo_rows = _array_rows(documents.get("bo"), ("business_objects", "BO", "rows", "items")) + parties: list[dict[str, Any]] = [] + collision_registry: dict[str, str] = {} + for index, row in enumerate(bo_rows): + if not isinstance(row, dict): + continue + explicit_bo = row.get("BO_ID") + for pointer, value in _candidate_values(row, ("party", "parties", "creditor", "debtor", "counterparty")): + stage1_party_id = value.get("party_id") if isinstance(value, dict) else None + identity = mint_context_occurrence_id( + "party", + "bo", + f"/{index}{pointer}", + value, + [str(explicit_bo)] if explicit_bo is not None else [], + stage1_canonical_id=str(stage1_party_id) if stage1_party_id is not None else None, + collision_registry=collision_registry, + ) + source_refs = identity["source_refs"] + parties.append( + { + "party_identity": identity, + "party_title": value.get("party_title") if isinstance(value, dict) else None, + "defendant_role": value.get("defendant_role") if isinstance(value, dict) else None, + "liability_context": value.get("liability_context") if isinstance(value, dict) else None, + "client_instruction": value.get("client_instruction") if isinstance(value, dict) else None, + "recovery_information": value.get("recovery_information") if isinstance(value, dict) else None, + "source_refs": source_refs, + } + ) + return { + "artifact_header": dict(artifact_header or _artifact_header()), + "parties": sorted(parties, key=lambda row: row["party_identity"]["party_context_id"]), + "source_refs": [_source_ref("bo", "", documents.get("bo"))], + "derivation": _derivation( + "party_title_derivation", + [row["party_identity"]["party_context_id"] for row in parties], + run_binding_digest, + ), + } + + + def build_slot_crosswalk( + documents: Mapping[str, Any], + domain_configs: Mapping[str, Any] | None = None, + *, + run_binding_digest: str, + artifact_header: Mapping[str, Any] | None = None, + ) -> dict[str, Any]: + domain_configs = domain_configs or {} + rows: list[dict[str, Any]] = [] + for domain_id in sorted(domain_configs): + config = domain_configs[domain_id] + adapter_id = "S2A-DOMAIN-CONFIG-V2" if domain_id in V2_DOMAIN_IDS else "S2A-DOMAIN-CONFIG-V1" + if not isinstance(config, dict): + continue + for slot_kind, key in ( + ("ELEMENT", "element_slots"), + ("OPPOSING_FACT", "opposing_fact_slots"), + ("DEFENSE", "defense_map"), + ): + values = config.get(key, []) + if isinstance(values, dict): + values = [{"slot_id": item_key, "value": item_value} for item_key, item_value in values.items()] + if not isinstance(values, list): + continue + for index, value in enumerate(values): + rows.append( + { + "domain_id": domain_id, + "slot_ref": str(value.get("slot_id", value.get("id", f"{domain_id}:{key}:{index}"))) if isinstance(value, dict) else f"{domain_id}:{key}:{index}", + "slot_kind": slot_kind, + "fact_ids": [], + "evidence_refs": [], + "evaluation_status": "UNEVALUABLE", + "proposed_new_slot": False, + "source_refs": [_source_ref(f"domain_config:{domain_id}", f"/{key}/{index}", value)], + "_adapter_id": adapter_id, + } + ) + adapter_ids = sorted({row.pop("_adapter_id") for row in rows}) + return { + "artifact_header": dict(artifact_header or _artifact_header()), + "domain_config_adapter_ids": adapter_ids, + "rows": sorted(rows, key=lambda row: (row["domain_id"], row["slot_kind"], row["slot_ref"])), + "rebuttal_slot_synthesis_count": 0, + "source_refs": _dedupe_source_refs( + [_source_ref("fact_ledger_base", "", documents.get("fact_ledger_base"))] + + [_source_ref(f"domain_config:{key}", "", domain_configs[key]) for key in sorted(domain_configs)] + ), + "derivation": _derivation("slot_crosswalk_derivation", [row["slot_ref"] for row in rows], run_binding_digest), + } + + + def _tarjan_scc(nodes: Sequence[str], edges: Sequence[tuple[str, str]]) -> list[list[str]]: + adjacency: dict[str, list[str]] = {node: [] for node in nodes} + for source, target in edges: + adjacency.setdefault(source, []).append(target) + adjacency.setdefault(target, []) + for value in adjacency.values(): + value.sort() + index = 0 + stack: list[str] = [] + on_stack: set[str] = set() + indices: dict[str, int] = {} + lowlink: dict[str, int] = {} + components: list[list[str]] = [] + + def visit(node: str) -> None: + nonlocal index + indices[node] = index + lowlink[node] = index + index += 1 + stack.append(node) + on_stack.add(node) + for neighbor in adjacency[node]: + if neighbor not in indices: + visit(neighbor) + lowlink[node] = min(lowlink[node], lowlink[neighbor]) + elif neighbor in on_stack: + lowlink[node] = min(lowlink[node], indices[neighbor]) + if lowlink[node] == indices[node]: + component: list[str] = [] + while True: + member = stack.pop() + on_stack.remove(member) + component.append(member) + if member == node: + break + components.append(sorted(component)) + + for node in sorted(adjacency): + if node not in indices: + visit(node) + return sorted(components, key=lambda component: component[0]) + + + def compile_cluster_plan( + members: Sequence[Mapping[str, Any]], + relations: Sequence[Mapping[str, Any]], + *, + algorithm_version: str = ALGORITHM_VERSION, + artifact_header: Mapping[str, Any] | None = None, + ) -> dict[str, Any]: + """Compile claim-neutral clusters using only explicit hard relations.""" + + by_id = {str(row["member_id"]): dict(row) for row in members} + if len(by_id) != len(members): + raise IngressError("DUPLICATE_CLUSTER_MEMBER_ID", "cluster member IDs must be unique") + parent = {member_id: member_id for member_id in by_id} + + def find(value: str) -> str: + while parent[value] != value: + parent[value] = parent[parent[value]] + value = parent[value] + return value + + def union(left: str, right: str) -> None: + a, b = find(left), find(right) + if a != b: + parent[max(a, b)] = min(a, b) + + normalized_relations: list[dict[str, Any]] = [] + relation_key_map = { + "SAME_BO_ID": ("EXPLICIT_SOURCE_RELATION", "SAME_BO_ID", True), + "SOURCE_BO_ATTACHMENT": ("EXPLICIT_SOURCE_RELATION", "LES_SOURCE_BO_ID", True), + "SAME_EVIDENCE_REF": ("EXPLICIT_SOURCE_RELATION", "SAME_EVIDENCE_REF", True), + "SAME_EVENT_REF": ("EXPLICIT_SOURCE_RELATION", "SAME_EVENT_REF", True), + "EXPLICIT_CASE_RELATION": ("EXPLICIT_SOURCE_RELATION", "APPROVED_EXPLICIT_CASE_RELATION", True), + "claim_precondition": ("CANDIDATE_RELATION", "CLAIM_PRECONDITION_CANDIDATE", False), + "accessory_of": ("CANDIDATE_RELATION", "ACCESSORY_OF_CANDIDATE", False), + "incompatible_with": ("CANDIDATE_RELATION", "INCOMPATIBLE_WITH_CANDIDATE", False), + "EXPLICIT_DEPENDENCY": ("CANDIDATE_RELATION", "CLAIM_PRECONDITION_CANDIDATE", False), + } + for relation in relations: + source = str(relation.get("source_member_id")) + target = str(relation.get("target_member_id")) + kind = str(relation.get("relation_kind")) + if source not in by_id or target not in by_id: + raise IngressError("DANGLING_CLUSTER_RELATION", f"relation references unknown member: {source}->{target}") + explicit_source_refs = relation.get("source_refs") + if not isinstance(explicit_source_refs, list) or not explicit_source_refs: + raise IngressError("RELATION_SOURCE_REF_MISSING", "every cluster relation needs explicit source refs") + edge_class, relation_key, hard_join_allowed = relation_key_map.get( + kind, + ("CANDIDATE_RELATION", "CLAIM_PRECONDITION_CANDIDATE", False), + ) + relation_source_refs = _dedupe_source_refs( + _coerce_source_ref(item) for item in explicit_source_refs + ) + edge_mint = mint_stage2_id( + "cluster_edge", + [source, target, kind, relation_source_refs], + prefix="EDGE", + ) + normalized = { + "source_member_id": source, + "target_member_id": target, + "source_relation_kind": kind, + "edge_id": edge_mint["id"], + "edge_class": edge_class, + "relation_key": relation_key, + "hard_join_allowed": hard_join_allowed, + "source_refs": relation_source_refs, + } + normalized_relations.append(normalized) + if hard_join_allowed and kind in HARD_RELATION_KINDS: + union(source, target) + groups: dict[str, list[str]] = defaultdict(list) + for member_id in sorted(by_id): + groups[find(member_id)].append(member_id) + collision_registry: dict[str, str] = {} + cluster_build_rows: list[dict[str, Any]] = [] + member_to_cluster: dict[str, str] = {} + for member_ids in sorted((sorted(value) for value in groups.values()), key=lambda value: value[0]): + source_refs = _dedupe_source_refs( + _coerce_source_ref(ref) + for member_id in member_ids + for ref in by_id[member_id].get("source_refs", []) + ) + relation_keys = sorted( + canonical_digest(relation) + for relation in normalized_relations + if relation["source_member_id"] in member_ids and relation["target_member_id"] in member_ids + ) + minted = mint_stage2_id( + "cluster", + [sorted(member_ids), source_refs, relation_keys], + prefix="CL", + algorithm_version=algorithm_version, + collision_registry=collision_registry, + ) + executable = bool(source_refs) and all( + by_id[member_id].get("scope_technical_disposition", "AVAILABLE") != "UNAVAILABLE" + and not by_id[member_id].get("residual_review_only", False) + for member_id in member_ids + ) + cluster = { + "cluster_id": minted["id"], + "mint_input_sha256": minted["mint_input_sha256"], + "member_ids": member_ids, + "source_refs": source_refs, + "cluster_status": "EXECUTABLE" if executable else "RESIDUAL_REVIEW_ONLY", + } + cluster_build_rows.append(cluster) + for member_id in member_ids: + member_to_cluster[member_id] = minted["id"] + cluster_edges: list[dict[str, Any]] = [] + edge_pairs: set[tuple[str, str]] = set() + for relation in normalized_relations: + source_cluster = member_to_cluster[relation["source_member_id"]] + target_cluster = member_to_cluster[relation["target_member_id"]] + if source_cluster == target_cluster: + continue + if relation["source_relation_kind"] not in CANDIDATE_RELATION_KINDS: + continue + edge = { + "source_cluster_id": source_cluster, + "target_cluster_id": target_cluster, + "edge_id": relation["edge_id"], + "edge_class": relation["edge_class"], + "relation_key": relation["relation_key"], + "hard_join_allowed": False, + "source_refs": relation["source_refs"], + } + cluster_edges.append(edge) + edge_pairs.add((source_cluster, target_cluster)) + cluster_ids = [row["cluster_id"] for row in cluster_build_rows] + sccs = _tarjan_scc(cluster_ids, sorted(edge_pairs)) + scc_rows: list[dict[str, Any]] = [] + cluster_to_scc: dict[str, str] = {} + for members_in_scc in sccs: + minted = mint_stage2_id("cluster_scc", members_in_scc, prefix="SCC") + cycle = len(members_in_scc) > 1 or any(left == right == members_in_scc[0] for left, right in edge_pairs) + row = { + "scc_id": minted["id"], + "member_cluster_ids": members_in_scc, + "cycle_preserved": cycle, + "mint_input_sha256": minted["mint_input_sha256"], + } + scc_rows.append(row) + for cluster_id in members_in_scc: + cluster_to_scc[cluster_id] = minted["id"] + scheduling_edges = sorted( + { + (cluster_to_scc[left], cluster_to_scc[right]) + for left, right in edge_pairs + if cluster_to_scc[left] != cluster_to_scc[right] + } + ) + scc_nodes = sorted(row["scc_id"] for row in scc_rows) + indegree = {node: 0 for node in scc_nodes} + adjacency: dict[str, set[str]] = {node: set() for node in scc_nodes} + for source, target in scheduling_edges: + if target not in adjacency[source]: + adjacency[source].add(target) + indegree[target] += 1 + scheduling_waves: list[list[str]] = [] + ready = sorted(node for node, degree in indegree.items() if degree == 0) + visited: set[str] = set() + while ready: + scheduling_waves.append(ready) + next_ready: list[str] = [] + for node in ready: + visited.add(node) + for target in sorted(adjacency[node]): + indegree[target] -= 1 + if indegree[target] == 0: + next_ready.append(target) + ready = sorted(set(next_ready)) + if len(visited) != len(scc_nodes): + raise IngressError("SCC_SCHEDULING_DAG_INVALID", "SCC condensation graph must be acyclic") + executable_ids = sorted( + row["cluster_id"] for row in cluster_build_rows if row["cluster_status"] == "EXECUTABLE" + ) + residual_ids = sorted(set(cluster_ids) - set(executable_ids)) + clusters: list[dict[str, Any]] = [] + for build_row in cluster_build_rows: + cluster_id = build_row["cluster_id"] + member_ids = build_row["member_ids"] + cluster_members = [ + { + "member_ref": str(by_id[member_id].get("member_ref", member_id)), + "member_kind": str(by_id[member_id].get("member_kind", "FACT")), + "source_refs": _dedupe_source_refs( + _coerce_source_ref(ref) for ref in by_id[member_id].get("source_refs", []) + ), + } + for member_id in member_ids + ] + internal_edges: list[dict[str, Any]] = [] + for relation in normalized_relations: + if relation["source_member_id"] in member_ids and relation["target_member_id"] in member_ids: + internal_edges.append( + { + "edge_id": relation["edge_id"], + "from_ref": relation["source_member_id"], + "to_ref": relation["target_member_id"], + "edge_class": relation["edge_class"], + "relation_key": relation["relation_key"], + "hard_join_allowed": relation["hard_join_allowed"], + "source_refs": relation["source_refs"], + } + ) + for edge in cluster_edges: + if cluster_id in {edge["source_cluster_id"], edge["target_cluster_id"]}: + internal_edges.append( + { + "edge_id": edge["edge_id"], + "from_ref": edge["source_cluster_id"], + "to_ref": edge["target_cluster_id"], + "edge_class": edge["edge_class"], + "relation_key": edge["relation_key"], + "hard_join_allowed": False, + "source_refs": edge["source_refs"], + } + ) + cluster = { + "cluster_id": cluster_id, + "cluster_status": build_row["cluster_status"], + "members": sorted(cluster_members, key=lambda row: row["member_ref"]), + "edges": sorted(internal_edges, key=lambda row: row["edge_id"]), + "scc_ids": [cluster_to_scc[cluster_id]], + "slice_ref": f"cluster_slices/{cluster_id}.json" if cluster_id in executable_ids else None, + "bundle_cohort_id": None, + "source_refs": build_row["source_refs"], + "derivation": { + "derivation_id": cluster_id, + "algorithm_version": algorithm_version, + "sorted_input_refs": sorted(member_ids), + "mint_input_sha256": build_row["mint_input_sha256"], + }, + } + clusters.append(cluster) + top_source_refs = _dedupe_source_refs( + ref for row in clusters for ref in row["source_refs"] + ) if clusters else [] + return { + "artifact_header": dict(artifact_header or _artifact_header()), + "clusters": sorted(clusters, key=lambda row: row["cluster_id"]), + "executable_cluster_ids": executable_ids, + "residual_review_cluster_ids": residual_ids, + "scheduling_waves": scheduling_waves, + "source_refs": top_source_refs, + "derivation": _derivation( + "cluster_plan_derivation", + cluster_ids or ["NO_CLUSTER"], + [scheduling_edges, [[row["scc_id"], row["member_cluster_ids"]] for row in scc_rows]], + ), + } + + + def compile_cluster_slices( + cluster_plan: Mapping[str, Any], + context_artifacts: Mapping[str, Mapping[str, Any]], + ) -> dict[str, dict[str, Any]]: + """Create immutable minimal source-ref slices for every cluster.""" + + projection_sources: dict[str, dict[str, Any]] = defaultdict(dict) + case_context = context_artifacts.get("case_context", {}) + for row in case_context.get("facts", []) if isinstance(case_context, dict) else []: + if isinstance(row, dict) and row.get("fact_id") is not None: + projection_sources["FACT"][str(row["fact_id"])] = row + evidence_inventory = context_artifacts.get("evidence_inventory", {}) + for row in evidence_inventory.get("items", []) if isinstance(evidence_inventory, dict) else []: + if not isinstance(row, dict): + continue + identifier = row.get("evidence_id", row.get("id")) + if identifier is not None: + projection_sources["EVIDENCE"][str(identifier)] = row + supplied_sources = context_artifacts.get("_projection_sources", {}) + if isinstance(supplied_sources, dict): + for kind, rows in supplied_sources.items(): + if isinstance(rows, dict): + projection_sources[str(kind)].update({str(key): value for key, value in rows.items()}) + + def materialize_projection( + kind: str, + source_id: str, + ) -> dict[str, Any]: + content = projection_sources.get(kind, {}).get(source_id) + if content is None: + raise IngressError( + "SLICE_PROJECTION_SOURCE_MISSING", + f"projection source is missing: {kind}:{source_id}", + ) + if not isinstance(content, dict): + raise IngressError( + "SLICE_PROJECTION_SOURCE_MISSING", + f"projection source is not an object: {kind}:{source_id}", + ) + raw = canonical_json_bytes(content) + if len(raw) > 262144: + raise IngressError("SLICE_PROJECTION_BUDGET_EXCEEDED", f"projection exceeds 262144 bytes: {kind}:{source_id}") + source_refs = _dedupe_source_refs(content.get("source_refs", [])) + matching_refs = [ + ref + for ref in source_refs + if ref.get("stage1_id") == source_id + and isinstance(ref.get("raw_value_sha256"), str) + and re.fullmatch(r"[a-f0-9]{64}", str(ref["raw_value_sha256"])) + ] + matching_hashes = {str(ref["raw_value_sha256"]) for ref in matching_refs} + if not matching_refs or len(matching_hashes) != 1: + raise IngressError( + "SLICE_PROJECTION_PROVENANCE_MISSING", + f"projection lacks one unambiguous source_id-bound raw hash: {kind}:{source_id}", + ) + raw_source_sha256 = next(iter(matching_hashes)) + projection_mint = mint_stage2_id("content_projection", [kind, source_id, content], prefix="CP") + return { + "projection_id": projection_mint["id"], + "projection_kind": kind, + "source_id": source_id, + "content_schema_id": None, + "content": content, + "raw_source_sha256": raw_source_sha256, + "canonical_content_sha256": canonical_digest(content), + "materialized_utf8_bytes": len(raw), + "projection_policy_id": "S2-SLICE-BOUNDED-PROJECTION-V1", + "truncated": False, + "source_refs": source_refs, + } + + projection_arrays = { + "FACT": "fact_projections", + "EVIDENCE": "evidence_projections", + "EVENT": "event_projections", + "LES_STRUCTURE": "les_structure_projections", + "SIGNAL_OCCURRENCE": "signal_occurrence_projections", + "REVIEW_ITEM": "review_item_projections", + "ACTIVE_PROFILE": "profile_projections", + } + slices: dict[str, dict[str, Any]] = {} + for cluster in cluster_plan.get("clusters", []): + cluster_id = cluster["cluster_id"] + source_refs = _dedupe_source_refs(cluster.get("source_refs", [])) + member_refs_by_kind: dict[str, list[str]] = defaultdict(list) + for member in cluster.get("members", []): + member_refs_by_kind[str(member.get("member_kind"))].append(str(member.get("member_ref"))) + minted = mint_stage2_id( + "cluster_slice", + [cluster_id, source_refs, sorted(member_refs_by_kind.items())], + prefix="SL", + ) + slice_body: dict[str, Any] = { + "artifact_header": dict(cluster_plan.get("artifact_header", _artifact_header())), + "cluster_id": cluster_id, + "slice_id": minted["id"], + "slice_schema_version": "stage2_cluster_slice.v1", + "immutable": True, + "source_refs": source_refs, + "fact_refs": sorted(set(member_refs_by_kind.get("FACT", []))), + "evidence_refs": sorted(set(member_refs_by_kind.get("EVIDENCE", []))), + "event_refs": sorted(set(member_refs_by_kind.get("EVENT", []))), + "les_structure_refs": sorted(set(member_refs_by_kind.get("LES_STRUCTURE", []))), + "signal_occurrence_refs": sorted(set(member_refs_by_kind.get("SIGNAL_OCCURRENCE", []))), + "review_keys": sorted(set(member_refs_by_kind.get("REVIEW_ITEM", []))), + "profile_refs": [], + "derivation": { + "derivation_id": minted["id"], + "algorithm_version": ALGORITHM_VERSION, + "sorted_input_refs": sorted( + str(member.get("member_ref")) for member in cluster.get("members", []) + ) or [cluster_id], + "mint_input_sha256": minted["mint_input_sha256"], + }, + } + all_projections: list[dict[str, Any]] = [] + for kind, array_name in projection_arrays.items(): + source_ids = sorted(set(member_refs_by_kind.get(kind, []))) + projections = [materialize_projection(kind, source_id) for source_id in source_ids] + slice_body[array_name] = projections + all_projections.extend(projections) + observed_projection_bytes = sum(row["materialized_utf8_bytes"] for row in all_projections) + if observed_projection_bytes > 2097152: + raise IngressError("SLICE_CONTENT_BUDGET_EXCEEDED", f"slice exceeds 2097152 bytes: {cluster_id}") + slice_body["projection_budget"] = { + "policy_id": "S2-SLICE-BOUNDED-PROJECTION-V1", + "max_projection_utf8_bytes": 262144, + "max_slice_content_utf8_bytes": 2097152, + "observed_projection_count": len(all_projections), + "observed_slice_content_utf8_bytes": observed_projection_bytes, + "budget_status": "PASS", + "raw_stage1_reread_required": False, + } + slice_body["slice_digest"] = canonical_digest(slice_body) + slices[cluster_id] = slice_body + return slices + + + _PII_PATTERNS: tuple[tuple[str, re.Pattern[str]], ...] = ( + ("KOREAN_RESIDENT_ID", re.compile(r"\b\d{6}-?[1-4]\d{6}\b")), + ("EMAIL", re.compile(r"\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b")), + ("KOREAN_PHONE", re.compile(r"\b01[016789]-?\d{3,4}-?\d{4}\b")), + ) + + + def _scan_pii(value: Any) -> list[str]: + rendered = canonical_json_bytes(value).decode("utf-8") + return [name for name, pattern in _PII_PATTERNS if pattern.search(rendered)] + + + def compile_bundle_plan( + cluster_plan: Mapping[str, Any], + cluster_slices: Mapping[str, Mapping[str, Any]], + release_lock: Mapping[str, Any], + *, + stage2_asset_root: str | os.PathLike[str] | None = None, + ) -> dict[str, Any]: + """Compile a data-only S2_10 handoff without materializing prompts.""" + + bundle = release_lock.get("bundle") + if not isinstance(bundle, dict): + raise IngressError("BUNDLE_RELEASE_CONTRACT_MISSING", "release bundle must be an object") + mode = bundle.get("mode") + release_class = release_lock.get("release_class") + if mode not in RELEASE_MODE or RELEASE_MODE[mode] != release_class: + raise IngressError("BUNDLE_RELEASE_CLASS_MISMATCH", f"{mode} is not valid for {release_class}") + handoff_contract_version = bundle.get("handoff_contract_version") + if handoff_contract_version != "S2_00_STRUCTURED_CONTEXT_HANDOFF_V2": + raise IngressError( + "HANDOFF_CONTRACT_VERSION_MISMATCH", + "S2_00 requires S2_00_STRUCTURED_CONTEXT_HANDOFF_V2", + ) + release_ref = str( + release_lock.get( + "release_id", + release_lock.get("stage2_release_digest", "stage2_release"), + ) + ) + verification_errors: list[str] = [] + + def normalize_sha256(value: Any, *, code: str) -> str: + if not isinstance(value, str) or re.fullmatch(r"[A-Fa-f0-9]{64}", value) is None: + verification_errors.append(code) + return "0" * 64 + return value.lower() + + def verify_opaque_ref(value: Any, *, label: str) -> str: + if not isinstance(value, dict): + verification_errors.append(f"{label}_REF_MISSING") + return "0" * 64 + try: + path = _safe_relative_path(value.get("path")).as_posix() + except (IngressError, TypeError): + verification_errors.append(f"{label}_REF_PATH_INVALID") + return "0" * 64 + expected = normalize_sha256( + value.get("sha256"), + code=f"{label}_HASH_UNBOUND", + ) + if stage2_asset_root is None: + verification_errors.append("STAGE2_ASSET_ROOT_MISSING") + return expected + try: + snapshot = open_bounded_snapshot( + stage2_asset_root, + path, + logical_input_id=label.lower(), + ) + except IngressError as exc: + verification_errors.append(f"{label}_ASSET_UNAVAILABLE") + return expected + if snapshot.raw_sha256 != expected: + verification_errors.append(f"{label}_HASH_MISMATCH") + return expected + + s2_10_agent_sha256 = verify_opaque_ref( + bundle.get("s2_10_agent_ref"), + label="S2_10_AGENT", + ) + s2_10_llm_binding_sha256 = verify_opaque_ref( + bundle.get("s2_10_llm_binding_ref"), + label="S2_10_LLM_BINDING", + ) + if normalize_sha256( + bundle.get("s2_10_agent_sha256"), + code="S2_10_AGENT_HASH_UNBOUND", + ) != s2_10_agent_sha256: + verification_errors.append("S2_10_AGENT_RELEASE_HASH_MISMATCH") + if normalize_sha256( + bundle.get("s2_10_llm_binding_sha256"), + code="S2_10_LLM_BINDING_HASH_UNBOUND", + ) != s2_10_llm_binding_sha256: + verification_errors.append("S2_10_LLM_BINDING_RELEASE_HASH_MISMATCH") + if bundle.get("context_cohort_policy_id") != "S2-CACHE-STRUCTURED-CONTEXT-HANDOFF-V2": + verification_errors.append("CONTEXT_COHORT_POLICY_MISMATCH") + if bundle.get("downstream_hybrid_status") != "HYBRID_RELEASE_BOUND": + verification_errors.append("S2_10_HYBRID_RELEASE_UNBOUND") + + selected_context_input = bundle.get("selected_context_refs", []) + if not isinstance(selected_context_input, list): + raise IngressError( + "SELECTED_CONTEXT_REFS_SHAPE", + "selected_context_refs must be an array", + ) + allowed_context_kinds = { + "ACTIVE_PROFILE", + "APPROVED_COMMON_AUTHORITY", + "LAW_VALUE_TOKEN", + "AUTHORITY_PROPOSITION", + } + selected_context_refs: list[dict[str, Any]] = [] + seen_context_ids: set[str] = set() + for index, row in enumerate(selected_context_input): + if not isinstance(row, dict): + raise IngressError( + "SELECTED_CONTEXT_REF_SHAPE", + f"selected context row {index} must be an object", + ) + context_ref_id = row.get("context_ref_id") + context_kind = row.get("context_kind") + if not isinstance(context_ref_id, str) or not context_ref_id: + raise IngressError( + "SELECTED_CONTEXT_REF_ID_INVALID", + f"selected context row {index} lacks context_ref_id", + ) + if context_ref_id in seen_context_ids: + verification_errors.append("SELECTED_CONTEXT_REF_ID_DUPLICATE") + seen_context_ids.add(context_ref_id) + if context_kind not in allowed_context_kinds: + raise IngressError( + "SELECTED_CONTEXT_KIND_INVALID", + f"unsupported context kind: {context_kind}", + ) + try: + relative_path = _safe_relative_path(row.get("path")).as_posix() + except (IngressError, TypeError) as exc: + raise IngressError( + "SELECTED_CONTEXT_PATH_INVALID", + f"invalid selected context path at row {index}", + ) from exc + raw_sha256 = normalize_sha256( + row.get("raw_sha256"), + code="SELECTED_CONTEXT_RAW_HASH_UNBOUND", + ) + canonical_sha256 = normalize_sha256( + row.get("canonical_sha256"), + code="SELECTED_CONTEXT_CANONICAL_HASH_UNBOUND", + ) + if row.get("order_index") != index: + verification_errors.append("SELECTED_CONTEXT_ORDER_INVALID") + observed_raw_sha256: str | None = None + observed_canonical_sha256: str | None = None + pii_codes: list[str] = [] + if stage2_asset_root is None: + verification_errors.append("STAGE2_ASSET_ROOT_MISSING") + else: + try: + snapshot = open_bounded_snapshot( + stage2_asset_root, + relative_path, + logical_input_id=f"selected_context:{context_ref_id}", + ) + observed_raw_sha256 = snapshot.raw_sha256 + try: + text = snapshot.raw.decode("utf-8", errors="strict") + canonical_text = unicodedata.normalize( + "NFC", + text.replace("\r\n", "\n").replace("\r", "\n"), + ) + observed_canonical_sha256 = hashlib.sha256( + canonical_text.encode("utf-8") + ).hexdigest() + pii_codes = _scan_pii(canonical_text) + except UnicodeDecodeError: + verification_errors.append("SELECTED_CONTEXT_NOT_UTF8") + except IngressError as exc: + verification_errors.append( + "SELECTED_CONTEXT_ASSET_UNAVAILABLE" + ) + if observed_raw_sha256 != raw_sha256: + verification_errors.append("SELECTED_CONTEXT_ASSET_HASH_MISMATCH") + if observed_canonical_sha256 != canonical_sha256: + verification_errors.append( + "SELECTED_CONTEXT_CANONICAL_HASH_MISMATCH" + ) + if pii_codes: + verification_errors.append("SELECTED_CONTEXT_PII") + selected_context_refs.append( + { + "context_ref_id": context_ref_id, + "context_kind": context_kind, + "path": relative_path, + "raw_sha256": raw_sha256, + "canonical_sha256": canonical_sha256, + "order_index": index, + "release_ref": str(row.get("release_ref", release_ref)), + } + ) + selected_context_ordered_refs_sha256 = canonical_digest( + selected_context_refs + ) + declared_context_digest = normalize_sha256( + bundle.get("selected_context_ordered_refs_sha256"), + code="SELECTED_CONTEXT_ORDERED_REFS_HASH_UNBOUND", + ) + if declared_context_digest != selected_context_ordered_refs_sha256: + verification_errors.append( + "SELECTED_CONTEXT_ORDERED_REFS_HASH_MISMATCH" + ) + if mode != "STRUCTURAL_FIXTURE" and not selected_context_refs: + verification_errors.append("SELECTED_CONTEXT_EMPTY") + + wave_by_scc: dict[str, int] = {} + for wave_ordinal, wave in enumerate( + cluster_plan.get("scheduling_waves", []) + ): + for scc_id in wave: + wave_by_scc[str(scc_id)] = wave_ordinal + cluster_rows = { + str(row.get("cluster_id")): row + for row in cluster_plan.get("clusters", []) + if isinstance(row, dict) and row.get("cluster_id") is not None + } + + valid_member_slices: list[dict[str, Any]] = [] + missing_member_slices: list[dict[str, Any]] = [] + for cluster_id in cluster_plan.get("executable_cluster_ids", []): + cluster_id = str(cluster_id) + cluster_row = cluster_rows.get(cluster_id, {}) + scc_ids = [ + str(value) + for value in cluster_row.get("scc_ids", []) + ] if isinstance(cluster_row, dict) else [] + dependency_wave_ordinal = min( + (wave_by_scc.get(value, 0) for value in scc_ids), + default=0, + ) + slice_body = cluster_slices.get(cluster_id) + member_slice = { + "cluster_id": cluster_id, + "path": f"context/cluster_slices/{cluster_id}.json", + "sha256": ( + canonical_digest(slice_body) + if isinstance(slice_body, Mapping) + else "0" * 64 + ), + "dependency_wave_ordinal": dependency_wave_ordinal, + } + if isinstance(slice_body, Mapping): + valid_member_slices.append(member_slice) + else: + missing_member_slices.append(member_slice) + + def build_cohort( + member_slices: Sequence[Mapping[str, Any]], + *, + extra_reason_codes: Sequence[str] = (), + ) -> dict[str, Any]: + cohort_member_cluster_ids = sorted( + str(row["cluster_id"]) for row in member_slices + ) + public_member_slices = sorted( + (dict(row) for row in member_slices), + key=lambda row: row["cluster_id"], + ) + cohort_membership_sha256 = canonical_digest( + cohort_member_cluster_ids + ) + member_slice_ref_set_sha256 = canonical_digest( + public_member_slices + ) + receipt_errors = sorted( + set(verification_errors).union(str(code) for code in extra_reason_codes) + ) + cohort_mint = mint_stage2_id( + "bundle_cohort", + [ + handoff_contract_version, + mode, + s2_10_agent_sha256, + s2_10_llm_binding_sha256, + selected_context_ordered_refs_sha256, + cohort_membership_sha256, + member_slice_ref_set_sha256, + ], + prefix="BC", + ) + materialization_input = { + "handoff_contract_version": handoff_contract_version, + "bundle_cohort_id": cohort_mint["id"], + "s2_10_agent_sha256": s2_10_agent_sha256, + "s2_10_llm_binding_sha256": s2_10_llm_binding_sha256, + "selected_context_ordered_refs_sha256": ( + selected_context_ordered_refs_sha256 + ), + "cohort_membership_sha256": cohort_membership_sha256, + "member_slice_ref_set_sha256": member_slice_ref_set_sha256, + } + receipt = { + "schema_version": ( + "stage2_s2_00_context_materialization_receipt.v2" + ), + "bundle_cohort_id": cohort_mint["id"], + "s2_10_agent_sha256": s2_10_agent_sha256, + "s2_10_llm_binding_sha256": s2_10_llm_binding_sha256, + "selected_context_ordered_refs_sha256": ( + selected_context_ordered_refs_sha256 + ), + "cohort_membership_sha256": cohort_membership_sha256, + "member_slice_ref_set_sha256": member_slice_ref_set_sha256, + "materialization_input_sha256": canonical_digest( + materialization_input + ), + "verification_status": ( + "PASS" if not receipt_errors else "NON_EXECUTABLE" + ), + "reason_codes": receipt_errors, + } + reason_codes = list(receipt_errors) + if mode == "STRUCTURAL_FIXTURE": + reason_codes.append("STRUCTURAL_FIXTURE_NOT_EXECUTABLE") + cohort_status = "STRUCTURAL_ONLY" + elif receipt_errors: + cohort_status = "NON_EXECUTABLE" + else: + cohort_status = "EXECUTABLE" + reason_codes = sorted(set(reason_codes)) + source_refs = _dedupe_source_refs( + ref + for cluster_id in cohort_member_cluster_ids + for ref in ( + cluster_rows.get(cluster_id, {}).get("source_refs", []) + if isinstance(cluster_rows.get(cluster_id), dict) + else [] + ) + ) + return { + "handoff_contract_version": handoff_contract_version, + "bundle_cohort_id": cohort_mint["id"], + "compile_mode": mode, + "release_class": release_class, + "cohort_status": cohort_status, + "selected_context_refs": list(selected_context_refs), + "selected_context_ordered_refs_sha256": ( + selected_context_ordered_refs_sha256 + ), + "member_slices": public_member_slices, + "cohort_member_cluster_ids": cohort_member_cluster_ids, + "s2_10_agent_sha256": s2_10_agent_sha256, + "s2_10_llm_binding_sha256": s2_10_llm_binding_sha256, + "context_materialization_receipt": receipt, + "forbidden_bulk_inputs_present": False, + "reason_codes": reason_codes, + "source_refs": source_refs, + "derivation": { + "derivation_id": cohort_mint["id"], + "algorithm_version": ALGORITHM_VERSION, + "sorted_input_refs": sorted( + cohort_member_cluster_ids + + [ + f"context:{selected_context_ordered_refs_sha256}", + f"agent:{s2_10_agent_sha256}", + f"binding:{s2_10_llm_binding_sha256}", + ] + ), + "mint_input_sha256": cohort_mint["mint_input_sha256"], + }, + } + + cohorts: list[dict[str, Any]] = [] + if valid_member_slices: + cohorts.append(build_cohort(valid_member_slices)) + for member_slice in missing_member_slices: + cohorts.append( + build_cohort( + [member_slice], + extra_reason_codes=["CLUSTER_SLICE_MISSING"], + ) + ) + public_cohorts = sorted( + cohorts, + key=lambda row: ( + row["cohort_member_cluster_ids"][0] + if row["cohort_member_cluster_ids"] + else row["bundle_cohort_id"] + ), + ) + executable_cohort_ids = sorted( + row["bundle_cohort_id"] + for row in public_cohorts + if row["cohort_status"] == "EXECUTABLE" + ) + non_executable_cohort_ids = sorted( + row["bundle_cohort_id"] + for row in public_cohorts + if row["cohort_status"] != "EXECUTABLE" + ) + top_refs = ( + _dedupe_source_refs( + ref for row in public_cohorts for ref in row["source_refs"] + ) + if public_cohorts + else list(cluster_plan.get("source_refs", [])) + ) + return { + "artifact_header": dict( + cluster_plan.get("artifact_header", _artifact_header()) + ), + "handoff_contract_version": handoff_contract_version, + "cohorts": public_cohorts, + "executable_bundle_cohort_ids": executable_cohort_ids, + "non_executable_bundle_cohort_ids": non_executable_cohort_ids, + "source_refs": top_refs, + "derivation": _derivation( + "bundle_plan_derivation", + [ + row["bundle_cohort_id"] for row in public_cohorts + ] or ["NO_BUNDLE_COHORT"], + [ + selected_context_ordered_refs_sha256, + s2_10_agent_sha256, + s2_10_llm_binding_sha256, + { + row["bundle_cohort_id"]: row["reason_codes"] + for row in public_cohorts + }, + ], + ), + } + + def validate_bundle_release_cohorts(bundle_plan: Mapping[str, Any]) -> dict[str, Any]: + """Recompute every cross-object invariant the JSON Schema cannot express.""" + + cohorts = bundle_plan.get("cohorts", []) + if not isinstance(cohorts, list): + raise IngressError( + "BUNDLE_COHORT_INVARIANT_FAILED", + "bundle_plan.cohorts must be an array", + ) + mode = cohorts[0].get("compile_mode") if cohorts else None + release_class = cohorts[0].get("release_class") if cohorts else None + invariant_errors: list[str] = [] + receipt_reason_codes = { + "S2_10_AGENT_REF_MISSING", + "S2_10_AGENT_REF_PATH_INVALID", + "S2_10_AGENT_HASH_UNBOUND", + "S2_10_AGENT_ASSET_UNAVAILABLE", + "S2_10_AGENT_HASH_MISMATCH", + "S2_10_AGENT_RELEASE_HASH_MISMATCH", + "S2_10_LLM_BINDING_REF_MISSING", + "S2_10_LLM_BINDING_REF_PATH_INVALID", + "S2_10_LLM_BINDING_HASH_UNBOUND", + "S2_10_LLM_BINDING_ASSET_UNAVAILABLE", + "S2_10_LLM_BINDING_HASH_MISMATCH", + "S2_10_LLM_BINDING_RELEASE_HASH_MISMATCH", + "S2_10_HYBRID_RELEASE_UNBOUND", + "STAGE2_ASSET_ROOT_MISSING", + "CONTEXT_COHORT_POLICY_MISMATCH", + "SELECTED_CONTEXT_REF_ID_DUPLICATE", + "SELECTED_CONTEXT_ORDER_INVALID", + "SELECTED_CONTEXT_RAW_HASH_UNBOUND", + "SELECTED_CONTEXT_CANONICAL_HASH_UNBOUND", + "SELECTED_CONTEXT_ASSET_UNAVAILABLE", + "SELECTED_CONTEXT_ASSET_HASH_MISMATCH", + "SELECTED_CONTEXT_CANONICAL_HASH_MISMATCH", + "SELECTED_CONTEXT_NOT_UTF8", + "SELECTED_CONTEXT_PII", + "SELECTED_CONTEXT_ORDERED_REFS_HASH_UNBOUND", + "SELECTED_CONTEXT_ORDERED_REFS_HASH_MISMATCH", + "SELECTED_CONTEXT_EMPTY", + "CLUSTER_SLICE_MISSING", + } + cohort_reason_codes = receipt_reason_codes | { + "STRUCTURAL_FIXTURE_NOT_EXECUTABLE" + } + + for index, row in enumerate(cohorts): + if not isinstance(row, dict): + invariant_errors.append(f"COHORT_{index}_SHAPE") + continue + row_mode = row.get("compile_mode") + row_release_class = row.get("release_class") + if row_mode not in RELEASE_MODE or RELEASE_MODE[row_mode] != row_release_class: + invariant_errors.append(f"COHORT_{index}_MODE_RELEASE_CLASS") + if mode is not None and row_mode != mode: + invariant_errors.append(f"COHORT_{index}_MODE_MIXED") + if release_class is not None and row_release_class != release_class: + invariant_errors.append(f"COHORT_{index}_RELEASE_CLASS_MIXED") + + selected_refs = row.get("selected_context_refs", []) + if not isinstance(selected_refs, list): + invariant_errors.append(f"COHORT_{index}_SELECTED_CONTEXT_SHAPE") + selected_refs = [] + if [item.get("order_index") for item in selected_refs if isinstance(item, dict)] != list( + range(len(selected_refs)) + ): + invariant_errors.append(f"COHORT_{index}_SELECTED_CONTEXT_ORDER") + selected_digest = canonical_digest(selected_refs) + if row.get("selected_context_ordered_refs_sha256") != selected_digest: + invariant_errors.append(f"COHORT_{index}_SELECTED_CONTEXT_DIGEST") + + member_ids = row.get("cohort_member_cluster_ids", []) + member_slices = row.get("member_slices", []) + if not isinstance(member_ids, list) or member_ids != sorted(member_ids): + invariant_errors.append(f"COHORT_{index}_MEMBER_ID_ORDER") + member_ids = list(member_ids) if isinstance(member_ids, list) else [] + if not isinstance(member_slices, list) or member_slices != sorted( + member_slices, + key=lambda item: str(item.get("cluster_id", "")) if isinstance(item, dict) else "", + ): + invariant_errors.append(f"COHORT_{index}_MEMBER_SLICE_ORDER") + member_slices = list(member_slices) if isinstance(member_slices, list) else [] + slice_ids = [ + item.get("cluster_id") + for item in member_slices + if isinstance(item, dict) + ] + if member_ids != slice_ids: + invariant_errors.append(f"COHORT_{index}_MEMBERSHIP_SET") + membership_digest = canonical_digest(member_ids) + slice_digest = canonical_digest(member_slices) + + receipt = row.get("context_materialization_receipt") + if not isinstance(receipt, dict): + invariant_errors.append(f"COHORT_{index}_RECEIPT_SHAPE") + receipt = {} + echoed_fields = { + "bundle_cohort_id": row.get("bundle_cohort_id"), + "s2_10_agent_sha256": row.get("s2_10_agent_sha256"), + "s2_10_llm_binding_sha256": row.get("s2_10_llm_binding_sha256"), + "selected_context_ordered_refs_sha256": selected_digest, + "cohort_membership_sha256": membership_digest, + "member_slice_ref_set_sha256": slice_digest, + } + for key, expected in echoed_fields.items(): + if receipt.get(key) != expected: + invariant_errors.append(f"COHORT_{index}_RECEIPT_{key.upper()}") + materialization_input = { + "handoff_contract_version": row.get("handoff_contract_version"), + **echoed_fields, + } + if receipt.get("materialization_input_sha256") != canonical_digest( + materialization_input + ): + invariant_errors.append(f"COHORT_{index}_MATERIALIZATION_DIGEST") + + expected_cohort = mint_stage2_id( + "bundle_cohort", + [ + row.get("handoff_contract_version"), + row_mode, + row.get("s2_10_agent_sha256"), + row.get("s2_10_llm_binding_sha256"), + selected_digest, + membership_digest, + slice_digest, + ], + prefix="BC", + )["id"] + if row.get("bundle_cohort_id") != expected_cohort: + invariant_errors.append(f"COHORT_{index}_ID_MINT") + + receipt_codes = receipt.get("reason_codes", []) + row_codes = row.get("reason_codes", []) + if not isinstance(receipt_codes, list) or not set(receipt_codes).issubset( + receipt_reason_codes + ): + invariant_errors.append(f"COHORT_{index}_RECEIPT_REASON_CODE") + if not isinstance(row_codes, list) or not set(row_codes).issubset( + cohort_reason_codes + ): + invariant_errors.append(f"COHORT_{index}_REASON_CODE") + + status = row.get("cohort_status") + receipt_status = receipt.get("verification_status") + if row_mode == "STRUCTURAL_FIXTURE": + if status != "STRUCTURAL_ONLY" or "STRUCTURAL_FIXTURE_NOT_EXECUTABLE" not in row_codes: + invariant_errors.append(f"COHORT_{index}_STRUCTURAL_STATUS") + elif receipt_status == "PASS": + if status != "EXECUTABLE" or row_codes: + invariant_errors.append(f"COHORT_{index}_EXECUTABLE_STATUS") + elif receipt_status == "NON_EXECUTABLE": + if status != "NON_EXECUTABLE" or not row_codes: + invariant_errors.append(f"COHORT_{index}_NON_EXECUTABLE_STATUS") + else: + invariant_errors.append(f"COHORT_{index}_RECEIPT_STATUS") + + expected_executable_ids = sorted( + str(row.get("bundle_cohort_id")) + for row in cohorts + if isinstance(row, dict) and row.get("cohort_status") == "EXECUTABLE" + ) + expected_non_executable_ids = sorted( + str(row.get("bundle_cohort_id")) + for row in cohorts + if isinstance(row, dict) and row.get("cohort_status") != "EXECUTABLE" + ) + if bundle_plan.get("executable_bundle_cohort_ids") != expected_executable_ids: + invariant_errors.append("TOP_EXECUTABLE_COHORT_PARTITION") + if bundle_plan.get("non_executable_bundle_cohort_ids") != expected_non_executable_ids: + invariant_errors.append("TOP_NON_EXECUTABLE_COHORT_PARTITION") + if invariant_errors: + raise IngressError( + "BUNDLE_COHORT_INVARIANT_FAILED", + ",".join(sorted(set(invariant_errors))), + details={"invariant_errors": sorted(set(invariant_errors))}, + ) + + mapping_pass = mode in RELEASE_MODE and RELEASE_MODE[mode] == release_class + executable = sorted( + cluster_id + for row in cohorts + if row.get("cohort_status") == "EXECUTABLE" + for cluster_id in row.get("cohort_member_cluster_ids", []) + ) + non_executable = sorted( + cluster_id + for row in cohorts + if row.get("cohort_status") == "NON_EXECUTABLE" + for cluster_id in row.get("cohort_member_cluster_ids", []) + ) + structural = sorted( + cluster_id + for row in cohorts + if row.get("cohort_status") == "STRUCTURAL_ONLY" + for cluster_id in row.get("cohort_member_cluster_ids", []) + ) + return { + "mapping_pass": mapping_pass, + "executable_cluster_ids": executable, + "non_executable_cluster_ids": non_executable, + "structural_cluster_ids": structural, + "passed": mapping_pass and (mode == "STRUCTURAL_FIXTURE" or bool(executable)), + } + + + def _write_fsynced(path: Path, payload: bytes) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL + if hasattr(os, "O_CLOEXEC"): + flags |= os.O_CLOEXEC + descriptor = os.open(path, flags, 0o600) + try: + view = memoryview(payload) + while view: + written = os.write(descriptor, view) + view = view[written:] + os.fsync(descriptor) + finally: + os.close(descriptor) + + + def _fsync_directory(path: Path) -> None: + flags = os.O_RDONLY + if hasattr(os, "O_DIRECTORY"): + flags |= os.O_DIRECTORY + descriptor = os.open(path, flags) + try: + os.fsync(descriptor) + finally: + os.close(descriptor) + + + def _published_tree_matches( + output_dir: Path, + run_binding_digest: str, + *, + stage2_asset_root: str | os.PathLike[str] | None = None, + ) -> bool: + status_path = output_dir / "ingress" / "ingress_status.json" + if not status_path.is_file() or status_path.is_symlink(): + return False + try: + snapshot = open_bounded_snapshot(output_dir, "ingress/ingress_status.json", logical_input_id="published_status") + status_value = load_json_strict(snapshot) + except IngressError: + return False + if not isinstance(status_value, dict): + return False + binding = status_value.get("run_binding_receipt", {}) + if not isinstance(binding, dict) or binding.get("run_binding_digest") != run_binding_digest: + return False + barrier = status_value.get("output_barrier", {}) + if not isinstance(barrier, dict) or barrier.get("written_last") is not True: + return False + expected_artifacts = barrier.get("artifacts", []) + if not isinstance(expected_artifacts, list): + return False + if barrier.get("artifact_set_digest") != canonical_digest(expected_artifacts): + return False + try: + schema_documents = _load_output_schemas( + Path(stage2_asset_root) + if stage2_asset_root is not None + else Path(__file__).resolve().parents[1] + ) + _validate_output_artifact( + "ingress/ingress_status.json", + status_value, + schema_documents, + ) + except IngressError: + return False + for row in expected_artifacts: + if not isinstance(row, dict): + return False + try: + artifact = open_bounded_snapshot(output_dir, row["path"], logical_input_id="published_artifact") + except IngressError: + return False + if artifact.raw_sha256 != row.get("raw_sha256"): + return False + try: + parsed_artifact = load_json_strict(artifact) + _validate_output_artifact(row["path"], parsed_artifact, schema_documents) + except IngressError: + return False + expected_paths = { + row.get("path") + for row in expected_artifacts + if isinstance(row, dict) and isinstance(row.get("path"), str) + } | {"ingress/ingress_status.json"} + observed_paths = { + path.relative_to(output_dir).as_posix() + for path in output_dir.rglob("*") + if path.is_file() and not path.is_symlink() + } + if expected_paths != observed_paths: + return False + return True + + + def _output_schema_id(relative_path: str) -> str: + if relative_path.startswith("context/"): + return CONTEXT_SCHEMA_ID + if relative_path.startswith("review/"): + return REVIEW_SCHEMA_ID + if relative_path.startswith("ingress/"): + return INGRESS_SCHEMA_ID + raise IngressError("OUTPUT_SCHEMA_FAMILY_UNKNOWN", f"no schema family for {relative_path}") + + + def publish_atomically( + output_dir: str | os.PathLike[str], + artifacts: Mapping[str, Any], + *, + run_binding_digest: str, + attempt_id: str, + stage2_asset_root: str | os.PathLike[str] | None = None, + ) -> dict[str, Any]: + """Publish a complete normal or diagnostic tree with one directory rename.""" + + output = Path(output_dir) + if not output.is_absolute(): + output = output.resolve() + if not re.fullmatch(r"[A-Za-z0-9_.-]{1,128}", attempt_id): + raise IngressError("ATTEMPT_ID_INVALID", "attempt_id contains forbidden characters") + if "ingress/ingress_status.json" not in artifacts: + raise IngressError("OUTPUT_BARRIER_MISSING", "ingress_status must be supplied") + safe_paths = {_safe_relative_path(path).as_posix(): value for path, value in artifacts.items()} + if len(safe_paths) != len(artifacts): + raise IngressError("DUPLICATE_OUTPUT_PATH", "duplicate output paths after canonicalization") + schema_root = ( + Path(stage2_asset_root) + if stage2_asset_root is not None + else Path(__file__).resolve().parents[1] + ) + schema_documents = _load_output_schemas(schema_root) + for relative_path, supplied_value in safe_paths.items(): + value = load_json_strict(supplied_value) if isinstance(supplied_value, bytes) else supplied_value + _validate_output_artifact(relative_path, value, schema_documents) + if output.exists(): + if output.is_dir() and _published_tree_matches( + output, + run_binding_digest, + stage2_asset_root=schema_root, + ): + return {"status": "IDEMPOTENT_SUCCESS", "output_dir": str(output), "run_binding_digest": run_binding_digest} + raise IngressError("RUN_TUPLE_CONFLICT", "published output exists with a different binding") + parent = output.parent + parent.mkdir(parents=True, exist_ok=True) + staging_parent = parent / ".staging" + staging_parent.mkdir(parents=True, exist_ok=True) + staging = staging_parent / f"{output.name}.{attempt_id}" + if staging.exists(): + if staging.is_symlink() or staging.parent.resolve() != staging_parent.resolve(): + raise IngressError("STAGING_PATH_UNSAFE", "staging path is not safe") + shutil.rmtree(staging) + staging.mkdir(mode=0o700) + try: + artifact_receipt: list[dict[str, Any]] = [] + for relative_path in sorted(path for path in safe_paths if path != "ingress/ingress_status.json"): + value = safe_paths[relative_path] + payload = value if isinstance(value, bytes) else canonical_json_bytes(value) + _write_fsynced(staging / relative_path, payload) + artifact_receipt.append( + { + "logical_artifact_id": relative_path, + "path": relative_path, + "schema_id": _output_schema_id(relative_path), + "raw_sha256": hashlib.sha256(payload).hexdigest(), + } + ) + status = dict(safe_paths["ingress/ingress_status.json"]) + binding = status.get("run_binding_receipt") + if not isinstance(binding, dict) or binding.get("run_binding_digest") != run_binding_digest: + raise IngressError("RUN_BINDING_RECEIPT_MISMATCH", "status run binding does not match publish binding") + status["output_barrier"] = { + "barrier_id": "S2_00_INGRESS_STATUS_BARRIER", + "barrier_path": "ingress/ingress_status.json", + "publish_semantics": "STATUS_LAST_LOGICAL_COMMIT", + "canonical_output_root": f"stage2_runs/by-binding/{run_binding_digest}/", + "branch": "DIAGNOSTIC" if "ingress/technical_diagnostic.json" in safe_paths else "NORMAL", + "artifacts": artifact_receipt, + "artifact_set_digest": canonical_digest(artifact_receipt), + "non_status_artifacts_read_back_verified": True, + "written_last": True, + } + _validate_output_artifact( + "ingress/ingress_status.json", + status, + schema_documents, + ) + _write_fsynced(staging / "ingress/ingress_status.json", canonical_json_bytes(status)) + directories = sorted((path for path in staging.rglob("*") if path.is_dir()), key=lambda path: len(path.parts), reverse=True) + for directory in directories: + _fsync_directory(directory) + _fsync_directory(staging) + _fsync_directory(staging_parent) + os.rename(staging, output) + _fsync_directory(parent) + except BaseException: + if staging.exists() and staging.parent == staging_parent: + shutil.rmtree(staging) + raise + return {"status": "PUBLISHED", "output_dir": str(output), "run_binding_digest": run_binding_digest} + + + def _load_case_snapshots( + stage1_root: Path, + contracts: Sequence[Mapping[str, Any]], + release_lock: Mapping[str, Any], + ) -> tuple[dict[str, Snapshot], list[dict[str, Any]]]: + snapshots: dict[str, Snapshot] = {} + issues: list[dict[str, Any]] = [] + aggregate = 0 + max_file = int(release_lock.get("limits", {}).get("max_file_bytes", MAX_FILE_BYTES)) + max_run = int(release_lock.get("limits", {}).get("max_run_bytes", MAX_RUN_BYTES)) + for contract in contracts: + logical_id = str(contract["logical_input_id"]) + try: + snapshot = open_bounded_snapshot( + stage1_root, + str(contract["observed_path"]), + logical_input_id=logical_id, + max_bytes=max_file, + ) + except IngressError as exc: + if exc.code != "SOURCE_MISSING": + issues.append(_issue(exc.code, source_refs=[logical_id], message=str(exc))) + continue + aggregate += snapshot.byte_length + if aggregate > max_run: + raise IngressError("AGGREGATE_RUN_SIZE_LIMIT", "aggregate Stage 1 input budget exceeded") + snapshots[logical_id] = snapshot + return snapshots, issues + + + def _load_bound_completion_seal( + stage1_root: Path, + release_lock: Mapping[str, Any], + ) -> tuple[Mapping[str, Any] | None, Snapshot | None]: + dependency = release_lock.get("dependency_locks", {}).get("stage1", {}) + ref = dependency.get("completion_seal_ref", {}) if isinstance(dependency, dict) else {} + if not isinstance(ref, dict): + return None, None + path = ref.get("path") + digest = ref.get("sha256") + if path in {None, "PENDING_SEQUENTIAL_BIND"} or digest in {None, "PENDING_SEQUENTIAL_BIND"}: + return None, None + if not isinstance(path, str) or not isinstance(digest, str) or re.fullmatch(r"[A-Fa-f0-9]{64}", digest) is None: + raise IngressError("COMPLETION_SEAL_UNBOUND", "completion seal ref is not exactly bound") + snapshot = open_bounded_snapshot(stage1_root, path, logical_input_id="stage1_completion_seal") + if snapshot.raw_sha256 != digest.lower(): + raise IngressError("COMPLETION_SEAL_HASH_MISMATCH", "completion seal raw hash differs from release") + document = load_json_strict(snapshot) + if not isinstance(document, dict): + raise IngressError("COMPLETION_SEAL_SHAPE", "completion seal must be an object") + return document, snapshot + + + def _load_stage1_deployment_closure( + stage1_deployment_root: Path, + release_lock: Mapping[str, Any], + ) -> tuple[dict[str, Snapshot], dict[str, Any], list[dict[str, Any]]]: + """Load only the release-enumerated Stage 1 deployment closure.""" + + dependency = release_lock.get("dependency_locks", {}).get("stage1", {}) + locked_rows = dependency.get("concrete_paths", []) if isinstance(dependency, dict) else [] + if not isinstance(locked_rows, list) or not locked_rows: + raise IngressError("STAGE1_DEPENDENCY_LOCK_MISSING", "Stage 1 deployment closure is not enumerated") + expected_count = dependency.get("expected_concrete_path_count") + if expected_count is not None and expected_count != len(locked_rows): + raise IngressError("STAGE1_DEPENDENCY_COUNT_MISMATCH", "Stage 1 dependency row count is not sealed") + snapshots: dict[str, Snapshot] = {} + documents: dict[str, Any] = {} + source_rows: list[dict[str, Any]] = [] + seen_paths: set[str] = set() + aggregate = 0 + max_file = int(release_lock.get("limits", {}).get("max_file_bytes", MAX_FILE_BYTES)) + max_run = int(release_lock.get("limits", {}).get("max_run_bytes", MAX_RUN_BYTES)) + for index, locked in enumerate(locked_rows): + if not isinstance(locked, dict) or not isinstance(locked.get("path"), str): + raise IngressError("STAGE1_DEPENDENCY_ROW_SHAPE", f"invalid dependency row {index}") + path = _safe_relative_path(locked["path"]).as_posix() + if path in seen_paths: + raise IngressError("STAGE1_DEPENDENCY_DUPLICATE_PATH", f"duplicate dependency path: {path}") + seen_paths.add(path) + expected_hash = locked.get("sha256") + if not isinstance(expected_hash, str) or not re.fullmatch(r"[A-Fa-f0-9]{64}", expected_hash): + raise IngressError("STAGE1_DEPENDENCY_UNBOUND", f"dependency hash is not bound: {path}") + lock_id = str(locked.get("lock_id", f"S1-DEPLOY-{index + 1:03d}")) + snapshot = open_bounded_snapshot( + stage1_deployment_root, + path, + logical_input_id=f"deployment:{lock_id}", + max_bytes=max_file, + ) + aggregate += snapshot.byte_length + if aggregate > max_run: + raise IngressError("AGGREGATE_DEPLOYMENT_SIZE_LIMIT", "Stage 1 deployment closure exceeds byte budget") + if snapshot.raw_sha256 != expected_hash: + raise IngressError("STAGE1_DEPENDENCY_HASH_MISMATCH", f"deployment hash mismatch: {path}") + document = load_json_strict( + snapshot, + max_depth=int(release_lock.get("limits", {}).get("max_json_depth", MAX_JSON_DEPTH)), + max_items=int(release_lock.get("limits", {}).get("max_json_items", MAX_JSON_ITEMS)), + ) + snapshots[lock_id] = snapshot + documents[path] = document + schema_id = locked.get("schema_id") + source_rows.append( + { + "logical_input_id": f"deployment:{lock_id}", + "requirement_class": "UPSTREAM_DEPLOYMENT", + "expected_path": path, + "observed_path": path, + "resolution_source": "RELEASE_BOUND_CONTRACT_MANIFEST", + "schema_id": schema_id, + "schema_sha256": None, + "producer_id": None, + "producer_alias_id": None, + "adapter_id": "S2A-UPSTREAM-DEPLOYMENT-V1", + "raw_sha256": snapshot.raw_sha256, + "byte_length": snapshot.byte_length, + "run_identity_ref": {"value": None, "disposition": "NOT_APPLICABLE", "source_ref": path}, + "transaction_identity_ref": {"value": None, "disposition": "NOT_APPLICABLE", "source_ref": path}, + "parse_status": "PASS", + "schema_status": "UNEVALUABLE" if schema_id else "NOT_APPLICABLE", + "seal_status": "PASS", + "scope_technical_disposition": "AVAILABLE", + "impact_scope": "GLOBAL", + "scope_refs": [f"deployment:{lock_id}"], + "source_contract_row_refs": [f"deployment:{lock_id}"], + "reason_codes": [], + "downstream_allowed_actions": [], + "issue_codes": [], + } + ) + return snapshots, documents, source_rows + + + def _signal_source_contract_rows( + signal_all: Mapping[str, Any], + release_lock: Mapping[str, Any], + ) -> list[dict[str, Any]]: + rows: list[dict[str, Any]] = [] + transaction_id = str(signal_all.get("manifest_transaction_id", "MISSING")) + family_contract = next( + ( + row + for row in _release_stage1_source_rows(release_lock) + if row.get("logical_input_id") == "signal_payload_family" + ), + {}, + ) + for file_row in signal_all.get("ordered_file_rows", []): + logical_id = f"signal_file:{int(file_row['manifest_index']):03d}" + hash_status = str(file_row.get("hash_status", "UNEVALUABLE")) + issue_codes = ["SIGNAL_FILE_HASH_MISMATCH"] if hash_status == "FAIL" else [] + rows.append( + { + "logical_input_id": logical_id, + "requirement_class": "SIGNAL_PAYLOAD", + "expected_path": str(file_row["physical_path"]), + "observed_path": str(file_row["physical_path"]), + "resolution_source": "RELEASE_BOUND_CONTRACT_MANIFEST", + "schema_id": None, + "schema_sha256": None, + "producer_id": family_contract.get("producer_id"), + "producer_alias_id": family_contract.get("producer_alias_id"), + "adapter_id": family_contract.get("adapter_id", "S2A-SIGNAL-PAYLOAD-FAMILY-V1"), + "raw_sha256": str(file_row["raw_sha256"]), + "byte_length": int(file_row["byte_length"]), + "run_identity_ref": {"value": None, "disposition": "MISSING", "source_ref": logical_id}, + "transaction_identity_ref": { + "value": transaction_id, + "disposition": "OBSERVED", + "source_ref": "signal_manifest", + }, + "parse_status": "PASS", + "schema_status": "UNEVALUABLE", + "seal_status": hash_status, + "scope_technical_disposition": "AVAILABLE_WITH_ISSUES" if issue_codes else "AVAILABLE", + "impact_scope": "SIGNAL", + "scope_refs": [logical_id], + "source_contract_row_refs": [logical_id], + "reason_codes": issue_codes, + "downstream_allowed_actions": [], + "issue_codes": issue_codes, + } + ) + return rows + + + def _snapshot_state_digest(snapshots: Sequence[Snapshot]) -> str: + """Digest stable source content, not hydration-local filesystem identity.""" + + return canonical_digest( + sorted( + [ + item.logical_input_id, + item.relative_path, + item.raw_sha256, + item.byte_length, + ] + for item in snapshots + ) + ) + + + def _verify_snapshot_state(snapshots: Sequence[Snapshot]) -> str: + rows: list[list[Any]] = [] + for item in snapshots: + path = Path(item.resolved_path) + if path.is_symlink(): + raise IngressError("SOURCE_SNAPSHOT_CHANGED", f"source became a symlink: {item.relative_path}") + try: + observed = path.stat(follow_symlinks=False) + except FileNotFoundError as exc: + raise IngressError("SOURCE_SNAPSHOT_CHANGED", f"source disappeared: {item.relative_path}") from exc + identity = (observed.st_dev, observed.st_ino, observed.st_size, observed.st_mtime_ns) + expected = (item.device, item.inode, item.byte_length, item.mtime_ns) + if identity != expected: + raise IngressError("SOURCE_SNAPSHOT_CHANGED", f"source changed: {item.relative_path}") + rows.append( + [ + item.logical_input_id, + item.relative_path, + item.raw_sha256, + item.byte_length, + ] + ) + return canonical_digest(sorted(rows)) + + + def _cluster_inputs(documents: Mapping[str, Any]) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: + ledger_rows = _array_rows(documents.get("fact_ledger_base"), ("facts", "fact_ledger", "rows", "items")) + bo_rows = _array_rows(documents.get("bo"), ("business_objects", "BO", "rows", "items")) + les_rows = _array_rows( + documents.get("legal_effect_structures"), + ("structures", "structure_records", "legal_effect_structures", "rows", "items"), + ) + evidence_rows = _array_rows(documents.get("evidence_indexed"), ("evidence", "evidence_items", "rows", "items")) + event_rows = _array_rows(documents.get("evidence_event_candidates"), ("events", "event_candidates", "rows", "items")) + members: list[dict[str, Any]] = [] + relations: list[dict[str, Any]] = [] + internal_by_kind_ref: dict[tuple[str, str], str] = {} + + def add_member( + kind: str, + member_ref: str, + logical_id: str, + pointer: str, + raw_value: Any, + ) -> str: + key = (kind, member_ref) + prior = internal_by_kind_ref.get(key) + if prior is not None: + return prior + internal_id = f"{kind}:{member_ref}" + internal_by_kind_ref[key] = internal_id + members.append( + { + "member_id": internal_id, + "member_ref": member_ref, + "member_kind": kind, + "source_refs": [_source_ref(logical_id, pointer, raw_value, stage1_id=member_ref)], + "scope_technical_disposition": "AVAILABLE", + } + ) + return internal_id + + bo_member_by_id: dict[str, str] = {} + for index, row in enumerate(bo_rows): + if not isinstance(row, dict): + continue + bo_id = str(row.get("BO_ID", f"BO-OCCURRENCE-{index}")) + bo_member_by_id[bo_id] = add_member("BO", bo_id, "bo", f"/business_objects/{index}", row) + + fact_member_by_id: dict[str, str] = {} + fact_members_by_bo: dict[str, list[str]] = defaultdict(list) + fact_members_by_evidence: dict[str, list[str]] = defaultdict(list) + fact_members_by_event: dict[str, list[str]] = defaultdict(list) + for index, row in enumerate(ledger_rows): + if not isinstance(row, dict): + continue + fact_id = str(row.get("fact_id", f"F-OCCURRENCE-{index}")) + member_id = add_member("FACT", fact_id, "fact_ledger_base", f"/facts/{index}", row) + fact_member_by_id[fact_id] = member_id + source_bo_id = row.get("source_bo_id") + if source_bo_id is not None: + fact_members_by_bo[str(source_bo_id)].append(member_id) + evidence_refs = row.get("evidence_refs", row.get("evidence_ids", [])) + if isinstance(evidence_refs, list): + for ref in evidence_refs: + fact_members_by_evidence[str(ref)].append(member_id) + event_refs = row.get("event_refs", row.get("event_ids", [])) + if isinstance(event_refs, list): + for ref in event_refs: + fact_members_by_event[str(ref)].append(member_id) + relation_arrays = [ + row.get(key) + for key in ("relations", "explicit_relations", "candidate_relations") + if isinstance(row.get(key), list) + ] + for relation_array in relation_arrays: + for relation_index, relation in enumerate(relation_array): + if not isinstance(relation, dict): + continue + target_ref = relation.get( + "target_fact_id", + relation.get("to_fact_id", relation.get("target_member_id")), + ) + relation_kind = str(relation.get("relation_kind", relation.get("kind", ""))) + if target_ref is None or relation_kind not in CANDIDATE_RELATION_KINDS | {"EXPLICIT_CASE_RELATION"}: + continue + relations.append( + { + "_deferred_source_fact_id": fact_id, + "_deferred_target_fact_id": str(target_ref), + "relation_kind": relation_kind, + "source_refs": [ + _source_ref( + "fact_ledger_base", + f"/facts/{index}/relations/{relation_index}", + relation, + ) + ], + } + ) + + for bo_id, fact_member_ids in sorted(fact_members_by_bo.items()): + bo_member = bo_member_by_id.get(bo_id) + if bo_member is None: + continue + for fact_member in sorted(set(fact_member_ids)): + relations.append( + { + "source_member_id": fact_member, + "target_member_id": bo_member, + "relation_kind": "SAME_BO_ID", + "source_refs": [_source_ref("bo", "", bo_id, stage1_id=bo_id)], + } + ) + + for index, row in enumerate(les_rows): + if not isinstance(row, dict): + continue + structure_ref = str(row.get("structure_id", row.get("legal_effect_structure_id", f"LES-OCCURRENCE-{index}"))) + les_member = add_member( + "LES_STRUCTURE", + structure_ref, + "legal_effect_structures", + f"/structures/{index}", + row, + ) + source_bo_ids = row.get("source_bo_ids", []) + if isinstance(source_bo_ids, list): + for bo_id in source_bo_ids: + bo_member = bo_member_by_id.get(str(bo_id)) + if bo_member is not None: + relations.append( + { + "source_member_id": les_member, + "target_member_id": bo_member, + "relation_kind": "SOURCE_BO_ATTACHMENT", + "source_refs": [ + _source_ref( + "legal_effect_structures", + f"/structures/{index}/source_bo_ids", + source_bo_ids, + ) + ], + } + ) + + for index, row in enumerate(evidence_rows): + if not isinstance(row, dict): + continue + evidence_id = str(row.get("evidence_id", row.get("id", f"EVIDENCE-OCCURRENCE-{index}"))) + evidence_member = add_member("EVIDENCE", evidence_id, "evidence_indexed", f"/items/{index}", row) + for fact_member in sorted(set(fact_members_by_evidence.get(evidence_id, []))): + relations.append( + { + "source_member_id": fact_member, + "target_member_id": evidence_member, + "relation_kind": "SAME_EVIDENCE_REF", + "source_refs": [_source_ref("fact_ledger_base", "", evidence_id, stage1_id=evidence_id)], + } + ) + + for index, row in enumerate(event_rows): + if not isinstance(row, dict): + continue + event_id = str(row.get("event_id", row.get("id", f"EVENT-OCCURRENCE-{index}"))) + event_member = add_member("EVENT", event_id, "evidence_event_candidates", f"/items/{index}", row) + for fact_member in sorted(set(fact_members_by_event.get(event_id, []))): + relations.append( + { + "source_member_id": fact_member, + "target_member_id": event_member, + "relation_kind": "SAME_EVENT_REF", + "source_refs": [_source_ref("fact_ledger_base", "", event_id, stage1_id=event_id)], + } + ) + + resolved_relations: list[dict[str, Any]] = [] + for relation in relations: + if "_deferred_source_fact_id" not in relation: + resolved_relations.append(relation) + continue + source_member = fact_member_by_id.get(str(relation["_deferred_source_fact_id"])) + target_member = fact_member_by_id.get(str(relation["_deferred_target_fact_id"])) + if source_member is None or target_member is None: + continue + resolved_relations.append( + { + "source_member_id": source_member, + "target_member_id": target_member, + "relation_kind": relation["relation_kind"], + "source_refs": relation["source_refs"], + } + ) + return members, resolved_relations + + + def _run_binding_digest( + snapshots: Mapping[str, Snapshot], + release_lock: Mapping[str, Any], + ) -> str: + input_set = [[key, snapshots[key].raw_sha256] for key in sorted(snapshots)] + binding = { + "input_set_digest": canonical_digest(input_set), + "stage2_release_digest": release_lock.get("_release_raw_sha256", release_lock.get("stage2_release_digest")), + "algorithm_digest": ALGORITHM_SEMANTIC_DIGEST, + "release_class": release_lock.get("release_class"), + } + return canonical_digest(binding) + + + def _input_set_digest( + snapshots: Mapping[str, Snapshot], + additional_snapshots: Sequence[Snapshot] = (), + ) -> str: + rows = [ + ["run", key, snapshots[key].relative_path, snapshots[key].raw_sha256] + for key in sorted(snapshots) + ] + rows.extend( + ["closure", item.logical_input_id, item.relative_path, item.raw_sha256] + for item in additional_snapshots + ) + return canonical_digest(sorted(rows, key=canonical_digest)) + + + def _make_run_binding_receipt( + snapshots: Mapping[str, Snapshot], + release_lock: Mapping[str, Any], + *, + request_id: str, + user_context_sha256: str, + workspace_context_sha256: str, + additional_snapshots: Sequence[Snapshot] = (), + ) -> dict[str, Any]: + input_digest = _input_set_digest(snapshots, additional_snapshots) + release_digest = str( + release_lock.get( + "_release_raw_sha256", + release_lock.get( + "release_digest", + release_lock.get("stage2_release_digest", "0" * 64), + ), + ) + ) + algorithm_digest = ALGORITHM_SEMANTIC_DIGEST + binding_digest = canonical_digest( + { + "input_set_digest": input_digest, + "stage2_release_digest": release_digest, + "algorithm_digest": algorithm_digest, + "release_class": release_lock.get("release_class"), + } + ) + return { + "schema_version": "stage2_s2_00_run_binding_receipt.v1.1", + "request_id": request_id, + "run_id": f"S2RUN-{binding_digest}", + "input_set_digest": input_digest, + "stage2_release_digest": release_digest, + "algorithm_digest": algorithm_digest, + "release_class": release_lock.get("release_class"), + "run_binding_digest": binding_digest, + "canonical_output_root": f"stage2_runs/by-binding/{binding_digest}/", + "run_identity_derivation": "RUN_ID_PREFIXED_FROM_RUN_BINDING_DIGEST", + "output_root_derivation": "stage2_runs/by-binding//", + "user_context_sha256": user_context_sha256, + "workspace_context_sha256": workspace_context_sha256, + } + + + def canonical_run_id(run_binding_digest: str) -> str: + """Derive the immutable run identifier from the complete binding digest.""" + + if re.fullmatch(r"[a-f0-9]{64}", run_binding_digest) is None: + raise IngressError( + "RUN_BINDING_DIGEST_INVALID", + "canonical run id requires one lowercase SHA-256 digest", + ) + return f"S2RUN-{run_binding_digest}" + + + def _compact_conservation_checks( + checks: Sequence[Mapping[str, Any]], + ) -> list[dict[str, Any]]: + compact: list[dict[str, Any]] = [] + for check in checks: + check_id = str(check.get("check_id", "UNNAMED_CONSERVATION_CHECK")) + status_value = str(check.get("status", "UNEVALUABLE")) + status = status_value if status_value in {"PASS", "FAIL", "UNEVALUABLE"} else "UNEVALUABLE" + observed_payload = {key: value for key, value in check.items() if key not in {"status"}} + left_digest = canonical_digest(observed_payload) if status != "UNEVALUABLE" else None + right_digest = left_digest if status == "PASS" else canonical_digest([check_id, "EXPECTED"]) if status == "FAIL" else None + left_count = next( + ( + int(check[key]) + for key in ("left_count", "bo_count", "source_bo_ref_count") + if isinstance(check.get(key), int) + ), + None, + ) + right_count = next( + ( + int(check[key]) + for key in ("right_count", "source_bo_ref_count", "bo_count") + if isinstance(check.get(key), int) + ), + None, + ) + compact.append( + { + "check_id": check_id, + "status": status, + "left_counter_digest": left_digest, + "right_counter_digest": right_digest, + "left_count": left_count, + "right_count": right_count, + "source_refs": [check_id], + "issue_codes": [f"{check_id}_FAILED"] if status == "FAIL" else [], + } + ) + return compact + + + def _compact_review_receipt( + normalized_reviews: Mapping[str, Any], + ) -> dict[str, Any]: + normalized = normalized_reviews.get("normalized_occurrences", []) + raw = normalized_reviews.get("raw_occurrences", []) + counts = dict(normalized_reviews.get("partition_counts", {})) + for key in ("SUPPORTED", "CONDITIONAL", "UNRESOLVED", "EXCLUDED", "UNMAPPED"): + counts.setdefault(key, 0) + return { + "schema_version": "stage2_review_normalization_receipt.v1", + "adapter_ids": list(normalized_reviews.get("adapter_ids", [])), + "raw_occurrence_count": len(raw), + "normalized_occurrence_count": len(normalized), + "partition_counts": counts, + "unmapped_occurrence_count": counts["UNMAPPED"], + "conservation_status": str(normalized_reviews.get("conservation_status", "FAIL")), + "ordered_occurrence_refs": [ + str(row.get("review_key", {}).get("value")) + for row in normalized + if row.get("review_key", {}).get("value") + ], + } + + + def _build_issue_ledger( + issues: Sequence[Mapping[str, Any]], + normalized_reviews: Mapping[str, Any], + *, + run_id: str, + input_set_digest: str, + ) -> dict[str, Any]: + issue_rows: list[dict[str, Any]] = [] + seen: set[str] = set() + for index, issue in enumerate(issues): + issue_code = str(issue.get("issue_code", "UNSPECIFIED_ISSUE")) + source_refs = sorted(set(str(item) for item in issue.get("scope_refs", []))) or ["S2_00"] + issue_mint = mint_stage2_id( + "issue", + [issue_code, source_refs, index], + prefix="ISS", + ) + if issue_mint["id"] in seen: + continue + seen.add(issue_mint["id"]) + raw_severity = str(issue.get("severity", "ERROR")).upper() + technical_severity = { + "INFO": "INFO", + "REVIEW": "REVIEW", + "WARNING": "HARD_WARNING", + "HARD_WARNING": "HARD_WARNING", + "ERROR": "BLOCK", + "BLOCK": "BLOCK", + }.get(raw_severity, "REVIEW") + issue_rows.append( + { + "issue_id": issue_mint["id"], + "issue_code": issue_code, + "technical_severity": technical_severity, + "review_partition": "UNRESOLVED", + "impact_scope": str(issue.get("impact_scope", "GLOBAL")), + "scope_refs": source_refs, + "source_contract_row_refs": sorted( + set(str(item) for item in issue.get("source_contract_row_refs", source_refs)) + ), + "reason_codes": sorted(set(str(item) for item in issue.get("reason_codes", [issue_code]))), + "downstream_allowed_actions": sorted( + set(str(item) for item in issue.get("downstream_allowed_actions", [])) + ), + "source_refs": source_refs, + "status": "OPEN", + "upstream_review_key": None, + "raw_item_sha256": None, + } + ) + for occurrence in normalized_reviews.get("normalized_occurrences", []): + if occurrence.get("partition") != "UNMAPPED": + continue + review_key = occurrence["review_key"]["value"] + issue_mint = mint_stage2_id( + "issue", + ["UNMAPPED_REVIEW_STATUS", review_key], + prefix="ISS", + ) + issue_rows.append( + { + "issue_id": issue_mint["id"], + "issue_code": "UNMAPPED_REVIEW_STATUS", + "technical_severity": "REVIEW", + "review_partition": "UNMAPPED", + "impact_scope": "REVIEW_ITEM", + "scope_refs": [review_key], + "source_contract_row_refs": occurrence["source_contract_row_refs"], + "reason_codes": ["UNMAPPED_REVIEW_STATUS"], + "downstream_allowed_actions": occurrence["downstream_allowed_actions"], + "source_refs": occurrence["source_contract_row_refs"], + "status": "OPEN", + "upstream_review_key": review_key, + "raw_item_sha256": occurrence["source_item_raw_sha256"], + } + ) + return { + "schema_version": "stage2_issue_ledger_base.v1", + "producer_id": "S2_00", + "run_id": run_id, + "input_set_digest": input_set_digest, + "mapping_table_version": "S2-REVIEW-MAP-V1", + "issues": sorted(issue_rows, key=lambda row: row["issue_id"]), + "raw_review_occurrence_count": len(normalized_reviews.get("raw_occurrences", [])), + "normalized_review_occurrence_count": len(normalized_reviews.get("normalized_occurrences", [])), + "review_conservation_status": str(normalized_reviews.get("conservation_status", "FAIL")), + } + + + def _route( + source_rows: Sequence[Mapping[str, Any]], + cluster_plan: Mapping[str, Any], + cohort_receipt: Mapping[str, Any], + issues: Sequence[Mapping[str, Any]], + ) -> str: + identity_unavailable = any( + row.get("requirement_class") == "IDENTITY_BACKBONE" + and row.get("scope_technical_disposition") == "UNAVAILABLE" + for row in source_rows + ) + diagnostic_issue_codes = { + "RELEASE_SOURCE_CONTRACT_MISSING", + "RELEASE_SOURCE_PATH_MISMATCH", + "RAW_HASH_MISMATCH", + "SCHEMA_HASH_MISMATCH", + "SCHEMA_ID_MISMATCH", + "RUN_IDENTITY_CONFLICT", + "TRANSACTION_IDENTITY_CONFLICT", + "BO_FACT_CONSERVATION_FAILED", + "FACT_ID_CONSERVATION_FAILED", + "CURRENT_V8_LEDGER_EXTENSION_MISSING", + } + observed_issue_codes = {str(item.get("issue_code")) for item in issues} + if identity_unavailable or observed_issue_codes & diagnostic_issue_codes or not cluster_plan.get("clusters"): + return "TO_S2_40_STATUS_ONLY" + if not cohort_receipt.get("executable_cluster_ids"): + return "TO_S2_40_STATUS_ONLY" + return "TO_S2_10_WITH_ISSUES" if issues else "TO_S2_10" + + + def execute_ingress( + stage1_run_root: str | os.PathLike[str], + release_lock: Mapping[str, Any], + *, + output_dir: str | os.PathLike[str] | None, + attempt_id: str, + run_id: str = "S2-00-REQUEST", + user_context_sha256: str = "0" * 64, + workspace_context_sha256: str = "0" * 64, + stage1_deployment_root: str | os.PathLike[str] | None = None, + stage2_asset_root: str | os.PathLike[str] | None = None, + contract_manifest: Mapping[str, Any] | None = None, + hydration_stability_receipt: Mapping[str, Any] | None = None, + ) -> dict[str, Any]: + """Execute C00→C05→C10→C15 for a canary or production case run.""" + + if release_lock.get("release_class") == "DEV_FIXTURE_RELEASE": + raise IngressError("DEV_FIXTURE_REAL_RUN_FORBIDDEN", "DEV fixture release cannot publish a case run") + if output_dir is None: + raise IngressError("OUTPUT_DIR_REQUIRED", "output_dir is required for canary/production") + if stage1_deployment_root is None: + raise IngressError("STAGE1_DEPLOYMENT_ROOT_REQUIRED", "Stage 1 deployment root is required") + if not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9._-]{0,127}", run_id): + raise IngressError("REQUEST_ID_INVALID", "request_id contains forbidden characters") + if not isinstance(hydration_stability_receipt, dict): + raise IngressError( + "HYDRATION_STABILITY_RECEIPT_REQUIRED", + "canary and production ingress require the remote two-pass hydration receipt", + ) + for label, value in ( + ("user_context_sha256", user_context_sha256), + ("workspace_context_sha256", workspace_context_sha256), + ): + if not re.fullmatch(r"[A-Fa-f0-9]{64}", value): + raise IngressError("RUN_CONTEXT_HASH_INVALID", f"{label} must be a SHA-256 digest") + root_arg = Path(stage1_run_root) + deployment_root_arg = Path(stage1_deployment_root) + asset_root_arg = Path(stage2_asset_root) if stage2_asset_root is not None else Path(__file__).resolve().parents[1] + for candidate, code in ( + (root_arg, "SYMLINK_ROOT_REJECTED"), + (deployment_root_arg, "SYMLINK_DEPLOYMENT_ROOT_REJECTED"), + (asset_root_arg, "SYMLINK_ASSET_ROOT_REJECTED"), + ): + if candidate.is_symlink(): + raise IngressError(code, "approved root itself may not be a symlink") + root = root_arg.resolve(strict=True) + deployment_root = deployment_root_arg.resolve(strict=True) + asset_root = ( + asset_root_arg.resolve(strict=True) + if stage2_asset_root is not None + else Path(__file__).resolve().parents[1] + ) + deployment_snapshots, deployment_documents, deployment_rows = _load_stage1_deployment_closure( + deployment_root, + release_lock, + ) + contracts = resolve_stage1_sources(root, contract_manifest) + snapshots, snapshot_issues = _load_case_snapshots(root, contracts, release_lock) + completion_seal, completion_seal_snapshot = _load_bound_completion_seal(root, release_lock) + ingress = validate_ingress_contracts( + snapshots, + contracts, + release_lock, + deployment_snapshots=deployment_snapshots, + deployment_documents=deployment_documents, + completion_seal=completion_seal, + contract_manifest=contract_manifest, + ) + documents = ingress["documents"] + issues = list(snapshot_issues) + list(ingress["issues"]) + signal_all: dict[str, Any] | None = None + if isinstance(documents.get("signal_manifest"), dict): + try: + max_run = int(release_lock.get("limits", {}).get("max_run_bytes", MAX_RUN_BYTES)) + already_loaded = sum(snapshot.byte_length for snapshot in snapshots.values()) + sum( + snapshot.byte_length for snapshot in deployment_snapshots.values() + ) + remaining = max_run - already_loaded + if remaining < 0: + raise IngressError("AGGREGATE_RUN_SIZE_LIMIT", "case and deployment closure exceed aggregate run budget") + signal_all = expand_stage2_signal_all( + root, + documents["signal_manifest"], + max_file_bytes=int(release_lock.get("limits", {}).get("max_file_bytes", MAX_FILE_BYTES)), + max_total_bytes=remaining, + signal_registry=deployment_documents.get("signals/signal_registry.v2.json"), + ) + signal_all = bind_signal_occurrences(signal_all, documents) + issues.extend(signal_all.get("issues", [])) + except IngressError as exc: + issues.append(_issue(exc.code, impact_scope="SIGNAL", source_refs=["signal_manifest"], message=str(exc))) + routing_activation = documents.get("domain_activation_manifest") + if isinstance(routing_activation, dict): + try: + parsed_signal_documents = signal_all.get("_parsed_documents_by_path", {}) if signal_all else {} + signal_activation = parsed_signal_documents.get("domain_activation_manifest.json") + if signal_activation is None: + raise IngressError( + "SG01_SIGNAL_ARTIFACT_MISSING", + "signal ALL does not contain domain_activation_manifest.json", + ) + signal_activation_row = next( + ( + row + for row in signal_all.get("ordered_file_rows", []) + if row.get("file_path") == "domain_activation_manifest.json" + ), + {}, + ) + verify_activation_projection( + routing_activation, + signal_activation, + routing_raw_sha256=snapshots.get("domain_activation_manifest").raw_sha256 if snapshots.get("domain_activation_manifest") else None, + signal_raw_sha256=signal_activation_row.get("raw_sha256"), + ) + except IngressError as exc: + issues.append(_issue(exc.code, impact_scope="SIGNAL", source_refs=["domain_activation_manifest", "signal_sg01_activation"], message=str(exc))) + seal_deployment_snapshots = { + "stage1_domain_registry_index": snapshot + for lock_id, snapshot in deployment_snapshots.items() + if snapshot.relative_path == "domains/_registry_index.json" + } + seals = verify_cross_artifact_seals(documents, snapshots, seal_deployment_snapshots) + issues.extend(seals["issues"]) + review_docs = {key: value for key, value in documents.items() if "review_handoff" in key or "soft_gate_handoff" in key} + normalized_reviews = normalize_review_items(review_docs, release_lock) + issues.extend(normalized_reviews.get("_issues", [])) + conservation = check_conservation( + documents, + signal_all=signal_all, + normalized_reviews=normalized_reviews, + source_snapshots=snapshots, + ) + issues.extend(conservation["issues"]) + signal_snapshots = list(signal_all.get("_payload_snapshots", [])) if signal_all else [] + closure_snapshots = list(deployment_snapshots.values()) + signal_snapshots + all_snapshots = list(snapshots.values()) + closure_snapshots + if completion_seal_snapshot is not None: + all_snapshots.append(completion_seal_snapshot) + snapshot_start_digest = _snapshot_state_digest(all_snapshots) + input_set_digest = _input_set_digest(snapshots, closure_snapshots) + run_binding_receipt = _make_run_binding_receipt( + snapshots, + release_lock, + request_id=run_id, + user_context_sha256=user_context_sha256, + workspace_context_sha256=workspace_context_sha256, + additional_snapshots=closure_snapshots, + ) + run_binding_digest = run_binding_receipt["run_binding_digest"] + run_id = canonical_run_id(run_binding_digest) + context_schema_snapshot = open_bounded_snapshot( + asset_root, + "schemas/context.schema.json", + logical_input_id="stage2_context_schema", + ) + artifact_header = _artifact_header( + schema_sha256=context_schema_snapshot.raw_sha256, + run_id=run_id, + input_set_digest=input_set_digest, + stage2_release_digest=run_binding_receipt["stage2_release_digest"], + algorithm_digest=run_binding_receipt["algorithm_digest"], + release_class=str(release_lock["release_class"]), + ) + domain_configs = { + match.group(1): value + for path, value in deployment_documents.items() + if (match := re.fullmatch(r"domains/([^/]+)/domain_config\.json", path)) is not None + } + routing_payload = ( + _activation_payload(documents["domain_activation_manifest"]) + if isinstance(documents.get("domain_activation_manifest"), dict) + else {} + ) + active_domain_ids = routing_payload.get("active_domain_ids", []) + if isinstance(active_domain_ids, list): + for domain_id in sorted(set(str(value) for value in active_domain_ids)): + if domain_id not in domain_configs: + issues.append( + _issue( + "ACTIVE_DOMAIN_CONFIG_MISSING", + impact_scope="CLUSTER", + source_refs=[f"domain_config:{domain_id}"], + ) + ) + case_context = build_case_context( + documents, + run_binding_digest=run_binding_digest, + issues=issues, + artifact_header=artifact_header, + signal_all=signal_all, + ) + evidence_inventory = build_evidence_inventory( + documents, + run_binding_digest=run_binding_digest, + artifact_header=artifact_header, + ) + object_registry = build_object_registry( + documents, + run_binding_digest=run_binding_digest, + artifact_header=artifact_header, + ) + party_context = build_party_and_title_context( + documents, + run_binding_digest=run_binding_digest, + artifact_header=artifact_header, + ) + slot_crosswalk = build_slot_crosswalk( + documents, + domain_configs, + run_binding_digest=run_binding_digest, + artifact_header=artifact_header, + ) + members, relations = _cluster_inputs(documents) + cluster_plan = compile_cluster_plan(members, relations, artifact_header=artifact_header) + contexts = { + "case_context": case_context, + "evidence_inventory": evidence_inventory, + "object_registry": object_registry, + "party_and_title_context": party_context, + "slot_crosswalk": slot_crosswalk, + "_projection_sources": { + "EVENT": { + str(row.get("event_id", row.get("id"))): row + for row in _array_rows(documents.get("evidence_event_candidates"), ("events", "event_candidates", "rows", "items")) + if isinstance(row, dict) and (row.get("event_id") is not None or row.get("id") is not None) + }, + "LES_STRUCTURE": { + str(row.get("structure_id", row.get("legal_effect_structure_id"))): row + for row in _array_rows(documents.get("legal_effect_structures"), ("structures", "structure_records", "rows", "items")) + if isinstance(row, dict) and (row.get("structure_id") is not None or row.get("legal_effect_structure_id") is not None) + }, + "SIGNAL_OCCURRENCE": { + str(row.get("occurrence_ref")): row + for row in (signal_all or {}).get("record_occurrences", []) + if row.get("occurrence_ref") is not None + }, + "REVIEW_ITEM": { + str(row.get("review_key", {}).get("value")): row + for row in normalized_reviews.get("normalized_occurrences", []) + if row.get("review_key", {}).get("value") is not None + }, + "ACTIVE_PROFILE": { + str(domain_id): config + for domain_id, config in domain_configs.items() + if domain_id in set(str(value) for value in active_domain_ids) + }, + }, + } + slices = compile_cluster_slices(cluster_plan, contexts) + effective_release_lock = dict(release_lock) + if not isinstance(effective_release_lock.get("bundle"), dict): + issues.append(_issue("BUNDLE_RELEASE_CONTRACT_MISSING", impact_scope="GLOBAL", source_refs=["stage2_release"])) + effective_release_lock["bundle"] = { + "mode": { + "DEV_FIXTURE_RELEASE": "STRUCTURAL_FIXTURE", + "SUBSET_CANARY_RELEASE": "SUBSET_CANARY", + "PRODUCTION_RELEASE": "PRODUCTION", + }.get(str(release_lock.get("release_class")), "STRUCTURAL_FIXTURE"), + "handoff_contract_version": "S2_00_STRUCTURED_CONTEXT_HANDOFF_V2", + "context_cohort_policy_id": "S2-CACHE-STRUCTURED-CONTEXT-HANDOFF-V2", + "s2_10_agent_ref": { + "path": "agent_scripts/Stage_2_S2_10.yml", + "sha256": "0" * 64, + }, + "s2_10_llm_binding_ref": { + "path": "deployment/stage2_s2_10_llm_binding.yml", + "sha256": "0" * 64, + }, + "s2_10_agent_sha256": "0" * 64, + "s2_10_llm_binding_sha256": "0" * 64, + "selected_context_refs": [], + "selected_context_ordered_refs_sha256": canonical_digest([]), + "downstream_hybrid_status": "PENDING_REIMPLEMENTATION_AND_RESEAL", + } + bundle_plan = compile_bundle_plan( + cluster_plan, + slices, + effective_release_lock, + stage2_asset_root=asset_root, + ) + cohort_by_cluster = { + cluster_id: row + for row in bundle_plan.get("cohorts", []) + for cluster_id in row.get("cohort_member_cluster_ids", []) + } + for cluster in cluster_plan.get("clusters", []): + cohort = cohort_by_cluster.get(cluster["cluster_id"]) + if cohort is not None: + cluster["bundle_cohort_id"] = cohort["bundle_cohort_id"] + if cohort["cohort_status"] != "EXECUTABLE" and cluster["cluster_status"] == "EXECUTABLE": + cluster["cluster_status"] = "NON_EXECUTABLE" + for code in cohort.get("reason_codes", []): + issues.append( + _issue( + str(code), + impact_scope="CLUSTER", + source_refs=[cluster["cluster_id"], cohort["bundle_cohort_id"]], + ) + ) + cohort_receipt = validate_bundle_release_cohorts(bundle_plan) + final_executable = cohort_receipt["executable_cluster_ids"] + final_residual = sorted( + set(cluster["cluster_id"] for cluster in cluster_plan.get("clusters", [])) + - set(final_executable) + ) + cluster_plan["executable_cluster_ids"] = final_executable + cluster_plan["residual_review_cluster_ids"] = final_residual + executable_scc_ids = { + scc_id + for cluster in cluster_plan.get("clusters", []) + if cluster["cluster_id"] in set(final_executable) + for scc_id in cluster.get("scc_ids", []) + } + cluster_plan["scheduling_waves"] = [ + [scc_id for scc_id in wave if scc_id in executable_scc_ids] + for wave in cluster_plan.get("scheduling_waves", []) + if any(scc_id in executable_scc_ids for scc_id in wave) + ] + route = _route(ingress["source_contract_rows"], cluster_plan, cohort_receipt, issues) + snapshot_end_digest = _verify_snapshot_state(all_snapshots) + if snapshot_end_digest != snapshot_start_digest: + raise IngressError("SOURCE_SNAPSHOT_CHANGED", "source closure changed after the one-read snapshot") + source_rows = list(ingress["source_contract_rows"]) + if signal_all: + source_rows.extend(_signal_source_contract_rows(signal_all, release_lock)) + source_rows.extend(deployment_rows) + source_rows = sorted(source_rows, key=lambda row: row["logical_input_id"]) + issue_codes = sorted(set(str(item["issue_code"]) for item in issues)) + source_counts = { + "declared": len(source_rows), + "observed": sum(row["raw_sha256"] is not None for row in source_rows), + "available": sum(row["scope_technical_disposition"] == "AVAILABLE" for row in source_rows), + "with_issues": sum(row["scope_technical_disposition"] == "AVAILABLE_WITH_ISSUES" for row in source_rows), + "unavailable": sum(row["scope_technical_disposition"] == "UNAVAILABLE" for row in source_rows), + } + intake_report = { + "schema_version": "stage2_s2_00_intake_report.v1", + "run_id": run_id, + "source_counts": source_counts, + "conservation_checks": _compact_conservation_checks(conservation["checks"]), + "review_normalization_receipt": _compact_review_receipt(normalized_reviews), + "minimum_coherent_package_possible": route != "TO_S2_40_STATUS_ONLY", + "scope_summary": [ + { + "scope_technical_disposition": row["scope_technical_disposition"], + "impact_scope": row["impact_scope"], + "scope_refs": row["scope_refs"], + "source_contract_row_refs": row["source_contract_row_refs"], + "reason_codes": row["reason_codes"], + "downstream_allowed_actions": row["downstream_allowed_actions"], + } + for row in source_rows + ], + "issue_codes": issue_codes, + } + manifest = { + "schema_version": "stage2_s2_00_stage1_input_manifest.v1.1", + "run_id": run_id, + "release_class": release_lock["release_class"], + "source_rows": source_rows, + "source_row_order": [row["logical_input_id"] for row in source_rows], + "signal_all_adapter_id": "S2A-SIGNAL-ALL-V1", + "dual_sg01_adapter_id": "S2A-DUAL-SG01-V1", + "hydration_stability_receipt": dict(hydration_stability_receipt), + "snapshot_start_digest": snapshot_start_digest, + "snapshot_end_digest": snapshot_end_digest, + "snapshot_status": "STABLE", + "input_set_digest": input_set_digest, + } + issue_ledger = _build_issue_ledger( + issues, + normalized_reviews, + run_id=run_id, + input_set_digest=input_set_digest, + ) + status = { + "schema_version": "stage2_s2_00_ingress_status.v1.1", + "run_id": run_id, + "run_binding_receipt": run_binding_receipt, + "route": route, + "executable_cluster_ids": final_executable if route != "TO_S2_40_STATUS_ONLY" else [], + "residual_review_cluster_ids": final_residual, + "issue_codes": issue_codes, + "output_barrier": { + "barrier_id": "S2_00_INGRESS_STATUS_BARRIER", + "barrier_path": "ingress/ingress_status.json", + "publish_semantics": "STATUS_LAST_LOGICAL_COMMIT", + "canonical_output_root": f"stage2_runs/by-binding/{run_binding_digest}/", + "branch": "DIAGNOSTIC" if route == "TO_S2_40_STATUS_ONLY" else "NORMAL", + "artifacts": [], + "artifact_set_digest": canonical_digest([]), + "non_status_artifacts_read_back_verified": True, + "written_last": True, + }, + } + if route == "TO_S2_40_STATUS_ONLY": + artifacts: dict[str, Any] = { + "ingress/stage1_input_manifest.json": manifest, + "ingress/intake_report.json": intake_report, + "ingress/technical_diagnostic.json": { + "schema_version": "stage2_s2_00_technical_diagnostic.v1", + "run_id": run_id, + "route": "TO_S2_40_STATUS_ONLY", + "minimum_coherent_package_possible": False, + "reason_codes": issue_codes or ["MINIMUM_COHERENT_PACKAGE_UNAVAILABLE"], + "source_contract_row_refs": sorted( + set( + ref + for item in issues + for ref in item.get("source_contract_row_refs", []) + ) + ) or ["S2_00"], + "context_published": False, + }, + "review/issue_ledger.base.json": issue_ledger, + "ingress/ingress_status.json": status, + } + else: + artifacts = { + "ingress/stage1_input_manifest.json": manifest, + "ingress/intake_report.json": intake_report, + "context/case_context.json": case_context, + "context/evidence_inventory.json": evidence_inventory, + "context/object_registry.json": object_registry, + "context/party_and_title_context.json": party_context, + "context/slot_crosswalk.json": slot_crosswalk, + "context/cluster_plan.json": cluster_plan, + "context/bundle_plan.json": bundle_plan, + "review/issue_ledger.base.json": issue_ledger, + "ingress/ingress_status.json": status, + } + for cluster_id, slice_body in slices.items(): + if cluster_id in set(final_executable): + artifacts[f"context/cluster_slices/{cluster_id}.json"] = slice_body + publish_receipt = publish_atomically( + output_dir, + artifacts, + run_binding_digest=run_binding_digest, + attempt_id=attempt_id, + stage2_asset_root=asset_root, + ) + return {"route": route, "run_binding_digest": run_binding_digest, "publish": publish_receipt} + + + def _execute_structural_fixture(descriptor_path: Path, release_lock: Mapping[str, Any]) -> dict[str, Any]: + snapshot = open_bounded_snapshot(descriptor_path.parent, descriptor_path.name, logical_input_id="fixture_descriptor") + descriptor = load_json_strict(snapshot) + if not isinstance(descriptor, dict) or descriptor.get("mode") != "STRUCTURAL_FIXTURE": + raise IngressError("FIXTURE_DESCRIPTOR_SHAPE", "fixture descriptor must declare STRUCTURAL_FIXTURE") + if release_lock.get("release_class") != "DEV_FIXTURE_RELEASE": + raise IngressError("FIXTURE_RELEASE_CLASS", "structural fixtures require DEV_FIXTURE_RELEASE") + required = {"fixture_id", "source_locator", "source_hashes", "source_semantics", "expected"} + missing = sorted(required - set(descriptor)) + if missing: + raise IngressError("FIXTURE_DESCRIPTOR_REQUIRED_FIELD", "fixture fields are missing", details={"missing": missing}) + if descriptor.get("source_semantics") not in {"EXPLICIT", "DERIVED"}: + raise IngressError("FIXTURE_SOURCE_SEMANTICS", "source_semantics must be EXPLICIT or DERIVED") + payloads = descriptor.get("fixture_payloads", {}) + source_hashes = descriptor.get("source_hashes", {}) + if not isinstance(payloads, dict) or not isinstance(source_hashes, dict): + raise IngressError("FIXTURE_SOURCE_HASH_SHAPE", "fixture payloads and source hashes must be objects") + for logical_id, payload in payloads.items(): + expected_hash = source_hashes.get(logical_id) + if not isinstance(expected_hash, str) or expected_hash != canonical_digest(payload): + raise IngressError( + "FIXTURE_SOURCE_HASH_MISMATCH", + f"fixture payload hash mismatch for {logical_id}", + ) + return { + "status": "STRUCTURAL_FIXTURE_VALIDATED", + "fixture_id": descriptor["fixture_id"], + "descriptor_sha256": snapshot.raw_sha256, + "canonical_descriptor_sha256": canonical_digest(descriptor), + "published": False, + } + + + def _inline_sha256(value: str, *, code: str) -> str: + if not isinstance(value, str) or re.fullmatch(r"[a-f0-9]{64}", value) is None: + raise IngressError(code, "expected one lowercase SHA-256 digest") + return value + + + def _inline_relative_path(value: str, *, code: str) -> str: + if not isinstance(value, str) or not value or "\x00" in value or "\\" in value: + raise IngressError(code, "logical path is empty or malformed") + if unicodedata.normalize("NFC", value) != value: + raise IngressError(code, "logical path must already be NFC") + path = PurePosixPath(value) + if path.is_absolute() or any(part in {"", ".", ".."} for part in path.parts): + raise IngressError(code, "logical path must be a contained relative path") + rendered = path.as_posix() + if rendered != value: + raise IngressError(code, "logical path is not canonical") + return rendered + + + def _inline_join(root: str, relative: str, *, code: str) -> str: + safe_root = _inline_relative_path(root, code=code) + safe_relative = _inline_relative_path(relative, code=code) + return _inline_relative_path(f"{safe_root}/{safe_relative}", code=code) + + + def _inline_parse_mcp_payload(raw: bytes, expected_id: int) -> Mapping[str, Any]: + """Parse one JSON or SSE JSON-RPC terminal response with an exact ID.""" + + candidates: list[Any] + try: + candidates = [load_json_strict(raw)] + except IngressError: + try: + text = raw.decode("utf-8", errors="strict") + except UnicodeDecodeError as exc: + raise IngressError("MCP_RESPONSE_UTF8", "MCP response is not strict UTF-8") from exc + events: list[bytes] = [] + data_lines: list[str] = [] + for line in text.replace("\r\n", "\n").replace("\r", "\n").split("\n"): + if line == "": + if data_lines: + events.append("\n".join(data_lines).encode("utf-8")) + data_lines = [] + continue + if line.startswith(":") or line.startswith("event:") or line.startswith("id:") or line.startswith("retry:"): + continue + if not line.startswith("data:"): + raise IngressError("MCP_SSE_SHAPE", "unexpected non-data SSE line") + payload = line[5:] + if payload.startswith(" "): + payload = payload[1:] + data_lines.append(payload) + if data_lines: + events.append("\n".join(data_lines).encode("utf-8")) + if not events: + raise IngressError("MCP_RESPONSE_SHAPE", "MCP response contains no JSON terminal event") + candidates = [load_json_strict(event) for event in events] + matching = [ + item + for item in candidates + if isinstance(item, dict) and item.get("id") == expected_id + ] + if len(matching) != 1: + raise IngressError( + "MCP_RESPONSE_ID_MISMATCH", + "MCP response must contain exactly one terminal result with the request ID", + ) + response = matching[0] + if response.get("jsonrpc") != "2.0": + raise IngressError("MCP_JSONRPC_VERSION", "MCP response jsonrpc must equal 2.0") + if response.get("error") is not None: + raise IngressError( + "MCP_JSONRPC_ERROR", + "MCP server returned a JSON-RPC error", + details={"rpc_error": response.get("error")}, + ) + if "result" not in response or not isinstance(response["result"], dict): + raise IngressError("MCP_RESULT_SHAPE", "MCP response result must be an object") + return response + + + def _inline_tool_text(result: Mapping[str, Any], tool_name: str) -> str: + if result.get("isError") is True: + content = result.get("content") + rendered = canonical_json_bytes(content).decode("utf-8", errors="replace") if content is not None else "" + lowered = rendered.lower() + code = ( + "LOCALDOCS_NOT_FOUND" + if any(marker in lowered for marker in ("not found", "does not exist", "no such file")) + else "MCP_TOOL_ERROR" + ) + raise IngressError(code, f"localdocs {tool_name} returned isError=true") + content = result.get("content") + if not isinstance(content, list) or len(content) != 1: + raise IngressError("MCP_CONTENT_CARDINALITY", "MCP tool result must contain exactly one content block") + block = content[0] + if not isinstance(block, dict) or block.get("type") != "text" or not isinstance(block.get("text"), str): + raise IngressError("MCP_CONTENT_SHAPE", "MCP tool result must contain one text block") + return block["text"] + + + def _inline_binary_envelope(text: str, logical_path: str) -> bytes: + value = load_json_strict(text) + if isinstance(value, dict) and "results" in value: + results = value.get("results") + if not isinstance(results, list) or len(results) != 1 or not isinstance(results[0], dict): + raise IngressError("LOCALDOCS_RESULT_CARDINALITY", "binary response must contain one result row") + inner: Any = results[0].get("content", results[0].get("text")) + value = load_json_strict(inner) if isinstance(inner, str) else inner + if not isinstance(value, dict) or not isinstance(value.get("content_base64"), str): + raise IngressError("LOCALDOCS_BINARY_ENVELOPE", "binary response lacks content_base64") + try: + payload = base64.b64decode(value["content_base64"].encode("ascii"), validate=True) + except (UnicodeEncodeError, binascii.Error, ValueError) as exc: + raise IngressError("LOCALDOCS_BASE64_INVALID", "binary response is not strict base64") from exc + declared_size = value.get("byte_length", value.get("size")) + if declared_size is not None and (not isinstance(declared_size, int) or declared_size != len(payload)): + raise IngressError("LOCALDOCS_BYTE_LENGTH_MISMATCH", f"binary length mismatch: {logical_path}") + declared_hash = value.get("sha256") + if declared_hash is not None and declared_hash != hashlib.sha256(payload).hexdigest(): + raise IngressError("LOCALDOCS_HASH_MISMATCH", f"binary hash mismatch: {logical_path}") + return payload + + + class _InlineLocaldocs: + """Minimal user/workspace-bound localdocs JSON-RPC client.""" + + def __init__( + self, + user_hash: str, + workspace_hash: str, + *, + client: Any | None = None, + timeout_seconds: int = 60, + ) -> None: + self.user_hash = _inline_sha256(user_hash, code="USER_CONTEXT_HASH_INVALID") + self.workspace_hash = _inline_sha256( + workspace_hash, + code="WORKSPACE_CONTEXT_HASH_INVALID", + ) + if client is None: + try: + import httpx # type: ignore + except ImportError as exc: + raise IngressError("HTTPX_UNAVAILABLE", "Code Executor must supply httpx==0.28.1") from exc + client = httpx.Client(timeout=timeout_seconds) + self.client = client + self.headers = { + "Content-Type": "application/json", + "Accept": "application/json, text/event-stream", + } + self._message_ids = itertools.count(10) + self._initialized = False + self._session_id: str | None = None + + def close(self) -> None: + close = getattr(self.client, "close", None) + if callable(close): + close() + + def _post(self, body: Mapping[str, Any], expected_id: int | None) -> Mapping[str, Any] | None: + try: + response = self.client.post(LOCALDOCS_URL, json=dict(body), headers=dict(self.headers)) + response.raise_for_status() + except Exception as exc: + raise IngressError("MCP_TRANSPORT_ERROR", "localdocs transport failed") from exc + session_id = response.headers.get("mcp-session-id") + if session_id: + if not isinstance(session_id, str) or not session_id.strip(): + raise IngressError("MCP_SESSION_ID_INVALID", "localdocs returned an invalid session ID") + normalized_session_id = session_id.strip() + if self._session_id is None: + if expected_id != 1: + raise IngressError( + "MCP_SESSION_ID_OUTSIDE_INITIALIZE", + "localdocs first bound a session outside initialize", + ) + self._session_id = normalized_session_id + elif normalized_session_id != self._session_id: + raise IngressError( + "MCP_SESSION_ID_CHANGED", + "localdocs changed the initialized session ID", + ) + self.headers["mcp-session-id"] = self._session_id + if expected_id is None: + return None + raw = response.content if isinstance(response.content, bytes) else bytes(response.content) + return _inline_parse_mcp_payload(raw, expected_id) + + def initialize(self) -> None: + response = self._post( + { + "jsonrpc": "2.0", + "id": 1, + "method": "initialize", + "params": { + "protocolVersion": MCP_PROTOCOL_VERSION, + "capabilities": {}, + "clientInfo": { + "name": INLINE_CLIENT_NAME, + "version": INLINE_CLIENT_VERSION, + "user_id": self.user_hash, + "workspace_id": self.workspace_hash, + }, + }, + }, + 1, + ) + if response is None: + raise IngressError("MCP_INITIALIZE_EMPTY", "localdocs initialize returned no result") + result = response.get("result") + if not isinstance(result, dict) or result.get("protocolVersion") != MCP_PROTOCOL_VERSION: + raise IngressError( + "MCP_PROTOCOL_VERSION_MISMATCH", + "localdocs did not negotiate the requested MCP protocol version", + ) + if self._session_id is None or "mcp-session-id" not in self.headers: + raise IngressError("MCP_SESSION_ID_MISSING", "localdocs initialize did not bind a session ID") + self._post( + {"jsonrpc": "2.0", "method": "notifications/initialized"}, + None, + ) + self._initialized = True + + def call(self, tool_name: str, arguments: Mapping[str, Any]) -> Mapping[str, Any]: + if not self._initialized: + raise IngressError("MCP_NOT_INITIALIZED", "localdocs session is not initialized") + message_id = next(self._message_ids) + response = self._post( + { + "jsonrpc": "2.0", + "id": message_id, + "method": "tools/call", + "params": {"name": tool_name, "arguments": dict(arguments)}, + }, + message_id, + ) + if response is None: + raise IngressError("MCP_TOOL_EMPTY", f"localdocs {tool_name} returned no result") + return response["result"] + + def read_binary(self, logical_path: str) -> bytes: + path = _inline_relative_path(logical_path, code="LOCALDOCS_READ_PATH_INVALID") + result = self.call("read_binary_doc", {"doc_name": path}) + return _inline_binary_envelope(_inline_tool_text(result, "read_binary_doc"), path) + + def read_binary_optional(self, logical_path: str) -> bytes | None: + try: + return self.read_binary(logical_path) + except IngressError as exc: + if exc.code == "LOCALDOCS_NOT_FOUND": + return None + raise + + def write_binary_verified(self, logical_path: str, payload: bytes) -> str: + path = _inline_relative_path(logical_path, code="LOCALDOCS_WRITE_PATH_INVALID") + encoded = base64.b64encode(payload).decode("ascii") + result = self.call( + "write_binary_file", + {"path": path, "content_base64": encoded, "overwrite": True}, + ) + _inline_tool_text(result, "write_binary_file") + observed = self.read_binary(path) + if observed != payload: + raise IngressError("LOCALDOCS_WRITE_READBACK_MISMATCH", f"read-back mismatch: {path}") + return hashlib.sha256(observed).hexdigest() + + + def _inline_validate_request(raw: bytes) -> dict[str, str]: + value = load_json_strict(raw) + if not isinstance(value, dict): + raise IngressError("RUN_REQUEST_SHAPE", "S2_00 request must be an object") + required = { + "schema_version", + "workflow_id", + "request_id", + "attempt_id", + "stage1_run_root_ref", + "stage1_deployment_root_ref", + } + if set(value) != required: + raise IngressError( + "RUN_REQUEST_CLOSED_SHAPE", + "S2_00 request has missing or unknown keys", + details={"missing": sorted(required - set(value)), "extra": sorted(set(value) - required)}, + ) + if value.get("schema_version") != "stage2_s2_00_execution_request.v1": + raise IngressError("RUN_REQUEST_SCHEMA_VERSION", "unsupported S2_00 request schema") + if value.get("workflow_id") != "S2_00": + raise IngressError("WORKFLOW_ID_MISMATCH", "S2_00 request workflow_id must equal S2_00") + request_id = value.get("request_id") + attempt_id = value.get("attempt_id") + if not isinstance(request_id, str) or re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9._-]{0,127}", request_id) is None: + raise IngressError("REQUEST_ID_INVALID", "request_id contains forbidden characters") + if not isinstance(attempt_id, str) or re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9._-]{0,127}", attempt_id) is None: + raise IngressError("ATTEMPT_ID_INVALID", "attempt_id contains forbidden characters") + run_root = _inline_relative_path( + value.get("stage1_run_root_ref"), + code="STAGE1_RUN_ROOT_REF_INVALID", + ) + deployment_root = _inline_relative_path( + value.get("stage1_deployment_root_ref"), + code="STAGE1_DEPLOYMENT_ROOT_REF_INVALID", + ) + return { + "schema_version": value["schema_version"], + "workflow_id": value["workflow_id"], + "request_id": request_id, + "attempt_id": attempt_id, + "stage1_run_root_ref": run_root, + "stage1_deployment_root_ref": deployment_root, + } + + + def _inline_bound_ref(row: Mapping[str, Any], *, code: str) -> tuple[str, str]: + path = _inline_relative_path(row.get("path"), code=code) + digest = _inline_sha256(str(row.get("sha256", "")), code=f"{code}_HASH") + if row.get("binding_status") not in {None, "BOUND"}: + raise IngressError(code, f"asset is not release-bound: {path}") + return path, digest + + + def _inline_release_materialization_plan( + request: Mapping[str, str], + release: Mapping[str, Any], + first_read: Callable[[str], bytes], + ) -> tuple[list[tuple[str, str, str]], dict[str, str]]: + """Return exact remote->temporary destinations and expected remote hashes.""" + + destinations: list[tuple[str, str, str]] = [] + expected_hashes: dict[str, str] = {} + seen_destinations: set[tuple[str, str]] = set() + + def add( + remote_path: str, + local_family: str, + relative_path: str, + *, + expected_sha256: str | None = None, + ) -> None: + remote = _inline_relative_path(remote_path, code="HYDRATION_REMOTE_PATH_INVALID") + relative = _inline_relative_path(relative_path, code="HYDRATION_LOCAL_PATH_INVALID") + key = (local_family, relative) + if key in seen_destinations: + raise IngressError("HYDRATION_DESTINATION_DUPLICATE", f"duplicate hydration destination: {relative}") + seen_destinations.add(key) + destinations.append((remote, local_family, relative)) + if expected_sha256 is not None: + digest = _inline_sha256(expected_sha256.lower(), code="HYDRATION_EXPECTED_HASH_INVALID") + prior = expected_hashes.get(remote) + if prior is not None and prior != digest: + raise IngressError("HYDRATION_HASH_CONFLICT", f"conflicting expected hash: {remote}") + expected_hashes[remote] = digest + + release_remote = INLINE_STAGE2_RELEASE_PATH + add( + release_remote, + "stage2_asset", + "manifest/stage2_release.json", + expected_sha256=EXPECTED_STAGE2_RELEASE_SHA256, + ) + + source_rows = _release_stage1_source_rows(release) + if not isinstance(source_rows, list): + raise IngressError("RELEASE_STAGE1_SOURCE_SHAPE", "release stage1_sources must be an array") + fixed_rows = [row for row in source_rows if row.get("logical_input_id") != "signal_payload_family"] + if len(fixed_rows) != 16: + raise IngressError("RELEASE_STAGE1_SOURCE_COUNT", "release must enumerate exactly 16 fixed inputs") + signal_manifest_relative: str | None = None + for row in fixed_rows: + if not isinstance(row, dict) or not isinstance(row.get("path"), str): + raise IngressError("RELEASE_STAGE1_SOURCE_ROW", "fixed source row lacks an exact path") + relative = _inline_relative_path(row["path"], code="STAGE1_SOURCE_PATH_INVALID") + remote = _inline_join(request["stage1_run_root_ref"], relative, code="STAGE1_SOURCE_PATH_INVALID") + add(remote, "stage1_run", relative) + first_read(remote) + if row.get("logical_input_id") == "signal_manifest": + signal_manifest_relative = relative + if signal_manifest_relative is None: + raise IngressError("SIGNAL_MANIFEST_RELEASE_ROW_MISSING", "release lacks signal_manifest") + signal_remote = _inline_join( + request["stage1_run_root_ref"], + signal_manifest_relative, + code="SIGNAL_MANIFEST_PATH_INVALID", + ) + signal_manifest = load_json_strict(first_read(signal_remote)) + if not isinstance(signal_manifest, dict) or not isinstance(signal_manifest.get("files"), list): + raise IngressError("SIGNAL_FILES_SHAPE", "signal manifest files must be an array") + for index, row in enumerate(signal_manifest["files"]): + if not isinstance(row, dict) or not isinstance(row.get("path"), str): + raise IngressError("SIGNAL_FILE_ROW_SHAPE", f"invalid signal row: {index}") + relative_payload = _inline_relative_path(row["path"], code="SIGNAL_FILE_PATH_INVALID") + if relative_payload.startswith("signals/"): + raise IngressError("SIGNAL_PATH_PREFIX_FORBIDDEN", "signal manifest path includes signals/") + relative = f"signals/{relative_payload}" + remote = _inline_join(request["stage1_run_root_ref"], relative, code="SIGNAL_FILE_PATH_INVALID") + expected = row.get("file_sha256", row.get("sha256", row.get("raw_sha256"))) + add(remote, "stage1_run", relative, expected_sha256=expected if isinstance(expected, str) else None) + first_read(remote) + + dependencies = release.get("dependency_locks") + stage1_dependency = dependencies.get("stage1") if isinstance(dependencies, dict) else None + closure = stage1_dependency.get("concrete_paths") if isinstance(stage1_dependency, dict) else None + if not isinstance(closure, list) or not closure: + raise IngressError("STAGE1_DEPENDENCY_LOCK_MISSING", "Stage 1 deployment closure is absent") + expected_count = stage1_dependency.get("expected_concrete_path_count") + if expected_count is not None and expected_count != len(closure): + raise IngressError("STAGE1_DEPENDENCY_COUNT_MISMATCH", "Stage 1 closure count differs from release") + for row in closure: + if not isinstance(row, dict): + raise IngressError("STAGE1_DEPENDENCY_ROW_SHAPE", "Stage 1 closure row must be an object") + relative, digest = _inline_bound_ref(row, code="STAGE1_DEPENDENCY_UNBOUND") + remote = _inline_join( + request["stage1_deployment_root_ref"], + relative, + code="STAGE1_DEPLOYMENT_PATH_INVALID", + ) + add(remote, "stage1_deployment", relative, expected_sha256=digest) + first_read(remote) + + contract_ref = stage1_dependency.get("contract_manifest_ref", {}) + if isinstance(contract_ref, dict) and isinstance(contract_ref.get("path"), str): + relative, digest = _inline_bound_ref(contract_ref, code="CONTRACT_MANIFEST_UNBOUND") + remote = _inline_join( + request["stage1_deployment_root_ref"], + relative, + code="CONTRACT_MANIFEST_PATH_INVALID", + ) + add(remote, "stage1_deployment", relative, expected_sha256=digest) + first_read(remote) + completion_ref = stage1_dependency.get("completion_seal_ref", {}) + if isinstance(completion_ref, dict) and completion_ref.get("path") not in {None, "PENDING_SEQUENTIAL_BIND"}: + relative, digest = _inline_bound_ref(completion_ref, code="COMPLETION_SEAL_UNBOUND") + remote = _inline_join(request["stage1_run_root_ref"], relative, code="COMPLETION_SEAL_PATH_INVALID") + add(remote, "stage1_run", relative, expected_sha256=digest) + first_read(remote) + + manifest_ref = release.get("module_manifest_ref") + if not isinstance(manifest_ref, dict): + raise IngressError("MODULE_MANIFEST_REF_MISSING", "release lacks module_manifest_ref") + manifest_relative, manifest_digest = _inline_bound_ref(manifest_ref, code="MODULE_MANIFEST_UNBOUND") + manifest_remote = _inline_join(INLINE_STAGE2_ASSET_ROOT, manifest_relative, code="MODULE_MANIFEST_PATH_INVALID") + add(manifest_remote, "stage2_asset", manifest_relative, expected_sha256=manifest_digest) + module_manifest = load_json_strict(first_read(manifest_remote)) + modules = module_manifest.get("modules") if isinstance(module_manifest, dict) else None + if not isinstance(modules, list): + raise IngressError("MODULE_MANIFEST_SHAPE", "module manifest lacks modules array") + found_schema_ids: set[str] = set() + for row in modules: + if not isinstance(row, dict) or row.get("module_id") not in INLINE_SCHEMA_MODULE_IDS: + continue + relative, digest = _inline_bound_ref(row, code="STAGE2_SCHEMA_UNBOUND") + remote = _inline_join(INLINE_STAGE2_ASSET_ROOT, relative, code="STAGE2_SCHEMA_PATH_INVALID") + add(remote, "stage2_asset", relative, expected_sha256=digest) + first_read(remote) + found_schema_ids.add(str(row["module_id"])) + if found_schema_ids != set(INLINE_SCHEMA_MODULE_IDS): + raise IngressError( + "STAGE2_SCHEMA_CLOSURE_INCOMPLETE", + "module manifest does not bind all S2_00 schemas", + details={"missing": sorted(set(INLINE_SCHEMA_MODULE_IDS) - found_schema_ids)}, + ) + + stage2_direct = dependencies.get("stage2_direct") if isinstance(dependencies, dict) else None + if not isinstance(stage2_direct, list): + raise IngressError("STAGE2_DIRECT_LOCK_MISSING", "release lacks Stage 2 direct closure") + for row in stage2_direct: + if not isinstance(row, dict): + raise IngressError("STAGE2_DIRECT_ROW_SHAPE", "Stage 2 direct row must be an object") + relative, digest = _inline_bound_ref(row, code="STAGE2_DIRECT_UNBOUND") + remote = _inline_join(INLINE_STAGE2_ASSET_ROOT, relative, code="STAGE2_DIRECT_PATH_INVALID") + add(remote, "stage2_asset", relative, expected_sha256=digest) + first_read(remote) + return destinations, expected_hashes + + + def _inline_hydrate( + localdocs: _InlineLocaldocs, + temp_root: Path, + ) -> tuple[dict[str, str], Mapping[str, Any], Path, Path, Path, dict[str, Any]]: + first_pass: dict[str, bytes] = {} + first_pass_bytes = 0 + + request_raw = localdocs.read_binary(INLINE_REQUEST_PATH) + if len(request_raw) > MAX_FILE_BYTES: + raise IngressError("SOURCE_SIZE_LIMIT", "S2_00 request exceeds the global pre-parse file limit") + first_pass[INLINE_REQUEST_PATH] = request_raw + first_pass_bytes += len(request_raw) + request = _inline_validate_request(request_raw) + + release_raw = localdocs.read_binary(INLINE_STAGE2_RELEASE_PATH) + if len(release_raw) > MAX_FILE_BYTES: + raise IngressError("SOURCE_SIZE_LIMIT", "Stage 2 release exceeds the global pre-parse file limit") + first_pass[INLINE_STAGE2_RELEASE_PATH] = release_raw + first_pass_bytes += len(release_raw) + if first_pass_bytes > MAX_RUN_BYTES: + raise IngressError("AGGREGATE_RUN_SIZE_LIMIT", "initial request and release exceed the global pre-parse budget") + if EXPECTED_STAGE2_RELEASE_SHA256 == "0" * 64: + raise IngressError( + "EXPECTED_STAGE2_RELEASE_SHA256_UNBOUND", + "build-time Stage 2 release pin has not been bound", + ) + expected_release = _inline_sha256( + EXPECTED_STAGE2_RELEASE_SHA256, + code="EXPECTED_STAGE2_RELEASE_SHA256_INVALID", + ) + if hashlib.sha256(release_raw).hexdigest() != expected_release: + raise IngressError("STAGE2_RELEASE_PIN_MISMATCH", "Stage 2 release bytes differ from the build-time pin") + release = load_json_strict(release_raw) + if not isinstance(release, dict): + raise IngressError("RELEASE_LOCK_SHAPE", "Stage 2 release must be an object") + limits = release.get("limits") if isinstance(release.get("limits"), dict) else {} + max_file = int(limits.get("max_file_bytes", MAX_FILE_BYTES)) + max_total = int(limits.get("max_run_bytes", MAX_RUN_BYTES)) + max_paths = int(limits.get("max_hydration_paths", 1024)) + if max_file <= 0 or max_file > MAX_FILE_BYTES: + raise IngressError("RELEASE_FILE_LIMIT_INVALID", "release max_file_bytes exceeds the build-time ceiling") + if max_total <= 0 or max_total > MAX_RUN_BYTES: + raise IngressError("RELEASE_RUN_LIMIT_INVALID", "release max_run_bytes exceeds the build-time ceiling") + if max_paths <= 0 or max_paths > 1024: + raise IngressError("RELEASE_PATH_LIMIT_INVALID", "release max_hydration_paths exceeds the build-time ceiling") + if len(request_raw) > max_file or len(release_raw) > max_file: + raise IngressError("SOURCE_SIZE_LIMIT", "initial request or release exceeds the release-bound file limit") + if first_pass_bytes > max_total: + raise IngressError("AGGREGATE_RUN_SIZE_LIMIT", "initial request and release exceed the release-bound budget") + + def first_read(path: str) -> bytes: + nonlocal first_pass_bytes + safe = _inline_relative_path(path, code="HYDRATION_REMOTE_PATH_INVALID") + if safe not in first_pass: + payload = localdocs.read_binary(safe) + if len(payload) > max_file: + raise IngressError("SOURCE_SIZE_LIMIT", f"remote source exceeds file limit: {safe}") + first_pass_bytes += len(payload) + if first_pass_bytes > max_total: + raise IngressError("AGGREGATE_RUN_SIZE_LIMIT", "first-pass hydration exceeds release budget") + first_pass[safe] = payload + return first_pass[safe] + + destinations, expected_hashes = _inline_release_materialization_plan( + request, + release, + first_read, + ) + if len(first_pass) > max_paths: + raise IngressError("HYDRATION_PATH_COUNT_LIMIT", "hydration path count exceeds release budget") + for remote, expected in expected_hashes.items(): + observed = hashlib.sha256(first_read(remote)).hexdigest() + if observed != expected: + raise IngressError("HYDRATION_BOUND_HASH_MISMATCH", f"release-bound hash mismatch: {remote}") + + second_pass_bytes = 0 + second_pass: dict[str, bytes] = {} + for remote in sorted(first_pass): + payload = localdocs.read_binary(remote) + second_pass_bytes += len(payload) + if len(payload) > max_file or second_pass_bytes > max_total: + raise IngressError("TWO_PASS_READ_BUDGET", "second-pass hydration exceeds release budget") + if payload != first_pass[remote] or hashlib.sha256(payload).digest() != hashlib.sha256(first_pass[remote]).digest(): + raise IngressError("SOURCE_SNAPSHOT_CHANGED", f"remote source changed between bounded reads: {remote}") + second_pass[remote] = payload + + pass_1_hash_rows = [ + [path, hashlib.sha256(first_pass[path]).hexdigest()] + for path in sorted(first_pass) + ] + pass_2_hash_rows = [ + [path, hashlib.sha256(second_pass[path]).hexdigest()] + for path in sorted(second_pass) + ] + hydration_receipt = { + "schema_version": "stage2_s2_00_two_pass_hydration_receipt.v1", + "transport": "localdocs.read_binary_doc", + "read_policy": "BOUNDED_TWO_PASS_BINARY_RAW_HASH_MAP_EQUALITY", + "read_pass_count": 2, + "max_read_passes": 2, + "pass_1_raw_hash_map_digest": canonical_digest(pass_1_hash_rows), + "pass_2_raw_hash_map_digest": canonical_digest(pass_2_hash_rows), + "source_receipts": [ + { + "logical_input_id": path, + "logical_path": path, + "pass_1_raw_sha256": hashlib.sha256(first_pass[path]).hexdigest(), + "pass_2_raw_sha256": hashlib.sha256(second_pass[path]).hexdigest(), + "pass_1_byte_length": len(first_pass[path]), + "pass_2_byte_length": len(second_pass[path]), + "stability_status": "STABLE", + } + for path in sorted(first_pass) + ], + "stability_status": "STABLE", + } + + roots = { + "stage1_run": temp_root / "stage1_run", + "stage1_deployment": temp_root / "stage1_deployment", + "stage2_asset": temp_root / "stage2_asset", + } + for root in roots.values(): + root.mkdir(parents=True, exist_ok=False) + for remote, family, relative in sorted(destinations): + target = roots[family] / PurePosixPath(relative) + target.parent.mkdir(parents=True, exist_ok=True) + if target.exists(): + raise IngressError("HYDRATION_DESTINATION_EXISTS", f"duplicate materialization: {relative}") + target.write_bytes(first_pass[remote]) + return ( + request, + release, + roots["stage1_run"], + roots["stage1_deployment"], + roots["stage2_asset"], + hydration_receipt, + ) + + + def _inline_output_files(output_root: Path, run_binding_digest: str) -> dict[str, bytes]: + if not output_root.is_dir() or output_root.is_symlink(): + raise IngressError("LOCAL_OUTPUT_TREE_MISSING", "pure core did not produce an output tree") + files: dict[str, bytes] = {} + for path in sorted(output_root.rglob("*")): + if path.is_symlink(): + raise IngressError("LOCAL_OUTPUT_SYMLINK", "pure core output contains a symlink") + if not path.is_file(): + continue + relative = _safe_relative_path(path.relative_to(output_root).as_posix()).as_posix() + files[relative] = path.read_bytes() + status_raw = files.get("ingress/ingress_status.json") + if status_raw is None: + raise IngressError("OUTPUT_BARRIER_MISSING", "pure core output lacks ingress_status") + status = load_json_strict(status_raw) + binding = status.get("run_binding_receipt") if isinstance(status, dict) else None + if not isinstance(binding, dict) or binding.get("run_binding_digest") != run_binding_digest: + raise IngressError("RUN_BINDING_RECEIPT_MISMATCH", "pure core status has a different binding") + barrier = status.get("output_barrier") + rows = barrier.get("artifacts") if isinstance(barrier, dict) else None + if not isinstance(rows, list) or barrier.get("written_last") is not True: + raise IngressError("OUTPUT_BARRIER_INVALID", "pure core output barrier is incomplete") + expected_paths = {"ingress/ingress_status.json"} + for row in rows: + if not isinstance(row, dict) or not isinstance(row.get("path"), str): + raise IngressError("OUTPUT_BARRIER_ROW_SHAPE", "output barrier row is malformed") + relative = _safe_relative_path(row["path"]).as_posix() + payload = files.get(relative) + if payload is None or hashlib.sha256(payload).hexdigest() != row.get("raw_sha256"): + raise IngressError("OUTPUT_BARRIER_HASH_MISMATCH", f"output barrier mismatch: {relative}") + expected_paths.add(relative) + if set(files) != expected_paths: + raise IngressError("OUTPUT_BARRIER_SET_MISMATCH", "output tree differs from its barrier set") + if barrier.get("artifact_set_digest") != canonical_digest(rows): + raise IngressError("OUTPUT_BARRIER_DIGEST_MISMATCH", "output barrier row digest is invalid") + return files + + + def _inline_verify_existing_remote( + localdocs: _InlineLocaldocs, + output_root: str, + files: Mapping[str, bytes], + run_binding_digest: str, + ) -> bool: + status_path = _inline_join(output_root, "ingress/ingress_status.json", code="OUTPUT_PATH_INVALID") + status_raw = localdocs.read_binary_optional(status_path) + if status_raw is None: + return False + status = load_json_strict(status_raw) + binding = status.get("run_binding_receipt") if isinstance(status, dict) else None + if not isinstance(binding, dict) or binding.get("run_binding_digest") != run_binding_digest: + raise IngressError("RUN_ID_BINDING_CONFLICT", "existing remote barrier has a different binding") + if status_raw != files.get("ingress/ingress_status.json"): + raise IngressError("IDEMPOTENT_STATUS_MISMATCH", "existing remote status is not byte-identical") + for relative, expected in sorted(files.items()): + observed = localdocs.read_binary(_inline_join(output_root, relative, code="OUTPUT_PATH_INVALID")) + if observed != expected: + raise IngressError("IDEMPOTENT_ARTIFACT_MISMATCH", f"existing artifact differs: {relative}") + return True + + + def _inline_publish_remote( + localdocs: _InlineLocaldocs, + files: Mapping[str, bytes], + run_binding_digest: str, + ) -> dict[str, Any]: + output_root = f"stage2_runs/by-binding/{run_binding_digest}" + status_raw = files["ingress/ingress_status.json"] + status = load_json_strict(status_raw) + barrier = status.get("output_barrier") if isinstance(status, dict) else None + branch = barrier.get("branch") if isinstance(barrier, dict) else None + if branch not in {"NORMAL", "DIAGNOSTIC"}: + raise IngressError("OUTPUT_BARRIER_BRANCH_INVALID", "pure core output barrier lacks a valid branch") + artifact_rows = [ + { + "logical_artifact_id": relative, + "path": relative, + "schema_id": _output_schema_id(relative), + "raw_sha256": hashlib.sha256(payload).hexdigest(), + "byte_length": len(payload), + } + for relative, payload in sorted(files.items()) + if relative != "ingress/ingress_status.json" + ] + if _inline_verify_existing_remote(localdocs, output_root, files, run_binding_digest): + publication_status = "IDEMPOTENT_SUCCESS" + else: + for relative in sorted(path for path in files if path != "ingress/ingress_status.json"): + localdocs.write_binary_verified( + _inline_join(output_root, relative, code="OUTPUT_PATH_INVALID"), + files[relative], + ) + status_path = _inline_join(output_root, "ingress/ingress_status.json", code="OUTPUT_PATH_INVALID") + localdocs.write_binary_verified(status_path, status_raw) + if localdocs.read_binary(status_path) != status_raw: + raise IngressError("OUTPUT_BARRIER_READBACK_MISMATCH", "remote ingress_status read-back failed") + publication_status = "PUBLISHED_STATUS_LAST" + return { + "schema_version": "stage2_logical_publish_receipt.v1", + "barrier_id": "S2_00_INGRESS_STATUS_BARRIER", + "barrier_path": "ingress/ingress_status.json", + "publish_semantics": "STATUS_LAST_LOGICAL_COMMIT", + "canonical_output_root": f"{output_root}/", + "run_binding_digest": run_binding_digest, + "branch": branch, + "artifacts": artifact_rows, + "artifact_set_digest": canonical_digest(artifact_rows), + "barrier_raw_sha256": hashlib.sha256(status_raw).hexdigest(), + "non_status_artifacts_read_back_verified": True, + "barrier_written_last": True, + "downstream_consumption_allowed": True, + "publication_status": publication_status, + } + + + def _inline_context_is_bound() -> bool: + return ( + re.fullmatch(r"[a-f0-9]{64}", INLINE_USER_HASH) is not None + and re.fullmatch(r"[a-f0-9]{64}", INLINE_WORKSPACE_HASH) is not None + ) + + + def run_inline_mcp(*, client: Any | None = None) -> int: + """Execute one complete MCP Code Executor S2_00 task and emit one JSON receipt.""" + + localdocs: _InlineLocaldocs | None = None + receipt: dict[str, Any] + exit_code = 0 + try: + with contextlib.redirect_stdout(io.StringIO()): + localdocs = _InlineLocaldocs( + INLINE_USER_HASH, + INLINE_WORKSPACE_HASH, + client=client, + ) + try: + localdocs.initialize() + with tempfile.TemporaryDirectory(prefix="liti-s2-00-") as directory: + temp_root = Path(directory) + ( + request, + _release, + stage1_root, + deployment_root, + asset_root, + hydration_receipt, + ) = _inline_hydrate(localdocs, temp_root) + release_lock = load_release_lock(asset_root / "manifest" / "stage2_release.json") + contract_manifest = _load_bound_contract_manifest(deployment_root, release_lock) + local_output = temp_root / "core_output" + result = execute_ingress( + stage1_root, + release_lock, + output_dir=local_output, + attempt_id=request["attempt_id"], + run_id=request["request_id"], + user_context_sha256=INLINE_USER_HASH, + workspace_context_sha256=INLINE_WORKSPACE_HASH, + stage1_deployment_root=deployment_root, + stage2_asset_root=asset_root, + contract_manifest=contract_manifest, + hydration_stability_receipt=hydration_receipt, + ) + run_binding_digest = _inline_sha256( + str(result.get("run_binding_digest", "")), + code="CORE_RUN_BINDING_INVALID", + ) + files = _inline_output_files(local_output, run_binding_digest) + publication = _inline_publish_remote(localdocs, files, run_binding_digest) + receipt = { + "schema_version": "stage2_s2_00_inner_receipt.v1", + "workflow_id": "S2_00", + "ok": True, + "status": ( + "DIAGNOSTIC_PUBLISHED" + if result["route"] == "TO_S2_40_STATUS_ONLY" + else "SUCCEEDED" + ), + "request_id": request["request_id"], + "route": result["route"], + "run_id": canonical_run_id(run_binding_digest), + "run_binding_digest": run_binding_digest, + "expected_release_sha256": EXPECTED_STAGE2_RELEASE_SHA256, + "algorithm_digest": ALGORITHM_SEMANTIC_DIGEST, + "artifact_set_digest": publication["artifact_set_digest"], + "ingress_status_sha256": publication["barrier_raw_sha256"], + "logical_publish_receipt": publication, + } + finally: + if localdocs is not None: + localdocs.close() + except Exception as exc: + error = ( + exc.as_dict() + if isinstance(exc, IngressError) + else {"code": "INLINE_RUNTIME_ERROR", "message": str(exc)} + ) + receipt = { + "schema_version": "stage2_s2_00_inner_receipt.v1", + "workflow_id": "S2_00", + "ok": False, + "status": "FAILED_NO_BARRIER", + "expected_release_sha256": EXPECTED_STAGE2_RELEASE_SHA256, + "error": error, + } + exit_code = 2 + sys.stdout.buffer.write(canonical_json_bytes(receipt)) + return exit_code + + + class _FixedArgvParser(argparse.ArgumentParser): + def error(self, message: str) -> None: + raise IngressError("CLI_ARGUMENT_ERROR", message) + + + def _parser() -> argparse.ArgumentParser: + parser = _FixedArgvParser( + description="S2_00 deterministic Stage 1 ingress", + add_help=False, + ) + parser.add_argument("--workflow-id", required=True) + parser.add_argument("--release-ref", required=True) + parser.add_argument("--run-id", required=True) + parser.add_argument("--attempt-id", required=True) + parser.add_argument("--user-context-sha256", required=True) + parser.add_argument("--workspace-context-sha256", required=True) + parser.add_argument("--stage1-run-root-ref", required=True) + parser.add_argument("--stage1-deployment-root-ref", required=True) + return parser + + + def _project_root_from_runtime() -> Path: + for candidate in Path(__file__).resolve().parents: + if candidate.name == "Case_02_Comparison_Research": + return candidate + raise IngressError("PROJECT_ROOT_NOT_FOUND", "runtime is not located below the canonical project root") + + + def _resolve_contained_ref( + approved_root: Path, + ref: str, + *, + code: str, + ) -> Path: + if not isinstance(ref, str) or not ref or "\x00" in ref: + raise IngressError(code, "workspace reference is empty or malformed") + candidate = Path(ref) + if not candidate.is_absolute(): + candidate = approved_root / candidate + if candidate.is_symlink(): + raise IngressError(code, "workspace reference may not be a symlink") + approved_resolved = approved_root.resolve(strict=True) + try: + lexical_relative = candidate.relative_to(approved_root) + except ValueError: + lexical_relative = None + if lexical_relative is not None: + _assert_no_symlink_components(approved_resolved, PurePosixPath(lexical_relative.as_posix())) + resolved = candidate.resolve(strict=True) + try: + resolved.relative_to(approved_resolved) + except ValueError as exc: + raise IngressError(code, "workspace reference escaped its approved root") from exc + return resolved + + + def _load_bound_contract_manifest( + stage1_deployment_root: Path, + release_lock: Mapping[str, Any], + ) -> Mapping[str, Any] | None: + dependency = release_lock.get("dependency_locks", {}).get("stage1", {}) + ref = dependency.get("contract_manifest_ref", {}) if isinstance(dependency, dict) else {} + if not isinstance(ref, dict) or not isinstance(ref.get("path"), str): + return None + expected_hash = ref.get("sha256") + if not isinstance(expected_hash, str) or not re.fullmatch(r"[A-Fa-f0-9]{64}", expected_hash): + raise IngressError("CONTRACT_MANIFEST_UNBOUND", "contract manifest hash is not bound") + snapshot = open_bounded_snapshot( + stage1_deployment_root, + ref["path"], + logical_input_id="stage1_contract_manifest", + ) + if snapshot.raw_sha256 != expected_hash: + raise IngressError("CONTRACT_MANIFEST_HASH_MISMATCH", "contract manifest raw hash differs from release") + value = load_json_strict(snapshot) + if not isinstance(value, dict): + raise IngressError("CONTRACT_MANIFEST_SHAPE", "contract manifest must be an object") + return value + + + def main(argv: Sequence[str] | None = None) -> int: + """CLI entrypoint. Stdout contains exactly one canonical JSON object.""" + + try: + args = _parser().parse_args(argv) + if args.workflow_id != "S2_00": + raise IngressError("WORKFLOW_ID_MISMATCH", "runtime accepts only workflow_id S2_00") + asset_root = Path(__file__).resolve().parents[1] + project_root = _project_root_from_runtime() + release_path = _resolve_contained_ref(asset_root, args.release_ref, code="RELEASE_REF_OUTSIDE_ROOT") + if not release_path.is_file(): + raise IngressError("RELEASE_REF_NOT_FILE", "release reference must identify one regular file") + stage1_run_root = _resolve_contained_ref( + project_root, + args.stage1_run_root_ref, + code="STAGE1_RUN_ROOT_OUTSIDE_WORKSPACE", + ) + stage1_deployment_root = _resolve_contained_ref( + project_root, + args.stage1_deployment_root_ref, + code="STAGE1_DEPLOYMENT_ROOT_OUTSIDE_WORKSPACE", + ) + if not stage1_run_root.is_dir() or not stage1_deployment_root.is_dir(): + raise IngressError("STAGE1_ROOT_NOT_DIRECTORY", "Stage 1 roots must be directories") + release_lock = load_release_lock(release_path) + contract_manifest = _load_bound_contract_manifest(stage1_deployment_root, release_lock) + output_dir = project_root / "stage2_runs" / args.run_id + result = execute_ingress( + stage1_run_root, + release_lock, + output_dir=output_dir, + attempt_id=args.attempt_id, + run_id=args.run_id, + user_context_sha256=args.user_context_sha256, + workspace_context_sha256=args.workspace_context_sha256, + stage1_deployment_root=stage1_deployment_root, + stage2_asset_root=asset_root, + contract_manifest=contract_manifest, + ) + sys.stdout.buffer.write(canonical_json_bytes({"ok": True, "result": result})) + return 0 + except (IngressError, FileNotFoundError, PermissionError, OSError) as exc: + error = exc.as_dict() if isinstance(exc, IngressError) else {"code": "OS_ERROR", "message": str(exc)} + sys.stdout.buffer.write(canonical_json_bytes({"ok": False, "error": error})) + return 2 + + + if __name__ == "__main__": + raise SystemExit(run_inline_mcp()) + task_procedure: + IN: + nexts: + - Task_S2_00_deterministic_ingress + wait_until: [] + Task_S2_00_deterministic_ingress: + nexts: + - OUT + wait_until: + - IN + OUT: + nexts: [] + wait_until: + - Task_S2_00_deterministic_ingress diff --git a/Case_02_Comparison_Research/YAML_Prompts/2. Stage_2/MEMORY.md b/Case_02_Comparison_Research/YAML_Prompts/2. Stage_2/MEMORY.md index 04da4dab..1bf0250c 100644 --- a/Case_02_Comparison_Research/YAML_Prompts/2. Stage_2/MEMORY.md +++ b/Case_02_Comparison_Research/YAML_Prompts/2. Stage_2/MEMORY.md @@ -16,6 +16,21 @@ - 하위 에이전트(subagents)를 사용하여 핵심 테마와 교훈을 식별하고 MEMORY.md에 섹션으로 저장하라. - 향후 세션 시작 시 MEMORY.md의 이전 섹션을 참조하라. + +## 2026-09-30 — S2_00 IO 분석·request 준비 계획 및 구현 경계 정정 + +한 줄 요약: S2_00 입출력 문서와 request 준비 계획·단순화 v1을 작성하고 원본 YAML을 보존했으며, request 생성 코드 구현과 backend 운영 연결 검증을 분리해야 한다는 판단으로 정정했다. + +- `Stage_2_00_IO_info.md`: 배포 YAML·분석서·release를 대조하여 DAG, 단계별 파일명·형식·경로, Stage 1 고정 입력 16개·배포 의존 55개·Stage 2 직접 의존 49개, 정상 11+E/diagnostic 5개 출력 및 임시/최종 저장을 문서화했다. 사건별 root·signal 목록·digest·cluster ID는 실제 값이 아닌 결정 규칙으로 구분했다. +- `../../plans/s2-00-request-preparation.md` 및 `_v1.md`: 두-task request 준비 계획을 작성하고, v1은 동일 불변 외부 인자에서 request를 재구성하여 대조하는 방식으로 별도 준비 receipt 전달·중복 검증을 줄였다. 구현 범위는 v1의 Objective 이하이며, 기존 6필드·canonical JSON·검증·localdocs write/read-back·후행 실패 차단·동시 실행 통제·관련 hash 일관성을 유지한다. +- `Default_Agent/Stage_2_Clean/agent_scripts/Stage_2_S2_00_outdated_9_09.yml`에 기존 배포본을 byte-identical로 보존했다(SHA-256 `94c2744d71d222cfb5cc1b2a0d33a6cda20079439823de431eef378a2fa5c943`). 기존 builder `--check`는 `PARITY_PASS`, mismatches=[]였으며 변경 전 기준선만 검증했다. 활성 YAML·authoring/runtime/binding은 변경하지 않았고 request writer 및 신규 YAML 구현·live 실행은 미완료다. +- Stage 1 Default_Agent 자산, v.8 YAML 4개 및 분석 보고서 4개를 확인했다. task 의존·동적 확장·workspace 치환에 관한 단서는 있으나 S2_00 외부 인자 공급·success-only·workspace 직렬화의 backend 구현 전체를 입증하지는 않는다. +- **정정·후속 원칙:** backend 정보 미확인을 이유로 생성 코드 구현까지 중단한 것은 과도했다. 기존 Stage 2 방식과 일관된 생성·검증·저장 함수를 네 외부 인자를 받도록 구현하고 offline 시험하는 데 backend 전체 구현·명세는 필요하지 않다. 실제 인자 주입·실패 차단·동시 실행의 운영 연결은 필요한 backend 정보만 확보하여 별도로 완성·검증한다. 미지원 템플릿·환경변수·잠금 기능을 임의로 가정하거나 offline 성공을 live-ready로 표현하지 않는다. + +## 2026-09-17 — Stage 2 외부 규칙 문서·Weaviate 절차에 관한 의견 평가 + +외부 문서 접근과 임베딩 검색 위임이 가능하다는 가정 아래 현행 배포 YAML 5종·분석서·관련 자산 및 Stage 1 인계 자료를 대조하여 `Evaluaton_on_hogyus_opinion.md`를 작성하였다. C25/C26의 사건종류·규칙 선택과 Markdown 원문 적재·해시 검증, C27~C29의 Weaviate `search_hybrid` 호출·요건사실 팩 구성은 존재하므로 “절차 부재”라는 평가는 정정하되, 승인된 규칙·검색 범위의 미충족, 규칙·청크 본문의 S2_30 전달 공백, 고정된 요건별 gap 자료, V12 필드와 S2_30→40 해시 대상 불일치는 접근 가능성만으로 해결되지 않음을 구분하였다. 외부 공용 자산의 부재를 추정하지 않았으며, 절차·출처 참조의 존재와 실제 내용 활용·요건 충족·실행 및 법률 검증을 분리하는 것이 핵심 교훈이다. 핵심 경로 독립 검토와 문서 링크·해시 확인을 수행하고 해시 전사 오류 1건을 수정하였으며, 실행 자산 변경·live LLM/MCP/Weaviate·E2E·법률검수·기존 회귀시험 재실행은 수행하지 않았다. + ## 2026-09-09 — Stage 2 배포 YAML 5종 전문 분석과 단계 간 계약 차이 기록 한 줄 요약: 배포 YAML 00/10/20/30/40 총 16,100행을 독립 병렬 분담해 전문 분석하고 각 분석서에 정확히 2회 검증·증분수정을 수행했으며, “구현/봉인/과거 offline PASS”와 “현재 의미 인계·live 실행·법률 승인”을 구별해야 한다는 교훈을 확인했다. diff --git a/Case_02_Comparison_Research/YAML_Prompts/2. Stage_2/Stage_2_00_IO_info.md b/Case_02_Comparison_Research/YAML_Prompts/2. Stage_2/Stage_2_00_IO_info.md new file mode 100644 index 00000000..23a492aa --- /dev/null +++ b/Case_02_Comparison_Research/YAML_Prompts/2. Stage_2/Stage_2_00_IO_info.md @@ -0,0 +1,332 @@ +# Stage_2_S2_00.yml 단계별 Input / Output 파일 정보 + +## 1. 범위·기준 및 경로 표기 + +작성 기준일: 2026-09-30. 사용자 지정 [Stage_2_00_Analysis_v1.md](Stage_2_00_Analysis_v1.md)의 DAG·IO·자산 분석을 기준으로, [배포 YAML](Default_Agent/Stage_2_Clean/agent_scripts/Stage_2_S2_00.yml)과 해당 YAML이 참조하는 release/module manifest를 대조하였다. main 폴더 바로 아래의 동명 YAML은 분석 대상이 아니다. `Y:행번호`는 배포 YAML의 행번호이다. + +이 문서는 **파일 계약에 대한 정적 분석**이다. 사건 run이나 MCP 실행을 수행하여 파일 생성 성공을 확인한 문서가 아니다. 현재 release는 `DEV_FIXTURE_RELEASE / DRAFT_NOT_EXECUTABLE`이며 core의 DEV guard가 실제 run을 거절한다. 아래 출력은 해당 실행 경로가 정상적으로 완료되는 경우의 코드상 산출물이다. + +| 기호 | 정확한 위치·결정 방식 | +|---|---| +| `M/` | `prompt_updates_sequential/Case_02_Comparison_Research/YAML_Prompts/2. Stage_2/` — 이 문서 저장 폴더 | +| `R/` | `M/Default_Agent/Stage_2_Clean/` — 로컬 배포 패키지 | +| `W/` | localdocs가 user/workspace session에 결속하는 논리 workspace root. 로컬 Dropbox 절대경로와 동일하다고 가정하지 않음 | +| `A/` | `W/Default_Agent/Stage_2_Clean/` — 실행 시 Stage 2 자산 source root. 로컬 대응 위치는 `R/` | +| `U/` | `W//` — Stage 1 사건 입력 root | +| `D/` | `W//` — Stage 1 고정 배포 source root | +| `O/` | `W/stage2_runs/by-binding//` — 최종 파일 저장 root | +| `T/` | Code Executor의 `TemporaryDirectory(prefix="liti-s2-00-")`로 생성한 임시 root | + +`U/`, `D/`, `T/`, ``, ``의 실제 값은 실행 입력·환경·내용에 따라 결정된다. 분석 자료만으로 실제 사건 폴더명이나 digest를 확정할 수 없으므로, 임의 값 대신 정확한 결정식을 기재한다. 각 표의 **folder + file name**을 이어 붙이면 전체 경로가 된다. 대소문자와 점·밑줄은 원문대로 유지하였다. + +근거: 분석서 §0.1·§2·§6, Y:194–227, 4544–4612, 5548–5591, 5957–5959, 6032–6059. + +분석 YAML SHA-256: `94c2744d71d222cfb5cc1b2a0d33a6cda20079439823de431eef378a2fa5c943`. 분석서에 기록된 YAML hash와 일치한다. 분석서 SHA-256: `905abf5d93d67b7ab7f9f55e0197e320a69c2d909859dcba145e173a13b65253`. + +## 2. 작업 DAG structure + +```mermaid +flowchart TD + IN[Agent IN] --> TASK[Task_S2_00_deterministic_ingress: 단일 code-executor.run_code] + TASK --> H0[MCP 세션 초기화 · request/release 읽기] + H0 --> H1[exact input closure 2회 읽기 · hash/bytes 대조 · 임시 파일 hydration] + H1 --> C00[C00: source/schema/producer/identity/seal · signal ALL 검증] + C00 --> C05[C05: review 정규화 · 보존검사] + C05 --> B[run binding · artifact header] + B --> C10[C10: case/evidence/object/party/slot context] + C10 --> C15A[C15: membership · cluster · SCC · scheduling wave] + C15A --> C15B[C15: slice · bundle/cohort · invariant] + C15B --> ROUTE[executable/residual 분할 · route · manifest/intake/issue/status 조립] + ROUTE --> N[정상: 공통 4 + context 7 + slice E] + ROUTE --> D[diagnostic: 공통 4 + technical diagnostic 1] + N --> L[원본 snapshot 재검사 결과 반영 · output schema 검사 · 임시 tree 저장/rename] + D --> L + L --> V[exact artifact set/hash 재검사] + V --> P[원격 non-status write/read-back · ingress_status 마지막 write/read-back] + P --> OUT[stdout JSON receipt · Agent OUT] + OUT --> EXT[외부 orchestrator: barrier 검증] + EXT --> S10[S2_10 또는 S2_10_WITH_ISSUES] + EXT --> S40[S2_40_STATUS_ONLY] +``` + +원본 snapshot 재검사는 실제 코드에서 route 결정 직후, manifest/intake/status 최종 조립 이전에 수행된다(Y:5114–5117). 도식의 저장 노드는 그 재검사 결과를 전제로 한다. + +Agent의 실제 task edge는 `IN → Task_S2_00_deterministic_ingress → OUT` 하나다. C00·C05·C10·C15는 단일 Python 호출 내부의 논리 단계이다. **C00–C15 사이에는 중간 JSON 파일을 저장하고 다음 단계가 다시 읽는 구조가 없다.** 메모리 객체를 전달하고 마지막에 선택된 출력 집합을 일괄 저장한다. S2_10/S2_40 호출은 YAML 외부 orchestrator의 작업이다. + +예외가 발생하면 `ok:false / FAILED_NO_BARRIER` receipt 및 exit 2로 빠질 수 있다. 이는 diagnostic 5파일이 자동 생성된다는 뜻도, 이미 저장한 원격 파일이 없다는 보장도 아니다. + +근거: 분석서 §1·§4·§6, Y:4801–5243, 5896–6103, 6230–6248. + +## 3. 전체 입력 파일 목록 + +### 3.1 제어·자산 메타데이터 및 출력 검증 schema + +| ID | 정확한 file name | file format | source folder | 사용 단계 | +|---|---|---|---|---| +| CTRL-1 | `s2_00_request.json` | JSON, UTF-8 | `W/stage2_control/` | H0/H1: 6개 request 필드와 source root 결정 | +| CTRL-2 | `stage2_release.json` | JSON | `A/manifest/` | H0/H1 및 C00–C15: raw hash pin·입력 및 bundle 계약 | +| META-1 | `module_manifest.json` | JSON | `A/manifest/` | H1: 아래 schema 3개 선택·hash 검증 | +| SCHEMA-1 | `ingress.schema.json` | JSON Schema (`.json`) | `A/schemas/` | hydration, manifest/intake/status/diagnostic 출력 검증 | +| SCHEMA-2 | `context.schema.json` | JSON Schema (`.json`) | `A/schemas/` | hydration, header 및 context/slice/bundle 검증 | +| SCHEMA-3 | `review_status.schema.json` | JSON Schema (`.json`) | `A/schemas/` | hydration, issue ledger 검증 | + +request는 정확히 `schema_version`, `workflow_id`, `request_id`, `attempt_id`, `stage1_run_root_ref`, `stage1_deployment_root_ref`의 6필드이다. 임의 output root를 받지 않는다. YAML 실행 정의 자체는 `R/agent_scripts/Stage_2_S2_00.yml`(YAML)이며 backend가 그 안의 Python code를 실행한다. 별도 `.py` 파일을 사건 입력으로 읽는 방식이 아니다. + +근거: Y:194–208, 637–682, 5548–5591, 5721–5753. + +### 3.2 Stage 1 사건 고정 입력 16개 + +아래 모두 strict UTF-8 JSON이다. H1에서 전부 필수 읽기 및 임시 복제하고 C00에서 계약 검증한다. 단계별 의미 사용은 §5에 정리한다. source folder는 원격 `U/` 기준이며 core가 읽는 대응 폴더는 `T/stage1_run/`이다. + +| ID | logical input ID | 정확한 file name | file format | source folder | +|---|---|---|---|---| +| I01 | `evidence_indexed` | `evidence_indexed.json` | JSON | `U/` | +| I02 | `evidence_event_candidates` | `evidence_event_candidates.json` | JSON | `U/` | +| I03 | `client_goal` | `client_goal.json` | JSON | `U/` | +| I04 | `domain_screening` | `domain_screening.json` | JSON | `U/routing/` | +| I05 | `domain_activation_manifest` | `domain_activation_manifest.json` | JSON | `U/routing/` | +| I06 | `b1_evidence_indexed_gate` | `B1_evidence_indexed_gate.json` | JSON | `U/quality_gates/` | +| I07 | `b2_event_candidates_gate` | `B2_event_candidates_gate.json` | JSON | `U/quality_gates/` | +| I08 | `stage1_part1_soft_gate_handoff` | `stage1_part1_soft_gate_handoff.json` | JSON | `U/quality_gates/` | +| I09 | `bo` | `BO.json` | JSON | `U/` | +| I10 | `signal_manifest` | `signal_manifest.json` | JSON | `U/signals/` | +| I11 | `stage1_part2_review_handoff` | `stage1_part2_review_handoff.json` | JSON | `U/quality_gates/` | +| I12 | `legal_effect_structures` | `legal_effect_structures.json` | JSON | `U/` | +| I13 | `stage1_part3_review_handoff` | `stage1_part3_review_handoff.json` | JSON | `U/quality_gates/` | +| I14 | `fact_ledger_base` | `Fact_Ledger_base.json` | JSON | `U/` | +| I15 | `fact_ledger_writer_report` | `fact_ledger_writer_report.json` | JSON | `U/stage1_tmp/fact_ledger/` | +| I16 | `stage1_part4_review_handoff` | `stage1_part4_review_handoff.json` | JSON | `U/quality_gates/` | + +### 3.3 manifest 전개 입력·조건부 입력 + +| 입력 | 정확한 file name / path 결정 규칙 | file format | source folder | 현재 확정 범위 | +|---|---|---|---|---| +| signal payload family | `basename(signal_manifest.files[i].path)` | JSON | `U/signals//`(dirname이 비어 있으면 `U/signals/`) | `U/signals/signal_manifest.json`의 모든 `files[]`를 전개. 전체 파일명 확정에는 실제 사건 manifest 필요 | +| SG-01 activation | `domain_activation_manifest.json` | JSON | `U/signals/` | 위 family 내부에서 core가 명시적으로 찾는 파일. `U/routing/`의 동명 파일과 별개 | +| contract manifest | `basename(release.dependency_locks.stage1.contract_manifest_ref.path)` | JSON | `D//` | 현 release는 path=null. 현재 고정 입력에는 추가하지 않음 | +| completion seal | `basename(release.dependency_locks.stage1.completion_seal_ref.path)` | JSON | `U//` | 현 release는 ref 없음. 현재 고정 입력에는 추가하지 않음 | + +signal의 정확한 전체 경로는 `U/signals/`. `files[i].path` 자체에 `signals/` prefix를 포함하면 금지된다. canonical/domain_signal은 의미 처리, compatibility_view는 무결성 검증 대상이다. 디렉터리 scan이나 추측한 고정 파일 목록으로 대체하지 않는다. + +근거: 분석서 §2.1, Y:1420–1591, 4895–4914, 5635–5720, 6162–6183. + +### 3.4 Stage 1 고정 배포 입력 55개 + +source-of-truth: [stage2_release.json](Default_Agent/Stage_2_Clean/manifest/stage2_release.json)의 `/dependency_locks/stage1/concrete_paths`. H1에서 **전부** raw hash를 검증하고 `T/stage1_deployment/`의 같은 상대경로로 복제한다. C00 검증과 C10 slot 구성에서 사용한다. schema 파일도 직렬화 형식은 JSON이다. 근거: 분석서 §3.2, Y:4164–4242, 5670–5698. + +| 번호 | 정확한 file name | file format | source folder | +|---|---|---|---| +| D01 | `runtime_manifest.json` | JSON | `D/` | +| D02 | `_registry_index.json` | JSON | `D/domains/` | +| D03 | `signal_registry.v2.json` | JSON | `D/signals/` | +| D04 | `domain_config.json` | JSON | `D/domains/E-00/` | +| D05 | `domain_config.json` | JSON | `D/domains/E-01/` | +| D06 | `domain_config.json` | JSON | `D/domains/E-02/` | +| D07 | `domain_config.json` | JSON | `D/domains/E-03/` | +| D08 | `domain_config.json` | JSON | `D/domains/E-04/` | +| D09 | `domain_config.json` | JSON | `D/domains/E-05/` | +| D10 | `domain_config.json` | JSON | `D/domains/E-06/` | +| D11 | `domain_config.json` | JSON | `D/domains/E-07/` | +| D12 | `domain_config.json` | JSON | `D/domains/E-08/` | +| D13 | `domain_config.json` | JSON | `D/domains/E-09/` | +| D14 | `domain_config.json` | JSON | `D/domains/E-10/` | +| D15 | `domain_config.json` | JSON | `D/domains/E-11/` | +| D16 | `domain_config.json` | JSON | `D/domains/E-12/` | +| D17 | `domain_config.json` | JSON | `D/domains/E-13/` | +| D18 | `domain_config.json` | JSON | `D/domains/E-14/` | +| D19 | `domain_config.json` | JSON | `D/domains/E-15/` | +| D20 | `domain_config.json` | JSON | `D/domains/E-16/` | +| D21 | `domain_config.json` | JSON | `D/domains/E-17/` | +| D22 | `domain_config.json` | JSON | `D/domains/E-18/` | +| D23 | `domain_config.json` | JSON | `D/domains/E-19/` | +| D24 | `domain_config.json` | JSON | `D/domains/E-20/` | +| D25 | `domain_config.json` | JSON | `D/domains/E-21/` | +| D26 | `domain_config.json` | JSON | `D/domains/EC-00/` | +| D27 | `domain_config.json` | JSON | `D/domains/X1/` | +| D28 | `domain_config.json` | JSON | `D/domains/X2/` | +| D29 | `domain_config.json` | JSON | `D/domains/X3/` | +| D30 | `client_goal_domain_profiles.schema.json` | JSON Schema | `D/platform/schemas/` | +| D31 | `domain_fanout_plan.schema.json` | JSON Schema | `D/platform/schemas/` | +| D32 | `domain_seed_output.schema.v3.json` | JSON Schema | `D/platform/schemas/` | +| D33 | `domain_slice.schema.v2.json` | JSON Schema | `D/platform/schemas/` | +| D34 | `fact_exception_pack.schema.json` | JSON Schema | `D/platform/schemas/` | +| D35 | `fact_ledger_base.schema.json` | JSON Schema | `D/platform/schemas/` | +| D36 | `fact_ledger_candidate_bundle.schema.json` | JSON Schema | `D/platform/schemas/` | +| D37 | `legal_effect_structures.schema.json` | JSON Schema | `D/platform/schemas/` | +| D38 | `structure_seed_bundle.schema.json` | JSON Schema | `D/platform/schemas/` | +| D39 | `evidence_slot_status.schema.json` | JSON Schema | `D/signals/_common/` | +| D40 | `signal_item.schema.json` | JSON Schema | `D/signals/_common/` | +| D41 | `domain_activation_manifest.schema.json` | JSON Schema | `D/signals/schemas/` | +| D42 | `procedural_posture_relief_signals.schema.json` | JSON Schema | `D/signals/schemas/` | +| D43 | `party_capacity_standing_signals.schema.json` | JSON Schema | `D/signals/schemas/` | +| D44 | `governing_law_version_signals.schema.json` | JSON Schema | `D/signals/schemas/` | +| D45 | `legal_relation_lifecycle_signals.schema.json` | JSON Schema | `D/signals/schemas/` | +| D46 | `timeline_notice_condition_signals.schema.json` | JSON Schema | `D/signals/schemas/` | +| D47 | `asset_right_state_signals.schema.json` | JSON Schema | `D/signals/schemas/` | +| D48 | `liability_causation_damage_signals.schema.json` | JSON Schema | `D/signals/schemas/` | +| D49 | `defense_exception_signals.schema.json` | JSON Schema | `D/signals/schemas/` | +| D50 | `evidence_proof_conflict_signals.schema.json` | JSON Schema | `D/signals/schemas/` | +| D51 | `calculation_requirements.schema.json` | JSON Schema | `D/signals/schemas/` | +| D52 | `remedy_enforcement_signals.schema.json` | JSON Schema | `D/signals/schemas/` | +| D53 | `legal_effect_routes.schema.json` | JSON Schema | `D/signals/schemas/` | +| D54 | `domain_signal_envelope.schema.v2.json` | JSON Schema | `D/signals/schemas/` | +| D55 | `signal_manifest.schema.json` | JSON Schema | `D/signals/schemas/` | + +### 3.5 Stage 2 직접 의존 입력 49개 + +source-of-truth: release `/dependency_locks/stage2_direct`. H1이 아래 **49개 전부를 읽고** `T/stage2_asset/`의 같은 상대경로에 복제한다. C15가 S2_10 Agent/binding의 opaque hash와 release가 선택한 context를 결속한다. 읽은 모든 profile을 후속 LLM에 제공한다는 뜻은 아니다. 현 release의 `/bundle/selected_context_refs`는 빈 배열이다. 근거: 분석서 §3.3, Y:3283–3475, 5741–5753. + +| 번호 | 정확한 file name | file format | source folder | +|---|---|---|---| +| A01 | `Stage_2_S2_10.yml` | YAML | `A/agent_scripts/` | +| A02 | `stage2_s2_10_llm_binding.yml` | YAML | `A/deployment/` | +| A03 | `authority_registry.yml` | YAML | `A/registry/authority/` | +| A04 | `authority_release.json` | JSON | `A/manifest/` | +| A05 | `EC-00_contract_general.yml` | YAML | `A/registry/substantive/` | +| A06 | `E-01_juristic_act_validity.yml` | YAML | `A/registry/substantive/` | +| A07 | `E-02_contract_money_claim.yml` | YAML | `A/registry/substantive/` | +| A08 | `E-03_parties_liability_succession.yml` | YAML | `A/registry/substantive/` | +| A09 | `E-04_unjust_enrichment.yml` | YAML | `A/registry/substantive/` | +| A10 | `E-05_tort_general.yml` | YAML | `A/registry/substantive/` | +| A11 | `E-06_professional_liability.yml` | YAML | `A/registry/substantive/` | +| A12 | `E-07_construction_defect.yml` | YAML | `A/registry/substantive/` | +| A13 | `E-08_lease_deposit.yml` | YAML | `A/registry/substantive/` | +| A14 | `E-09_registry_transfer_claims.yml` | YAML | `A/registry/substantive/` | +| A15 | `E-10_secured_registry.yml` | YAML | `A/registry/substantive/` | +| A16 | `E-11_possession_vindication.yml` | YAML | `A/registry/substantive/` | +| A17 | `E-12_co_ownership_boundary.yml` | YAML | `A/registry/substantive/` | +| A18 | `E-13_creditor_preservation.yml` | YAML | `A/registry/substantive/` | +| A19 | `E-14_execution_linked_claims.yml` | YAML | `A/registry/substantive/` | +| A20 | `E-15_succession_family_property.yml` | YAML | `A/registry/substantive/` | +| A21 | `E-16_negotiable_instruments.yml` | YAML | `A/registry/substantive/` | +| A22 | `E-17_labor_wage_claims.yml` | YAML | `A/registry/substantive/` | +| A23 | `E-18_org_resolution_status.yml` | YAML | `A/registry/substantive/` | +| A24 | `E-19_insurance_claims.yml` | YAML | `A/registry/substantive/` | +| A25 | `E-20_ip_claims.yml` | YAML | `A/registry/substantive/` | +| A26 | `E-21_media_personality_rights.yml` | YAML | `A/registry/substantive/` | +| A27 | `E-00_residual_unrouted.yml` | YAML | `A/profiles/crosscut/` | +| A28 | `X1_notice_lifecycle.yml` | YAML | `A/profiles/crosscut/` | +| A29 | `X2_asset_identity_lineage.yml` | YAML | `A/profiles/crosscut/` | +| A30 | `X3_procedure_standing_relief.yml` | YAML | `A/profiles/crosscut/` | +| A31 | `X4_response_admission_defense.yml` | YAML | `A/profiles/crosscut/` | +| A32 | `SL-AUTO_motor_vehicle.yml` | YAML | `A/profiles/special_law/` | +| A33 | `SL-INDUSTRIAL_ACCIDENT.yml` | YAML | `A/profiles/special_law/` | +| A34 | `SL-PRODUCT_LIABILITY.yml` | YAML | `A/profiles/special_law/` | +| A35 | `SL-RESIDENTIAL_LEASE.yml` | YAML | `A/profiles/special_law/` | +| A36 | `SL-COMMERCIAL_LEASE.yml` | YAML | `A/profiles/special_law/` | +| A37 | `SL-LABOR.yml` | YAML | `A/profiles/special_law/` | +| A38 | `SL-STATE_LIABILITY.yml` | YAML | `A/profiles/special_law/` | +| A39 | `SL-IP-PATENT.yml` | YAML | `A/profiles/special_law/` | +| A40 | `SL-IP-COPYRIGHT.yml` | YAML | `A/profiles/special_law/` | +| A41 | `SL-IP-OTHER.yml` | YAML | `A/profiles/special_law/` | +| A42 | `SL-MEDIA.yml` | YAML | `A/profiles/special_law/` | +| A43 | `SL-TRANSPORT_MARITIME.yml` | YAML | `A/profiles/special_law/` | +| A44 | `SL-CONSUMER_CONTRACT.yml` | YAML | `A/profiles/special_law/` | +| A45 | `ACTIO-MORTGAGE.yml` | YAML | `A/profiles/overlays/` | +| A46 | `ACTIO-MORTGAGE-CREATION.yml` | YAML | `A/profiles/overlays/` | +| A47 | `ACTIO-ENCUMBERED-TRANSFER.yml` | YAML | `A/profiles/overlays/` | +| A48 | `ACTIO-PRESERVED-CLAIM-BUNDLE.yml` | YAML | `A/profiles/overlays/` | +| A49 | `ACTIO-DEFENSE-MAP.yml` | YAML | `A/profiles/overlays/` | + +## 4. 최종 출력 파일 목록 + +모든 출력 형식은 **canonical UTF-8 JSON**이다. 표의 `O/`는 최종 원격 저장 root이며, 같은 상대경로가 먼저 `T/core_output/` 아래에 저장된다. `E`는 cohort 검증 이후의 **최종 executable cluster 수**이다. 정상 branch(`TO_S2_10`, `TO_S2_10_WITH_ISSUES`)는 11+E개, diagnostic branch(`TO_S2_40_STATUS_ONLY`)는 정확히 5개다. + +| ID | 정확한 file name | file format | 최종 저장 folder | branch | 생성 책임·내용 | +|---|---|---|---|---|---| +| O01 | `stage1_input_manifest.json` | JSON | `O/ingress/` | 공통 | C00 검사 결과·source snapshot·hydration receipt를 마지막에 조립 | +| O02 | `intake_report.json` | JSON | `O/ingress/` | 공통 | C00/C05 결과·source count·compact review/conservation report | +| O03 | `issue_ledger.base.json` | JSON | `O/review/` | 공통 | 누적 technical issue와 UNMAPPED review를 조립 | +| O04 | `ingress_status.json` | JSON | `O/ingress/` | 공통·마지막 write | route·run binding·exact artifact hash 목록·barrier | +| O05 | `case_context.json` | JSON | `O/context/` | 정상 | C10: fact/BO/LES/evidence/event/signal 등의 참조 context | +| O06 | `evidence_inventory.json` | JSON | `O/context/` | 정상 | C10: evidence-event-fact 연결·provenance | +| O07 | `object_registry.json` | JSON | `O/context/` | 정상 | C10: BO object occurrence·ID·label·lineage | +| O08 | `party_and_title_context.json` | JSON | `O/context/` | 정상 | C10: party occurrence와 title/role 등 context | +| O09 | `slot_crosswalk.json` | JSON | `O/context/` | 정상 | C10: domain config의 slot skeleton | +| O10 | `cluster_plan.json` | JSON | `O/context/` | 정상 | C15: cluster·SCC·wave·executable/residual | +| O11 | `bundle_plan.json` | JSON | `O/context/` | 정상 | C15: selected context·cohort·slice·S2_10 binding | +| O12 | `.json` | JSON | `O/context/cluster_slices/` | 정상·E개 | C15: 최종 executable cluster별 immutable slice | +| O13 | `technical_diagnostic.json` | JSON | `O/ingress/` | diagnostic만 | route 조립: reason·source ref·context_published=false | + +O12의 실제 basename은 코드가 계산한 `cluster_id`에 `.json`을 붙인 값이다. residual/non-executable cluster의 slice는 최종 출력 집합에 포함하지 않는다. diagnostic branch에는 O05–O12가 없다. + +근거: 분석서 §2.2, Y:5180–5243. + +## 5. DAG 각 단계별 Input / Output 연결 + +아래 ID는 §3·§4의 정확한 이름·형식·폴더 행을 참조한다. C00 이후의 '입력'은 원칙적으로 hydration된 파일을 파싱한 메모리 객체이다. '출력 Oxx'는 그 단계에서 준비하는 데이터의 **최종 파일 대응**이며 해당 단계 즉시 저장을 뜻하지 않는다. + +| 단계·작업명 | Input file / 전달 객체 | Output file / 전달 객체와 저장 위치 | 근거 Y | +|---|---|---|---| +| Agent IN·단일 executor 호출 | `R/agent_scripts/Stage_2_S2_00.yml`(YAML); user/workspace hash 및 인증·실행환경은 파일이 아닌 외부 주입 값 | Python 실행 시작. 별도 업무 출력 파일 없음 | 1–43, 6230–6248 | +| H0 MCP 초기화·제어 확정 | CTRL-1, CTRL-2 | request/release 메모리 객체. 별도 request 복사 파일 없음 | 5548–5591, 5757–5824 | +| H1 exact closure 2회 읽기·hydration | CTRL-1/2, META-1, SCHEMA-1–3, I01–I16, signal family, D01–D55, A01–A49; 조건이 충족되면 contract/completion | §6.1에 명시한 임시 파일 복제. `hydration_stability_receipt`는 메모리 객체, 이후 O01에 포함 | 5602–5893 | +| C00 source/schema/producer/identity | I01–I16, release, D01–D55 및 조건부 contract/completion | source contract rows·snapshots·issues. 이후 O01/O02/O03/O04로 조립, 이 단계 저장 없음 | 1084–1392, 4801–4888 | +| C00 signal ALL·activation | I10 `signal_manifest.json`, manifest payload 전부, I05 routing activation, `D/signals/signal_registry.v2.json`, 사건 documents | signal occurrence·semantic/integrity rows·activation 비교 결과. 별도 signal output 파일 없음 | 1420–1726, 4881–4923 | +| C00 cross-artifact seal | 사건 documents/snapshots, 특히 I04/I06/I07/I08 등의 P1 guard와 `D/domains/_registry_index.json`, P2–P4 및 ledger/report | seal 검사 결과·issues. 이후 O01/O02/O03에 반영 | 1740–1782, 4924–4931 | +| C05 review 정규화 | I08/I11/I13/I16(4개 review handoff), CTRL-2의 review mapping | normalized review occurrence·issues; 독립 `normalized_reviews.json`은 없음. compact receipt→O02, UNMAPPED issue→O03 | 1897–2065, 4932–4934 | +| C05 보존검사 | I01–I16 파싱 documents와 snapshots, signal ALL 결과, normalized reviews | conservation checks·issues. compact checks→O02, issue→O03 | 2068–2436, 4935–4941 | +| run binding·artifact header | source/signal/deployment snapshots, release, request/context 값, SCHEMA-2 | binding/header 메모리 객체. O04 내 run_binding_receipt 및 context header 등에 포함 | 4942–4971 | +| C10 case context | 사건 documents(주요 I05/I09/I12/I14 및 evidence/event/source refs), signal ALL, binding/header/issues | O05용 `case_context` 객체 → 최종 `O/context/case_context.json` | 2523–2630, 4988–4994 | +| C10 evidence inventory | I01 `evidence_indexed.json`, I02 `evidence_event_candidates.json` 등 documents·header | O06용 객체 → 최종 `O/context/evidence_inventory.json` | 2633–2685, 4995–4999 | +| C10 object registry | I09 `BO.json`, header | O07용 객체 → 최종 `O/context/object_registry.json` | 2701–2731, 5000–5004 | +| C10 party/title context | I09 `BO.json`, header | O08용 객체 → 최종 `O/context/party_and_title_context.json` | 2734–2779, 5005–5009 | +| C10 slot crosswalk | 사건 documents, D/ 아래 domain config 26개(E-00–E-21, EC-00, X1–X3), header | O09용 객체 → 최종 `O/context/slot_crosswalk.json`; 현재 fact/evidence 배열은 빈 skeleton | 2782–2831, 4972–4977, 5010–5015 | +| C15 membership·cluster·SCC·wave | I09 BO, I14 ledger, I12 LES, I01 evidence, I02 event의 명시 member/relation | O10용 `cluster_plan` 객체. 최종 executable/wave는 cohort 검사 후 갱신 | 2834–3129, 4339–4527, 5016–5017 | +| C15 bounded slice | cluster_plan 및 C10 context 객체, EVENT/LES 원문 row, signal/review/domain-config projection source map | O12용 slice 객체들. 최종 executable만 `O/context/cluster_slices/.json` | 3132–3269, 5018–5054 | +| C15 bundle/cohort·invariant | cluster_plan·slice 객체, CTRL-2 bundle 계약, `A/agent_scripts/Stage_2_S2_10.yml`, `A/deployment/stage2_s2_10_llm_binding.yml`, release-selected context refs(현재 없음) | O11용 bundle_plan, O10 executable/residual/wave 갱신. 입력 Agent/binding은 hash 결속 대상 | 3283–3906, 5055–5113 | +| route·snapshot 재검사·최종 조립 | source rows, C00/C05 검사·review·issue, C10/C15 객체, 원본 임시 snapshot | 정상 O01–O11+O12 E개 또는 diagnostic O01–O04+O13를 메모리 artifact map으로 확정 | 5114–5243 | +| output schema 검사·로컬 tree 발행 | 선택된 artifact map, `T/stage2_asset/schemas/`의 SCHEMA-1–3 | 모든 선택 파일을 `T/.staging/core_output./`에 저장/fsync, status 마지막 저장, `T/core_output/`로 rename | 4015–4108, 5235–5242 | +| exact artifact set/hash 재검사 | `T/core_output/ingress/ingress_status.json` 및 해당 tree의 전체 파일 | 파일 bytes map. 새로운 별도 출력 파일 없음 | 5896–5931 | +| 원격 status-last 발행 | 검증된 local 파일 bytes; 기존 `O/ingress/ingress_status.json`이 있으면 기존 O 전체 파일도 읽어 대조 | 선택된 동일 파일 집합을 O에 저장/read-back; O04 마지막. 기존 집합과 bytes 동일하면 재사용 | 5934–6008 | +| 단일 receipt·Agent OUT | core 결과·logical publication receipt 또는 예외 | stdout JSON 객체 1개. 고정 파일명·저장 폴더 없음 | 6018–6103 | +| 외부 downstream dispatch | O04 barrier 및 O10/O11/O12 등 | S2_10 동적 item 또는 S2_40 status-only 호출. S2_00이 추가 파일을 쓰지 않음 | 분석서 §1·§2.2·§5 | + +C15 표의 projection source map에 signal/review/profile이 준비된다는 것과 실제 slice에 모두 포함된다는 것은 다르다. 현 구현의 member 생성은 BO/FACT/LES/EVIDENCE/EVENT 중심이며, 이 문서는 전체 내용의 무손실 인계를 보증하지 않는다(분석서 §8). + +## 6. 임시 저장·최종 저장·파일이 아닌 출력의 구별 + +### 6.1 hydration 파일의 source → 임시 output + +| 원격 Input | 임시 Output 경로 | file name / format | +|---|---|---| +| `U/<상대경로>`: I01–I16 및 signal family, 조건부 completion | `T/stage1_run/<동일 상대경로>` | 원본 basename 및 JSON bytes 그대로 | +| `D/<상대경로>`: D01–D55, 조건부 contract manifest | `T/stage1_deployment/<동일 상대경로>` | 원본 basename 및 JSON bytes 그대로 | +| `A/<상대경로>`: release/module manifest·schema 3개·direct 49개 | `T/stage2_asset/<동일 상대경로>` | 원본 basename 및 JSON/YAML bytes 그대로 | +| `W/stage2_control/s2_00_request.json` | 임시 파일로 저장하지 않음 | 메모리 request 및 두 read-pass 비교에 사용 | + +두 read-pass는 임시 복제 전에 같은 원격 경로의 bytes를 비교한다. 임시 폴더는 `TemporaryDirectory` 수명 내에서만 사용한다. 여기의 input 복제는 최종 업무 output 개수(11+E 또는 5)에 포함하지 않는다. 근거: Y:5602–5753, 5825–5893. + +### 6.2 최종 저장 순서 + +1. 선택된 모든 output에 schema 검사를 수행한다. +2. `T/.staging/core_output./<출력 상대경로>`에 non-status를 저장/fsync하고 `ingress/ingress_status.json`을 마지막에 저장한다. +3. staging tree를 `T/core_output/`으로 rename한다. +4. exact file set와 raw hash를 다시 검사한다. +5. 기존 원격 barrier가 있으면 binding·status 및 전체 파일 bytes 일치를 검사한다. 일치하면 `IDEMPOTENT_SUCCESS`이다. +6. 기존 barrier가 없으면 `O/<출력 상대경로>`에 non-status 파일별 write/read-back을 마친 뒤 `O/ingress/ingress_status.json`을 마지막에 write/read-back한다. + +로컬 rename과 원격 `STATUS_LAST_LOGICAL_COMMIT`은 서로 다른 저장 보장이다. 저장 후 read-back 또는 close 실패가 발생하면 실패 receipt와 일부 원격 파일이 함께 남을 수 있다. 근거: Y:4015–4108, 5896–6103. + +### 6.3 별도 파일로 오인하면 안 되는 값 + +| 이름 | 실제 형태·위치 | +|---|---| +| `hydration_stability_receipt` | 메모리 객체 및 O01 내부 객체 | +| `run_binding_receipt` | O04 등의 내부 객체 | +| `context_materialization_receipt` | bundle 관련 내부 객체 | +| `output_barrier` | O04 내부 객체 | +| `logical_publish_receipt` | stdout inner receipt 내부 객체 | +| `stage2_s2_00_inner_receipt.v1` | stdout 객체의 schema_version; basename이 아님 | +| `selected_legal_context_json`, `cluster_case_payload_json` | 외부 orchestrator가 만드는 S2_10 item의 문자열 필드 | + +S2_00은 `control/run_status.json`이나 최종 청구취지·청구원인 문안을 생성하지 않는다. 또한 YAML 안에 보존된 별도 `main(argv)` CLI의 `project_root/stage2_runs/` 경로를 현재 inline MCP 실행의 O 경로와 혼동하면 안 된다. 실제 `__main__`은 `run_inline_mcp()`를 호출한다(Y:6190–6232). + +## 7. 식별 결과와 검증 범위 + +- 16개 Stage 1 고정 사건 입력, 55개 Stage 1 배포 입력, 49개 Stage 2 직접 의존 입력의 이름·형식·source folder를 현재 release와 대조하였다. +- 제어 request/release, module manifest 및 출력 검증 schema 3개를 별도로 식별하였다. +- 정상·diagnostic 분기의 출력 목록과 개수, 임시 staging/core_output 및 최종 O 경로를 YAML의 artifact map·publication 코드와 대조하였다. +- signal family의 모든 basename, 조건부 입력의 실제 경로, U/D의 실제 사건 경로, output digest 및 cluster별 basename은 실제 request/manifest/run이 없으므로 값 자체는 미확정이다. 코드가 정한 결정 규칙을 명시하였다. +- `runtime/s2_00_ingress.py`, `.txt` mirror, workflow YAML, offline build/test 파일은 실행·검증 자산이지만 이 inline task가 별도 업무 입력으로 import/실행하는 파일이 아니다. `deployment.schema.json`, prompt cache policy, P00/P10 Markdown 등도 위 실제 hydration 집합에 포함하지 않았다(분석서 §3.1·§3.3). +- 원본 YAML·분석서·release 및 실행 자산은 수정하지 않았다. 문서의 경로·집합 대조는 정적 검증이며, MCP 실행 성공·후속 Agent 실행·법률적 적정성 검증을 뜻하지 않는다. diff --git a/Case_02_Comparison_Research/plans/s2-00-request-preparation.md b/Case_02_Comparison_Research/plans/s2-00-request-preparation.md new file mode 100644 index 00000000..7e969966 --- /dev/null +++ b/Case_02_Comparison_Research/plans/s2-00-request-preparation.md @@ -0,0 +1,262 @@ +# S2_00 request 생성·검증·저장 선행 task 신설 계획 + +## Objective + +`Stage_2_S2_00.yml`의 맨 앞에 `Task_S2_00_prepare_request`를 신설하여 외부 실행 인자를 검증하고 `stage2_control/s2_00_request.json`을 저장·read-back한 뒤, 성공한 동일 실행만 기존 `Task_S2_00_deterministic_ingress`로 진행하도록 한다. workspace 단위 동시 실행 통제, 두 task 간 request identity 결속, 빌드·배포·hash 계약 갱신을 하나의 변경으로 다룬다. + +작성일: 2026-09-30. 상태: **구현 전 계획**. 이 문서 작성 과정에서는 실행 YAML·runtime·배포 자산을 변경하지 않았다. + +## Deliverables + +| 산출물 | 내용 | +|---|---| +| 외부 실행 인자·host 계약 | 인자 전달 방식, workspace 권한·경로 결속, 직렬화, task 실패 차단, receipt 전달 | +| 2-task authoring 및 배포 YAML | prepare → ingress 순서와 success-only 실행 | +| prepare 구현 및 runtime mirror | 입력 검증, request 생성, write/read-back, 준비 receipt | +| ingress 보강 | host가 전달한 준비 receipt의 request hash·실행 identity와 실제 입력 대조 | +| workflow/schema/binding/builder 갱신 | task별 쓰기 권한, 2-task 추출·검증·receipt·mirror 계약 | +| 회귀·통합 검증 증거 | mock, 동시 실행, 실제 backend를 구분한 결과 | +| 분석·IO 문서 개정 | 새로운 선행 단계, 외부 인자 및 실패·잠금 수명 반영 | + +## Scope and Non-Scope + +- 범위: request 준비와 기존 ingress를 연결하는 구조 및 이를 배포 가능한 상태로 검증하는 데 필요한 계약 변경. +- 유지: request 파일의 기존 6필드, 고정 경로, 기존 C00/C05/C10/C15의 업무 책임, 정상/diagnostic 산출물 집합. +- 제외: Stage 1 사건 데이터 수정, S2_10 이후 업무 로직 변경, 기존 DEV release의 자동 production 승격, 기존 보존·projection 결함의 일괄 수정. +- 계획 작성이 구현 또는 live 배포 승인을 뜻하지 않는다. 실행 제한 해제는 request 생성 구현과 별개의 작업이다. + +## Known Inputs + +`M = Case_02_Comparison_Research/YAML_Prompts/2. Stage_2/`, `R = M/Default_Agent/Stage_2_Clean/`, `W = localdocs가 결속한 논리 workspace root`. + +| 확인한 현 상태 | 근거 | +|---|---| +| 현재 IN → ingress → OUT, 단일 task | `R/agent_scripts/Stage_2_S2_00.yml`: 17–43, 6230–6248 | +| request는 읽기 입력이며 6필드 폐쇄형 검증 | 같은 YAML: 5548–5591, 5757 이후 | +| 입력 closure를 두 번 읽고 bytes 비교 후 temp hydration | 같은 YAML: 5825–5893 | +| 빌더가 정확히 1 task·1 run_code·고정 DAG를 강제 | `R/offline_build/build_s2_00_inline_projection.py`: 283–378 | +| authoring 정본은 `M/Stage_2_S2_00_v.2.yml`; 배포 YAML과 mirror는 파생본 | 같은 builder: 30–38, 680–755 | +| 단일 code pointer와 code hash·mirror를 receipt에 기록 | 같은 builder: 680–755 | +| 현재 S2_00 쓰기 root는 `stage2_runs/by-binding//` | `R/deployment/stage2_code_executor_binding.yml`: 47–66 | +| localdocs 계약상 허용 도구는 read_binary_doc/write_binary_file | 같은 binding: 47–66. CAS·분산 잠금 기능이 있다는 근거는 없음 | +| 외부 인자 주입 및 성공 receipt의 task 간 전달에 필요한 backend 구체 API | 현재 확인 범위에서 미확정. 구현 1단계에서 확인해야 함 | + +사용자 지정 기존 분석 자료: [Stage_2_00_Analysis_v1.md](../YAML_Prompts/2.%20Stage_2/Stage_2_00_Analysis_v1.md), [Stage_2_00_IO_info.md](../YAML_Prompts/2.%20Stage_2/Stage_2_00_IO_info.md). + +## Material Assumptions + +1. request는 LLM이 추론하거나 폴더를 scan해서 만드는 값이 아니다. host가 확정한 네 값과 코드상 두 상수로 생성한다. +2. task 프로세스·임시 디렉터리·메모리는 서로 공유된다고 가정하지 않는다. backend가 인증된 실행 context와 선행 receipt를 후행 task에 전달해야 한다. +3. `wait_until`은 실행 순서를 기술하지만, 실패한 선행 task의 후속 실행까지 자동 차단한다는 보장은 별도로 검증한다. +4. 파일 읽기→없으면 lock 파일 쓰기는 원자적 잠금이 아니다. read-back·hash 비교도 상호 배제를 대신하지 않는다. +5. 권고 기본안은 **host가 같은 user/workspace의 전체 S2_00 실행을 직렬화**하는 것이다. 최소 요구 구간인 request 저장 전부터 hydration 완료까지보다 길게 잠금을 유지하여, 초기 구현에서 중간 해제 프로토콜을 추가하지 않는다. + +## Questions That Could Change the Outcome + +구현 착수 시 아래를 backend 문서·코드·시험으로 먼저 확인한다. 답을 추측해 YAML 문법이나 도구 기능을 만들지 않는다. + +| 확인 사항 | 결정 기준·미지원 시 처리 | +|---|---| +| 외부 structured 인자 전달 방식 | 실제 지원하는 executor 입력 채널 사용. 고정된 템플릿 문법·미지원 run_code parameter를 임의 추가하지 않음 | +| 선행 stdout receipt를 후행 입력으로 전달하는 방식 | backend가 실행 identity에 결속해 전달해야 함. 동일 고정 파일에 별도 receipt를 쓴다는 방식만으로 신뢰 경계를 대신하지 않음 | +| task 성공 판정 | transport 성공뿐 아니라 exit code·receipt schema·ok·실행 ID 일치 확인 가능해야 함 | +| workspace 직렬화와 취소·worker 종료 보장 | 여러 backend worker에 걸친 전역 보장 필요. 한 프로세스의 mutex만 있으면 불충분 | +| 저장 권한 강제 | prepare에 정확한 request 경로만 쓰기 허용; ingress 결과 쓰기 권한과 분리 가능한지 확인 | + +미지원 항목이 있으면 해당 backend 기능을 선행 구현 대상으로 기록한다. 로컬/mock 작업은 계속할 수 있으나 backend 증거 없이 live-ready로 표시하지 않는다. + +## Workstreams and Dependencies + +### 1. 기준선 및 변경 영향 확정 + +- 관련 authoring·배포 YAML, runtime mirror, workflow, schema, binding, builder·tests의 hash와 Git 변경 상태를 기록한다. 기존 사용자 변경은 보존한다. +- 문자열뿐 아니라 task ID·code pointer·code hash를 참조하는 소비자를 찾아 변경 목록을 닫는다. +- 기존 검사 결과를 기준선으로 기록하여 이번 변경의 회귀와 기존 실패를 구분한다. +- 위 backend 확인 사항을 검증하고 실제 인자·receipt 주입 방식 및 잠금 구현 위치를 확정한다. + +완료 기준: 정확한 변경 파일 목록, backend 연동 계약, 기존 실패 목록이 기록되어야 한다. + +### 2. 외부 인자 및 request 계약 확정 + +request 출력은 다음 6필드로 고정한다. + +| 필드 | 공급자·형식 | 검증 | +|---|---|---| +| `schema_version` | prepare 코드 상수 | `stage2_s2_00_execution_request.v1` | +| `workflow_id` | prepare 코드 상수 | `S2_00` | +| `request_id` | host 문자열 | `[A-Za-z0-9][A-Za-z0-9._-]{0,127}` | +| `attempt_id` | host 문자열 | 위와 동일. 재시도 정책에 따라 새 시도 식별자 부여 | +| `stage1_run_root_ref` | host 문자열 | W 아래 승인된 Stage 1 사건 root에 결속한 canonical 상대경로 | +| `stage1_deployment_root_ref` | host 문자열 | W 아래 승인된 Stage 1 배포 root에 결속한 canonical 상대경로 | + +- 누락·추가 필드, 빈 값, 잘못된 타입, absolute path, `..`, 역슬래시, NUL, 비정규 NFC 경로 등을 거절한다. 기존 `_inline_relative_path`와 검증 의미를 맞춘다. +- 문법적으로 올바른 경로라는 사실과 해당 실행이 접근할 권한이 있다는 사실을 구별한다. host가 승인한 user/workspace 및 Stage 1 완료 실행과 두 root를 결속한다. +- request 값은 다른 업무 run에서 남은 파일, 작업 디렉터리 추측, LLM 판단으로 채우지 않는다. +- user/workspace·host execution ID·lock ownership·준비 receipt는 별도의 실행 envelope에 담는다. 기존 request JSON에 7번째 필드를 추가하지 않는다. +- structured 전달을 우선한다. 코드 템플릿에 전달해야 하는 환경이라면 검증된 직렬화·안전한 encoding을 사용하고, 입력 문자열을 Python source에 직접 삽입하지 않는다. + +완료 기준: 허용·거절 예제와 인자 전달 계약이 양쪽 task 및 backend 검증에서 일치한다. + +### 3. host 직렬화 및 실패 수명 구현 + +잠금 key는 인증된 `(user, workspace, S2_00 request 고정 경로)`로 정한다. 서로 다른 request_id라도 동일 workspace의 고정 파일을 공유하면 같은 key를 사용한다. + +```text +host admission → workspace 실행 소유권 획득 + → prepare 시작 → request 저장/read-back + → 성공 receipt를 동일 실행 ingress에 전달 + → ingress request 1차/2차 읽기 → temp hydration + → core 실행·원격 출력 발행 → 종료/취소 정리 + → 두 task와 미완료 I/O의 종료 확인 → 소유권 해제 +``` + +- 기본안은 전체 S2_00 종료까지 직렬화한다. 따라서 최소 보호 구간인 hydration 완료까지는 확실히 포함된다. +- prepare에서 획득한 프로세스 로컬 lock을 task 종료 때 해제하는 구조는 금지한다. 소유권은 host 실행 전체에 귀속된다. +- 모든 request writer와 재시도 경로가 같은 통제에 참여해야 한다. 통제 밖 writer는 권한으로 차단한다. +- timeout·취소·host 재시작 시 오래된 worker와 진행 중 쓰기가 끝났는지 확인한 후 다음 실행을 허용한다. TTL 만료만으로 새 소유자를 실행하지 않는다. +- lease 방식이 불가피하다면 stale owner의 쓰기/읽기를 실제로 거절하는 fencing 또는 동등한 host 보장을 구현한다. JSON에 token만 넣는 것은 fencing이 아니다. +- 이후 성능 최적화로 hydration 직후 해제하려면 authenticated hydration-complete 신호, 이후 request 재읽기 없음, 원본 Stage 1 run의 불변성 등을 별도 입증한다. 초기 버전에는 적용하지 않는다. + +완료 기준: 두 host worker의 경쟁·지연·취소 시험에서도 request 교차 소비가 없어야 한다. + +### 4. 선행 task 구현 + +신규 task ID: `Task_S2_00_prepare_request`. + +1. 인증된 host envelope, 실행 소유권, 네 외부 인자를 검증한다. +2. 고정 두 상수를 추가하여 6필드 request 객체를 만든다. +3. strict JSON/schema 및 경로·ID 검증을 수행한다. schema 자산을 읽는 경우 release/module manifest의 hash 결속된 정확한 closure를 사용한다. +4. 기존 canonical JSON 규칙(UTF-8, 정렬 key, 고정 구분자, NaN 금지, 종료 newline)으로 직렬화하고 raw SHA-256을 계산한다. +5. 소유권을 유지한 상태에서 `W/stage2_control/s2_00_request.json`에 binary write를 수행한다. 현재 계약에 없는 원격 rename·원자적 파일 교체 기능을 가정하지 않는다. +6. 같은 경로를 binary read-back하여 bytes가 동일한지 확인하고, 다시 parse/schema 검증한다. +7. 성공일 때만 stdout 준비 receipt를 반환한다. 권고 schema는 신규 `stage2_s2_00_prepare_request_receipt.v1`이며 `ok`, request path/hash/byte length, request_id/attempt_id, host execution binding을 포함한다. 이것은 **신규 설계**이며 기존 파일이라고 표현하지 않는다. +8. 오류·write 응답 유실·read-back 불일치·close 실패 등은 성공 receipt로 취급하지 않는다. ingress는 실행하지 않는다. 부분 파일이 남아 있어도 성공한 것으로 간주하지 않는다. + +동일 시도 재실행은 같은 bytes 및 host identity가 확인되면 재사용 가능하다. 새 시도/다른 request로 교체할 때에는 이전 실행 종료 및 새 소유권이 먼저 확인되어야 한다. 성공 후 무조건 request를 삭제하는 정리 코드는 두지 않는다. + +완료 기준: 실제 저장 bytes와 receipt hash가 일치하고, 모든 실패 경로가 후행 실행을 차단한다. + +### 5. ingress 결속 및 DAG 변경 + +```text +IN + → Task_S2_00_prepare_request + → [host: exit=0 + 유효한 ok:true receipt + 동일 실행 binding 확인] + → Task_S2_00_deterministic_ingress + → OUT +``` + +- YAML `task_procedure`에 prepare의 `wait_until: [IN]`, ingress의 `wait_until: [Task_S2_00_prepare_request]`를 설정하고 nexts를 대응시킨다. stage prevs/nexts는 기존 독립 stage 구조를 유지한다. +- host의 success-only 조건은 YAML edge와 별도로 계약 및 시험에 포함한다. 실패한 prepare가 '완료'되었다는 이유로 ingress를 시작하지 않는다. +- ingress는 host가 제공한 준비 receipt가 없거나 잘못되면 시작을 거절한다. 최초 request bytes와 expected hash, request_id/attempt_id를 대조한다. +- 기존 두 read-pass의 동등성 검사를 유지한다. 두 번째 request bytes도 준비 receipt hash와 일치해야 한다. 동일하게 바뀐 다른 request를 두 번 읽어 통과하는 상황까지 차단한다. +- 신뢰할 수 없는 stale receipt·다른 workspace receipt·다른 attempt의 receipt를 거절한다. +- 기존 C00–C15, output barrier·route 계약은 유지한다. wrapper 실패와 core diagnostic 분기를 혼동하지 않는다. + +완료 기준: prepare와 ingress가 같은 request를 사용했다는 검증 가능한 연결이 생긴다. + +### 6. authoring·빌드·배포 계약 갱신 + +| 파일·영역 | 작업 | +|---|---| +| `M/Stage_2_S2_00_v.2.yml` | authoring 정본에서 2-task 구조·설명·version·inline source 갱신 | +| `R/agent_scripts/Stage_2_S2_00.yml` | 수정한 builder로 배포본 생성. 배포본만 수작업 수정하지 않음 | +| `R/offline_build/build_s2_00_inline_projection.py/.txt` | '정확히 1개' 검증을 **정확히 두 개의 지정 task와 지정 순서**로 변경; 추가 임의 task는 거절 | +| 같은 builder의 code 검사 | 공통 보안·endpoint·import 검사는 유지; ingress 전용 token과 prepare 전용 token을 분리 | +| `R/runtime/s2_00_ingress.py/.txt` | ingress task code의 파생 mirror로 유지 | +| 신규 `R/runtime/s2_00_prepare_request.py/.txt` | prepare code의 파생 mirror 제안. 두 task를 결합한 실행 파일로 만들지 않음 | +| `R/manifest/s2_00_inline_code_receipt.json` | task별 code pointer·hash·mirror, 전체 task semantics hash로 확장; schema version/소비자 동시 갱신 | +| `R/workflows/S2_00_stage1_ingress_normalize_and_bundle_compile.yml` | 선행 단계, 외부 인자, receipt·success gate·직렬화 계약 반영 | +| `R/deployment/stage2_code_executor_binding.yml` | task별 executor 및 권한: prepare는 request 고정 경로, ingress는 기존 결과 root. schema/release 읽기 경로도 task별 명시 | +| `R/schemas/ingress.schema.json` | 기존 6필드 execution_request 유지, 신규 prepare/envelope receipt 검증 정의 | +| `R/schemas/deployment.schema.json` 및 관련 schema | 두 task·task별 code binding·receipt 계약과 실제 소비자 호환성 반영 | +| `R/manifest/module_manifest.json`, `stage2_release.json`, release validator | 변경 자산의 등록·hash와 추가된 계약 검증 반영 | +| `M/Stage_2_00_Analysis_v1.md`, `Stage_2_00_IO_info.md` | 구현 완료 후 실제 구조·IO·검증 상태로 개정 | + +`task index=0`에 ingress가 있다고 가정하는 코드를 모두 확인한다. task_name으로 추출하고 receipt에는 실제 YAML pointer도 남긴다. version 값은 관련 schema와 소비자 호환성을 확인해 함께 올린다. + +### 7. hash 재결속 및 패키지 확정 + +- 변경 자산 → module/release → inline release pin → authoring projection/mirror/receipt → executor binding/detached admission의 실제 참조 그래프를 먼저 만든다. +- 현재 builder는 projection 이후 binding 재결속을 전제로 한다. 기존 detached 설계를 유지하고, parent release에 자신을 포함한 code hash를 되먹이는 순환 의존을 새로 만들지 않는다. +- 참조 그래프의 위상 순서대로 재생성한다. 순환이 발견되면 hash를 반복 덮어쓰는 대신 detached boundary를 설계·검증한다. +- 현재 `expected_release_sha256`, task별 code hash, Agent raw hash, workflow/schema/module hash와 receipt pointer를 함께 대조한다. +- 재생성 뒤 빌더 `--check`, release validator, mirror parity를 실행한다. 두 번째 빌드에서 bytes 차이가 없어야 한다. +- 테스트용 release와 실제 배포 release를 분리한다. DEV guard를 삭제하거나 production 표기를 붙여 검사를 통과시키지 않는다. + +완료 기준: 참조된 변경 자산의 hash가 모두 일치하고, 반복 빌드가 추가 변경을 만들지 않는다. + +## Source and Tool Plan + +로컬 authoring·builder·runtime·tests를 우선 읽는다. backend 관련 문서·코드는 필요한 범위에서 확인하고 지원 여부를 증거로 남긴다. 신규 의존성 설치나 서비스 호출은 이 계획 작성에 포함하지 않는다. 구현 시 기존 도구·검증 runner를 확인해 사용하고, 라이브 시험은 별도 격리 workspace와 비실사건 fixture로 수행한다. + +## Validation Plan + +| 시험 | 통과 기준 | +|---|---| +| 유효한 인자 | request 정확히 6필드, canonical bytes/hash/read-back 일치 | +| 누락·추가·타입·ID·경로 오류 | 저장 전 거절, ingress 호출 0회 | +| 권한 밖 경로·다른 workspace | host 결속 검증 실패, 업무 입력 읽기 불가 | +| 저장 실패·응답 유실·read-back mismatch | 성공 receipt 없음, ingress 호출 0회 | +| 선행 task exit 0이나 ok:false/깨진 receipt | host가 후행 task를 실행하지 않음 | +| 준비 후 request 변조 | 첫 읽기 expected hash 검증 또는 두 pass 비교에서 거절 | +| 두 pass 모두 동일한 타 요청으로 교체 | 준비 receipt hash/identity 검증에서 거절 | +| 같은 workspace 두 요청·두 worker | 단일 소유자만 진입, 다른 요청 bytes를 소비하지 않음 | +| 다른 workspace 병렬 실행 | 불필요한 전역 직렬화 없이 독립 진행 | +| cancel·timeout·host crash·오래된 worker 재개 | 이전 쓰기 가능성이 사라지기 전 새 owner 쓰기 금지; stale owner 차단 | +| 동일 시도 retry / 새 attempt | 정의한 재사용/교체 정책 준수, 기존 output idempotency 계약 유지 | +| task shape / builder | 지정 두 task와 순서만 허용; task 누락·추가·우회 edge 거절 | +| 두 code mirror·receipt pointer·hash | authoring에서 추출한 bytes와 정확히 일치 | +| 기존 core 정상·diagnostic 회귀 | fixture 기준 11+E/5 출력 및 barrier 계약 보존 | +| 실제 backend 통합 | 인자 주입·receipt 전달·failure gate·workspace 격리를 실제 환경에서 확인 | + +mock 통과는 실제 host 잠금·권한 강제 증거가 아니다. `STATIC_PASS`, `OFFLINE_TEST_PASS`, `BACKEND_INTEGRATION_PASS`, `LIVE_ADMISSION_PENDING`을 구분해 보고한다. 현재 DEV release로 정상 core 성공을 주장하지 않는다. + +## Approval Boundaries + +현재 요청은 작업 계획 작성이다. 이 단계에서는 계획 문서만 생성한다. 이후 구현 요청 시 범위 내 로컬 수정·시험을 수행하고, 실제 서비스 배포·외부 쓰기·live activation은 그때의 승인 범위를 확인한다. backend 기능 미지원은 기술적 미완료 조건으로 기록하며 이를 단순 승인 문제로 대체하지 않는다. + +## Progress + +- [x] 현재 단일 task·request read·builder 제약 확인 +- [x] 현재 write root와 localdocs 도구 계약 확인 +- [x] 작업 순서·의존성·동시 실행·검증 계획 작성 +- [ ] backend 전달·직렬화 기능 확인 및 연동 방식 확정 +- [ ] schema·prepare·ingress·DAG 구현 +- [ ] builder·binding·hash 재결속 +- [ ] offline·동시 실행·backend 통합 검증 +- [ ] 분석·IO 문서 갱신 및 배포 적격성 판정 + +## Decision Log + +| 날짜 | 결정 | 근거 | 결과 | +|---|---|---|---| +| 2026-09-30 | request의 기존 6필드 유지 | 현재 strict validator와 사용자 요구 | 실행 보조 정보는 host envelope로 분리 | +| 2026-09-30 | 첫 버전은 전체 S2_00 host 직렬화 | 두 독립 task 사이 lock 수명 및 고정 request 경로 | hydration 완료 전 해제 위험과 중간 callback 의존 감소 | +| 2026-09-30 | 준비 receipt hash를 ingress에 전달 | 두 read-pass 비교만으로 다른 요청 교체를 모두 검출할 수 없음 | 소비할 request를 선행 실행에 결속 | +| 2026-09-30 | authoring부터 재생성 | builder가 정본 및 mirror parity를 관리 | 배포 YAML 단독 변경 방지 | + +## Evidence Ledger + +| 명제 | 확인 근거 | 상태 | +|---|---|---| +| 기존 builder는 두 task를 거절한다 | extract_run_code_task의 EXACTLY_ONE_TASK_REQUIRED·EXACTLY_ONE_RUN_CODE_REQUIRED | 확인 | +| 현재 request 저장은 쓰기 허용 root 밖이다 | S2_00 binding의 write_root_rule | 계약에서 확인; live 강제 여부는 미확인 | +| localdocs 원자적 lock/CAS 지원 | 현재 allowlist에 read/write만 있음 | 지원 근거 없음, 사용 가능하다고 가정하지 않음 | +| 선행 task를 현재 backend가 어떻게 인자 결속하는가 | 구체 backend 구현을 아직 확인하지 않음 | 미확정 | +| request writer의 생성 필요 | 현재 소비 코드 및 이전 폴더 전체 검색 결과 | 현 패키지에 실사용 writer 없음 | + +## Risks and Failure Modes + +- YAML만 바꾸면 builder·code pointer·권한·hash 계약에서 실패한다. +- `wait_until`만 추가하면 prepare 실패 후 ingress가 진행될 수 있다. +- 고정 request 파일에 write/read-back만 구현하면 동시 요청이 서로 덮어쓸 수 있다. +- lock TTL 이후 오래된 writer가 살아 있으면 새 요청을 오염시킬 수 있다. +- 기존 release 제한을 이번 변경의 실패로 혼동하거나, 이를 제거해 정상 실행을 과장할 수 있다. +- 다른 request가 같은 run binding으로 귀결될 때 기존 status bytes 일치 요건은 여전히 적용된다. 기존 출력 충돌을 자동 덮어쓰기로 해소하지 않는다. + +롤백은 두 task 실행을 먼저 정지하고 진행 중 I/O 종료를 확인한 뒤, authoring·배포본·mirror·schema·binding·receipt를 검증된 동일 버전 묶음으로 복원한다. 이전 버전도 외부 request 공급 없이는 실행되지 않으므로 롤백 자체를 서비스 정상화로 보지 않는다. 사건 입력·이미 발행된 결과는 일괄 삭제하지 않는다. + +## Results and Residual Uncertainty + +계획은 작성 완료하였고, 구현은 미착수다. 가장 중요한 미확정 항목은 backend의 structured 인자/receipt 전달과 workspace 단위 직렬화·worker 종료 통제이다. 이 기능이 확인되거나 구현되어야 안전한 2-task 실행이 완성된다. 완료 판정은 request 생성 성공뿐 아니라 **동일 요청의 후행 소비, 실패 시 차단, 동시 실행 안전성, 빌드·hash 일치**를 모두 통과하는 것으로 한다. 실제 backend 시험 전에는 live-ready로 판정하지 않는다. diff --git a/Case_02_Comparison_Research/plans/s2-00-request-preparation_v1.md b/Case_02_Comparison_Research/plans/s2-00-request-preparation_v1.md new file mode 100644 index 00000000..4ea4c4c4 --- /dev/null +++ b/Case_02_Comparison_Research/plans/s2-00-request-preparation_v1.md @@ -0,0 +1,206 @@ +# S2_00 request 준비 계획 v1 — 필수 조건을 유지한 단순화 + +작성일: 2026-09-30. 상태: **분석·계획 작성 완료 / 구현 미착수**. +원본: [s2-00-request-preparation.md](s2-00-request-preparation.md). 원본은 보존한다. + +## 1. Occam's Razor 분석 + +**기존 계획은 안전성에 필요한 항목과 선택적인 구현을 함께 필수화하여, 가장 단순한 계획이라고 보기는 어렵다.** 같은 요구를 충족하면서 새 프로토콜·검증 중복·변경 범위를 줄일 수 있다. 다만 backend 기능이 미확정이므로 '절대적으로 가장 효율적'이라고 단정하지 않고, 확인된 코드와 사용자 제약 아래의 권고안을 제시한다. 여기서 효율성은 우선 구현·검증·유지보수 비용이며, 실행시간 개선 수치는 측정하지 않았다. + +### 1.1 제약을 지키는 단순화와 구조 변경의 구분 + +| 대안 | 장점·비용 | 판단 | +|---|---|---| +| 원본의 독립 2-task + 준비 receipt 전달 | 별도 receipt schema·실행 envelope·후행 전달·task별 검증 필요 | 요청 identity 확인은 타당하지만 구체 수단을 줄일 수 있음 | +| **2-task + 동일한 불변 host 인자 공급** | 두 task가 같은 값으로 request를 재구성; 별도 준비 receipt 전달 불필요. 2-task 빌드 변경은 여전히 필요 | **v1 채택**: 기존 '선행 task 신설' 요구 유지 | +| 기존 단일 task 맨 앞의 준비 함수 | task 간 전달·실패 gate·2-task builder 변경을 제거할 수 있어 구조적으로 더 단순 | 독립 선행 task 요구를 완화하는 대안. v1에 몰래 적용하지 않음 | +| 외부 host에서 request 생성 | YAML 구조 변경이 작음 | 'YAML 맨 앞의 선행 task' 요구와 다르므로 기본안 제외 | + +v1은 요청 목적을 바꾸지 않고 독립 선행 task를 유지한다. 단일 task 내부 함수 방식은 추가 구조를 가장 크게 줄이지만, 기존에 명시한 두-task 조건과 구별해야 한다. + +### 1.2 줄일 수 있는 항목과 유지할 항목 + +| 원본 계획 | v1 결정 | 이유·적용 조건 | +|---|---|---| +| 신규 prepare receipt schema와 host→ingress receipt 전달 | **제거**. host가 확정한 동일 네 인자를 두 task에 제공하고 ingress가 expected bytes를 재구성 | 별도 파생 자료를 전달하지 않아도 요청 일치 검증 가능. host 인자는 실행 중 불변이어야 함 | +| 새로운 execution envelope·lock token을 task마다 검증 | **별도 형식 신설 안 함**. 기존 인증된 실행 context와 host 직렬화 사용 | 기존 기능의 실제 지원 여부부터 확인. 없는 기능을 있다고 가정하지 않음 | +| 검증한 bytes를 write/read-back한 뒤 동일 bytes를 다시 parse/schema 검사 | **반복 검사 제거** | 처음 검증한 canonical bytes와 read-back bytes의 완전 일치가 확인되면 같은 내용을 다시 검증할 필요 없음 | +| 선행 단계에서 schema/release closure 별도 hydration | **기본안에서 제외** | 이미 존재하는 폐쇄형 request validator와 schema의 일치를 offline test로 확인. ingress의 기존 release 검증은 유지 | +| 신규 분산 lock·lease·fencing 체계 가능성까지 초기 구현에 포함 | **기존 host 직렬화 재사용 우선**, 별도 체계는 실제 미지원일 때만 재설계 | 잠금 안전성은 필수이나 새 잠금 서비스를 만드는 것이 필수는 아님 | +| 동일 attempt 재사용 최적화 | **초기 제외**. 소유권 획득 후 전체 준비 작업 재실행 | 작은 JSON의 재쓰기보다 별도 재사용 분기의 관리 비용이 클 수 있음. 기존 결과 idempotency는 유지 | +| schema·manifest 전체 개정 | **실제 구조·hash 참조가 바뀌는 항목만 갱신** | 파일을 더 적게 수정하려고 필요한 hash/권한 갱신을 생략하지 않음 | +| 정확히 2 task·순서 및 두 code hash 검증 | **유지** | 현 builder가 1 task를 강제하므로 피할 수 없는 변경 | +| 저장 실패 시 차단·request 2회 읽기·host 직렬화 | **유지** | 요구된 실행 안전성. bytes 비교는 상호 배제를 대신하지 않음 | + +특히 runtime의 `write_binary_verified()`에는 이미 write 후 read-back bytes 비교가 있다. 같은 일을 하는 새 IO 계층을 만들 필요가 없다. `_inline_validate_request()`도 이미 정확히 6필드·ID·경로 검증을 수행한다. 근거는 아래 Evidence Ledger에 기록한다. + +## Objective + +`IN → Task_S2_00_prepare_request → Task_S2_00_deterministic_ingress → OUT`을 구현하되, 새 준비 receipt 전달 프로토콜 없이 동일한 외부 인자와 기존 검증·IO 기능으로 request 생성 및 소비를 연결한다. + +## Deliverables + +- 기존 authoring을 정본으로 하는 2-task YAML·파생 배포본 및 필요한 code mirror/빌드 receipt 변경. +- 최소 준비 함수, ingress의 expected request bytes 대조, 외부 인자·직렬화·success-only 계약. +- 실제 변경된 binding·schema·manifest/hash, 관련 시험 및 분석·IO 문서 개정. + +## Scope and Non-Scope + +request의 6필드·고정 경로와 기존 ingress의 C00–C15·출력 계약을 유지한다. 신규 receipt 파일·범용 workflow framework·독립 lock 서비스·release 전체 재설계는 기본 범위에 넣지 않는다. 기존 DEV release 제한과 core 결함 수정은 별개다. 이 문서는 구현 계획이며 실행 자산을 변경하지 않는다. + +## Known Inputs + +`M = Case_02_Comparison_Research/YAML_Prompts/2. Stage_2/`, `R = M/Default_Agent/Stage_2_Clean/`, `W = 인증된 localdocs workspace`. + +현재 authoring은 `M/Stage_2_S2_00_v.2.yml`, 배포본은 `R/agent_scripts/Stage_2_S2_00.yml`이다. 빌더는 1 task·1 code pointer를 전제로 하고, S2_00 binding은 결과 root만 쓰기 허용한다. request writer를 추가하려면 이 두 계약 변경은 불가피하다. backend의 실제 인자 전달·직렬화 구현은 **UNVERIFIED**다. + +## Material Assumptions + +- host가 권한을 확인한 네 인자를 실행 시작 때 확정하고 두 task에 동일하게 공급할 수 있어야 한다. YAML 문법·환경변수 이름·run_code parameter를 임의로 발명하지 않는다. +- 같은 `(user, workspace)`의 S2_00 실행은 host가 전역 직렬화한다. 단일 worker의 로컬 mutex만으로 다중 worker 안전성을 주장하지 않는다. +- task 간 메모리 공유는 가정하지 않는다. request bytes는 각 task가 동일한 규칙으로 재구성한다. +- 동일한 bytes의 request가 남아 있다는 사실만으로 prepare 성공을 대신하지 않는다. 현재 host 실행에서 prepare 성공 판정이 먼저 있어야 한다. + +## Questions That Could Change the Outcome + +구현 전에 **세 가지**만 먼저 확인한다: ① 동일 불변 인자를 두 task에 공급하는 실제 방식, ② exit code와 task 결과에 따른 success-only 후행 실행, ③ 취소·worker 장애·진행 중 I/O까지 포함한 workspace 직렬화. 세 조건이 지원되면 아래 최소안을 진행한다. 미지원이면 필요한 backend 보강을 명시하며, receipt나 lock 파일을 추가하는 것으로 문제가 해결되었다고 간주하지 않는다. 기존 지원만으로 가능하다는 현재의 실증 근거는 없다. + +## Workstreams and Dependencies + +### 1단계 — 외부 입력과 host 경계 확정 + +외부에서 공급하는 값은 다음 네 문자열이다. + +| 필드 | 검증 | +|---|---| +| `request_id` | `[A-Za-z0-9][A-Za-z0-9._-]{0,127}` | +| `attempt_id` | 위와 동일; host 재시도 정책으로 결정 | +| `stage1_run_root_ref` | W 아래 승인된 Stage 1 사건 root의 NFC canonical 상대경로 | +| `stage1_deployment_root_ref` | W 아래 승인된 Stage 1 배포 root의 NFC canonical 상대경로 | + +준비 task가 `schema_version=stage2_s2_00_execution_request.v1`, `workflow_id=S2_00`을 추가하여 정확히 6필드를 만든다. 누락·추가 외부 키와 잘못된 타입을 거절한다. 기존 path validator로 absolute path·`..`·역슬래시·NUL·비정규 경로를 거절하고, 접근 권한과 Stage 1 run 선택은 host가 결속한다. shell/Python source에 원문 인자를 직접 삽입하지 않는다. + +host는 준비 task 시작 전 실행 소유권을 획득하여 **S2_00 전체 종료 및 미완료 I/O 종료까지** 유지한다. 이는 hydration 완료까지의 최소 보호 요구를 충족한다. 중간 해제 callback은 추가하지 않는다. 모든 writer·retry가 같은 통제에 참여해야 하며, 종료 여부가 불명확한 경우 다음 실행을 보류한다. TTL 만료만으로 takeover하지 않는다. host에 이 보장이 없으면 live-ready로 판정하지 않는다. + +### 2단계 — 최소 준비 함수와 ingress 비교 구현 + +아래는 의미를 설명하는 pseudocode이며 현재 실행 가능한 API가 아니다. + +```python +# 양쪽 task에 host가 동일하게 공급한 네 값 +expected = build_request(immutable_host_args) # 두 상수 추가, 정확한 외부 키 확인 +expected_raw = canonical_json_bytes(expected) +_inline_validate_request(expected_raw) + +# prepare task: 인증된 localdocs 세션 및 host 실행 소유권 아래 수행 +localdocs.write_binary_verified(INLINE_REQUEST_PATH, expected_raw) +# 세션 종료까지 성공한 후 task 성공을 반환한다. + +# ingress task: 현재 실행의 host_args로 expected_raw를 독립 재구성 +actual_raw = localdocs.read_binary(INLINE_REQUEST_PATH) # 기존 첫 읽기에 통합 +if actual_raw != expected_raw: + raise IngressError("RUN_REQUEST_INPUT_MISMATCH", "request differs from host input") +# 이후 기존 strict validation 및 2회 읽기/hydration을 계속한다. +``` + +- `build_request`는 네 값에 두 상수를 추가하는 작은 함수로 제한한다. 별도의 범용 serializer나 validator framework를 만들지 않는다. +- prepare는 기존 validator·canonical serializer·MCP IO helper의 동작을 재사용한다. 두 독립 inline payload에 필요한 helper는 authoring/build 시 포함한다. 외부 `.py` runtime import나 ingress 전체 실행 코드를 복제해 실행하는 방식은 피한다. 중복 포함된 helper의 일치는 좁은 parity 검사로 확인하며 새 공통 모듈 배포 체계까지 만들지 않는다. +- `write_binary_verified`가 bytes 일치를 확인하므로 추가 read-back/parse를 중복 수행하지 않는다. task 성공 결과에는 backend가 요구하는 최소 `ok`/error만 사용하며 새 버전 receipt schema·추가 파일은 만들지 않는다. +- ingress의 expected bytes 검사는 기존 첫 request 읽기에 결합한다. 이미 첫 읽기가 expected와 같고 두 번째가 첫 번째와 같은지 검사하므로 두 번째에서 동일한 expected hash를 다시 계산할 필요는 없다. +- 이 대조는 request ID만이 아니라 여섯 필드 전체를 확인한다. 서로 다른 요청으로 두 pass 모두 교체되더라도 거절한다. 다만 상호 배제는 여전히 host 책임이다. +- 실패 시 준비 작업 전체를 다시 실행한다. 남아 있는 파일을 신뢰해 ingress만 우회 실행하지 않는다. 정리 목적으로 request를 무조건 삭제하지 않는다. + +### 3단계 — 두 task의 실행 순서와 실패 차단 + +```text +host: 동일 workspace 실행 소유권 확보, 네 인자 확정 + → IN + → prepare: 검증 → 저장/read-back → 세션 종료 → 성공 + → host: 현재 실행의 prepare 성공일 때만 ingress 시작 + → ingress: expected bytes 일치 → 기존 2회 읽기/hydration → 기존 core/발행 + → OUT 또는 실패 정리 + → host: task와 미완료 I/O 종료 확인 후 소유권 해제 +``` + +YAML의 nexts/wait_until은 위 순서를 정확히 기술한다. transport 응답 성공만으로 업무 성공을 판정하지 않는다. exit code 및 backend가 해석하는 task 결과에서 실패·잘못된 결과·timeout이면 ingress를 호출하지 않는다. 후행에 준비 receipt를 전달할 필요는 없지만, host의 task 성공 판정 자체는 생략할 수 없다. 직접 ingress 호출은 host admission으로 차단한다. + +### 4단계 — 실제 영향 범위만 빌드·배포 갱신 + +| 항목 | 최소 변경 | +|---|---| +| `M/Stage_2_S2_00_v.2.yml` | prepare task와 지정 DAG, ingress expected bytes 검사 추가 | +| `R/offline_build/build_s2_00_inline_projection.py/.txt` | 정확히 두 task를 ID로 추출·검증. 0번 task가 ingress라는 가정 제거. 공통 코드 검사와 task별 필수 검사 분리 | +| `R/agent_scripts/Stage_2_S2_00.yml`, runtime mirror, `R/manifest/s2_00_inline_code_receipt.json` | 기존 파생 절차로 재생성. 두 code pointer/hash를 결속. 별도 runtime 준비 receipt와 혼동하지 않음 | +| prepare code mirror | 기존 mirror 정책을 유지하는 데 필요한 `runtime/s2_00_prepare_request.py/.txt`만 파생 생성. 새로운 독립 서비스는 아님 | +| workflow 및 executor binding | 외부 네 인자, success-only, 직렬화 책임과 prepare의 정확한 request 쓰기 경로 추가. ingress의 결과 root 쓰기 범위 유지 | +| 관련 schema | 2-task binding/빌드 receipt 표현이 바뀌어 기존 schema와 충돌하는 부분만 수정. 기존 execution_request의 6필드 schema는 유지 | +| module/release·code pin·Agent/workflow hash | 실제 변경 자산 및 이를 참조하는 hash만 재결속. 순환 결속을 새로 만들지 않고 기존 detached 절차 유지 | +| 기존 분석서·IO 문서 | 구현 완료 후 실제 두-task 구조, 인자 및 request 저장 시점을 반영 | + +스키마를 안 바꾼다는 이유로 필요한 code hash를 누락하거나, 단순화를 이유로 기존 검증기를 무력화하지 않는다. 두 task 구성이 요구되는 한 builder·mirror·권한·hash 비용은 남는다. 별도의 준비 receipt schema와 전달 소비자만 제거하는 것이다. 빌드 재실행 시 동일 bytes인지 확인한다. + +### 5단계 — 필요한 시험과 완료 판정 + +아래 Validation Plan을 실행하여 기준선에 있던 실패와 이번 변경의 회귀를 구분한다. host 지원 여부가 불확실하면 offline 구현·시험과 실제 운영 적격성을 별도로 보고한다. DEV guard를 해제해서 성공 결과를 만들지 않는다. + +## Source and Tool Plan + +원본 계획, 현재 authoring/runtime/builder/binding을 우선 근거로 삼는다. 외부 제품 권고나 일반론이 아닌 로컬 코드의 중복·의존성을 분석한다. 구현 시 backend 코드·문서로 세 선행 조건을 확인하고 기존 테스트 runner 및 빌드 `--check`를 사용한다. 실제 backend 통합 시험은 격리된 fixture 환경에서 수행한다. + +## Validation Plan + +| 시험 묶음 | 통과 기준 | +|---|---| +| 정상·잘못된 외부 입력 | 정상은 정확한 6필드 canonical bytes; 누락/추가/ID/path/type 오류는 저장 전 거절. 기존 request schema와 validator 수용 범위 일치 | +| 저장·read-back·종료 실패 | 후행 ingress 호출 0회. `ok:false`·비정상 exit·잘못된 성공 결과도 차단 | +| request 교체·잘못된 host 입력 결속 | 첫 읽기의 전체 bytes 대조 및 기존 두 pass 비교로 거절. stale request·다른 attempt·다른 roots 포함 | +| 경쟁·취소·재시도 | 같은 workspace의 두 worker는 직렬 실행; 이전 writer/I/O가 남은 동안 다음 실행 금지. 다른 workspace는 독립 실행 | +| 2-task 빌드·회귀 | 지정 두 task/순서만 허용, code/mirror/receipt/hash 일치, 반복 빌드 동일. 기존 정상 11+E/diagnostic 5파일 fixture 계약 유지 | +| backend 통합 | 동일 불변 인자 공급, success-only 실행, 전역 직렬화·권한 강제가 실제 작동 | + +별도 준비 receipt, TTL/fencing 프로토콜, 재사용 최적화를 만들지 않으므로 이들의 전용 시험도 만들지 않는다. 단, worker 장애 이후 교차 쓰기가 없다는 기능 시험은 필수로 남는다. 신규 기능의 fixture 회귀와 실제 MCP/backend 실행 증거를 구분한다. + +## Approval Boundaries + +이번 요청 범위는 분석과 v1 계획 파일 작성이다. 원본 계획과 실행 자산을 보존한다. 이 문서 작성만으로 구현·배포·live 검증을 수행한 것으로 표현하지 않는다. + +## Progress + +- [x] 원본 계획과 현재 validator·IO·builder·binding 대조 +- [x] 제약 유지안과 제약 완화 대안 구별 +- [x] 중복 프로토콜·검증·최적화 제거 및 v1 작성 +- [ ] backend 세 조건 확인 +- [ ] 준비 task·ingress·빌드/배포 계약 구현 +- [ ] 회귀·경쟁·backend 통합 검증 및 문서 갱신 + +## Decision Log + +| 결정 | 근거·효과 | +|---|---| +| 독립 선행 task 유지 | 앞선 사용자 요구를 바꾸지 않음. 절대 최소 구조와 제약 내 최소안을 구분 | +| 동일 host 인자로 expected bytes 재구성 | 별도 준비 receipt·schema·전달을 제거하면서 전체 request 일치 검증 유지 | +| 기존 verified write/validator 재사용 | 현재 존재하는 동작을 새 계층으로 중복 구현하지 않음 | +| 전체 실행 직렬화 유지 | 중간 해제 신호보다 구현이 단순함. 동일 workspace 처리량 최적화는 이번 목표에서 제외 | +| 조건부 backend 확인 | 미확정 host 기능을 기정사실로 삼는 것이 가장 큰 숨은 복잡성이므로 먼저 검증 | + +## Evidence Ledger + +행번호는 2026-09-30 확인본 기준이다. runtime mirror를 구현 대조 자료로 사용했으며 authoring의 정본 지위를 바꾸지 않는다. + +| 분석 명제 | 현재 확인 근거 | 판정 | +|---|---|---| +| 중복 없는 writer/validator 재사용 가능 | `R/runtime/s2_00_ingress.py`의 `write_binary_verified`(5492행), `_inline_validate_request`(5505행) | write/read-back 및 6필드 검증 존재 확인 | +| ingress 첫 읽기에 expected 대조 추가 가능 | 같은 runtime `_inline_hydrate`(5712행 이후) | 기존 첫 읽기 및 검증 순서 확인; 변경은 제안 | +| 2-task 빌드 변경은 생략 불가 | `R/offline_build/build_s2_00_inline_projection.py`: 326–378 | 단일 task·DAG 제약 확인 | +| 준비 receipt 제거의 전제 | 두 task가 동일한 인증된 불변 host 입력을 받는다는 새 계약 | 설계 제안; backend 실제 지원 UNVERIFIED | +| request 쓰기 권한 추가 필요 | `R/deployment/stage2_code_executor_binding.yml`: 47–66 | 기존 write root는 결과 폴더 | + +## Risks and Failure Modes + +직렬화가 실제로 보장되지 않으면 receipt를 제거하든 유지하든 고정 request 경로의 안전한 실행은 완성되지 않는다. task별 입력이 서로 다르거나 조작될 수 있다면 expected bytes 방식의 전제가 무너진다. 이 경우 host 입력 결속을 먼저 수정해야 한다. 인증된 인자·권한·잠금은 JSON을 단순하게 만드는 것과 별개의 필수 조건이다. + +실패한 변경의 롤백은 진행 중 실행과 I/O 종료 후 검증된 기존 authoring·배포본·mirror·binding/hash 묶음으로 복원한다. 사건 자료·발행 결과는 삭제하지 않는다. 기존 버전도 request 외부 공급이 필요하므로 롤백이 자동 정상화를 뜻하지 않는다. + +## Results and Residual Uncertainty + +**기존 계획은 단순화할 수 있다.** v1은 두-task 요구를 유지하면서 신규 준비 receipt/전달 schema·독립 실행 envelope 설계·중복 read-back 검증·불필요한 사전 closure 읽기·재사용 최적화를 기본 범위에서 제거했다. 필수 인자 검증·저장 확인·후행 차단·동일 요청 소비·동시 실행 통제·hash 일치는 유지했다. 이는 코드 근거에 따른 설계 평가이며 성능 벤치마크 결과가 아니다. backend 세 조건의 실제 지원과 구현 완료 여부는 미확정·미착수다.