Planner Autonomy: вывести catalog-alignment в replay artifacts

This commit is contained in:
2026-05-01 15:38:06 +03:00
parent 8b5104a2c6
commit a63742f0d6
7 changed files with 89 additions and 3 deletions
+3
View File
@@ -1727,6 +1727,9 @@ def build_scenario_step_state(
"selected_recipe": debug.get("selected_recipe"),
"capability_id": debug.get("capability_id"),
"capability_route_mode": debug.get("capability_route_mode"),
"mcp_discovery_catalog_chain_alignment_status": debug.get("mcp_discovery_catalog_chain_alignment_status"),
"mcp_discovery_catalog_chain_top_match": debug.get("mcp_discovery_catalog_chain_top_match"),
"mcp_discovery_catalog_chain_selected_matches_top": debug.get("mcp_discovery_catalog_chain_selected_matches_top"),
"route_expectation_status": debug.get("route_expectation_status"),
"result_mode": debug.get("result_mode"),
"response_type": debug.get("response_type"),
+3
View File
@@ -679,6 +679,9 @@ def build_truth_review_markdown(spec: dict[str, Any], scenario_state: dict[str,
f"intent: `{step_state.get('detected_intent') or 'n/a'}`",
f"recipe: `{step_state.get('selected_recipe') or 'n/a'}`",
f"capability: `{step_state.get('capability_id') or 'n/a'}`",
f"catalog_alignment_status: `{step_state.get('mcp_discovery_catalog_chain_alignment_status') or 'n/a'}`",
f"catalog_top_match: `{step_state.get('mcp_discovery_catalog_chain_top_match') or 'n/a'}`",
f"catalog_selected_matches_top: `{step_state.get('mcp_discovery_catalog_chain_selected_matches_top')}`",
f"limited_reason_category: `{step_state.get('limited_reason_category') or 'n/a'}`",
f"filters: `{dump_json(step_state.get('extracted_filters') or {})}`",
f"direct_answer: {step_state.get('actual_direct_answer') or 'n/a'}",
+6
View File
@@ -198,6 +198,9 @@ def build_scenario_acceptance_matrix(
"reply_type": step_state.get("reply_type"),
"detected_intent": step_state.get("detected_intent"),
"capability_id": step_state.get("capability_id"),
"mcp_discovery_catalog_chain_alignment_status": step_state.get("mcp_discovery_catalog_chain_alignment_status"),
"mcp_discovery_catalog_chain_top_match": step_state.get("mcp_discovery_catalog_chain_top_match"),
"mcp_discovery_catalog_chain_selected_matches_top": step_state.get("mcp_discovery_catalog_chain_selected_matches_top"),
"selected_object_step": _has_selected_object_signal(step),
"meta_context_step": _has_meta_context_signal(step),
"highest_unresolved_priority": highest_priority,
@@ -330,6 +333,9 @@ def build_scenario_acceptance_matrix_markdown(acceptance_matrix: dict[str, Any])
f" review_status: `{row.get('review_status')}`",
f" criticality: `{row.get('criticality')}`",
f" semantic_tags: {', '.join(row.get('semantic_tags') or []) or 'none'}",
f" catalog_alignment_status: `{row.get('mcp_discovery_catalog_chain_alignment_status') or 'n/a'}`",
f" catalog_top_match: `{row.get('mcp_discovery_catalog_chain_top_match') or 'n/a'}`",
f" catalog_selected_matches_top: `{row.get('mcp_discovery_catalog_chain_selected_matches_top')}`",
f" highest_unresolved_priority: `{row.get('highest_unresolved_priority')}`",
f" selected_object_step: `{row.get('selected_object_step')}`",
f" meta_context_step: `{row.get('meta_context_step')}`",
@@ -0,0 +1,54 @@
from __future__ import annotations
import sys
import unittest
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
import domain_case_loop as dcl
class DomainCaseLoopStepStateTests(unittest.TestCase):
def test_preserves_mcp_catalog_alignment_debug_fields(self) -> None:
step_state = dcl.build_scenario_step_state(
scenario_id="planner_alignment_demo",
domain="planner_autonomy",
step={
"step_id": "step_01",
"title": "Alignment visibility",
"depends_on": [],
"question_template": "show planner alignment",
},
step_index=1,
question_resolved="show planner alignment",
analysis_context={},
turn_artifact={
"assistant_message": {
"reply_type": "factual",
"text": "Confirmed answer",
"message_id": "msg-1",
"trace_id": "trace-1",
},
"technical_debug_payload": {
"detected_mode": "address_query",
"detected_intent": "counterparty_turnover",
"selected_recipe": "counterparty_turnover_by_period",
"capability_id": "confirmed_counterparty_turnover",
"mcp_discovery_catalog_chain_alignment_status": "selected_matches_top",
"mcp_discovery_catalog_chain_top_match": "value_flow",
"mcp_discovery_catalog_chain_selected_matches_top": True,
},
"session_summary": {},
},
entries=[],
)
self.assertEqual(step_state["mcp_discovery_catalog_chain_alignment_status"], "selected_matches_top")
self.assertEqual(step_state["mcp_discovery_catalog_chain_top_match"], "value_flow")
self.assertTrue(step_state["mcp_discovery_catalog_chain_selected_matches_top"])
if __name__ == "__main__":
unittest.main()
@@ -84,6 +84,9 @@ class ScenarioAcceptancePolicyTests(unittest.TestCase):
"reply_type": "factual",
"detected_intent": "inventory_on_hand_as_of_date",
"capability_id": "confirmed_inventory_on_hand_as_of_date",
"mcp_discovery_catalog_chain_alignment_status": "selected_matches_top",
"mcp_discovery_catalog_chain_top_match": "inventory_stock_snapshot",
"mcp_discovery_catalog_chain_selected_matches_top": True,
"review_findings": [],
}
},
@@ -104,6 +107,15 @@ class ScenarioAcceptancePolicyTests(unittest.TestCase):
self.assertTrue(pack_state["acceptance_gate_passed"])
self.assertTrue(pack_state["critical_path_green"])
self.assertTrue(all(pack_state["invariants"].values()))
self.assertEqual(
acceptance_matrix["rows"][0]["mcp_discovery_catalog_chain_alignment_status"],
"selected_matches_top",
)
self.assertEqual(
acceptance_matrix["rows"][0]["mcp_discovery_catalog_chain_top_match"],
"inventory_stock_snapshot",
)
self.assertTrue(acceptance_matrix["rows"][0]["mcp_discovery_catalog_chain_selected_matches_top"])
def test_flags_meta_context_integrity_when_meta_step_leaks_technical_answer_shape(self) -> None:
spec = {