Усилить агентную валидацию phase12 replay

This commit is contained in:
dctouch 2026-06-02 10:20:20 +03:00
parent 7995980215
commit 5c3ef746c0
6 changed files with 756 additions and 11 deletions

View File

@ -1,4 +1,92 @@
[
{
"generation_id": "gen-ag06020717-08d833",
"created_at": "2026-06-02T07:17:06+00:00",
"mode": "saved_user_sessions",
"title": "AGENT | Phase 12 wider multi-trajectory authority replay",
"count": 20,
"domain": "address_phase12_wider_saved_session_pool",
"questions": [
"приветик - че как там дела",
"Альтернатива Плюс",
"расскажи что можешь интересного",
"кайф - что там на складе по остаткам?",
"а исторические остатки на другие даты умеешь?",
"март 2016",
"По выбранному объекту \"Рабочая станция универсального специалиста (индивидуальное изготовление)\": где взяли это?",
"а кому продали?",
"ндс можешь прикинуть на дату покупки рабочей станции?",
"кто у нас самый доходный клиент за все время",
"кто нам должен денег на май 2017",
"а какой ндс мы должны примерно заплатить за этот период?",
"мы должны комуто денег на сегодня?",
"а нам?",
"ты умеешь считать дельту по договорам?",
"по чепурнову покажи все доки",
"а по свк",
"хвосты покажи по счету 60 на август 2022",
"Есть ли остатки товара, которые закупались очень давно",
"а по Альтернативе Плюс сколько лет активности в базе 1С?"
],
"generated_by": "codex_agent",
"saved_case_set_file": "assistant_autogen_saved_user_sessions_20260602071706_gen-ag06020717-08d833.json",
"context": {
"llm_provider": null,
"model": null,
"assistant_prompt_version": null,
"decomposition_prompt_version": null,
"prompt_fingerprint": null,
"autogen_personality_id": null,
"autogen_personality_prompt": null,
"source_session_id": null,
"saved_session_file": "assistant_saved_session_20260602071706_gen-ag06020717-08d833.json",
"saved_case_set_kind": "agent_semantic_scenario",
"agent_run": true,
"agent_focus": "Phase12 wide replay: organization authority, inventory selected-object, VAT/date carryover, payables/receivables today reset, counterparty docs, old stock and activity age.",
"architecture_phase": "turnaround_11",
"source_spec_file": "X:\\1C\\NDC_1C\\docs\\orchestration\\address_truth_harness_phase12_wider_saved_session_pool.json",
"scenario_id": "address_truth_harness_phase12_wider_saved_session_pool",
"semantic_tags": [
"aggregate_revenue",
"bridge_inventory_to_vat",
"capability_over_followup",
"company_authority",
"company_selected",
"counterparty_followup",
"counterparty_root",
"cross_domain_pivot",
"date_carryover",
"display_label_integrity",
"documents",
"historical_date_anchor",
"human_answer_quality",
"inventory_aging",
"inventory_provenance",
"inventory_root",
"inventory_sale_trace",
"late_session_stability",
"meta_capability",
"meta_historical_capability",
"meta_smalltalk",
"organization_activity_age",
"organization_authority",
"proactive_scope_offer",
"same_date_restore",
"same_period_restore",
"selected_object",
"settlements_account_60",
"settlements_mirror_followup",
"settlements_payables",
"settlements_receivables",
"tail_authority_proof",
"today_scope",
"vat_followup"
],
"validation_status": "accepted_domain_case_loop_scenario",
"validated_run_dir": "artifacts\\domain_runs\\address_truth_harness_phase12_wider_saved_session_pool_20260602_p10",
"saved_after_validated_replay": true
}
},
{
"generation_id": "gen-ag06020614-947fd1",
"created_at": "2026-06-02T06:14:41+00:00",

View File

@ -0,0 +1,363 @@
{
"saved_at": "2026-06-02T07:17:06+00:00",
"generation_id": "gen-ag06020717-08d833",
"mode": "saved_user_sessions",
"title": "AGENT | Phase 12 wider multi-trajectory authority replay",
"agent_run": true,
"questions": [
"приветик - че как там дела",
"Альтернатива Плюс",
"расскажи что можешь интересного",
"кайф - что там на складе по остаткам?",
"а исторические остатки на другие даты умеешь?",
"март 2016",
"По выбранному объекту \"Рабочая станция универсального специалиста (индивидуальное изготовление)\": где взяли это?",
"а кому продали?",
"ндс можешь прикинуть на дату покупки рабочей станции?",
"кто у нас самый доходный клиент за все время",
"кто нам должен денег на май 2017",
"а какой ндс мы должны примерно заплатить за этот период?",
"мы должны комуто денег на сегодня?",
"а нам?",
"ты умеешь считать дельту по договорам?",
"по чепурнову покажи все доки",
"а по свк",
"хвосты покажи по счету 60 на август 2022",
"Есть ли остатки товара, которые закупались очень давно",
"а по Альтернативе Плюс сколько лет активности в базе 1С?"
],
"metadata": {
"assistant_prompt_version": null,
"decomposition_prompt_version": null,
"prompt_fingerprint": null,
"agent_focus": "Phase12 wide replay: organization authority, inventory selected-object, VAT/date carryover, payables/receivables today reset, counterparty docs, old stock and activity age.",
"architecture_phase": "turnaround_11",
"source_spec_file": "X:\\1C\\NDC_1C\\docs\\orchestration\\address_truth_harness_phase12_wider_saved_session_pool.json",
"scenario_id": "address_truth_harness_phase12_wider_saved_session_pool",
"semantic_tags": [
"aggregate_revenue",
"bridge_inventory_to_vat",
"capability_over_followup",
"company_authority",
"company_selected",
"counterparty_followup",
"counterparty_root",
"cross_domain_pivot",
"date_carryover",
"display_label_integrity",
"documents",
"historical_date_anchor",
"human_answer_quality",
"inventory_aging",
"inventory_provenance",
"inventory_root",
"inventory_sale_trace",
"late_session_stability",
"meta_capability",
"meta_historical_capability",
"meta_smalltalk",
"organization_activity_age",
"organization_authority",
"proactive_scope_offer",
"same_date_restore",
"same_period_restore",
"selected_object",
"settlements_account_60",
"settlements_mirror_followup",
"settlements_payables",
"settlements_receivables",
"tail_authority_proof",
"today_scope",
"vat_followup"
],
"validation_status": "accepted_domain_case_loop_scenario",
"validated_run_dir": "artifacts\\domain_runs\\address_truth_harness_phase12_wider_saved_session_pool_20260602_p10",
"saved_after_validated_replay": true,
"save_gate": {
"schema_version": "agent_semantic_save_gate_v1",
"validation_status": "accepted_domain_case_loop_scenario",
"validated_run_dir": "artifacts\\domain_runs\\address_truth_harness_phase12_wider_saved_session_pool_20260602_p10",
"final_status": "accepted",
"execution_status": "partial",
"scenario_id": "address_truth_harness_phase12_wider_saved_session_pool_20260602_p10",
"steps_total": 20,
"effective_runtime": {
"manifest_path": "artifacts\\domain_runs\\address_truth_harness_phase12_wider_saved_session_pool_20260602_p10\\effective_runtime.json",
"runner": "domain_case_loop.run-scenario",
"git_sha": "7995980215b6a9840c42c1c7ef73d6faa6612c37",
"backend_url": "http://127.0.0.1:8787",
"mcp_proxy_url": "http://127.0.0.1:6003",
"llm_provider": "local",
"llm_model": "unsloth/qwen3-30b-a3b-instruct-2507",
"temperature": 0.8,
"max_output_tokens": 900,
"prompt_version": "normalizer_v2_0_2",
"prompt_source": "file",
"prompt_hash": "f36e36a5e491cd24511b380e0c0059cb01aea5e2af738fb9f01671cd291735c1",
"prompt_registry_status": "pass"
},
"saved_after_validated_replay": true
}
},
"source_session_id": null,
"session": {
"session_id": null,
"mode": "agent_semantic_run",
"items": [
{
"message_id": "agent-user-001",
"role": "user",
"text": "приветик - че как там дела",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-002",
"role": "user",
"text": "Альтернатива Плюс",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-003",
"role": "user",
"text": "расскажи что можешь интересного",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-004",
"role": "user",
"text": "кайф - что там на складе по остаткам?",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-005",
"role": "user",
"text": "а исторические остатки на другие даты умеешь?",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-006",
"role": "user",
"text": "март 2016",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-007",
"role": "user",
"text": "По выбранному объекту \"Рабочая станция универсального специалиста (индивидуальное изготовление)\": где взяли это?",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-008",
"role": "user",
"text": "а кому продали?",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-009",
"role": "user",
"text": "ндс можешь прикинуть на дату покупки рабочей станции?",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-010",
"role": "user",
"text": "кто у нас самый доходный клиент за все время",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-011",
"role": "user",
"text": "кто нам должен денег на май 2017",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-012",
"role": "user",
"text": "а какой ндс мы должны примерно заплатить за этот период?",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-013",
"role": "user",
"text": "мы должны комуто денег на сегодня?",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-014",
"role": "user",
"text": "а нам?",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-015",
"role": "user",
"text": "ты умеешь считать дельту по договорам?",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-016",
"role": "user",
"text": "по чепурнову покажи все доки",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-017",
"role": "user",
"text": "а по свк",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-018",
"role": "user",
"text": "хвосты покажи по счету 60 на август 2022",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-019",
"role": "user",
"text": "Есть ли остатки товара, которые закупались очень давно",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
},
{
"message_id": "agent-user-020",
"role": "user",
"text": "а по Альтернативе Плюс сколько лет активности в базе 1С?",
"created_at": "2026-06-02T07:17:06+00:00",
"reply_type": null,
"trace_id": null,
"debug": null
}
],
"agent_run": true,
"metadata": {
"assistant_prompt_version": null,
"decomposition_prompt_version": null,
"prompt_fingerprint": null,
"agent_focus": "Phase12 wide replay: organization authority, inventory selected-object, VAT/date carryover, payables/receivables today reset, counterparty docs, old stock and activity age.",
"architecture_phase": "turnaround_11",
"source_spec_file": "X:\\1C\\NDC_1C\\docs\\orchestration\\address_truth_harness_phase12_wider_saved_session_pool.json",
"scenario_id": "address_truth_harness_phase12_wider_saved_session_pool",
"semantic_tags": [
"aggregate_revenue",
"bridge_inventory_to_vat",
"capability_over_followup",
"company_authority",
"company_selected",
"counterparty_followup",
"counterparty_root",
"cross_domain_pivot",
"date_carryover",
"display_label_integrity",
"documents",
"historical_date_anchor",
"human_answer_quality",
"inventory_aging",
"inventory_provenance",
"inventory_root",
"inventory_sale_trace",
"late_session_stability",
"meta_capability",
"meta_historical_capability",
"meta_smalltalk",
"organization_activity_age",
"organization_authority",
"proactive_scope_offer",
"same_date_restore",
"same_period_restore",
"selected_object",
"settlements_account_60",
"settlements_mirror_followup",
"settlements_payables",
"settlements_receivables",
"tail_authority_proof",
"today_scope",
"vat_followup"
],
"validation_status": "accepted_domain_case_loop_scenario",
"validated_run_dir": "artifacts\\domain_runs\\address_truth_harness_phase12_wider_saved_session_pool_20260602_p10",
"saved_after_validated_replay": true,
"save_gate": {
"schema_version": "agent_semantic_save_gate_v1",
"validation_status": "accepted_domain_case_loop_scenario",
"validated_run_dir": "artifacts\\domain_runs\\address_truth_harness_phase12_wider_saved_session_pool_20260602_p10",
"final_status": "accepted",
"execution_status": "partial",
"scenario_id": "address_truth_harness_phase12_wider_saved_session_pool_20260602_p10",
"steps_total": 20,
"effective_runtime": {
"manifest_path": "artifacts\\domain_runs\\address_truth_harness_phase12_wider_saved_session_pool_20260602_p10\\effective_runtime.json",
"runner": "domain_case_loop.run-scenario",
"git_sha": "7995980215b6a9840c42c1c7ef73d6faa6612c37",
"backend_url": "http://127.0.0.1:8787",
"mcp_proxy_url": "http://127.0.0.1:6003",
"llm_provider": "local",
"llm_model": "unsloth/qwen3-30b-a3b-instruct-2507",
"temperature": 0.8,
"max_output_tokens": 900,
"prompt_version": "normalizer_v2_0_2",
"prompt_source": "file",
"prompt_hash": "f36e36a5e491cd24511b380e0c0059cb01aea5e2af738fb9f01671cd291735c1",
"prompt_registry_status": "pass"
},
"saved_after_validated_replay": true
}
}
}
}

View File

@ -0,0 +1,85 @@
{
"suite_id": "assistant_saved_session_gen-ag06020717-08d833",
"suite_version": "0.1.0",
"schema_version": "assistant_saved_session_suite_v0_1",
"generated_at": "2026-06-02T07:17:06+00:00",
"generation_id": "gen-ag06020717-08d833",
"mode": "saved_user_sessions",
"title": "AGENT | Phase 12 wider multi-trajectory authority replay",
"domain": "address_phase12_wider_saved_session_pool",
"scenario_count": 1,
"case_ids": [
"SAVED-001"
],
"cases": [
{
"case_id": "SAVED-001",
"scenario_tag": "agent_saved_user_sessions",
"title": "AGENT | Phase 12 wider multi-trajectory authority replay",
"question_type": "followup",
"broadness_level": "medium",
"turns": [
{
"user_message": "приветик - че как там дела"
},
{
"user_message": "Альтернатива Плюс"
},
{
"user_message": "расскажи что можешь интересного"
},
{
"user_message": "кайф - что там на складе по остаткам?"
},
{
"user_message": "а исторические остатки на другие даты умеешь?"
},
{
"user_message": "март 2016"
},
{
"user_message": "По выбранному объекту \"Рабочая станция универсального специалиста (индивидуальное изготовление)\": где взяли это?"
},
{
"user_message": "а кому продали?"
},
{
"user_message": "ндс можешь прикинуть на дату покупки рабочей станции?"
},
{
"user_message": "кто у нас самый доходный клиент за все время"
},
{
"user_message": "кто нам должен денег на май 2017"
},
{
"user_message": "а какой ндс мы должны примерно заплатить за этот период?"
},
{
"user_message": "мы должны комуто денег на сегодня?"
},
{
"user_message": "а нам?"
},
{
"user_message": "ты умеешь считать дельту по договорам?"
},
{
"user_message": "по чепурнову покажи все доки"
},
{
"user_message": "а по свк"
},
{
"user_message": "хвосты покажи по счету 60 на август 2022"
},
{
"user_message": "Есть ли остатки товара, которые закупались очень давно"
},
{
"user_message": "а по Альтернативе Плюс сколько лет активности в базе 1С?"
}
]
}
]
}

View File

@ -9,7 +9,7 @@ import subprocess
import sys
import textwrap
import time
from datetime import datetime, timezone
from datetime import date, datetime, timezone
from pathlib import Path
from typing import Any
from urllib.error import HTTPError, URLError
@ -712,6 +712,7 @@ def merge_scenario_date_scope(
def question_resets_temporal_scope(value: Any) -> bool:
normalized = str(value or "").casefold().replace("ё", "е")
repaired_normalized = repair_text_mojibake(str(value or "")).casefold().replace("ё", "е")
same_scope_markers = (
"за тот же период",
"на тот же период",
@ -727,6 +728,8 @@ def question_resets_temporal_scope(value: Any) -> bool:
)
if any(marker in normalized for marker in same_scope_markers):
return False
if any(marker in repaired_normalized for marker in ("сегодня", "на сегодня", "текущую дату", "сейчас")):
return True
markers = (
"за все доступное время",
"за все время",
@ -779,6 +782,41 @@ def normalize_bindings(raw_bindings: Any) -> dict[str, Any]:
return {str(key): normalize_binding_value(value) for key, value in raw_bindings.items()}
def build_runtime_bindings() -> dict[str, str]:
today = date.today()
today_iso = today.isoformat()
today_dot = today.strftime("%d.%m.%Y")
return {
"today_iso": today_iso,
"today_dot": today_dot,
"today_iso_regex": re.escape(today_iso),
"today_dot_regex": re.escape(today_dot),
}
def resolve_runtime_placeholders(value: Any) -> Any:
if not isinstance(value, str) or "{{" not in value:
return value
runtime_bindings = build_runtime_bindings()
def replace(match: re.Match[str]) -> str:
placeholder = match.group(1).strip()
if not placeholder.startswith("runtime."):
return match.group(0)
key = placeholder.split(".", 1)[1]
return runtime_bindings.get(key, match.group(0))
return re.sub(r"{{\s*([^{}]+?)\s*}}", replace, value)
def normalize_pattern_list(*raw_values: Any) -> list[str]:
for raw_value in raw_values:
values = normalize_string_list(raw_value)
if values:
return [str(resolve_runtime_placeholders(value)) for value in values]
return []
def normalize_string_list(raw_values: Any) -> list[str]:
if isinstance(raw_values, str):
value = raw_values.strip()
@ -820,12 +858,17 @@ def normalize_validation_filters(raw_filters: Any) -> dict[str, str]:
key = str(raw_key or "").strip()
if not key:
continue
resolved_value = resolve_runtime_placeholders(raw_value)
if key in {"as_of_date", "period_from", "period_to"}:
normalized_value = normalize_iso_date(raw_value)
normalized_value = normalize_iso_date(resolved_value)
if normalized_value:
normalized[key] = normalized_value
continue
text_value = str(raw_value or "").strip()
text_value = str(resolved_value or "").strip()
if text_value:
normalized[key] = text_value
continue
text_value = str(resolved_value or "").strip()
if text_value:
normalized[key] = text_value
return normalized
@ -1710,9 +1753,18 @@ def normalize_step_definition(index: int, raw_step: Any) -> dict[str, Any]:
"required_answer_shape": (
str(raw_step.get("required_answer_shape") or raw_step.get("expected_answer_shape") or "").strip() or None
),
"forbidden_answer_patterns": normalize_string_list(raw_step.get("forbidden_answer_patterns")),
"required_answer_patterns_any": normalize_string_list(raw_step.get("required_answer_patterns_any")),
"required_answer_patterns_all": normalize_string_list(raw_step.get("required_answer_patterns_all")),
"forbidden_answer_patterns": normalize_pattern_list(
raw_step.get("forbidden_answer_patterns"),
raw_step.get("forbidden_direct_answer_patterns"),
),
"required_answer_patterns_any": normalize_pattern_list(
raw_step.get("required_answer_patterns_any"),
raw_step.get("required_direct_answer_patterns_any"),
),
"required_answer_patterns_all": normalize_pattern_list(
raw_step.get("required_answer_patterns_all"),
raw_step.get("required_direct_answer_patterns_all"),
),
"semantic_tags": normalize_string_list(raw_step.get("semantic_tags")),
"required_carryover_invariants": normalize_string_list(raw_step.get("required_carryover_invariants")),
"invariant_severity": normalize_invariant_severity(raw_step.get("invariant_severity")),
@ -3236,17 +3288,23 @@ def save_scenario_step_bundle(
write_text(step_dir / "resolved_question.txt", f"{step_state['question_resolved']}\n")
def is_effectively_complete_partial_step(step_output: dict[str, Any]) -> bool:
def is_effectively_complete_validated_step(step_output: dict[str, Any]) -> bool:
execution_status = str(step_output.get("execution_status") or step_output.get("status") or "").strip()
if execution_status == "exact":
return True
if execution_status != "partial":
return False
if str(step_output.get("acceptance_status") or "").strip() != "validated":
return False
if execution_status == "needs_exact_capability":
return step_output.get("clean_meta_chat_answer_validated") is True
if execution_status != "partial":
return False
return (
step_output.get("missing_axis_clarification_validated") is True
or step_output.get("clarification_answer_validated") is True
or step_output.get("bounded_mcp_answer_validated") is True
or step_output.get("memory_checkpoint_validated") is True
or step_output.get("runtime_factual_answer_validated") is True
or step_output.get("guarded_insufficiency_validated") is True
)
@ -3256,9 +3314,14 @@ def derive_scenario_execution_status(step_outputs: dict[str, dict[str, Any]]) ->
return "blocked"
if any(status == "blocked" for status in statuses):
return "blocked"
if any(status == "needs_exact_capability" for status in statuses):
if any(
status == "needs_exact_capability" and not is_effectively_complete_validated_step(item)
for status, item in zip(statuses, step_outputs.values())
):
return "needs_exact_capability"
if any(not is_effectively_complete_partial_step(item) for item in step_outputs.values()):
if any(not is_effectively_complete_validated_step(item) for item in step_outputs.values()):
return "partial"
if any(status != "exact" for status in statuses):
return "partial"
return "exact"
@ -4302,6 +4365,8 @@ def derive_repair_target_severity(step_output: dict[str, Any]) -> str:
return "P0"
if route_candidate_status == "needs_route_enablement":
return "P1"
if is_effectively_complete_validated_step(step_output):
return "P2"
if acceptance_status in {"rejected", "needs_exact_capability"}:
return "P1"
if execution_status in {"partial", "needs_exact_capability"} or reply_type == "partial_coverage":

View File

@ -333,6 +333,90 @@ def validate_domain_case_loop_pack_dir(run_dir: Path) -> dict[str, Any]:
}
def parse_markdown_status_field(markdown: str, field_name: str) -> str:
pattern = re.compile(rf"^-\s*{re.escape(field_name)}:\s*`([^`]+)`\s*$", re.MULTILINE)
match = pattern.search(markdown)
return str(match.group(1) if match else "").strip()
def step_has_validated_partial_or_meta_gate(step_state: dict[str, Any]) -> bool:
execution_status = str(step_state.get("execution_status") or step_state.get("status") or "").strip()
if execution_status == "exact":
return True
if str(step_state.get("acceptance_status") or step_state.get("status") or "").strip() != "validated":
return False
if execution_status == "needs_exact_capability":
return step_state.get("clean_meta_chat_answer_validated") is True
if execution_status != "partial":
return False
return any(
step_state.get(flag_name) is True
for flag_name in (
"bounded_mcp_answer_validated",
"memory_checkpoint_validated",
"runtime_factual_answer_validated",
"guarded_insufficiency_validated",
"clarification_answer_validated",
"missing_axis_clarification_validated",
)
)
def validate_domain_case_loop_scenario_dir(run_dir: Path) -> dict[str, Any]:
run_dir = run_dir.resolve()
effective_runtime = require_effective_runtime_manifest(run_dir)
scenario_state = load_json_object(run_dir / "scenario_state.json", "Validated domain-case-loop scenario_state.json")
scenario_manifest = load_json_object(
run_dir / "scenario_manifest.json",
"Validated domain-case-loop scenario_manifest.json",
)
final_status_markdown = (run_dir / "final_status.md").read_text(encoding="utf-8")
final_status = parse_markdown_status_field(final_status_markdown, "status")
execution_status = parse_markdown_status_field(final_status_markdown, "execution_status")
step_outputs = scenario_state.get("step_outputs") if isinstance(scenario_state.get("step_outputs"), dict) else {}
manifest_steps = scenario_manifest.get("steps") if isinstance(scenario_manifest.get("steps"), list) else []
problems: list[str] = []
assert_status(final_status, "accepted", "final_status.status", problems)
if not step_outputs:
problems.append("scenario_state.step_outputs=empty")
if manifest_steps and len(step_outputs) != len(manifest_steps):
problems.append(f"step_count={len(step_outputs)}<manifest_steps={len(manifest_steps)}")
for step_id, raw_step_state in step_outputs.items():
if not isinstance(raw_step_state, dict):
problems.append(f"{step_id}.state=invalid")
continue
step_acceptance = str(raw_step_state.get("acceptance_status") or raw_step_state.get("status") or "").strip()
step_execution = str(raw_step_state.get("execution_status") or "").strip()
if step_acceptance != "validated":
problems.append(f"{step_id}.acceptance_status={step_acceptance or 'missing'}")
if raw_step_state.get("hard_fail") is True:
problems.append(f"{step_id}.hard_fail=true")
if step_execution == "blocked":
problems.append(f"{step_id}.execution_status=blocked")
if not step_has_validated_partial_or_meta_gate(raw_step_state):
problems.append(f"{step_id}.validated_gate=missing")
if problems:
raise RuntimeError(
"Refusing to save AGENT autorun because the validated domain-case-loop scenario is not clean: "
+ ", ".join(problems)
)
return {
"schema_version": VALIDATED_AGENT_SAVE_SCHEMA_VERSION,
"validation_status": "accepted_domain_case_loop_scenario",
"validated_run_dir": repo_relative(run_dir),
"final_status": final_status,
"execution_status": execution_status,
"scenario_id": scenario_state.get("scenario_id") or scenario_manifest.get("scenario_id"),
"steps_total": len(step_outputs),
"effective_runtime": build_effective_runtime_save_summary(effective_runtime, run_dir),
"saved_after_validated_replay": True,
}
def validate_accepted_run_dir(run_dir: Path) -> dict[str, Any]:
run_dir = run_dir.resolve()
if (run_dir / "loop_state.json").exists():
@ -341,6 +425,8 @@ def validate_accepted_run_dir(run_dir: Path) -> dict[str, Any]:
return validate_truth_harness_run_dir(run_dir)
if (run_dir / "pack_state.json").exists() and (run_dir / "repair_targets.json").exists():
return validate_domain_case_loop_pack_dir(run_dir)
if (run_dir / "scenario_state.json").exists() and (run_dir / "scenario_manifest.json").exists():
return validate_domain_case_loop_scenario_dir(run_dir)
return validate_truth_harness_run_dir(run_dir)

View File

@ -86,6 +86,64 @@ class DomainCaseLoopStepStateTests(unittest.TestCase):
self.assertEqual(step_state["mcp_discovery_route_candidate_provided_axes"], ["period"])
self.assertFalse(step_state["mcp_discovery_route_candidate_executable_now"])
def test_scenario_execution_status_treats_validated_meta_and_guarded_partial_as_complete(self) -> None:
step_outputs = {
"smalltalk": {
"execution_status": "needs_exact_capability",
"acceptance_status": "validated",
"clean_meta_chat_answer_validated": True,
},
"exact_inventory": {
"execution_status": "exact",
"acceptance_status": "validated",
},
"account_60_boundary": {
"execution_status": "partial",
"acceptance_status": "validated",
"runtime_factual_answer_validated": True,
},
}
self.assertEqual(dcl.derive_scenario_execution_status(step_outputs), "partial")
self.assertEqual(dcl.derive_scenario_status(step_outputs), "accepted")
def test_today_scope_required_filter_and_direct_patterns_are_enforced(self) -> None:
self.assertTrue(dcl.question_resets_temporal_scope("мы должны комуто денег на сегодня?"))
step = dcl.normalize_step_definition(
1,
{
"step_id": "payables_today",
"title": "Payables today",
"question": "мы должны комуто денег на сегодня?",
"expected_intents": ["payables_confirmed_as_of_date"],
"required_filters": {"as_of_date": "{{runtime.today_iso}}"},
"required_direct_answer_patterns_any": ["{{runtime.today_dot_regex}}"],
},
)
runtime_today = dcl.build_runtime_bindings()["today_iso"]
runtime_today_pattern = dcl.build_runtime_bindings()["today_dot_regex"]
self.assertEqual(step["required_filters"]["as_of_date"], runtime_today)
self.assertEqual(step["required_answer_patterns_any"], [runtime_today_pattern])
step_state = dcl.validate_step_contract(
{
"execution_status": "exact",
"reply_type": "factual",
"detected_intent": "payables_confirmed_as_of_date",
"required_filters": step["required_filters"],
"required_answer_patterns_any": step["required_answer_patterns_any"],
"extracted_filters": {"as_of_date": "2017-05-31"},
"assistant_text": "на 31.05.2017 мы должны 3.433.472,35 ₽.",
"actual_direct_answer": "на 31.05.2017 мы должны 3.433.472,35 ₽.",
"top_non_empty_lines": ["на 31.05.2017 мы должны 3.433.472,35 ₽."],
}
)
self.assertIn("wrong_as_of_date", step_state["violated_invariants"])
self.assertIn("required_answer_patterns_any_missing", step_state["violated_invariants"])
self.assertEqual(step_state["acceptance_status"], "rejected")
def test_repair_targets_promote_route_candidate_enablement_gaps(self) -> None:
repair_targets = dcl.build_deterministic_repair_targets(
{"pack_id": "route_candidate_pack", "domain": "open_world", "final_status": "accepted"},