From 5c3ef746c0a83d90823198e1e45712470d6c4468 Mon Sep 17 00:00:00 2001 From: dctouch Date: Tue, 2 Jun 2026 10:20:20 +0300 Subject: [PATCH] =?UTF-8?q?=D0=A3=D1=81=D0=B8=D0=BB=D0=B8=D1=82=D1=8C=20?= =?UTF-8?q?=D0=B0=D0=B3=D0=B5=D0=BD=D1=82=D0=BD=D1=83=D1=8E=20=D0=B2=D0=B0?= =?UTF-8?q?=D0=BB=D0=B8=D0=B4=D0=B0=D1=86=D0=B8=D1=8E=20phase12=20replay?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../data/autorun_generators/history.json | 88 +++++ ..._20260602071706_gen-ag06020717-08d833.json | 363 ++++++++++++++++++ ..._20260602071706_gen-ag06020717-08d833.json | 85 ++++ scripts/domain_case_loop.py | 87 ++++- scripts/save_agent_semantic_run.py | 86 +++++ scripts/test_domain_case_loop_step_state.py | 58 +++ 6 files changed, 756 insertions(+), 11 deletions(-) create mode 100644 llm_normalizer/data/autorun_generators/saved_sessions/assistant_saved_session_20260602071706_gen-ag06020717-08d833.json create mode 100644 llm_normalizer/data/eval_cases/assistant_autogen_saved_user_sessions_20260602071706_gen-ag06020717-08d833.json diff --git a/llm_normalizer/data/autorun_generators/history.json b/llm_normalizer/data/autorun_generators/history.json index cb183d7..3a59a6f 100644 --- a/llm_normalizer/data/autorun_generators/history.json +++ b/llm_normalizer/data/autorun_generators/history.json @@ -1,4 +1,92 @@ [ + { + "generation_id": "gen-ag06020717-08d833", + "created_at": "2026-06-02T07:17:06+00:00", + "mode": "saved_user_sessions", + "title": "AGENT | Phase 12 wider multi-trajectory authority replay", + "count": 20, + "domain": "address_phase12_wider_saved_session_pool", + "questions": [ + "приветик - че как там дела", + "Альтернатива Плюс", + "расскажи что можешь интересного", + "кайф - что там на складе по остаткам?", + "а исторические остатки на другие даты умеешь?", + "март 2016", + "По выбранному объекту \"Рабочая станция универсального специалиста (индивидуальное изготовление)\": где взяли это?", + "а кому продали?", + "ндс можешь прикинуть на дату покупки рабочей станции?", + "кто у нас самый доходный клиент за все время", + "кто нам должен денег на май 2017", + "а какой ндс мы должны примерно заплатить за этот период?", + "мы должны комуто денег на сегодня?", + "а нам?", + "ты умеешь считать дельту по договорам?", + "по чепурнову покажи все доки", + "а по свк", + "хвосты покажи по счету 60 на август 2022", + "Есть ли остатки товара, которые закупались очень давно", + "а по Альтернативе Плюс сколько лет активности в базе 1С?" + ], + "generated_by": "codex_agent", + "saved_case_set_file": "assistant_autogen_saved_user_sessions_20260602071706_gen-ag06020717-08d833.json", + "context": { + "llm_provider": null, + "model": null, + "assistant_prompt_version": null, + "decomposition_prompt_version": null, + "prompt_fingerprint": null, + "autogen_personality_id": null, + "autogen_personality_prompt": null, + "source_session_id": null, + "saved_session_file": "assistant_saved_session_20260602071706_gen-ag06020717-08d833.json", + "saved_case_set_kind": "agent_semantic_scenario", + "agent_run": true, + "agent_focus": "Phase12 wide replay: organization authority, inventory selected-object, VAT/date carryover, payables/receivables today reset, counterparty docs, old stock and activity age.", + "architecture_phase": "turnaround_11", + "source_spec_file": "X:\\1C\\NDC_1C\\docs\\orchestration\\address_truth_harness_phase12_wider_saved_session_pool.json", + "scenario_id": "address_truth_harness_phase12_wider_saved_session_pool", + "semantic_tags": [ + "aggregate_revenue", + "bridge_inventory_to_vat", + "capability_over_followup", + "company_authority", + "company_selected", + "counterparty_followup", + "counterparty_root", + "cross_domain_pivot", + "date_carryover", + "display_label_integrity", + "documents", + "historical_date_anchor", + "human_answer_quality", + "inventory_aging", + "inventory_provenance", + "inventory_root", + "inventory_sale_trace", + "late_session_stability", + "meta_capability", + "meta_historical_capability", + "meta_smalltalk", + "organization_activity_age", + "organization_authority", + "proactive_scope_offer", + "same_date_restore", + "same_period_restore", + "selected_object", + "settlements_account_60", + "settlements_mirror_followup", + "settlements_payables", + "settlements_receivables", + "tail_authority_proof", + "today_scope", + "vat_followup" + ], + "validation_status": "accepted_domain_case_loop_scenario", + "validated_run_dir": "artifacts\\domain_runs\\address_truth_harness_phase12_wider_saved_session_pool_20260602_p10", + "saved_after_validated_replay": true + } + }, { "generation_id": "gen-ag06020614-947fd1", "created_at": "2026-06-02T06:14:41+00:00", diff --git a/llm_normalizer/data/autorun_generators/saved_sessions/assistant_saved_session_20260602071706_gen-ag06020717-08d833.json b/llm_normalizer/data/autorun_generators/saved_sessions/assistant_saved_session_20260602071706_gen-ag06020717-08d833.json new file mode 100644 index 0000000..5c8ff47 --- /dev/null +++ b/llm_normalizer/data/autorun_generators/saved_sessions/assistant_saved_session_20260602071706_gen-ag06020717-08d833.json @@ -0,0 +1,363 @@ +{ + "saved_at": "2026-06-02T07:17:06+00:00", + "generation_id": "gen-ag06020717-08d833", + "mode": "saved_user_sessions", + "title": "AGENT | Phase 12 wider multi-trajectory authority replay", + "agent_run": true, + "questions": [ + "приветик - че как там дела", + "Альтернатива Плюс", + "расскажи что можешь интересного", + "кайф - что там на складе по остаткам?", + "а исторические остатки на другие даты умеешь?", + "март 2016", + "По выбранному объекту \"Рабочая станция универсального специалиста (индивидуальное изготовление)\": где взяли это?", + "а кому продали?", + "ндс можешь прикинуть на дату покупки рабочей станции?", + "кто у нас самый доходный клиент за все время", + "кто нам должен денег на май 2017", + "а какой ндс мы должны примерно заплатить за этот период?", + "мы должны комуто денег на сегодня?", + "а нам?", + "ты умеешь считать дельту по договорам?", + "по чепурнову покажи все доки", + "а по свк", + "хвосты покажи по счету 60 на август 2022", + "Есть ли остатки товара, которые закупались очень давно", + "а по Альтернативе Плюс сколько лет активности в базе 1С?" + ], + "metadata": { + "assistant_prompt_version": null, + "decomposition_prompt_version": null, + "prompt_fingerprint": null, + "agent_focus": "Phase12 wide replay: organization authority, inventory selected-object, VAT/date carryover, payables/receivables today reset, counterparty docs, old stock and activity age.", + "architecture_phase": "turnaround_11", + "source_spec_file": "X:\\1C\\NDC_1C\\docs\\orchestration\\address_truth_harness_phase12_wider_saved_session_pool.json", + "scenario_id": "address_truth_harness_phase12_wider_saved_session_pool", + "semantic_tags": [ + "aggregate_revenue", + "bridge_inventory_to_vat", + "capability_over_followup", + "company_authority", + "company_selected", + "counterparty_followup", + "counterparty_root", + "cross_domain_pivot", + "date_carryover", + "display_label_integrity", + "documents", + "historical_date_anchor", + "human_answer_quality", + "inventory_aging", + "inventory_provenance", + "inventory_root", + "inventory_sale_trace", + "late_session_stability", + "meta_capability", + "meta_historical_capability", + "meta_smalltalk", + "organization_activity_age", + "organization_authority", + "proactive_scope_offer", + "same_date_restore", + "same_period_restore", + "selected_object", + "settlements_account_60", + "settlements_mirror_followup", + "settlements_payables", + "settlements_receivables", + "tail_authority_proof", + "today_scope", + "vat_followup" + ], + "validation_status": "accepted_domain_case_loop_scenario", + "validated_run_dir": "artifacts\\domain_runs\\address_truth_harness_phase12_wider_saved_session_pool_20260602_p10", + "saved_after_validated_replay": true, + "save_gate": { + "schema_version": "agent_semantic_save_gate_v1", + "validation_status": "accepted_domain_case_loop_scenario", + "validated_run_dir": "artifacts\\domain_runs\\address_truth_harness_phase12_wider_saved_session_pool_20260602_p10", + "final_status": "accepted", + "execution_status": "partial", + "scenario_id": "address_truth_harness_phase12_wider_saved_session_pool_20260602_p10", + "steps_total": 20, + "effective_runtime": { + "manifest_path": "artifacts\\domain_runs\\address_truth_harness_phase12_wider_saved_session_pool_20260602_p10\\effective_runtime.json", + "runner": "domain_case_loop.run-scenario", + "git_sha": "7995980215b6a9840c42c1c7ef73d6faa6612c37", + "backend_url": "http://127.0.0.1:8787", + "mcp_proxy_url": "http://127.0.0.1:6003", + "llm_provider": "local", + "llm_model": "unsloth/qwen3-30b-a3b-instruct-2507", + "temperature": 0.8, + "max_output_tokens": 900, + "prompt_version": "normalizer_v2_0_2", + "prompt_source": "file", + "prompt_hash": "f36e36a5e491cd24511b380e0c0059cb01aea5e2af738fb9f01671cd291735c1", + "prompt_registry_status": "pass" + }, + "saved_after_validated_replay": true + } + }, + "source_session_id": null, + "session": { + "session_id": null, + "mode": "agent_semantic_run", + "items": [ + { + "message_id": "agent-user-001", + "role": "user", + "text": "приветик - че как там дела", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-002", + "role": "user", + "text": "Альтернатива Плюс", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-003", + "role": "user", + "text": "расскажи что можешь интересного", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-004", + "role": "user", + "text": "кайф - что там на складе по остаткам?", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-005", + "role": "user", + "text": "а исторические остатки на другие даты умеешь?", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-006", + "role": "user", + "text": "март 2016", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-007", + "role": "user", + "text": "По выбранному объекту \"Рабочая станция универсального специалиста (индивидуальное изготовление)\": где взяли это?", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-008", + "role": "user", + "text": "а кому продали?", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-009", + "role": "user", + "text": "ндс можешь прикинуть на дату покупки рабочей станции?", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-010", + "role": "user", + "text": "кто у нас самый доходный клиент за все время", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-011", + "role": "user", + "text": "кто нам должен денег на май 2017", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-012", + "role": "user", + "text": "а какой ндс мы должны примерно заплатить за этот период?", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-013", + "role": "user", + "text": "мы должны комуто денег на сегодня?", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-014", + "role": "user", + "text": "а нам?", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-015", + "role": "user", + "text": "ты умеешь считать дельту по договорам?", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-016", + "role": "user", + "text": "по чепурнову покажи все доки", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-017", + "role": "user", + "text": "а по свк", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-018", + "role": "user", + "text": "хвосты покажи по счету 60 на август 2022", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-019", + "role": "user", + "text": "Есть ли остатки товара, которые закупались очень давно", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + }, + { + "message_id": "agent-user-020", + "role": "user", + "text": "а по Альтернативе Плюс сколько лет активности в базе 1С?", + "created_at": "2026-06-02T07:17:06+00:00", + "reply_type": null, + "trace_id": null, + "debug": null + } + ], + "agent_run": true, + "metadata": { + "assistant_prompt_version": null, + "decomposition_prompt_version": null, + "prompt_fingerprint": null, + "agent_focus": "Phase12 wide replay: organization authority, inventory selected-object, VAT/date carryover, payables/receivables today reset, counterparty docs, old stock and activity age.", + "architecture_phase": "turnaround_11", + "source_spec_file": "X:\\1C\\NDC_1C\\docs\\orchestration\\address_truth_harness_phase12_wider_saved_session_pool.json", + "scenario_id": "address_truth_harness_phase12_wider_saved_session_pool", + "semantic_tags": [ + "aggregate_revenue", + "bridge_inventory_to_vat", + "capability_over_followup", + "company_authority", + "company_selected", + "counterparty_followup", + "counterparty_root", + "cross_domain_pivot", + "date_carryover", + "display_label_integrity", + "documents", + "historical_date_anchor", + "human_answer_quality", + "inventory_aging", + "inventory_provenance", + "inventory_root", + "inventory_sale_trace", + "late_session_stability", + "meta_capability", + "meta_historical_capability", + "meta_smalltalk", + "organization_activity_age", + "organization_authority", + "proactive_scope_offer", + "same_date_restore", + "same_period_restore", + "selected_object", + "settlements_account_60", + "settlements_mirror_followup", + "settlements_payables", + "settlements_receivables", + "tail_authority_proof", + "today_scope", + "vat_followup" + ], + "validation_status": "accepted_domain_case_loop_scenario", + "validated_run_dir": "artifacts\\domain_runs\\address_truth_harness_phase12_wider_saved_session_pool_20260602_p10", + "saved_after_validated_replay": true, + "save_gate": { + "schema_version": "agent_semantic_save_gate_v1", + "validation_status": "accepted_domain_case_loop_scenario", + "validated_run_dir": "artifacts\\domain_runs\\address_truth_harness_phase12_wider_saved_session_pool_20260602_p10", + "final_status": "accepted", + "execution_status": "partial", + "scenario_id": "address_truth_harness_phase12_wider_saved_session_pool_20260602_p10", + "steps_total": 20, + "effective_runtime": { + "manifest_path": "artifacts\\domain_runs\\address_truth_harness_phase12_wider_saved_session_pool_20260602_p10\\effective_runtime.json", + "runner": "domain_case_loop.run-scenario", + "git_sha": "7995980215b6a9840c42c1c7ef73d6faa6612c37", + "backend_url": "http://127.0.0.1:8787", + "mcp_proxy_url": "http://127.0.0.1:6003", + "llm_provider": "local", + "llm_model": "unsloth/qwen3-30b-a3b-instruct-2507", + "temperature": 0.8, + "max_output_tokens": 900, + "prompt_version": "normalizer_v2_0_2", + "prompt_source": "file", + "prompt_hash": "f36e36a5e491cd24511b380e0c0059cb01aea5e2af738fb9f01671cd291735c1", + "prompt_registry_status": "pass" + }, + "saved_after_validated_replay": true + } + } + } +} diff --git a/llm_normalizer/data/eval_cases/assistant_autogen_saved_user_sessions_20260602071706_gen-ag06020717-08d833.json b/llm_normalizer/data/eval_cases/assistant_autogen_saved_user_sessions_20260602071706_gen-ag06020717-08d833.json new file mode 100644 index 0000000..2bd1a84 --- /dev/null +++ b/llm_normalizer/data/eval_cases/assistant_autogen_saved_user_sessions_20260602071706_gen-ag06020717-08d833.json @@ -0,0 +1,85 @@ +{ + "suite_id": "assistant_saved_session_gen-ag06020717-08d833", + "suite_version": "0.1.0", + "schema_version": "assistant_saved_session_suite_v0_1", + "generated_at": "2026-06-02T07:17:06+00:00", + "generation_id": "gen-ag06020717-08d833", + "mode": "saved_user_sessions", + "title": "AGENT | Phase 12 wider multi-trajectory authority replay", + "domain": "address_phase12_wider_saved_session_pool", + "scenario_count": 1, + "case_ids": [ + "SAVED-001" + ], + "cases": [ + { + "case_id": "SAVED-001", + "scenario_tag": "agent_saved_user_sessions", + "title": "AGENT | Phase 12 wider multi-trajectory authority replay", + "question_type": "followup", + "broadness_level": "medium", + "turns": [ + { + "user_message": "приветик - че как там дела" + }, + { + "user_message": "Альтернатива Плюс" + }, + { + "user_message": "расскажи что можешь интересного" + }, + { + "user_message": "кайф - что там на складе по остаткам?" + }, + { + "user_message": "а исторические остатки на другие даты умеешь?" + }, + { + "user_message": "март 2016" + }, + { + "user_message": "По выбранному объекту \"Рабочая станция универсального специалиста (индивидуальное изготовление)\": где взяли это?" + }, + { + "user_message": "а кому продали?" + }, + { + "user_message": "ндс можешь прикинуть на дату покупки рабочей станции?" + }, + { + "user_message": "кто у нас самый доходный клиент за все время" + }, + { + "user_message": "кто нам должен денег на май 2017" + }, + { + "user_message": "а какой ндс мы должны примерно заплатить за этот период?" + }, + { + "user_message": "мы должны комуто денег на сегодня?" + }, + { + "user_message": "а нам?" + }, + { + "user_message": "ты умеешь считать дельту по договорам?" + }, + { + "user_message": "по чепурнову покажи все доки" + }, + { + "user_message": "а по свк" + }, + { + "user_message": "хвосты покажи по счету 60 на август 2022" + }, + { + "user_message": "Есть ли остатки товара, которые закупались очень давно" + }, + { + "user_message": "а по Альтернативе Плюс сколько лет активности в базе 1С?" + } + ] + } + ] +} diff --git a/scripts/domain_case_loop.py b/scripts/domain_case_loop.py index 863f38b..5f328c8 100644 --- a/scripts/domain_case_loop.py +++ b/scripts/domain_case_loop.py @@ -9,7 +9,7 @@ import subprocess import sys import textwrap import time -from datetime import datetime, timezone +from datetime import date, datetime, timezone from pathlib import Path from typing import Any from urllib.error import HTTPError, URLError @@ -712,6 +712,7 @@ def merge_scenario_date_scope( def question_resets_temporal_scope(value: Any) -> bool: normalized = str(value or "").casefold().replace("ё", "е") + repaired_normalized = repair_text_mojibake(str(value or "")).casefold().replace("ё", "е") same_scope_markers = ( "за тот же период", "на тот же период", @@ -727,6 +728,8 @@ def question_resets_temporal_scope(value: Any) -> bool: ) if any(marker in normalized for marker in same_scope_markers): return False + if any(marker in repaired_normalized for marker in ("сегодня", "на сегодня", "текущую дату", "сейчас")): + return True markers = ( "за все доступное время", "за все время", @@ -779,6 +782,41 @@ def normalize_bindings(raw_bindings: Any) -> dict[str, Any]: return {str(key): normalize_binding_value(value) for key, value in raw_bindings.items()} +def build_runtime_bindings() -> dict[str, str]: + today = date.today() + today_iso = today.isoformat() + today_dot = today.strftime("%d.%m.%Y") + return { + "today_iso": today_iso, + "today_dot": today_dot, + "today_iso_regex": re.escape(today_iso), + "today_dot_regex": re.escape(today_dot), + } + + +def resolve_runtime_placeholders(value: Any) -> Any: + if not isinstance(value, str) or "{{" not in value: + return value + runtime_bindings = build_runtime_bindings() + + def replace(match: re.Match[str]) -> str: + placeholder = match.group(1).strip() + if not placeholder.startswith("runtime."): + return match.group(0) + key = placeholder.split(".", 1)[1] + return runtime_bindings.get(key, match.group(0)) + + return re.sub(r"{{\s*([^{}]+?)\s*}}", replace, value) + + +def normalize_pattern_list(*raw_values: Any) -> list[str]: + for raw_value in raw_values: + values = normalize_string_list(raw_value) + if values: + return [str(resolve_runtime_placeholders(value)) for value in values] + return [] + + def normalize_string_list(raw_values: Any) -> list[str]: if isinstance(raw_values, str): value = raw_values.strip() @@ -820,12 +858,17 @@ def normalize_validation_filters(raw_filters: Any) -> dict[str, str]: key = str(raw_key or "").strip() if not key: continue + resolved_value = resolve_runtime_placeholders(raw_value) if key in {"as_of_date", "period_from", "period_to"}: - normalized_value = normalize_iso_date(raw_value) + normalized_value = normalize_iso_date(resolved_value) if normalized_value: normalized[key] = normalized_value + continue + text_value = str(resolved_value or "").strip() + if text_value: + normalized[key] = text_value continue - text_value = str(raw_value or "").strip() + text_value = str(resolved_value or "").strip() if text_value: normalized[key] = text_value return normalized @@ -1710,9 +1753,18 @@ def normalize_step_definition(index: int, raw_step: Any) -> dict[str, Any]: "required_answer_shape": ( str(raw_step.get("required_answer_shape") or raw_step.get("expected_answer_shape") or "").strip() or None ), - "forbidden_answer_patterns": normalize_string_list(raw_step.get("forbidden_answer_patterns")), - "required_answer_patterns_any": normalize_string_list(raw_step.get("required_answer_patterns_any")), - "required_answer_patterns_all": normalize_string_list(raw_step.get("required_answer_patterns_all")), + "forbidden_answer_patterns": normalize_pattern_list( + raw_step.get("forbidden_answer_patterns"), + raw_step.get("forbidden_direct_answer_patterns"), + ), + "required_answer_patterns_any": normalize_pattern_list( + raw_step.get("required_answer_patterns_any"), + raw_step.get("required_direct_answer_patterns_any"), + ), + "required_answer_patterns_all": normalize_pattern_list( + raw_step.get("required_answer_patterns_all"), + raw_step.get("required_direct_answer_patterns_all"), + ), "semantic_tags": normalize_string_list(raw_step.get("semantic_tags")), "required_carryover_invariants": normalize_string_list(raw_step.get("required_carryover_invariants")), "invariant_severity": normalize_invariant_severity(raw_step.get("invariant_severity")), @@ -3236,17 +3288,23 @@ def save_scenario_step_bundle( write_text(step_dir / "resolved_question.txt", f"{step_state['question_resolved']}\n") -def is_effectively_complete_partial_step(step_output: dict[str, Any]) -> bool: +def is_effectively_complete_validated_step(step_output: dict[str, Any]) -> bool: execution_status = str(step_output.get("execution_status") or step_output.get("status") or "").strip() if execution_status == "exact": return True - if execution_status != "partial": - return False if str(step_output.get("acceptance_status") or "").strip() != "validated": return False + if execution_status == "needs_exact_capability": + return step_output.get("clean_meta_chat_answer_validated") is True + if execution_status != "partial": + return False return ( step_output.get("missing_axis_clarification_validated") is True or step_output.get("clarification_answer_validated") is True + or step_output.get("bounded_mcp_answer_validated") is True + or step_output.get("memory_checkpoint_validated") is True + or step_output.get("runtime_factual_answer_validated") is True + or step_output.get("guarded_insufficiency_validated") is True ) @@ -3256,9 +3314,14 @@ def derive_scenario_execution_status(step_outputs: dict[str, dict[str, Any]]) -> return "blocked" if any(status == "blocked" for status in statuses): return "blocked" - if any(status == "needs_exact_capability" for status in statuses): + if any( + status == "needs_exact_capability" and not is_effectively_complete_validated_step(item) + for status, item in zip(statuses, step_outputs.values()) + ): return "needs_exact_capability" - if any(not is_effectively_complete_partial_step(item) for item in step_outputs.values()): + if any(not is_effectively_complete_validated_step(item) for item in step_outputs.values()): + return "partial" + if any(status != "exact" for status in statuses): return "partial" return "exact" @@ -4302,6 +4365,8 @@ def derive_repair_target_severity(step_output: dict[str, Any]) -> str: return "P0" if route_candidate_status == "needs_route_enablement": return "P1" + if is_effectively_complete_validated_step(step_output): + return "P2" if acceptance_status in {"rejected", "needs_exact_capability"}: return "P1" if execution_status in {"partial", "needs_exact_capability"} or reply_type == "partial_coverage": diff --git a/scripts/save_agent_semantic_run.py b/scripts/save_agent_semantic_run.py index 5b38e06..875ebd3 100644 --- a/scripts/save_agent_semantic_run.py +++ b/scripts/save_agent_semantic_run.py @@ -333,6 +333,90 @@ def validate_domain_case_loop_pack_dir(run_dir: Path) -> dict[str, Any]: } +def parse_markdown_status_field(markdown: str, field_name: str) -> str: + pattern = re.compile(rf"^-\s*{re.escape(field_name)}:\s*`([^`]+)`\s*$", re.MULTILINE) + match = pattern.search(markdown) + return str(match.group(1) if match else "").strip() + + +def step_has_validated_partial_or_meta_gate(step_state: dict[str, Any]) -> bool: + execution_status = str(step_state.get("execution_status") or step_state.get("status") or "").strip() + if execution_status == "exact": + return True + if str(step_state.get("acceptance_status") or step_state.get("status") or "").strip() != "validated": + return False + if execution_status == "needs_exact_capability": + return step_state.get("clean_meta_chat_answer_validated") is True + if execution_status != "partial": + return False + return any( + step_state.get(flag_name) is True + for flag_name in ( + "bounded_mcp_answer_validated", + "memory_checkpoint_validated", + "runtime_factual_answer_validated", + "guarded_insufficiency_validated", + "clarification_answer_validated", + "missing_axis_clarification_validated", + ) + ) + + +def validate_domain_case_loop_scenario_dir(run_dir: Path) -> dict[str, Any]: + run_dir = run_dir.resolve() + effective_runtime = require_effective_runtime_manifest(run_dir) + scenario_state = load_json_object(run_dir / "scenario_state.json", "Validated domain-case-loop scenario_state.json") + scenario_manifest = load_json_object( + run_dir / "scenario_manifest.json", + "Validated domain-case-loop scenario_manifest.json", + ) + final_status_markdown = (run_dir / "final_status.md").read_text(encoding="utf-8") + final_status = parse_markdown_status_field(final_status_markdown, "status") + execution_status = parse_markdown_status_field(final_status_markdown, "execution_status") + step_outputs = scenario_state.get("step_outputs") if isinstance(scenario_state.get("step_outputs"), dict) else {} + manifest_steps = scenario_manifest.get("steps") if isinstance(scenario_manifest.get("steps"), list) else [] + + problems: list[str] = [] + assert_status(final_status, "accepted", "final_status.status", problems) + if not step_outputs: + problems.append("scenario_state.step_outputs=empty") + if manifest_steps and len(step_outputs) != len(manifest_steps): + problems.append(f"step_count={len(step_outputs)} dict[str, Any]: run_dir = run_dir.resolve() if (run_dir / "loop_state.json").exists(): @@ -341,6 +425,8 @@ def validate_accepted_run_dir(run_dir: Path) -> dict[str, Any]: return validate_truth_harness_run_dir(run_dir) if (run_dir / "pack_state.json").exists() and (run_dir / "repair_targets.json").exists(): return validate_domain_case_loop_pack_dir(run_dir) + if (run_dir / "scenario_state.json").exists() and (run_dir / "scenario_manifest.json").exists(): + return validate_domain_case_loop_scenario_dir(run_dir) return validate_truth_harness_run_dir(run_dir) diff --git a/scripts/test_domain_case_loop_step_state.py b/scripts/test_domain_case_loop_step_state.py index 5928a79..a05c1c2 100644 --- a/scripts/test_domain_case_loop_step_state.py +++ b/scripts/test_domain_case_loop_step_state.py @@ -86,6 +86,64 @@ class DomainCaseLoopStepStateTests(unittest.TestCase): self.assertEqual(step_state["mcp_discovery_route_candidate_provided_axes"], ["period"]) self.assertFalse(step_state["mcp_discovery_route_candidate_executable_now"]) + def test_scenario_execution_status_treats_validated_meta_and_guarded_partial_as_complete(self) -> None: + step_outputs = { + "smalltalk": { + "execution_status": "needs_exact_capability", + "acceptance_status": "validated", + "clean_meta_chat_answer_validated": True, + }, + "exact_inventory": { + "execution_status": "exact", + "acceptance_status": "validated", + }, + "account_60_boundary": { + "execution_status": "partial", + "acceptance_status": "validated", + "runtime_factual_answer_validated": True, + }, + } + + self.assertEqual(dcl.derive_scenario_execution_status(step_outputs), "partial") + self.assertEqual(dcl.derive_scenario_status(step_outputs), "accepted") + + def test_today_scope_required_filter_and_direct_patterns_are_enforced(self) -> None: + self.assertTrue(dcl.question_resets_temporal_scope("мы должны комуто денег на сегодня?")) + + step = dcl.normalize_step_definition( + 1, + { + "step_id": "payables_today", + "title": "Payables today", + "question": "мы должны комуто денег на сегодня?", + "expected_intents": ["payables_confirmed_as_of_date"], + "required_filters": {"as_of_date": "{{runtime.today_iso}}"}, + "required_direct_answer_patterns_any": ["{{runtime.today_dot_regex}}"], + }, + ) + runtime_today = dcl.build_runtime_bindings()["today_iso"] + runtime_today_pattern = dcl.build_runtime_bindings()["today_dot_regex"] + self.assertEqual(step["required_filters"]["as_of_date"], runtime_today) + self.assertEqual(step["required_answer_patterns_any"], [runtime_today_pattern]) + + step_state = dcl.validate_step_contract( + { + "execution_status": "exact", + "reply_type": "factual", + "detected_intent": "payables_confirmed_as_of_date", + "required_filters": step["required_filters"], + "required_answer_patterns_any": step["required_answer_patterns_any"], + "extracted_filters": {"as_of_date": "2017-05-31"}, + "assistant_text": "на 31.05.2017 мы должны 3.433.472,35 ₽.", + "actual_direct_answer": "на 31.05.2017 мы должны 3.433.472,35 ₽.", + "top_non_empty_lines": ["на 31.05.2017 мы должны 3.433.472,35 ₽."], + } + ) + + self.assertIn("wrong_as_of_date", step_state["violated_invariants"]) + self.assertIn("required_answer_patterns_any_missing", step_state["violated_invariants"]) + self.assertEqual(step_state["acceptance_status"], "rejected") + def test_repair_targets_promote_route_candidate_enablement_gaps(self) -> None: repair_targets = dcl.build_deterministic_repair_targets( {"pack_id": "route_candidate_pack", "domain": "open_world", "final_status": "accepted"},