312 lines
14 KiB
Python
312 lines
14 KiB
Python
# #region Test.Scenario.Compiler [C:3] [TYPE Module] [SEMANTICS testing,scenario,compiler,determinism]
|
|
# @defgroup Test.Scenario Compiler determinism and safety tests.
|
|
# @LAYER Test
|
|
# @RELATION BINDS_TO -> [ScenarioGraph.Compiler.Compile]
|
|
# @RATIONALE The compiler is the determinism and safety boundary — byte-stable output and no unsafe steps.
|
|
# @REJECTED Testing only single-shot compile — would hide non-determinism and selector/baseline regressions.
|
|
|
|
from __future__ import annotations
|
|
|
|
from src.services.dashboard_testing.scenario.compiler import CompileScenarioRequest, compile_scenario
|
|
from src.services.dashboard_testing.scenario.templates import REGISTERED_ACTIONS
|
|
from src.services.dashboard_testing.scenario.validator import validate_scenario
|
|
|
|
FULL = {
|
|
"browser": True, # browser is base infrastructure, always available in test env
|
|
"native_filters": True, "text_filter": True, "table_filter": True, "pagination": True,
|
|
"row_edit": True, "bulk_edit": True, "persistence_refresh": True, "time_rollover": True,
|
|
"xlsx_export": True, "cross_dashboard": True, "superset_metric": True,
|
|
"dataset_field_read": True, "screenshot": True, "repository_write": True,
|
|
"safe_test_data": True, "safe_clock_fixture": True,
|
|
}
|
|
|
|
|
|
def _req(selected: list[str] | None = None, parameters: dict | None = None, capabilities: dict | None = None) -> CompileScenarioRequest:
|
|
return CompileScenarioRequest(
|
|
agent_run_id="550e8400-e29b-41d4-a716-446655440000",
|
|
objective={"goal": "verify filters metric xlsx", "selected_case_ids": selected or ["B01", "C04", "C05", "T01"], "rationale": "rc"},
|
|
query_model={"dashboard_key": "fi-0080"},
|
|
checklist_catalog_version=1,
|
|
baseline_version="2026-07-01",
|
|
capabilities=capabilities or FULL,
|
|
parameters=parameters or {"test_date": {"type": "date"}, "counterparty": {"type": "string"}},
|
|
has_dataset_fields=True,
|
|
environment_id="env-prod-01",
|
|
dashboard_id=80,
|
|
dashboard_name="FI-0080",
|
|
)
|
|
|
|
|
|
def test_repeated_compile_byte_identical() -> None:
|
|
a = compile_scenario(_req())
|
|
b = compile_scenario(_req())
|
|
assert a.scenario.revision_hash == b.scenario.revision_hash
|
|
assert a.scenario.canonical_bytes() == b.scenario.canonical_bytes()
|
|
|
|
|
|
def test_shuffled_selected_case_ids_identical() -> None:
|
|
base = compile_scenario(_req(["B01", "C04", "C05", "T01"]))
|
|
shuffled = compile_scenario(_req(["T01", "C05", "C04", "B01"]))
|
|
assert base.scenario.revision_hash == shuffled.scenario.revision_hash
|
|
|
|
|
|
def test_shuffled_parameter_order_identical() -> None:
|
|
base = compile_scenario(_req(parameters={"test_date": {"type": "date"}, "counterparty": {"type": "string"}}))
|
|
shuffled = compile_scenario(_req(parameters={"counterparty": {"type": "string"}, "test_date": {"type": "date"}}))
|
|
assert base.scenario.revision_hash == shuffled.scenario.revision_hash
|
|
|
|
|
|
def test_all_steps_use_registered_actions() -> None:
|
|
result = compile_scenario(_req())
|
|
for step in result.scenario.steps:
|
|
assert step.action in REGISTERED_ACTIONS, f"unregistered action {step.action}"
|
|
assert REGISTERED_ACTIONS[step.action]["tool"] == step.tool
|
|
|
|
|
|
def test_coverage_contains_all_selected_cases() -> None:
|
|
result = compile_scenario(_req(["B01", "C04", "C05", "T01"]))
|
|
covered = {c.case_id for c in result.scenario.checklist_coverage}
|
|
assert {"B01", "C04", "C05", "T01"} <= covered
|
|
|
|
|
|
def test_missing_selector_produces_blocker() -> None:
|
|
caps = dict(FULL, safe_test_data=True)
|
|
result = compile_scenario(_req(["B02"], parameters={}, capabilities=caps))
|
|
codes = {b.code for b in result.blockers}
|
|
assert "NEEDS_SELECTOR" in codes
|
|
# the interaction step itself is flagged needs_selector
|
|
assert any(s.automation_status == "needs_selector" for s in result.scenario.steps)
|
|
|
|
|
|
def test_no_sql_action_emitted() -> None:
|
|
result = compile_scenario(_req(["T01", "T02", "T03"]))
|
|
for step in result.scenario.steps:
|
|
assert step.action not in {"raw_sql", "execute_sql"}
|
|
assert "sql" not in step.action.lower()
|
|
|
|
|
|
def test_unsupported_case_warned_not_dropped() -> None:
|
|
caps = dict(FULL, xlsx_export=False, safe_test_data=True)
|
|
result = compile_scenario(_req(["C04"], capabilities=caps))
|
|
covered = {c.case_id for c in result.scenario.checklist_coverage}
|
|
assert "C04" in covered # still present in coverage with rationale
|
|
assert any(b.code == "UNSUPPORTED_ACTION" for b in result.blockers)
|
|
assert result.scenario.steps == []
|
|
assert not any(w.code == "UNSUPPORTED_CASE" for w in result.warnings)
|
|
|
|
|
|
def test_cyrillic_goal_produces_ascii_scenario_id() -> None:
|
|
"""A Russian objective goal must NOT leak into the technical scenario_id.
|
|
|
|
Python str.isalnum() is Unicode-aware and would keep Cyrillic in the slug,
|
|
yielding an id that violates the ASCII _STEP_ID_RE contract (compile 422),
|
|
forcing the agent to retry. The id generator must sanitize to ASCII so the
|
|
first compile always validates.
|
|
"""
|
|
import re as _re
|
|
|
|
from src.services.dashboard_testing.scenario.models import _STEP_ID_RE
|
|
|
|
result = compile_scenario(_req_with_goal("Проверить фильтры метрики и экспорт в Excel"))
|
|
scenario_id = result.scenario.scenario_id
|
|
assert _re.fullmatch(_STEP_ID_RE.pattern, scenario_id), f"scenario_id not ASCII/valid: {scenario_id!r}"
|
|
assert not any(ord(c) > 127 for c in scenario_id), f"scenario_id contains non-ASCII: {scenario_id!r}"
|
|
# The Russian objective must still live in the content, not the identifier.
|
|
assert "Проверить" in str(result.scenario.objective)
|
|
|
|
|
|
def test_cyrillic_only_goal_still_yields_valid_id() -> None:
|
|
"""Even a fully-Cyrillic goal must produce a pattern-valid id (no 422)."""
|
|
import re as _re
|
|
|
|
from src.services.dashboard_testing.scenario.models import _STEP_ID_RE
|
|
|
|
result = compile_scenario(_req_with_goal("Проверить отчётность"))
|
|
assert _re.fullmatch(_STEP_ID_RE.pattern, result.scenario.scenario_id)
|
|
|
|
|
|
def _req_with_goal(goal: str) -> CompileScenarioRequest:
|
|
from dataclasses import replace
|
|
|
|
return replace(_req(), objective=dict(_req().objective, goal=goal))
|
|
# #endregion Test.Scenario.Compiler
|
|
|
|
|
|
# #region Test.Scenario.Compiler.Branches [C:3] [TYPE Module]
|
|
# @defgroup Scalar parameter specs, unsupported/human-checkpoint cases, build-step automation branches.
|
|
|
|
def test_scalar_parameter_spec() -> None:
|
|
"""A bare scalar parameter spec (not a dict) compiles into a string parameter."""
|
|
result = compile_scenario(_req(parameters={"my_param": "default-value"}))
|
|
names = [p.name for p in result.scenario.parameters]
|
|
assert "my_param" in names
|
|
p = next(p for p in result.scenario.parameters if p.name == "my_param")
|
|
assert p.status == "resolved"
|
|
assert p.value == "default-value"
|
|
|
|
|
|
def _selector_params() -> dict:
|
|
return {
|
|
"test_date": {"type": "date", "default": "2026-07-01"},
|
|
"counterparty": {"type": "string", "default": "ACME"},
|
|
"selector_hint": {
|
|
"type": "selector_hint", "default": "#native-filter", "required": False,
|
|
"affected_step_ids": ["phase-2-B01-apply_native_filter"],
|
|
},
|
|
}
|
|
|
|
|
|
def test_visual_case_emits_canonical_capture_compare_chain() -> None:
|
|
caps = dict(FULL, baseline=True)
|
|
result = compile_scenario(_req(["B01"], parameters=_selector_params(), capabilities=caps))
|
|
steps = result.scenario.steps
|
|
assert [s.action for s in steps] == ["apply_native_filter", "capture_screenshot", "compare_to_baseline"]
|
|
assert [s.tool for s in steps] == ["browser", "screenshot", "assertion"]
|
|
assert steps[0].depends_on == []
|
|
assert steps[1].depends_on == [steps[0].id]
|
|
assert steps[2].depends_on == [steps[1].id]
|
|
assert "text_filter" not in {s.action for s in steps}
|
|
assert "apply_filters" not in {s.action for s in steps}
|
|
assert "register_artifact" not in {s.action for s in steps}
|
|
validation = validate_scenario(result.scenario)
|
|
assert validation.valid is True
|
|
assert not validation.errors
|
|
assert not validation.blockers
|
|
|
|
|
|
def test_text_filter_alias_is_not_emitted() -> None:
|
|
caps = dict(FULL, baseline=True)
|
|
result = compile_scenario(_req(["B02"], parameters=_selector_params(), capabilities=caps))
|
|
actions = [s.action for s in result.scenario.steps]
|
|
assert "text_filter" not in actions
|
|
assert actions[0] == "apply_native_filter"
|
|
assert actions == ["apply_native_filter", "capture_screenshot", "compare_to_baseline"]
|
|
|
|
|
|
def test_selected_unsupported_case_blocks_without_steps() -> None:
|
|
caps = dict(FULL, xlsx_export=False, safe_test_data=True)
|
|
result = compile_scenario(_req(["C04"], capabilities=caps))
|
|
assert result.scenario.steps == []
|
|
assert any(b.code == "UNSUPPORTED_ACTION" for b in result.blockers)
|
|
assert {c.case_id for c in result.scenario.checklist_coverage} >= {"C04"}
|
|
|
|
|
|
def test_unselected_unsupported_does_not_block_selected_chain() -> None:
|
|
caps = dict(FULL, xlsx_export=False, baseline=True)
|
|
result = compile_scenario(_req(["B01"], parameters=_selector_params(), capabilities=caps))
|
|
assert not any(b.code == "UNSUPPORTED_ACTION" for b in result.blockers)
|
|
covered = {c.case_id: c.classification for c in result.scenario.checklist_coverage}
|
|
assert covered["C04"] == "unsupported"
|
|
assert [s.action for s in result.scenario.steps] == [
|
|
"apply_native_filter", "capture_screenshot", "compare_to_baseline",
|
|
]
|
|
|
|
|
|
def test_visual_compare_without_screenshot_is_blocker() -> None:
|
|
caps = dict(FULL, screenshot=False, baseline=True)
|
|
result = compile_scenario(_req(["B01"], parameters=_selector_params(), capabilities=caps))
|
|
actions = [s.action for s in result.scenario.steps]
|
|
assert actions == ["apply_native_filter"]
|
|
assert "capture_screenshot" not in actions
|
|
assert "compare_to_baseline" not in actions
|
|
assert any(b.code == "MISSING_SCREENSHOT" for b in result.blockers)
|
|
|
|
|
|
# #region Test.Scenario.Compiler.SelectorBinding [C:2] [TYPE Function] [SEMANTICS test,selector,binding]
|
|
def test_selector_hint_only_resolves_its_bound_step() -> None:
|
|
result = compile_scenario(_req(
|
|
["B01", "B02"],
|
|
parameters={"selector": {
|
|
"type": "selector_hint", "default": "#filter", "required": False,
|
|
"affected_step_ids": ["phase-2-B01-apply_native_filter"],
|
|
}},
|
|
capabilities=dict(FULL, baseline=True),
|
|
))
|
|
|
|
statuses = {step.id: step.automation_status for step in result.scenario.steps}
|
|
assert statuses["phase-2-B01-apply_native_filter"] == "ready"
|
|
assert statuses["phase-2-B02-apply_native_filter"] == "needs_selector"
|
|
assert [item.step_id for item in result.blockers if item.code == "NEEDS_SELECTOR"] == [
|
|
"phase-2-B02-apply_native_filter",
|
|
]
|
|
# #endregion Test.Scenario.Compiler.SelectorBinding
|
|
|
|
|
|
def test_metric_case_appends_compare_when_baseline_present() -> None:
|
|
caps = dict(FULL, baseline=True)
|
|
result = compile_scenario(_req(["T01"], capabilities=caps))
|
|
steps = result.scenario.steps
|
|
assert [s.action for s in steps] == ["dataset_field_assert", "compare_to_baseline"]
|
|
assert steps[1].depends_on == [steps[0].id]
|
|
assert "capture_screenshot" not in {s.action for s in steps}
|
|
|
|
|
|
def test_unsafe_mutation_case_is_dropped_not_human() -> None:
|
|
# AGSCN-FR-019/020: B05 without safe_test_data maps to unsupported — the compiler
|
|
# emits no steps for it (no human step, no screenshot, nothing scheduled).
|
|
caps = dict(FULL, safe_test_data=False)
|
|
result = compile_scenario(_req(["B05"], capabilities=caps))
|
|
assert [s.action for s in result.scenario.steps] == []
|
|
assert "capture_screenshot" not in {s.action for s in result.scenario.steps}
|
|
assert "human_checkpoint" not in {s.action for s in result.scenario.steps}
|
|
|
|
|
|
def _evaluation_spec():
|
|
from src.services.dashboard_testing.execution.decision_policy import DecisionPolicy
|
|
from src.services.dashboard_testing.scenario.models import (
|
|
AgentEvaluationSpec,
|
|
EvaluationCriterion,
|
|
EvaluationLimits,
|
|
)
|
|
|
|
return AgentEvaluationSpec.model_validate({
|
|
"schema_version": 1,
|
|
"spec_id": "e5e5e5e5-e5e5-4e5e-8e5e-e5e5e5e5e5e5",
|
|
"provider_id": "llm-provider-a",
|
|
"provider_version": "1",
|
|
"model_id": "vision-model",
|
|
"model_version": "2026-08",
|
|
"prompt_template_id": "agent-evaluation-prompt",
|
|
"prompt_template_version": "1.0.0",
|
|
"prompt_template_hash": "a" * 64,
|
|
"evidence_refs": ["artifact-1"],
|
|
"comparison_refs": ["44444444-4444-4444-8444-444444444444"],
|
|
"output_schema": "agent-evaluation.schema.json",
|
|
"decision_policy": DecisionPolicy(policy_id="baseline-semantic", version="1.0.0"),
|
|
"limits": EvaluationLimits(
|
|
timeout_ms=60000, max_images=4, max_input_tokens=32000,
|
|
max_output_tokens=2000, max_cost="1.00", currency="USD",
|
|
),
|
|
"trust_policy_hash": "b" * 64,
|
|
"criteria": [
|
|
EvaluationCriterion(
|
|
criterion_id="crit-visual", criterion_kind="semantic",
|
|
description="visual layout matches", comparison_id=None,
|
|
),
|
|
],
|
|
})
|
|
|
|
|
|
def test_evaluate_declared_spec_emitted_only_when_request_has_spec() -> None:
|
|
from dataclasses import replace
|
|
|
|
caps = dict(FULL, baseline=True)
|
|
without_spec = compile_scenario(_req(["B01"], parameters=_selector_params(), capabilities=caps))
|
|
assert "evaluate_declared_spec" not in {s.action for s in without_spec.scenario.steps}
|
|
|
|
spec = _evaluation_spec()
|
|
with_spec = compile_scenario(replace(
|
|
_req(["B01"], parameters=_selector_params(), capabilities=caps),
|
|
agent_evaluation_spec=spec,
|
|
))
|
|
steps = with_spec.scenario.steps
|
|
assert [s.action for s in steps] == [
|
|
"apply_native_filter", "capture_screenshot", "compare_to_baseline", "evaluate_declared_spec",
|
|
]
|
|
assert steps[3].depends_on == [steps[2].id]
|
|
assert steps[3].agent_evaluation_spec == spec
|
|
assert steps[3].decision_policy is None
|
|
|
|
|
|
# #endregion Test.Scenario.Compiler.Branches
|