Files
ss-tools/backend/tests/services/dashboard_testing/scenario/test_compiler.py

312 lines
14 KiB
Python

# #region Test.Scenario.Compiler [C:3] [TYPE Module] [SEMANTICS testing,scenario,compiler,determinism]
# @defgroup Test.Scenario Compiler determinism and safety tests.
# @LAYER Test
# @RELATION BINDS_TO -> [ScenarioGraph.Compiler.Compile]
# @RATIONALE The compiler is the determinism and safety boundary — byte-stable output and no unsafe steps.
# @REJECTED Testing only single-shot compile — would hide non-determinism and selector/baseline regressions.
from __future__ import annotations
from src.services.dashboard_testing.scenario.compiler import CompileScenarioRequest, compile_scenario
from src.services.dashboard_testing.scenario.templates import REGISTERED_ACTIONS
from src.services.dashboard_testing.scenario.validator import validate_scenario
FULL = {
"browser": True, # browser is base infrastructure, always available in test env
"native_filters": True, "text_filter": True, "table_filter": True, "pagination": True,
"row_edit": True, "bulk_edit": True, "persistence_refresh": True, "time_rollover": True,
"xlsx_export": True, "cross_dashboard": True, "superset_metric": True,
"dataset_field_read": True, "screenshot": True, "repository_write": True,
"safe_test_data": True, "safe_clock_fixture": True,
}
def _req(selected: list[str] | None = None, parameters: dict | None = None, capabilities: dict | None = None) -> CompileScenarioRequest:
return CompileScenarioRequest(
agent_run_id="550e8400-e29b-41d4-a716-446655440000",
objective={"goal": "verify filters metric xlsx", "selected_case_ids": selected or ["B01", "C04", "C05", "T01"], "rationale": "rc"},
query_model={"dashboard_key": "fi-0080"},
checklist_catalog_version=1,
baseline_version="2026-07-01",
capabilities=capabilities or FULL,
parameters=parameters or {"test_date": {"type": "date"}, "counterparty": {"type": "string"}},
has_dataset_fields=True,
environment_id="env-prod-01",
dashboard_id=80,
dashboard_name="FI-0080",
)
def test_repeated_compile_byte_identical() -> None:
a = compile_scenario(_req())
b = compile_scenario(_req())
assert a.scenario.revision_hash == b.scenario.revision_hash
assert a.scenario.canonical_bytes() == b.scenario.canonical_bytes()
def test_shuffled_selected_case_ids_identical() -> None:
base = compile_scenario(_req(["B01", "C04", "C05", "T01"]))
shuffled = compile_scenario(_req(["T01", "C05", "C04", "B01"]))
assert base.scenario.revision_hash == shuffled.scenario.revision_hash
def test_shuffled_parameter_order_identical() -> None:
base = compile_scenario(_req(parameters={"test_date": {"type": "date"}, "counterparty": {"type": "string"}}))
shuffled = compile_scenario(_req(parameters={"counterparty": {"type": "string"}, "test_date": {"type": "date"}}))
assert base.scenario.revision_hash == shuffled.scenario.revision_hash
def test_all_steps_use_registered_actions() -> None:
result = compile_scenario(_req())
for step in result.scenario.steps:
assert step.action in REGISTERED_ACTIONS, f"unregistered action {step.action}"
assert REGISTERED_ACTIONS[step.action]["tool"] == step.tool
def test_coverage_contains_all_selected_cases() -> None:
result = compile_scenario(_req(["B01", "C04", "C05", "T01"]))
covered = {c.case_id for c in result.scenario.checklist_coverage}
assert {"B01", "C04", "C05", "T01"} <= covered
def test_missing_selector_produces_blocker() -> None:
caps = dict(FULL, safe_test_data=True)
result = compile_scenario(_req(["B02"], parameters={}, capabilities=caps))
codes = {b.code for b in result.blockers}
assert "NEEDS_SELECTOR" in codes
# the interaction step itself is flagged needs_selector
assert any(s.automation_status == "needs_selector" for s in result.scenario.steps)
def test_no_sql_action_emitted() -> None:
result = compile_scenario(_req(["T01", "T02", "T03"]))
for step in result.scenario.steps:
assert step.action not in {"raw_sql", "execute_sql"}
assert "sql" not in step.action.lower()
def test_unsupported_case_warned_not_dropped() -> None:
caps = dict(FULL, xlsx_export=False, safe_test_data=True)
result = compile_scenario(_req(["C04"], capabilities=caps))
covered = {c.case_id for c in result.scenario.checklist_coverage}
assert "C04" in covered # still present in coverage with rationale
assert any(b.code == "UNSUPPORTED_ACTION" for b in result.blockers)
assert result.scenario.steps == []
assert not any(w.code == "UNSUPPORTED_CASE" for w in result.warnings)
def test_cyrillic_goal_produces_ascii_scenario_id() -> None:
"""A Russian objective goal must NOT leak into the technical scenario_id.
Python str.isalnum() is Unicode-aware and would keep Cyrillic in the slug,
yielding an id that violates the ASCII _STEP_ID_RE contract (compile 422),
forcing the agent to retry. The id generator must sanitize to ASCII so the
first compile always validates.
"""
import re as _re
from src.services.dashboard_testing.scenario.models import _STEP_ID_RE
result = compile_scenario(_req_with_goal("Проверить фильтры метрики и экспорт в Excel"))
scenario_id = result.scenario.scenario_id
assert _re.fullmatch(_STEP_ID_RE.pattern, scenario_id), f"scenario_id not ASCII/valid: {scenario_id!r}"
assert not any(ord(c) > 127 for c in scenario_id), f"scenario_id contains non-ASCII: {scenario_id!r}"
# The Russian objective must still live in the content, not the identifier.
assert "Проверить" in str(result.scenario.objective)
def test_cyrillic_only_goal_still_yields_valid_id() -> None:
"""Even a fully-Cyrillic goal must produce a pattern-valid id (no 422)."""
import re as _re
from src.services.dashboard_testing.scenario.models import _STEP_ID_RE
result = compile_scenario(_req_with_goal("Проверить отчётность"))
assert _re.fullmatch(_STEP_ID_RE.pattern, result.scenario.scenario_id)
def _req_with_goal(goal: str) -> CompileScenarioRequest:
from dataclasses import replace
return replace(_req(), objective=dict(_req().objective, goal=goal))
# #endregion Test.Scenario.Compiler
# #region Test.Scenario.Compiler.Branches [C:3] [TYPE Module]
# @defgroup Scalar parameter specs, unsupported/human-checkpoint cases, build-step automation branches.
def test_scalar_parameter_spec() -> None:
"""A bare scalar parameter spec (not a dict) compiles into a string parameter."""
result = compile_scenario(_req(parameters={"my_param": "default-value"}))
names = [p.name for p in result.scenario.parameters]
assert "my_param" in names
p = next(p for p in result.scenario.parameters if p.name == "my_param")
assert p.status == "resolved"
assert p.value == "default-value"
def _selector_params() -> dict:
return {
"test_date": {"type": "date", "default": "2026-07-01"},
"counterparty": {"type": "string", "default": "ACME"},
"selector_hint": {
"type": "selector_hint", "default": "#native-filter", "required": False,
"affected_step_ids": ["phase-2-B01-apply_native_filter"],
},
}
def test_visual_case_emits_canonical_capture_compare_chain() -> None:
caps = dict(FULL, baseline=True)
result = compile_scenario(_req(["B01"], parameters=_selector_params(), capabilities=caps))
steps = result.scenario.steps
assert [s.action for s in steps] == ["apply_native_filter", "capture_screenshot", "compare_to_baseline"]
assert [s.tool for s in steps] == ["browser", "screenshot", "assertion"]
assert steps[0].depends_on == []
assert steps[1].depends_on == [steps[0].id]
assert steps[2].depends_on == [steps[1].id]
assert "text_filter" not in {s.action for s in steps}
assert "apply_filters" not in {s.action for s in steps}
assert "register_artifact" not in {s.action for s in steps}
validation = validate_scenario(result.scenario)
assert validation.valid is True
assert not validation.errors
assert not validation.blockers
def test_text_filter_alias_is_not_emitted() -> None:
caps = dict(FULL, baseline=True)
result = compile_scenario(_req(["B02"], parameters=_selector_params(), capabilities=caps))
actions = [s.action for s in result.scenario.steps]
assert "text_filter" not in actions
assert actions[0] == "apply_native_filter"
assert actions == ["apply_native_filter", "capture_screenshot", "compare_to_baseline"]
def test_selected_unsupported_case_blocks_without_steps() -> None:
caps = dict(FULL, xlsx_export=False, safe_test_data=True)
result = compile_scenario(_req(["C04"], capabilities=caps))
assert result.scenario.steps == []
assert any(b.code == "UNSUPPORTED_ACTION" for b in result.blockers)
assert {c.case_id for c in result.scenario.checklist_coverage} >= {"C04"}
def test_unselected_unsupported_does_not_block_selected_chain() -> None:
caps = dict(FULL, xlsx_export=False, baseline=True)
result = compile_scenario(_req(["B01"], parameters=_selector_params(), capabilities=caps))
assert not any(b.code == "UNSUPPORTED_ACTION" for b in result.blockers)
covered = {c.case_id: c.classification for c in result.scenario.checklist_coverage}
assert covered["C04"] == "unsupported"
assert [s.action for s in result.scenario.steps] == [
"apply_native_filter", "capture_screenshot", "compare_to_baseline",
]
def test_visual_compare_without_screenshot_is_blocker() -> None:
caps = dict(FULL, screenshot=False, baseline=True)
result = compile_scenario(_req(["B01"], parameters=_selector_params(), capabilities=caps))
actions = [s.action for s in result.scenario.steps]
assert actions == ["apply_native_filter"]
assert "capture_screenshot" not in actions
assert "compare_to_baseline" not in actions
assert any(b.code == "MISSING_SCREENSHOT" for b in result.blockers)
# #region Test.Scenario.Compiler.SelectorBinding [C:2] [TYPE Function] [SEMANTICS test,selector,binding]
def test_selector_hint_only_resolves_its_bound_step() -> None:
result = compile_scenario(_req(
["B01", "B02"],
parameters={"selector": {
"type": "selector_hint", "default": "#filter", "required": False,
"affected_step_ids": ["phase-2-B01-apply_native_filter"],
}},
capabilities=dict(FULL, baseline=True),
))
statuses = {step.id: step.automation_status for step in result.scenario.steps}
assert statuses["phase-2-B01-apply_native_filter"] == "ready"
assert statuses["phase-2-B02-apply_native_filter"] == "needs_selector"
assert [item.step_id for item in result.blockers if item.code == "NEEDS_SELECTOR"] == [
"phase-2-B02-apply_native_filter",
]
# #endregion Test.Scenario.Compiler.SelectorBinding
def test_metric_case_appends_compare_when_baseline_present() -> None:
caps = dict(FULL, baseline=True)
result = compile_scenario(_req(["T01"], capabilities=caps))
steps = result.scenario.steps
assert [s.action for s in steps] == ["dataset_field_assert", "compare_to_baseline"]
assert steps[1].depends_on == [steps[0].id]
assert "capture_screenshot" not in {s.action for s in steps}
def test_unsafe_mutation_case_is_dropped_not_human() -> None:
# AGSCN-FR-019/020: B05 without safe_test_data maps to unsupported — the compiler
# emits no steps for it (no human step, no screenshot, nothing scheduled).
caps = dict(FULL, safe_test_data=False)
result = compile_scenario(_req(["B05"], capabilities=caps))
assert [s.action for s in result.scenario.steps] == []
assert "capture_screenshot" not in {s.action for s in result.scenario.steps}
assert "human_checkpoint" not in {s.action for s in result.scenario.steps}
def _evaluation_spec():
from src.services.dashboard_testing.execution.decision_policy import DecisionPolicy
from src.services.dashboard_testing.scenario.models import (
AgentEvaluationSpec,
EvaluationCriterion,
EvaluationLimits,
)
return AgentEvaluationSpec.model_validate({
"schema_version": 1,
"spec_id": "e5e5e5e5-e5e5-4e5e-8e5e-e5e5e5e5e5e5",
"provider_id": "llm-provider-a",
"provider_version": "1",
"model_id": "vision-model",
"model_version": "2026-08",
"prompt_template_id": "agent-evaluation-prompt",
"prompt_template_version": "1.0.0",
"prompt_template_hash": "a" * 64,
"evidence_refs": ["artifact-1"],
"comparison_refs": ["44444444-4444-4444-8444-444444444444"],
"output_schema": "agent-evaluation.schema.json",
"decision_policy": DecisionPolicy(policy_id="baseline-semantic", version="1.0.0"),
"limits": EvaluationLimits(
timeout_ms=60000, max_images=4, max_input_tokens=32000,
max_output_tokens=2000, max_cost="1.00", currency="USD",
),
"trust_policy_hash": "b" * 64,
"criteria": [
EvaluationCriterion(
criterion_id="crit-visual", criterion_kind="semantic",
description="visual layout matches", comparison_id=None,
),
],
})
def test_evaluate_declared_spec_emitted_only_when_request_has_spec() -> None:
from dataclasses import replace
caps = dict(FULL, baseline=True)
without_spec = compile_scenario(_req(["B01"], parameters=_selector_params(), capabilities=caps))
assert "evaluate_declared_spec" not in {s.action for s in without_spec.scenario.steps}
spec = _evaluation_spec()
with_spec = compile_scenario(replace(
_req(["B01"], parameters=_selector_params(), capabilities=caps),
agent_evaluation_spec=spec,
))
steps = with_spec.scenario.steps
assert [s.action for s in steps] == [
"apply_native_filter", "capture_screenshot", "compare_to_baseline", "evaluate_declared_spec",
]
assert steps[3].depends_on == [steps[2].id]
assert steps[3].agent_evaluation_spec == spec
assert steps[3].decision_policy is None
# #endregion Test.Scenario.Compiler.Branches