Merge feat/038-step-inputs: typed action_inputs op, per-action validation, embedded-metric-literal validator (AGSCN-FR-016/023)

This commit is contained in:
2026-09-18 10:49:58 +03:00
6 changed files with 585 additions and 5 deletions

View File

@@ -19,8 +19,10 @@ from .ops import (
SetAssertionOp,
SetDependencyOp,
SetParameterDefinitionOp,
SetStepInputsOp,
)
from src.services.dashboard_testing.scenario.models import DashboardTestScenario, Expected, ScenarioStep, sha256_hex
from src.services.dashboard_testing.scenario.step_inputs import assert_step_inputs
from src.services.dashboard_testing.scenario.templates import STEP_TEMPLATES, REGISTERED_ACTIONS
from src.services.dashboard_testing.scenario.validator import validate_scenario
from src.services.dashboard_testing.scenario.sql_guard import contains_unsafe_free_text
@@ -30,6 +32,14 @@ def _contains_unsafe_text(value: Any) -> bool:
return contains_unsafe_free_text(value)
def _contains_unsafe_inputs(value: Any) -> bool:
if isinstance(value, dict):
return any(_contains_unsafe_inputs(item) for item in value.values())
if isinstance(value, list):
return any(_contains_unsafe_inputs(item) for item in value)
return _contains_unsafe_text(value)
# #region ScenarioEditor.Apply.ValidateAssertion [C:3] [TYPE Function] [SEMANTICS scenario,editor,assertion,constrain]
# @ingroup ScenarioEditor
# @BRIEF Validate a registered comparison and approved baseline reference.
@@ -53,7 +63,10 @@ def validate_assertion(edit: SetAssertionOp) -> dict[str, Any]:
# @POST Returns a new graph plus normalized validation findings; input graph is not mutated.
def apply_ops(base_graph: dict[str, Any], ops: list[EditOperation]) -> dict[str, Any]:
for operation in ops:
if any(contains_unsafe_free_text(getattr(operation, field, None)) for field in ("baseline_ref", "value")):
if isinstance(operation, SetStepInputsOp):
if _contains_unsafe_inputs(operation.action_inputs):
raise ValueError("unsafe edit operation")
elif any(contains_unsafe_free_text(getattr(operation, field, None)) for field in ("baseline_ref", "value")):
raise ValueError("unsafe edit operation")
try:
scenario = DashboardTestScenario.model_validate(base_graph)
@@ -112,6 +125,12 @@ def _apply_legacy_ops(base_graph: dict[str, Any], ops: list[EditOperation]) -> d
edge for edge in graph["dependencies"]
if operation.logical_step_id not in {edge.get("source"), edge.get("target")}
]
elif isinstance(operation, SetStepInputsOp):
step = next((item for item in graph["steps"]
if item.get("logical_step_id", item.get("id")) == operation.logical_step_id), None)
if step is None:
raise ValueError("step not found")
step["action_inputs"] = assert_step_inputs(step.get("action"), operation.action_inputs)
elif isinstance(operation, SetDependencyOp):
if "dependencies" not in graph:
raise ValueError("legacy graph has no dependencies collection")
@@ -142,6 +161,11 @@ def _apply_operation(graph: dict[str, Any], operation: EditOperation) -> None:
step["expected"] = {"kind": "baseline_ref", "ref": operation.baseline_ref, "predicate": operation.comparison}
if operation.threshold is not None:
step["expected"]["description"] = f"threshold {operation.threshold}"
elif isinstance(operation, SetStepInputsOp):
step = next((s for s in steps if s["id"] == operation.logical_step_id), None)
if step is None:
raise ValueError("step not found")
step["action_inputs"] = assert_step_inputs(step.get("action"), operation.action_inputs)
elif isinstance(operation, AddStepOp):
template = STEP_TEMPLATES.get(operation.template)
if template is None:

View File

@@ -7,7 +7,7 @@
# @REJECTED Accepting arbitrary JSON graph patches was rejected — each mutation must carry a typed operation.
from __future__ import annotations
from typing import Annotated, Literal
from typing import Annotated, Any, Literal
from pydantic import BaseModel, Field
@@ -59,8 +59,22 @@ class SetDependencyOp(BaseModel):
# #endregion ScenarioEditor.Ops.SetDependency
# #region ScenarioEditor.Ops.SetStepInputs [C:2] [TYPE Class] [SEMANTICS scenario,editor,ops,step,inputs]
# @ingroup ScenarioEditor
# @BRIEF Pin bounded per-action inputs on an existing step; values are validated per action before apply.
# @INVARIANT Raw numeric business literals never pass — only structural bounds and baseline references are representable.
# @RATIONALE Pinned action arguments use a distinct optional field so canonical `inputs` remains the list[Ref] data contract.
# @REJECTED Overloading ScenarioStep.inputs was rejected because it changes ref semantics and legacy graph bytes.
class SetStepInputsOp(BaseModel):
model_config = {"extra": "forbid"}
op: Literal["set_step_inputs"]
logical_step_id: str = Field(..., min_length=1, max_length=128)
action_inputs: dict[str, Any]
# #endregion ScenarioEditor.Ops.SetStepInputs
EditOperation = Annotated[
SetParameterDefinitionOp | SetAssertionOp | AddStepOp | RemoveStepOp | SetDependencyOp,
SetParameterDefinitionOp | SetAssertionOp | AddStepOp | RemoveStepOp | SetDependencyOp | SetStepInputsOp,
Field(discriminator="op"),
]

View File

@@ -208,6 +208,7 @@ class ScenarioStep(BaseModel):
tool: Literal["browser", "superset_api", "sql_evidence", "transform", "xlsx", "assertion", "screenshot", "report", "artifact", "human", "agent_evaluation"]
action: str = Field(pattern=r"^[a-z][a-z0-9_]{1,63}$")
inputs: list[Ref] = Field(default_factory=list)
action_inputs: dict[str, Any] | None = None
outputs: list[Ref] = Field(default_factory=list)
expected: Expected
depends_on: list[str] = Field(default_factory=list)
@@ -220,6 +221,7 @@ class ScenarioStep(BaseModel):
vlm_analysis: VlmAnalysis | None = None
agent_evaluation_spec: AgentEvaluationSpec | None = None
decision_policy: DecisionPolicy | None = None
# #endregion ScenarioGraph.Models.ScenarioStep
@@ -336,7 +338,13 @@ class DashboardTestScenario(BaseModel):
# @BRIEF Serialize to canonical bytes for revision hashing (stable key order, no volatile fields).
# @POST Returns byte-stable canonical representation excluding revision fields.
def canonical_bytes(self) -> bytes:
return canonical_dump(self.model_dump(exclude={"revision_hash", "parent_revision_hash"}))
logger.reason("Serializing canonical bytes", src="ScenarioGraph.Models.CanonicalDump")
with belief_scope("ScenarioGraph.Models.CanonicalDump", "Canonical serialization"):
data = self.model_dump(exclude={"revision_hash", "parent_revision_hash"})
for step in data.get("steps", []):
if step.get("action_inputs") is None:
step.pop("action_inputs", None)
return canonical_dump(data)
# #endregion ScenarioGraph.Models.DashboardTestScenario.CanonicalBytes
# #endregion ScenarioGraph.Models.DashboardTestScenario

View File

@@ -0,0 +1,307 @@
# #region ScenarioGraph.StepInputs [C:3] [TYPE Module] [SEMANTICS scenario,step,inputs,validation,bounded]
# @defgroup ScenarioGraph Bounded per-action step inputs and embedded business-literal rejection.
# @LAYER Service
# @RELATION CALLED_BY -> [ScenarioGraph.Validator.CheckStepInputs]
# @RELATION CALLED_BY -> [ScenarioEditor.Apply]
# @INVARIANT Only declared keys of a known action pass; every value is type/length/bound/enum checked.
# @INVARIANT Numeric business expected literals are rejected; structural bounds and baseline_ref tolerance are allowed.
# @RATIONALE Pinned per-action inputs are executable step content: the bounded input contract must fail closed before compile/apply, and AGSCN-FR-016 forbids embedding metric truth inside a step.
# @REJECTED Accepting an arbitrary input dict was rejected — untyped keys/values would smuggle raw metric literals and free-form payloads past the validator.
from __future__ import annotations
import re
from dataclasses import dataclass, field
from typing import Any
_NUMERIC_STR_RE = re.compile(r"^-?\d+(\.\d+)?$")
_DATE_RE = r"^\d{4}-\d{2}-\d{2}(T\d{2}:\d{2}(:\d{2})?)?$"
_SCALAR_ITEM = (str, int, float, bool)
_TEXT_KINDS = (str,)
_INT_KINDS = (int,)
_NUM_KINDS = (int, float)
_BOOL_KINDS = (bool,)
_ID_KINDS = (str, int)
_LIST_KINDS = (list,)
_STRUCT_KINDS = (dict,)
# Business metric truth may only live in an approved baseline; structural bounds are scenario-owned inputs.
BUSINESS_LITERAL_KEYS = frozenset({
"expected", "expected_value", "expected_total", "expected_sum", "expected_count",
"business_value", "metric_value", "metric", "total", "sum", "amount",
})
BASELINE_TOLERANCE_KEYS = frozenset({"threshold", "tolerance"})
BASELINE_REF_KEYS = frozenset({"baseline_ref", "candidate_ref"})
STRUCTURAL_NUMERIC_KEYS = frozenset({
"element_count", "min_count", "max_count", "count_threshold",
"max_rows", "max_columns", "limit", "page_size", "timeout_ms", "chart_id",
})
# #region ScenarioGraph.StepInputs.InputField [C:1] [TYPE Class] [SEMANTICS scenario,step,inputs,schema]
# @ingroup ScenarioGraph
# @BRIEF Declarative bound for one declared action-input key.
@dataclass(frozen=True)
class InputField:
kinds: tuple[type, ...]
min_length: int | None = None
max_length: int | None = None
minimum: float | None = None
maximum: float | None = None
pattern: str | None = None
enum: tuple[str, ...] | None = None
item_kinds: tuple[type, ...] | None = None
# #endregion ScenarioGraph.StepInputs.InputField
# #region ScenarioGraph.StepInputs.Finding [C:1] [TYPE Class] [SEMANTICS scenario,step,inputs,finding]
# @ingroup ScenarioGraph
# @BRIEF One deterministic step-input finding with its code, message, and offending key.
@dataclass
class StepInputsFinding:
code: str
message: str
key: str | None = None
# #endregion ScenarioGraph.StepInputs.Finding
# #region ScenarioGraph.StepInputs.Validation [C:1] [TYPE Class] [SEMANTICS scenario,step,inputs,finding]
# @ingroup ScenarioGraph
# @BRIEF Aggregate step-input validation verdict and findings.
@dataclass
class StepInputsValidation:
valid: bool
findings: list[StepInputsFinding] = field(default_factory=list)
# #endregion ScenarioGraph.StepInputs.Validation
def _text(min_length: int = 1, max_length: int = 200) -> InputField:
return InputField(kinds=_TEXT_KINDS, min_length=min_length, max_length=max_length)
def _id(max_length: int = 200) -> InputField:
return InputField(kinds=_ID_KINDS, max_length=max_length)
def _enum(*values: str) -> InputField:
return InputField(kinds=_TEXT_KINDS, enum=values)
def _int(minimum: int, maximum: int) -> InputField:
return InputField(kinds=_INT_KINDS, minimum=minimum, maximum=maximum)
def _bool() -> InputField:
return InputField(kinds=_BOOL_KINDS)
def _struct() -> InputField:
return InputField(kinds=_STRUCT_KINDS)
def _list(max_items: int, item_kinds: tuple[type, ...] = _SCALAR_ITEM) -> InputField:
return InputField(kinds=_LIST_KINDS, max_length=max_items, item_kinds=item_kinds)
def _date() -> InputField:
return InputField(kinds=_TEXT_KINDS, min_length=8, max_length=40, pattern=_DATE_RE)
def _number(minimum: float = -1_000_000_000.0, maximum: float = 1_000_000_000.0) -> InputField:
return InputField(kinds=_NUM_KINDS, minimum=minimum, maximum=maximum)
_COMPARISON_INPUTS = {
"baseline_ref": _text(1, 256), "candidate_ref": _text(1, 256),
"tolerance": _number(0.0), "threshold": _number(),
}
_NAVIGATE_TAB = {"tab": _text(), "tab_identifier": _text(), "checkpoint": _bool()}
_APPLY_NATIVE_FILTER = {
"filter_id": _text(), "filter_name": _text(), "column": _text(), "values": _list(100),
"search_text": _text(0, 200), "mode": _enum("set", "clear"), "date": _date(), "wait_state": _text(),
}
_APPLY_TABLE_FILTER = {
"chart_id": _id(), "column": _text(), "values": _list(100),
"column_ref": _text(1, 256), "predicate_ref": _text(1, 256), "mode": _enum("set", "clear"),
}
_CLICK = {"selector": _text(1, 500), "selector_ref": _text(1, 500), "effect_class": _enum("read_only")}
_ASSERT_DOM = {
"selector": _text(1, 500), "text_present": _text(0, 500), "text_absent": _text(0, 500),
"href": _text(0, 2000), "element_count": _int(0, 1_000_000), "min_count": _int(0, 1_000_000),
"max_count": _int(0, 1_000_000), "count_threshold": _int(0, 1_000_000),
"sticky": _bool(), "bounding_box": _struct(),
}
_INSPECT_FILTER_OPTIONS = {
"filter_id": _text(), "filter_name": _text(), "column": _text(), "limit": _int(1, 500),
}
_WAIT_FOR_SELECTOR = {
"selector": _text(1, 500), "text": _text(1, 500),
"timeout_ms": _int(1, 120_000), "state": _enum("visible", "attached", "hidden"),
}
_EXTRACT_TABLE = {
"chart_id": _id(), "table": _text(), "max_rows": _int(1, 10_000), "max_columns": _int(1, 100),
}
# #region ScenarioGraph.StepInputs.Schema [C:2] [TYPE Model] [SEMANTICS scenario,step,inputs,schema,registry]
# @ingroup ScenarioGraph
# @BRIEF Closed per-action key schema and required-key groups for the P0 read-only browser actions.
# @INVARIANT An action absent from this mapping can never carry pinned inputs.
ACTION_INPUT_SCHEMAS: dict[str, dict[str, InputField]] = {
"navigate_tab": {**_COMPARISON_INPUTS, **_NAVIGATE_TAB},
"navigate_tabs": {**_COMPARISON_INPUTS, **_NAVIGATE_TAB},
"apply_native_filter": {**_COMPARISON_INPUTS, **_APPLY_NATIVE_FILTER},
"apply_table_filter": {**_COMPARISON_INPUTS, **_APPLY_TABLE_FILTER},
"click": {**_COMPARISON_INPUTS, **_CLICK},
"scroll_to": {**_COMPARISON_INPUTS, **_CLICK},
"assert_dom": {**_COMPARISON_INPUTS, **_ASSERT_DOM},
"inspect_filter_options": {**_COMPARISON_INPUTS, **_INSPECT_FILTER_OPTIONS},
"wait_for_selector": {**_COMPARISON_INPUTS, **_WAIT_FOR_SELECTOR},
"extract_table": {**_COMPARISON_INPUTS, **_EXTRACT_TABLE},
}
REQUIRED_ANY: dict[str, tuple[tuple[str, ...], ...]] = {
"navigate_tab": (("tab", "tab_identifier"),),
"navigate_tabs": (("tab", "tab_identifier"),),
"click": (("selector", "selector_ref"),),
"scroll_to": (("selector", "selector_ref"),),
"assert_dom": (("selector",),),
"inspect_filter_options": (("filter_id", "filter_name", "column"),),
"wait_for_selector": (("selector", "text"),),
"extract_table": (("chart_id", "table"),),
}
# #endregion ScenarioGraph.StepInputs.Schema
# #region ScenarioGraph.StepInputs.EmbeddedLiterals [C:2] [TYPE Function] [SEMANTICS scenario,step,inputs,literal,baseline]
# @ingroup ScenarioGraph
# @BRIEF Reject numeric business expected literals while allowing baseline references, tolerance, and structural bounds.
# @POST Returns EMBEDDED_METRIC_LITERAL findings for numeric truth under business keys or unpinned tolerance.
def find_embedded_metric_literals(inputs: dict[str, Any]) -> list[StepInputsFinding]:
findings: list[StepInputsFinding] = []
baseline_pinned = any(key in inputs for key in BASELINE_REF_KEYS)
for key in sorted(inputs):
if key in BASELINE_REF_KEYS or key in STRUCTURAL_NUMERIC_KEYS:
continue
raw = inputs[key]
candidates = raw if isinstance(raw, list) else [raw]
for item in candidates:
if not (_is_number(item) or _is_numeric_string(item)):
continue
if key in BUSINESS_LITERAL_KEYS:
findings.append(StepInputsFinding(
"EMBEDDED_METRIC_LITERAL",
f"numeric business literal {item!r} must resolve through an approved baseline", key,
))
break
if key in BASELINE_TOLERANCE_KEYS and not baseline_pinned:
findings.append(StepInputsFinding(
"EMBEDDED_METRIC_LITERAL",
f"numeric {key} requires a baseline_ref or candidate_ref", key,
))
break
return findings
# #endregion ScenarioGraph.StepInputs.EmbeddedLiterals
def _is_number(value: Any) -> bool:
return isinstance(value, (int, float)) and not isinstance(value, bool)
def _is_numeric_string(value: Any) -> bool:
return isinstance(value, str) and bool(_NUMERIC_STR_RE.match(value.strip()))
def _type_ok(value: Any, kinds: tuple[type, ...]) -> bool:
if kinds is _BOOL_KINDS:
return isinstance(value, bool)
if kinds is _INT_KINDS:
return isinstance(value, int) and not isinstance(value, bool)
if kinds is _NUM_KINDS:
return _is_number(value)
if kinds is _ID_KINDS:
return (isinstance(value, str) and bool(value)) or (isinstance(value, int) and not isinstance(value, bool))
return isinstance(value, kinds)
# #region ScenarioGraph.StepInputs.CheckField [C:2] [TYPE Function] [SEMANTICS scenario,step,inputs,bounds]
# @ingroup ScenarioGraph
# @BRIEF Validate one declared key against its type, length, bound, pattern, and enum contract.
# @POST Appends at most one typed finding per key and returns after the first violation.
def _check_field(key: str, value: Any, spec: InputField, findings: list[StepInputsFinding]) -> None:
if not _type_ok(value, spec.kinds):
findings.append(StepInputsFinding("INVALID_STEP_INPUT_TYPE", f"{key} has invalid type", key))
return
if isinstance(value, str):
if spec.min_length is not None and len(value) < spec.min_length:
findings.append(StepInputsFinding("INVALID_STEP_INPUT_TYPE", f"{key} must be at least {spec.min_length} characters", key))
return
if spec.max_length is not None and len(value) > spec.max_length:
findings.append(StepInputsFinding("STEP_INPUT_OUT_OF_BOUNDS", f"{key} exceeds {spec.max_length} characters", key))
return
if spec.pattern is not None and not re.match(spec.pattern, value):
findings.append(StepInputsFinding("INVALID_STEP_INPUT_TYPE", f"{key} has an invalid format", key))
return
if spec.enum is not None and value not in spec.enum:
findings.append(StepInputsFinding("STEP_INPUT_OUT_OF_BOUNDS", f"{key} must be one of {spec.enum}", key))
return
if _is_number(value):
if spec.minimum is not None and value < spec.minimum:
findings.append(StepInputsFinding("STEP_INPUT_OUT_OF_BOUNDS", f"{key} is below {spec.minimum}", key))
return
if spec.maximum is not None and value > spec.maximum:
findings.append(StepInputsFinding("STEP_INPUT_OUT_OF_BOUNDS", f"{key} exceeds {spec.maximum}", key))
return
if isinstance(value, list):
if spec.max_length is not None and len(value) > spec.max_length:
findings.append(StepInputsFinding("STEP_INPUT_OUT_OF_BOUNDS", f"{key} exceeds {spec.max_length} items", key))
return
for item in value:
if spec.item_kinds is not None and not _type_ok(item, spec.item_kinds):
findings.append(StepInputsFinding("INVALID_STEP_INPUT_TYPE", f"{key} contains an invalid item", key))
return
# #endregion ScenarioGraph.StepInputs.CheckField
# #region ScenarioGraph.StepInputs.Validate [C:3] [TYPE Function] [SEMANTICS scenario,step,inputs,validation]
# @ingroup ScenarioGraph
# @BRIEF Validate a pinned input mapping against the closed per-action schema and embedded-literal rule.
# @PRE action is the step action; inputs is the pinned mapping (or any JSON value at the boundary).
# @POST Returns a deterministic StepInputsValidation; unknown actions and keys never pass.
def validate_step_inputs(action: Any, inputs: Any) -> StepInputsValidation:
if not isinstance(inputs, dict):
return StepInputsValidation(False, [StepInputsFinding("INVALID_STEP_INPUT_TYPE", "inputs must be a mapping")])
schema = ACTION_INPUT_SCHEMAS.get(action) if isinstance(action, str) else None
if schema is None:
return StepInputsValidation(False, [StepInputsFinding(
"UNSUPPORTED_STEP_INPUT_ACTION", f"no bounded input schema for action {action!r}",
)])
findings: list[StepInputsFinding] = list(find_embedded_metric_literals(inputs))
literal_keys = {finding.key for finding in findings}
for key in sorted(inputs):
spec = schema.get(key)
if spec is None:
if key not in literal_keys:
findings.append(StepInputsFinding("UNKNOWN_STEP_INPUT", f"unknown input {key!r} for action {action!r}", key))
continue
_check_field(key, inputs[key], spec, findings)
for group in REQUIRED_ANY.get(action, ()):
if not any(key in inputs for key in group):
findings.append(StepInputsFinding("REQUIRED_STEP_INPUT", f"one of {group} is required for action {action!r}"))
return StepInputsValidation(not findings, findings)
# #endregion ScenarioGraph.StepInputs.Validate
# #region ScenarioGraph.StepInputs.Assert [C:2] [TYPE Function] [SEMANTICS scenario,step,inputs,fail-closed]
# @ingroup ScenarioGraph
# @BRIEF Fail closed on invalid pinned inputs and return the accepted mapping.
# @POST Raises ValueError with the first typed finding code; otherwise returns a copy of the inputs.
def assert_step_inputs(action: Any, inputs: Any) -> dict[str, Any]:
result = validate_step_inputs(action, inputs)
if not result.valid:
first = result.findings[0]
raise ValueError(f"{first.code}: {first.message}")
return dict(inputs)
# #endregion ScenarioGraph.StepInputs.Assert
# #endregion ScenarioGraph.StepInputs

View File

@@ -19,6 +19,7 @@ from src.core.logger import logger
from src.core.logger import belief_scope
from src.services.dashboard_testing.scenario.models import DashboardTestScenario, Finding
from src.services.dashboard_testing.scenario.step_inputs import validate_step_inputs
from src.services.dashboard_testing.scenario.templates import REGISTERED_ACTIONS, TOOLS
from src.services.dashboard_testing.scenario.sql_guard import contains_sql_statement
@@ -100,6 +101,7 @@ def _validate_checks(scenario: DashboardTestScenario) -> ScenarioValidationResul
_check_structure(steps, id_set, result)
_check_refs(steps, result)
_check_tools(steps, result)
_check_step_inputs(steps, result)
_check_safety(steps, result)
_check_dashboard_context(scenario, result)
_check_expectations(steps, result)
@@ -150,7 +152,10 @@ def _check_refs(steps: list[dict[str, Any]], result: ScenarioValidationResult) -
producers[name] = s["id"]
produced = set(producers)
for s in steps:
for inp in s.get("inputs", []):
declared = s.get("inputs")
if not isinstance(declared, list):
continue
for inp in declared:
name = inp["name"]
if name.startswith("step.") and name not in produced:
result.errors.append(_err("MISSING_REF", f"step {s['id']} consumes missing output ref {name}", step_id=s["id"]))
@@ -173,6 +178,21 @@ def _check_tools(steps: list[dict[str, Any]], result: ScenarioValidationResult)
# #endregion ScenarioGraph.Validator.CheckTools
# #region ScenarioGraph.Validator.CheckStepInputs [C:2] [TYPE Function] [SEMANTICS scenario,validator,step,inputs,literal]
# @ingroup ScenarioGraph
# @BRIEF Validate pinned per-action step inputs and reject embedded numeric business literals.
# @POST Appends one error per typed step-input finding; legacy list refs are left to CheckRefs.
def _check_step_inputs(steps: list[dict[str, Any]], result: ScenarioValidationResult) -> None:
for s in steps:
action_inputs = s.get("action_inputs")
if not isinstance(action_inputs, dict) or not action_inputs:
continue
for finding in validate_step_inputs(s.get("action"), action_inputs).findings:
detail = f"step {s['id']} input {finding.key}: {finding.message}" if finding.key else f"step {s['id']}: {finding.message}"
result.errors.append(_err(finding.code, detail, step_id=s["id"]))
# #endregion ScenarioGraph.Validator.CheckStepInputs
# #region ScenarioGraph.Validator.CheckSafety [C:2] [TYPE Function] [SEMANTICS scenario,validator,safety]
# @ingroup ScenarioGraph
# @BRIEF Check SQL, executable code, and forbidden actions in step text and tooling.

View File

@@ -0,0 +1,207 @@
# #region Test.ScenarioEditor.StepInputs [C:3] [TYPE Module] [SEMANTICS test,scenario,editor,step,inputs,mcp,parity]
# @defgroup Typed set_step_inputs application, bounded per-action validation, and MCP propose_graph_revision parity.
# @LAYER Test
# @RELATION BINDS_TO -> [ScenarioEditor.Apply]
# @RELATION BINDS_TO -> [ScenarioGraph.StepInputs]
# @TEST_EDGE: embedded_business_literal -> EMBEDDED_METRIC_LITERAL rejected before apply
# @TEST_EDGE: unsupported_action_input -> UNSUPPORTED_STEP_INPUT_ACTION rejected
# @TEST_EDGE: legacy_step_without_inputs -> canonical dump keeps no inputs key
# @TEST_INVARIANT AGSCN-FR-016 -> VERIFIED_BY: test_validator_flags_embedded_metric_literal, test_apply_ops_rejects_business_literal
from __future__ import annotations
import pytest
from pydantic import TypeAdapter
from src.mcp_server.scenario_inputs import GraphRevisionInput
from src.models.scenario_registry import ScenarioEditProposal, ScenarioRevision
from src.services.dashboard_testing.editor.agent import agent_propose
from src.services.dashboard_testing.editor.apply import apply_ops
from src.services.dashboard_testing.editor.ops import EditOperation, SetStepInputsOp
from src.services.dashboard_testing.scenario.models import DashboardTestScenario, Ref, ScenarioStep
from src.services.dashboard_testing.scenario.step_inputs import validate_step_inputs
from src.services.dashboard_testing.scenario.validator import validate_scenario
_SCENARIO_ID = "11111111-1111-4111-8111-111111111111"
_BASE_REVISION_ID = "22222222-2222-4222-8222-222222222222"
_OPS_ADAPTER = TypeAdapter(list[EditOperation])
_FILTER_INPUTS = {"filter_id": "region", "mode": "set", "values": ["EU"]}
# #region Test.ScenarioEditor.StepInputs.Graph [C:1] [TYPE Function] [SEMANTICS test,scenario,step,inputs,fixture]
# @ingroup Test.ScenarioEditor.StepInputs
# @BRIEF Hardcoded canonical graph with one bounded-action step and one extraction step.
# @TEST_FIXTURE step_inputs_graph -> INLINE_JSON
def _step(step_id: str, action: str) -> dict:
return {
"id": step_id, "phase": "interact", "title": action, "tool": "browser", "action": action,
"inputs": [], "outputs": [], "expected": {"kind": "structural", "description": f"{action} completes"},
"depends_on": [], "automation_status": "ready", "risk": "browser_interaction",
}
def _canonical_graph() -> dict:
return {
"schema_version": 1, "compiler_version": "038.1.0", "template_version": "v1",
"scenario_id": "sc-step-inputs", "revision_hash": "a" * 64,
"dashboard_context": {"environment_id": "env-preprod-02", "dashboard_id": 80},
"objective": {"goal": "verify filters", "selected_case_ids": ["B01"]},
"input_fingerprints": {"query_model": "c" * 64},
"parameters": [], "phases": ["interact"],
"steps": [_step("step-filter-1", "apply_native_filter"), _step("step-table-1", "extract_table")],
"outputs": [], "artifact_plan": [], "checklist_coverage": [],
"warnings": [], "blockers": [], "risk_summary": {},
}
# #endregion Test.ScenarioEditor.StepInputs.Graph
# #region Test.ScenarioEditor.StepInputs.CanonicalPin [C:2] [TYPE Function] [SEMANTICS test,scenario,step,inputs,canonical]
# @ingroup Test.ScenarioEditor.StepInputs
# @BRIEF Parsed steps preserve canonical refs and explicit empty refs.
def test_step_inputs_refs_are_canonical_list():
absent = ScenarioStep.model_validate(_step("step-filter-1", "apply_native_filter"))
assert absent.inputs == []
assert absent.model_dump()["inputs"] == []
present = ScenarioStep.model_validate({**_step("step-filter-1", "apply_native_filter"), "inputs": []})
assert present.model_dump()["inputs"] == []
# @BRIEF Legacy refs remain list[Ref], while action arguments use a distinct optional field.
def test_inputs_refs_remain_typed_and_action_inputs_are_distinct():
graph = _canonical_graph()
graph["steps"][0]["inputs"] = [{"name": "step.previous", "kind": "step_output"}]
scenario = DashboardTestScenario.model_validate(graph)
assert scenario.steps[0].inputs == [Ref(name="step.previous", kind="step_output")]
assert scenario.steps[0].action_inputs is None
# #endregion Test.ScenarioEditor.StepInputs.CanonicalPin
# @BRIEF Canonical bytes for an old graph remain unchanged after schema recovery.
def test_legacy_canonical_bytes_remain_unchanged():
graph = _canonical_graph()
graph["steps"][0]["inputs"] = []
expected = DashboardTestScenario.model_validate(graph).canonical_bytes()
round_tripped = DashboardTestScenario.model_validate(
DashboardTestScenario.model_validate(graph).model_dump(mode="json")
).canonical_bytes()
assert round_tripped == expected
# #region Test.ScenarioEditor.StepInputs.Apply [C:2] [TYPE Function] [SEMANTICS test,scenario,editor,step,inputs]
# @ingroup Test.ScenarioEditor.StepInputs
# @BRIEF set_step_inputs pins the validated mapping on the target step and leaves other steps untouched.
def test_apply_ops_pins_validated_inputs():
ops = _OPS_ADAPTER.validate_python([{
"op": "set_step_inputs", "logical_step_id": "step-filter-1", "action_inputs": _FILTER_INPUTS,
}])
assert isinstance(ops[0], SetStepInputsOp)
result = apply_ops(_canonical_graph(), ops)
assert result["steps"][0]["action_inputs"] == _FILTER_INPUTS
assert result["steps"][0]["inputs"] == []
assert result["steps"][1]["action_inputs"] is None
# #endregion Test.ScenarioEditor.StepInputs.Apply
# #region Test.ScenarioEditor.StepInputs.RejectMetric [C:2] [TYPE Function] [SEMANTICS test,scenario,editor,step,inputs,literal]
# @ingroup Test.ScenarioEditor.StepInputs
# @BRIEF Raw numeric business truth is refused before it can be pinned on a step.
def test_apply_ops_rejects_business_literal():
ops = _OPS_ADAPTER.validate_python([{
"op": "set_step_inputs", "logical_step_id": "step-filter-1", "action_inputs": {"expected_total": 1234.5},
}])
with pytest.raises(ValueError, match="EMBEDDED_METRIC_LITERAL"):
apply_ops(_canonical_graph(), ops)
def test_apply_ops_rejects_unsupported_step_action_inputs():
ops = _OPS_ADAPTER.validate_python([{
"op": "set_step_inputs", "logical_step_id": "step-filter-1", "action_inputs": {"selector": "#x"},
}])
with pytest.raises(ValueError, match="UNKNOWN_STEP_INPUT"):
apply_ops(_canonical_graph(), ops)
# #endregion Test.ScenarioEditor.StepInputs.RejectMetric
# #region Test.ScenarioEditor.StepInputs.PerActionSchema [C:2] [TYPE Function] [SEMANTICS test,scenario,step,inputs,schema]
# @ingroup Test.ScenarioEditor.StepInputs
# @BRIEF Every bounded action accepts its structural inputs, and unsupported actions have no schema.
def test_per_action_schema_accepts_structural_inputs():
cases = {
"navigate_tab": {"tab": "Overview"},
"apply_native_filter": {"filter_id": "region", "search_text": "EU", "mode": "clear"},
"apply_table_filter": {"column": "revenue", "values": ["x"]},
"click": {"selector": "#tab-1", "effect_class": "read_only"},
"scroll_to": {"selector": "#table", "baseline_ref": "baseline:scroll"},
"assert_dom": {"selector": "#t", "element_count": 5, "sticky": True},
"inspect_filter_options": {"filter_id": "region", "limit": 25},
"wait_for_selector": {"selector": "#spinner", "timeout_ms": 5000, "state": "hidden"},
"extract_table": {"chart_id": 128, "max_rows": 100, "max_columns": 10},
}
for action, inputs in cases.items():
assert validate_step_inputs(action, inputs).valid, action
def test_unsupported_action_and_unknown_key_are_rejected():
assert [f.code for f in validate_step_inputs("open_dashboard", {"selector": "#x"}).findings] == [
"UNSUPPORTED_STEP_INPUT_ACTION"
]
assert "REQUIRED_STEP_INPUT" in [f.code for f in validate_step_inputs("navigate_tab", {}).findings]
# #endregion Test.ScenarioEditor.StepInputs.PerActionSchema
# #region Test.ScenarioEditor.StepInputs.ValidatorFinding [C:2] [TYPE Function] [SEMANTICS test,scenario,validator,step,inputs,literal]
# @ingroup Test.ScenarioEditor.StepInputs
# @BRIEF The validator rejects embedded metric truth while accepting structural bounds and baseline tolerance.
def test_validator_flags_embedded_metric_literal():
graph = _canonical_graph()
graph["steps"][0]["action_inputs"] = {"selector": "#metric", "expected_total": 98765.43}
codes = [f.code for f in validate_scenario(DashboardTestScenario.model_validate(graph)).errors]
assert "EMBEDDED_METRIC_LITERAL" in codes
def test_validator_accepts_structural_bounds_and_baseline_tolerance():
graph = _canonical_graph()
graph["steps"][0]["action_inputs"] = {
"filter_id": "region", "mode": "set", "tolerance": 0.01, "baseline_ref": "baseline:filter-state",
}
assert [f.code for f in validate_scenario(DashboardTestScenario.model_validate(graph)).errors] == []
# #endregion Test.ScenarioEditor.StepInputs.ValidatorFinding
# #region Test.ScenarioEditor.StepInputs.McpParity [C:3] [TYPE Function] [SEMANTICS test,mcp,authoring,step,inputs,parity]
# @ingroup Test.ScenarioEditor.StepInputs
# @BRIEF The MCP propose_graph_revision boundary accepts the same typed operation and derives the same graph.
# @INVARIANT MCP operations are passed verbatim into the shared EditOperation union; no MCP-only or editor-only type exists.
def test_mcp_propose_graph_revision_parity(seeded_registry):
revision = seeded_registry.query(ScenarioRevision).filter_by(revision_id=_BASE_REVISION_ID).one()
revision.graph_snapshot = _canonical_graph()
seeded_registry.commit()
operation = {"op": "set_step_inputs", "logical_step_id": "step-filter-1", "action_inputs": _FILTER_INPUTS}
request = GraphRevisionInput(
workspace_id="ws-step-inputs", idempotency_key="idem-step-inputs-1", expected_cas_version=0,
request_text="pin filter inputs", operations=[operation],
)
assert request.operations == [operation]
result = agent_propose(
seeded_registry, _SCENARIO_ID, _BASE_REVISION_ID, request.request_text, request.operations,
created_by="agent-bot", agent_action_id="agent-action-step-inputs",
)
seeded_registry.commit()
proposal = seeded_registry.query(ScenarioEditProposal).filter_by(proposal_id=result["proposal_id"]).one()
assert proposal.proposed_graph["steps"][0]["action_inputs"] == _FILTER_INPUTS
assert proposal.operations == [operation]
def test_mcp_propose_graph_revision_rejects_business_literal(seeded_registry):
revision = seeded_registry.query(ScenarioRevision).filter_by(revision_id=_BASE_REVISION_ID).one()
revision.graph_snapshot = _canonical_graph()
seeded_registry.commit()
with pytest.raises(ValueError, match="EMBEDDED_METRIC_LITERAL"):
agent_propose(
seeded_registry, _SCENARIO_ID, _BASE_REVISION_ID, "pin raw truth",
[{"op": "set_step_inputs", "logical_step_id": "step-filter-1", "action_inputs": {"expected_total": 42}}],
created_by="agent-bot",
)
# #endregion Test.ScenarioEditor.StepInputs.McpParity
# #endregion Test.ScenarioEditor.StepInputs