Merge feat/038-step-inputs: typed action_inputs op, per-action validation, embedded-metric-literal validator (AGSCN-FR-016/023)
This commit is contained in:
@@ -19,8 +19,10 @@ from .ops import (
|
||||
SetAssertionOp,
|
||||
SetDependencyOp,
|
||||
SetParameterDefinitionOp,
|
||||
SetStepInputsOp,
|
||||
)
|
||||
from src.services.dashboard_testing.scenario.models import DashboardTestScenario, Expected, ScenarioStep, sha256_hex
|
||||
from src.services.dashboard_testing.scenario.step_inputs import assert_step_inputs
|
||||
from src.services.dashboard_testing.scenario.templates import STEP_TEMPLATES, REGISTERED_ACTIONS
|
||||
from src.services.dashboard_testing.scenario.validator import validate_scenario
|
||||
from src.services.dashboard_testing.scenario.sql_guard import contains_unsafe_free_text
|
||||
@@ -30,6 +32,14 @@ def _contains_unsafe_text(value: Any) -> bool:
|
||||
return contains_unsafe_free_text(value)
|
||||
|
||||
|
||||
def _contains_unsafe_inputs(value: Any) -> bool:
|
||||
if isinstance(value, dict):
|
||||
return any(_contains_unsafe_inputs(item) for item in value.values())
|
||||
if isinstance(value, list):
|
||||
return any(_contains_unsafe_inputs(item) for item in value)
|
||||
return _contains_unsafe_text(value)
|
||||
|
||||
|
||||
# #region ScenarioEditor.Apply.ValidateAssertion [C:3] [TYPE Function] [SEMANTICS scenario,editor,assertion,constrain]
|
||||
# @ingroup ScenarioEditor
|
||||
# @BRIEF Validate a registered comparison and approved baseline reference.
|
||||
@@ -53,7 +63,10 @@ def validate_assertion(edit: SetAssertionOp) -> dict[str, Any]:
|
||||
# @POST Returns a new graph plus normalized validation findings; input graph is not mutated.
|
||||
def apply_ops(base_graph: dict[str, Any], ops: list[EditOperation]) -> dict[str, Any]:
|
||||
for operation in ops:
|
||||
if any(contains_unsafe_free_text(getattr(operation, field, None)) for field in ("baseline_ref", "value")):
|
||||
if isinstance(operation, SetStepInputsOp):
|
||||
if _contains_unsafe_inputs(operation.action_inputs):
|
||||
raise ValueError("unsafe edit operation")
|
||||
elif any(contains_unsafe_free_text(getattr(operation, field, None)) for field in ("baseline_ref", "value")):
|
||||
raise ValueError("unsafe edit operation")
|
||||
try:
|
||||
scenario = DashboardTestScenario.model_validate(base_graph)
|
||||
@@ -112,6 +125,12 @@ def _apply_legacy_ops(base_graph: dict[str, Any], ops: list[EditOperation]) -> d
|
||||
edge for edge in graph["dependencies"]
|
||||
if operation.logical_step_id not in {edge.get("source"), edge.get("target")}
|
||||
]
|
||||
elif isinstance(operation, SetStepInputsOp):
|
||||
step = next((item for item in graph["steps"]
|
||||
if item.get("logical_step_id", item.get("id")) == operation.logical_step_id), None)
|
||||
if step is None:
|
||||
raise ValueError("step not found")
|
||||
step["action_inputs"] = assert_step_inputs(step.get("action"), operation.action_inputs)
|
||||
elif isinstance(operation, SetDependencyOp):
|
||||
if "dependencies" not in graph:
|
||||
raise ValueError("legacy graph has no dependencies collection")
|
||||
@@ -142,6 +161,11 @@ def _apply_operation(graph: dict[str, Any], operation: EditOperation) -> None:
|
||||
step["expected"] = {"kind": "baseline_ref", "ref": operation.baseline_ref, "predicate": operation.comparison}
|
||||
if operation.threshold is not None:
|
||||
step["expected"]["description"] = f"threshold {operation.threshold}"
|
||||
elif isinstance(operation, SetStepInputsOp):
|
||||
step = next((s for s in steps if s["id"] == operation.logical_step_id), None)
|
||||
if step is None:
|
||||
raise ValueError("step not found")
|
||||
step["action_inputs"] = assert_step_inputs(step.get("action"), operation.action_inputs)
|
||||
elif isinstance(operation, AddStepOp):
|
||||
template = STEP_TEMPLATES.get(operation.template)
|
||||
if template is None:
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
# @REJECTED Accepting arbitrary JSON graph patches was rejected — each mutation must carry a typed operation.
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Annotated, Literal
|
||||
from typing import Annotated, Any, Literal
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
@@ -59,8 +59,22 @@ class SetDependencyOp(BaseModel):
|
||||
# #endregion ScenarioEditor.Ops.SetDependency
|
||||
|
||||
|
||||
# #region ScenarioEditor.Ops.SetStepInputs [C:2] [TYPE Class] [SEMANTICS scenario,editor,ops,step,inputs]
|
||||
# @ingroup ScenarioEditor
|
||||
# @BRIEF Pin bounded per-action inputs on an existing step; values are validated per action before apply.
|
||||
# @INVARIANT Raw numeric business literals never pass — only structural bounds and baseline references are representable.
|
||||
# @RATIONALE Pinned action arguments use a distinct optional field so canonical `inputs` remains the list[Ref] data contract.
|
||||
# @REJECTED Overloading ScenarioStep.inputs was rejected because it changes ref semantics and legacy graph bytes.
|
||||
class SetStepInputsOp(BaseModel):
|
||||
model_config = {"extra": "forbid"}
|
||||
op: Literal["set_step_inputs"]
|
||||
logical_step_id: str = Field(..., min_length=1, max_length=128)
|
||||
action_inputs: dict[str, Any]
|
||||
# #endregion ScenarioEditor.Ops.SetStepInputs
|
||||
|
||||
|
||||
EditOperation = Annotated[
|
||||
SetParameterDefinitionOp | SetAssertionOp | AddStepOp | RemoveStepOp | SetDependencyOp,
|
||||
SetParameterDefinitionOp | SetAssertionOp | AddStepOp | RemoveStepOp | SetDependencyOp | SetStepInputsOp,
|
||||
Field(discriminator="op"),
|
||||
]
|
||||
|
||||
|
||||
@@ -208,6 +208,7 @@ class ScenarioStep(BaseModel):
|
||||
tool: Literal["browser", "superset_api", "sql_evidence", "transform", "xlsx", "assertion", "screenshot", "report", "artifact", "human", "agent_evaluation"]
|
||||
action: str = Field(pattern=r"^[a-z][a-z0-9_]{1,63}$")
|
||||
inputs: list[Ref] = Field(default_factory=list)
|
||||
action_inputs: dict[str, Any] | None = None
|
||||
outputs: list[Ref] = Field(default_factory=list)
|
||||
expected: Expected
|
||||
depends_on: list[str] = Field(default_factory=list)
|
||||
@@ -220,6 +221,7 @@ class ScenarioStep(BaseModel):
|
||||
vlm_analysis: VlmAnalysis | None = None
|
||||
agent_evaluation_spec: AgentEvaluationSpec | None = None
|
||||
decision_policy: DecisionPolicy | None = None
|
||||
|
||||
# #endregion ScenarioGraph.Models.ScenarioStep
|
||||
|
||||
|
||||
@@ -336,7 +338,13 @@ class DashboardTestScenario(BaseModel):
|
||||
# @BRIEF Serialize to canonical bytes for revision hashing (stable key order, no volatile fields).
|
||||
# @POST Returns byte-stable canonical representation excluding revision fields.
|
||||
def canonical_bytes(self) -> bytes:
|
||||
return canonical_dump(self.model_dump(exclude={"revision_hash", "parent_revision_hash"}))
|
||||
logger.reason("Serializing canonical bytes", src="ScenarioGraph.Models.CanonicalDump")
|
||||
with belief_scope("ScenarioGraph.Models.CanonicalDump", "Canonical serialization"):
|
||||
data = self.model_dump(exclude={"revision_hash", "parent_revision_hash"})
|
||||
for step in data.get("steps", []):
|
||||
if step.get("action_inputs") is None:
|
||||
step.pop("action_inputs", None)
|
||||
return canonical_dump(data)
|
||||
# #endregion ScenarioGraph.Models.DashboardTestScenario.CanonicalBytes
|
||||
# #endregion ScenarioGraph.Models.DashboardTestScenario
|
||||
|
||||
|
||||
307
backend/src/services/dashboard_testing/scenario/step_inputs.py
Normal file
307
backend/src/services/dashboard_testing/scenario/step_inputs.py
Normal file
@@ -0,0 +1,307 @@
|
||||
# #region ScenarioGraph.StepInputs [C:3] [TYPE Module] [SEMANTICS scenario,step,inputs,validation,bounded]
|
||||
# @defgroup ScenarioGraph Bounded per-action step inputs and embedded business-literal rejection.
|
||||
# @LAYER Service
|
||||
# @RELATION CALLED_BY -> [ScenarioGraph.Validator.CheckStepInputs]
|
||||
# @RELATION CALLED_BY -> [ScenarioEditor.Apply]
|
||||
# @INVARIANT Only declared keys of a known action pass; every value is type/length/bound/enum checked.
|
||||
# @INVARIANT Numeric business expected literals are rejected; structural bounds and baseline_ref tolerance are allowed.
|
||||
# @RATIONALE Pinned per-action inputs are executable step content: the bounded input contract must fail closed before compile/apply, and AGSCN-FR-016 forbids embedding metric truth inside a step.
|
||||
# @REJECTED Accepting an arbitrary input dict was rejected — untyped keys/values would smuggle raw metric literals and free-form payloads past the validator.
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any
|
||||
|
||||
_NUMERIC_STR_RE = re.compile(r"^-?\d+(\.\d+)?$")
|
||||
_DATE_RE = r"^\d{4}-\d{2}-\d{2}(T\d{2}:\d{2}(:\d{2})?)?$"
|
||||
_SCALAR_ITEM = (str, int, float, bool)
|
||||
_TEXT_KINDS = (str,)
|
||||
_INT_KINDS = (int,)
|
||||
_NUM_KINDS = (int, float)
|
||||
_BOOL_KINDS = (bool,)
|
||||
_ID_KINDS = (str, int)
|
||||
_LIST_KINDS = (list,)
|
||||
_STRUCT_KINDS = (dict,)
|
||||
|
||||
# Business metric truth may only live in an approved baseline; structural bounds are scenario-owned inputs.
|
||||
BUSINESS_LITERAL_KEYS = frozenset({
|
||||
"expected", "expected_value", "expected_total", "expected_sum", "expected_count",
|
||||
"business_value", "metric_value", "metric", "total", "sum", "amount",
|
||||
})
|
||||
BASELINE_TOLERANCE_KEYS = frozenset({"threshold", "tolerance"})
|
||||
BASELINE_REF_KEYS = frozenset({"baseline_ref", "candidate_ref"})
|
||||
STRUCTURAL_NUMERIC_KEYS = frozenset({
|
||||
"element_count", "min_count", "max_count", "count_threshold",
|
||||
"max_rows", "max_columns", "limit", "page_size", "timeout_ms", "chart_id",
|
||||
})
|
||||
|
||||
|
||||
# #region ScenarioGraph.StepInputs.InputField [C:1] [TYPE Class] [SEMANTICS scenario,step,inputs,schema]
|
||||
# @ingroup ScenarioGraph
|
||||
# @BRIEF Declarative bound for one declared action-input key.
|
||||
@dataclass(frozen=True)
|
||||
class InputField:
|
||||
kinds: tuple[type, ...]
|
||||
min_length: int | None = None
|
||||
max_length: int | None = None
|
||||
minimum: float | None = None
|
||||
maximum: float | None = None
|
||||
pattern: str | None = None
|
||||
enum: tuple[str, ...] | None = None
|
||||
item_kinds: tuple[type, ...] | None = None
|
||||
# #endregion ScenarioGraph.StepInputs.InputField
|
||||
|
||||
|
||||
# #region ScenarioGraph.StepInputs.Finding [C:1] [TYPE Class] [SEMANTICS scenario,step,inputs,finding]
|
||||
# @ingroup ScenarioGraph
|
||||
# @BRIEF One deterministic step-input finding with its code, message, and offending key.
|
||||
@dataclass
|
||||
class StepInputsFinding:
|
||||
code: str
|
||||
message: str
|
||||
key: str | None = None
|
||||
# #endregion ScenarioGraph.StepInputs.Finding
|
||||
|
||||
|
||||
# #region ScenarioGraph.StepInputs.Validation [C:1] [TYPE Class] [SEMANTICS scenario,step,inputs,finding]
|
||||
# @ingroup ScenarioGraph
|
||||
# @BRIEF Aggregate step-input validation verdict and findings.
|
||||
@dataclass
|
||||
class StepInputsValidation:
|
||||
valid: bool
|
||||
findings: list[StepInputsFinding] = field(default_factory=list)
|
||||
# #endregion ScenarioGraph.StepInputs.Validation
|
||||
|
||||
|
||||
def _text(min_length: int = 1, max_length: int = 200) -> InputField:
|
||||
return InputField(kinds=_TEXT_KINDS, min_length=min_length, max_length=max_length)
|
||||
|
||||
|
||||
def _id(max_length: int = 200) -> InputField:
|
||||
return InputField(kinds=_ID_KINDS, max_length=max_length)
|
||||
|
||||
|
||||
def _enum(*values: str) -> InputField:
|
||||
return InputField(kinds=_TEXT_KINDS, enum=values)
|
||||
|
||||
|
||||
def _int(minimum: int, maximum: int) -> InputField:
|
||||
return InputField(kinds=_INT_KINDS, minimum=minimum, maximum=maximum)
|
||||
|
||||
|
||||
def _bool() -> InputField:
|
||||
return InputField(kinds=_BOOL_KINDS)
|
||||
|
||||
|
||||
def _struct() -> InputField:
|
||||
return InputField(kinds=_STRUCT_KINDS)
|
||||
|
||||
|
||||
def _list(max_items: int, item_kinds: tuple[type, ...] = _SCALAR_ITEM) -> InputField:
|
||||
return InputField(kinds=_LIST_KINDS, max_length=max_items, item_kinds=item_kinds)
|
||||
|
||||
|
||||
def _date() -> InputField:
|
||||
return InputField(kinds=_TEXT_KINDS, min_length=8, max_length=40, pattern=_DATE_RE)
|
||||
|
||||
|
||||
def _number(minimum: float = -1_000_000_000.0, maximum: float = 1_000_000_000.0) -> InputField:
|
||||
return InputField(kinds=_NUM_KINDS, minimum=minimum, maximum=maximum)
|
||||
|
||||
|
||||
_COMPARISON_INPUTS = {
|
||||
"baseline_ref": _text(1, 256), "candidate_ref": _text(1, 256),
|
||||
"tolerance": _number(0.0), "threshold": _number(),
|
||||
}
|
||||
|
||||
|
||||
_NAVIGATE_TAB = {"tab": _text(), "tab_identifier": _text(), "checkpoint": _bool()}
|
||||
_APPLY_NATIVE_FILTER = {
|
||||
"filter_id": _text(), "filter_name": _text(), "column": _text(), "values": _list(100),
|
||||
"search_text": _text(0, 200), "mode": _enum("set", "clear"), "date": _date(), "wait_state": _text(),
|
||||
}
|
||||
_APPLY_TABLE_FILTER = {
|
||||
"chart_id": _id(), "column": _text(), "values": _list(100),
|
||||
"column_ref": _text(1, 256), "predicate_ref": _text(1, 256), "mode": _enum("set", "clear"),
|
||||
}
|
||||
_CLICK = {"selector": _text(1, 500), "selector_ref": _text(1, 500), "effect_class": _enum("read_only")}
|
||||
_ASSERT_DOM = {
|
||||
"selector": _text(1, 500), "text_present": _text(0, 500), "text_absent": _text(0, 500),
|
||||
"href": _text(0, 2000), "element_count": _int(0, 1_000_000), "min_count": _int(0, 1_000_000),
|
||||
"max_count": _int(0, 1_000_000), "count_threshold": _int(0, 1_000_000),
|
||||
"sticky": _bool(), "bounding_box": _struct(),
|
||||
}
|
||||
_INSPECT_FILTER_OPTIONS = {
|
||||
"filter_id": _text(), "filter_name": _text(), "column": _text(), "limit": _int(1, 500),
|
||||
}
|
||||
_WAIT_FOR_SELECTOR = {
|
||||
"selector": _text(1, 500), "text": _text(1, 500),
|
||||
"timeout_ms": _int(1, 120_000), "state": _enum("visible", "attached", "hidden"),
|
||||
}
|
||||
_EXTRACT_TABLE = {
|
||||
"chart_id": _id(), "table": _text(), "max_rows": _int(1, 10_000), "max_columns": _int(1, 100),
|
||||
}
|
||||
|
||||
# #region ScenarioGraph.StepInputs.Schema [C:2] [TYPE Model] [SEMANTICS scenario,step,inputs,schema,registry]
|
||||
# @ingroup ScenarioGraph
|
||||
# @BRIEF Closed per-action key schema and required-key groups for the P0 read-only browser actions.
|
||||
# @INVARIANT An action absent from this mapping can never carry pinned inputs.
|
||||
ACTION_INPUT_SCHEMAS: dict[str, dict[str, InputField]] = {
|
||||
"navigate_tab": {**_COMPARISON_INPUTS, **_NAVIGATE_TAB},
|
||||
"navigate_tabs": {**_COMPARISON_INPUTS, **_NAVIGATE_TAB},
|
||||
"apply_native_filter": {**_COMPARISON_INPUTS, **_APPLY_NATIVE_FILTER},
|
||||
"apply_table_filter": {**_COMPARISON_INPUTS, **_APPLY_TABLE_FILTER},
|
||||
"click": {**_COMPARISON_INPUTS, **_CLICK},
|
||||
"scroll_to": {**_COMPARISON_INPUTS, **_CLICK},
|
||||
"assert_dom": {**_COMPARISON_INPUTS, **_ASSERT_DOM},
|
||||
"inspect_filter_options": {**_COMPARISON_INPUTS, **_INSPECT_FILTER_OPTIONS},
|
||||
"wait_for_selector": {**_COMPARISON_INPUTS, **_WAIT_FOR_SELECTOR},
|
||||
"extract_table": {**_COMPARISON_INPUTS, **_EXTRACT_TABLE},
|
||||
}
|
||||
|
||||
REQUIRED_ANY: dict[str, tuple[tuple[str, ...], ...]] = {
|
||||
"navigate_tab": (("tab", "tab_identifier"),),
|
||||
"navigate_tabs": (("tab", "tab_identifier"),),
|
||||
"click": (("selector", "selector_ref"),),
|
||||
"scroll_to": (("selector", "selector_ref"),),
|
||||
"assert_dom": (("selector",),),
|
||||
"inspect_filter_options": (("filter_id", "filter_name", "column"),),
|
||||
"wait_for_selector": (("selector", "text"),),
|
||||
"extract_table": (("chart_id", "table"),),
|
||||
}
|
||||
# #endregion ScenarioGraph.StepInputs.Schema
|
||||
|
||||
|
||||
# #region ScenarioGraph.StepInputs.EmbeddedLiterals [C:2] [TYPE Function] [SEMANTICS scenario,step,inputs,literal,baseline]
|
||||
# @ingroup ScenarioGraph
|
||||
# @BRIEF Reject numeric business expected literals while allowing baseline references, tolerance, and structural bounds.
|
||||
# @POST Returns EMBEDDED_METRIC_LITERAL findings for numeric truth under business keys or unpinned tolerance.
|
||||
def find_embedded_metric_literals(inputs: dict[str, Any]) -> list[StepInputsFinding]:
|
||||
findings: list[StepInputsFinding] = []
|
||||
baseline_pinned = any(key in inputs for key in BASELINE_REF_KEYS)
|
||||
for key in sorted(inputs):
|
||||
if key in BASELINE_REF_KEYS or key in STRUCTURAL_NUMERIC_KEYS:
|
||||
continue
|
||||
raw = inputs[key]
|
||||
candidates = raw if isinstance(raw, list) else [raw]
|
||||
for item in candidates:
|
||||
if not (_is_number(item) or _is_numeric_string(item)):
|
||||
continue
|
||||
if key in BUSINESS_LITERAL_KEYS:
|
||||
findings.append(StepInputsFinding(
|
||||
"EMBEDDED_METRIC_LITERAL",
|
||||
f"numeric business literal {item!r} must resolve through an approved baseline", key,
|
||||
))
|
||||
break
|
||||
if key in BASELINE_TOLERANCE_KEYS and not baseline_pinned:
|
||||
findings.append(StepInputsFinding(
|
||||
"EMBEDDED_METRIC_LITERAL",
|
||||
f"numeric {key} requires a baseline_ref or candidate_ref", key,
|
||||
))
|
||||
break
|
||||
return findings
|
||||
# #endregion ScenarioGraph.StepInputs.EmbeddedLiterals
|
||||
|
||||
|
||||
def _is_number(value: Any) -> bool:
|
||||
return isinstance(value, (int, float)) and not isinstance(value, bool)
|
||||
|
||||
|
||||
def _is_numeric_string(value: Any) -> bool:
|
||||
return isinstance(value, str) and bool(_NUMERIC_STR_RE.match(value.strip()))
|
||||
|
||||
|
||||
def _type_ok(value: Any, kinds: tuple[type, ...]) -> bool:
|
||||
if kinds is _BOOL_KINDS:
|
||||
return isinstance(value, bool)
|
||||
if kinds is _INT_KINDS:
|
||||
return isinstance(value, int) and not isinstance(value, bool)
|
||||
if kinds is _NUM_KINDS:
|
||||
return _is_number(value)
|
||||
if kinds is _ID_KINDS:
|
||||
return (isinstance(value, str) and bool(value)) or (isinstance(value, int) and not isinstance(value, bool))
|
||||
return isinstance(value, kinds)
|
||||
|
||||
|
||||
# #region ScenarioGraph.StepInputs.CheckField [C:2] [TYPE Function] [SEMANTICS scenario,step,inputs,bounds]
|
||||
# @ingroup ScenarioGraph
|
||||
# @BRIEF Validate one declared key against its type, length, bound, pattern, and enum contract.
|
||||
# @POST Appends at most one typed finding per key and returns after the first violation.
|
||||
def _check_field(key: str, value: Any, spec: InputField, findings: list[StepInputsFinding]) -> None:
|
||||
if not _type_ok(value, spec.kinds):
|
||||
findings.append(StepInputsFinding("INVALID_STEP_INPUT_TYPE", f"{key} has invalid type", key))
|
||||
return
|
||||
if isinstance(value, str):
|
||||
if spec.min_length is not None and len(value) < spec.min_length:
|
||||
findings.append(StepInputsFinding("INVALID_STEP_INPUT_TYPE", f"{key} must be at least {spec.min_length} characters", key))
|
||||
return
|
||||
if spec.max_length is not None and len(value) > spec.max_length:
|
||||
findings.append(StepInputsFinding("STEP_INPUT_OUT_OF_BOUNDS", f"{key} exceeds {spec.max_length} characters", key))
|
||||
return
|
||||
if spec.pattern is not None and not re.match(spec.pattern, value):
|
||||
findings.append(StepInputsFinding("INVALID_STEP_INPUT_TYPE", f"{key} has an invalid format", key))
|
||||
return
|
||||
if spec.enum is not None and value not in spec.enum:
|
||||
findings.append(StepInputsFinding("STEP_INPUT_OUT_OF_BOUNDS", f"{key} must be one of {spec.enum}", key))
|
||||
return
|
||||
if _is_number(value):
|
||||
if spec.minimum is not None and value < spec.minimum:
|
||||
findings.append(StepInputsFinding("STEP_INPUT_OUT_OF_BOUNDS", f"{key} is below {spec.minimum}", key))
|
||||
return
|
||||
if spec.maximum is not None and value > spec.maximum:
|
||||
findings.append(StepInputsFinding("STEP_INPUT_OUT_OF_BOUNDS", f"{key} exceeds {spec.maximum}", key))
|
||||
return
|
||||
if isinstance(value, list):
|
||||
if spec.max_length is not None and len(value) > spec.max_length:
|
||||
findings.append(StepInputsFinding("STEP_INPUT_OUT_OF_BOUNDS", f"{key} exceeds {spec.max_length} items", key))
|
||||
return
|
||||
for item in value:
|
||||
if spec.item_kinds is not None and not _type_ok(item, spec.item_kinds):
|
||||
findings.append(StepInputsFinding("INVALID_STEP_INPUT_TYPE", f"{key} contains an invalid item", key))
|
||||
return
|
||||
# #endregion ScenarioGraph.StepInputs.CheckField
|
||||
|
||||
|
||||
# #region ScenarioGraph.StepInputs.Validate [C:3] [TYPE Function] [SEMANTICS scenario,step,inputs,validation]
|
||||
# @ingroup ScenarioGraph
|
||||
# @BRIEF Validate a pinned input mapping against the closed per-action schema and embedded-literal rule.
|
||||
# @PRE action is the step action; inputs is the pinned mapping (or any JSON value at the boundary).
|
||||
# @POST Returns a deterministic StepInputsValidation; unknown actions and keys never pass.
|
||||
def validate_step_inputs(action: Any, inputs: Any) -> StepInputsValidation:
|
||||
if not isinstance(inputs, dict):
|
||||
return StepInputsValidation(False, [StepInputsFinding("INVALID_STEP_INPUT_TYPE", "inputs must be a mapping")])
|
||||
schema = ACTION_INPUT_SCHEMAS.get(action) if isinstance(action, str) else None
|
||||
if schema is None:
|
||||
return StepInputsValidation(False, [StepInputsFinding(
|
||||
"UNSUPPORTED_STEP_INPUT_ACTION", f"no bounded input schema for action {action!r}",
|
||||
)])
|
||||
findings: list[StepInputsFinding] = list(find_embedded_metric_literals(inputs))
|
||||
literal_keys = {finding.key for finding in findings}
|
||||
for key in sorted(inputs):
|
||||
spec = schema.get(key)
|
||||
if spec is None:
|
||||
if key not in literal_keys:
|
||||
findings.append(StepInputsFinding("UNKNOWN_STEP_INPUT", f"unknown input {key!r} for action {action!r}", key))
|
||||
continue
|
||||
_check_field(key, inputs[key], spec, findings)
|
||||
for group in REQUIRED_ANY.get(action, ()):
|
||||
if not any(key in inputs for key in group):
|
||||
findings.append(StepInputsFinding("REQUIRED_STEP_INPUT", f"one of {group} is required for action {action!r}"))
|
||||
return StepInputsValidation(not findings, findings)
|
||||
# #endregion ScenarioGraph.StepInputs.Validate
|
||||
|
||||
|
||||
# #region ScenarioGraph.StepInputs.Assert [C:2] [TYPE Function] [SEMANTICS scenario,step,inputs,fail-closed]
|
||||
# @ingroup ScenarioGraph
|
||||
# @BRIEF Fail closed on invalid pinned inputs and return the accepted mapping.
|
||||
# @POST Raises ValueError with the first typed finding code; otherwise returns a copy of the inputs.
|
||||
def assert_step_inputs(action: Any, inputs: Any) -> dict[str, Any]:
|
||||
result = validate_step_inputs(action, inputs)
|
||||
if not result.valid:
|
||||
first = result.findings[0]
|
||||
raise ValueError(f"{first.code}: {first.message}")
|
||||
return dict(inputs)
|
||||
# #endregion ScenarioGraph.StepInputs.Assert
|
||||
|
||||
# #endregion ScenarioGraph.StepInputs
|
||||
@@ -19,6 +19,7 @@ from src.core.logger import logger
|
||||
|
||||
from src.core.logger import belief_scope
|
||||
from src.services.dashboard_testing.scenario.models import DashboardTestScenario, Finding
|
||||
from src.services.dashboard_testing.scenario.step_inputs import validate_step_inputs
|
||||
from src.services.dashboard_testing.scenario.templates import REGISTERED_ACTIONS, TOOLS
|
||||
from src.services.dashboard_testing.scenario.sql_guard import contains_sql_statement
|
||||
|
||||
@@ -100,6 +101,7 @@ def _validate_checks(scenario: DashboardTestScenario) -> ScenarioValidationResul
|
||||
_check_structure(steps, id_set, result)
|
||||
_check_refs(steps, result)
|
||||
_check_tools(steps, result)
|
||||
_check_step_inputs(steps, result)
|
||||
_check_safety(steps, result)
|
||||
_check_dashboard_context(scenario, result)
|
||||
_check_expectations(steps, result)
|
||||
@@ -150,7 +152,10 @@ def _check_refs(steps: list[dict[str, Any]], result: ScenarioValidationResult) -
|
||||
producers[name] = s["id"]
|
||||
produced = set(producers)
|
||||
for s in steps:
|
||||
for inp in s.get("inputs", []):
|
||||
declared = s.get("inputs")
|
||||
if not isinstance(declared, list):
|
||||
continue
|
||||
for inp in declared:
|
||||
name = inp["name"]
|
||||
if name.startswith("step.") and name not in produced:
|
||||
result.errors.append(_err("MISSING_REF", f"step {s['id']} consumes missing output ref {name}", step_id=s["id"]))
|
||||
@@ -173,6 +178,21 @@ def _check_tools(steps: list[dict[str, Any]], result: ScenarioValidationResult)
|
||||
# #endregion ScenarioGraph.Validator.CheckTools
|
||||
|
||||
|
||||
# #region ScenarioGraph.Validator.CheckStepInputs [C:2] [TYPE Function] [SEMANTICS scenario,validator,step,inputs,literal]
|
||||
# @ingroup ScenarioGraph
|
||||
# @BRIEF Validate pinned per-action step inputs and reject embedded numeric business literals.
|
||||
# @POST Appends one error per typed step-input finding; legacy list refs are left to CheckRefs.
|
||||
def _check_step_inputs(steps: list[dict[str, Any]], result: ScenarioValidationResult) -> None:
|
||||
for s in steps:
|
||||
action_inputs = s.get("action_inputs")
|
||||
if not isinstance(action_inputs, dict) or not action_inputs:
|
||||
continue
|
||||
for finding in validate_step_inputs(s.get("action"), action_inputs).findings:
|
||||
detail = f"step {s['id']} input {finding.key}: {finding.message}" if finding.key else f"step {s['id']}: {finding.message}"
|
||||
result.errors.append(_err(finding.code, detail, step_id=s["id"]))
|
||||
# #endregion ScenarioGraph.Validator.CheckStepInputs
|
||||
|
||||
|
||||
# #region ScenarioGraph.Validator.CheckSafety [C:2] [TYPE Function] [SEMANTICS scenario,validator,safety]
|
||||
# @ingroup ScenarioGraph
|
||||
# @BRIEF Check SQL, executable code, and forbidden actions in step text and tooling.
|
||||
|
||||
@@ -0,0 +1,207 @@
|
||||
# #region Test.ScenarioEditor.StepInputs [C:3] [TYPE Module] [SEMANTICS test,scenario,editor,step,inputs,mcp,parity]
|
||||
# @defgroup Typed set_step_inputs application, bounded per-action validation, and MCP propose_graph_revision parity.
|
||||
# @LAYER Test
|
||||
# @RELATION BINDS_TO -> [ScenarioEditor.Apply]
|
||||
# @RELATION BINDS_TO -> [ScenarioGraph.StepInputs]
|
||||
# @TEST_EDGE: embedded_business_literal -> EMBEDDED_METRIC_LITERAL rejected before apply
|
||||
# @TEST_EDGE: unsupported_action_input -> UNSUPPORTED_STEP_INPUT_ACTION rejected
|
||||
# @TEST_EDGE: legacy_step_without_inputs -> canonical dump keeps no inputs key
|
||||
# @TEST_INVARIANT AGSCN-FR-016 -> VERIFIED_BY: test_validator_flags_embedded_metric_literal, test_apply_ops_rejects_business_literal
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
from pydantic import TypeAdapter
|
||||
|
||||
from src.mcp_server.scenario_inputs import GraphRevisionInput
|
||||
from src.models.scenario_registry import ScenarioEditProposal, ScenarioRevision
|
||||
from src.services.dashboard_testing.editor.agent import agent_propose
|
||||
from src.services.dashboard_testing.editor.apply import apply_ops
|
||||
from src.services.dashboard_testing.editor.ops import EditOperation, SetStepInputsOp
|
||||
from src.services.dashboard_testing.scenario.models import DashboardTestScenario, Ref, ScenarioStep
|
||||
from src.services.dashboard_testing.scenario.step_inputs import validate_step_inputs
|
||||
from src.services.dashboard_testing.scenario.validator import validate_scenario
|
||||
|
||||
_SCENARIO_ID = "11111111-1111-4111-8111-111111111111"
|
||||
_BASE_REVISION_ID = "22222222-2222-4222-8222-222222222222"
|
||||
_OPS_ADAPTER = TypeAdapter(list[EditOperation])
|
||||
_FILTER_INPUTS = {"filter_id": "region", "mode": "set", "values": ["EU"]}
|
||||
|
||||
|
||||
# #region Test.ScenarioEditor.StepInputs.Graph [C:1] [TYPE Function] [SEMANTICS test,scenario,step,inputs,fixture]
|
||||
# @ingroup Test.ScenarioEditor.StepInputs
|
||||
# @BRIEF Hardcoded canonical graph with one bounded-action step and one extraction step.
|
||||
# @TEST_FIXTURE step_inputs_graph -> INLINE_JSON
|
||||
def _step(step_id: str, action: str) -> dict:
|
||||
return {
|
||||
"id": step_id, "phase": "interact", "title": action, "tool": "browser", "action": action,
|
||||
"inputs": [], "outputs": [], "expected": {"kind": "structural", "description": f"{action} completes"},
|
||||
"depends_on": [], "automation_status": "ready", "risk": "browser_interaction",
|
||||
}
|
||||
|
||||
|
||||
def _canonical_graph() -> dict:
|
||||
return {
|
||||
"schema_version": 1, "compiler_version": "038.1.0", "template_version": "v1",
|
||||
"scenario_id": "sc-step-inputs", "revision_hash": "a" * 64,
|
||||
"dashboard_context": {"environment_id": "env-preprod-02", "dashboard_id": 80},
|
||||
"objective": {"goal": "verify filters", "selected_case_ids": ["B01"]},
|
||||
"input_fingerprints": {"query_model": "c" * 64},
|
||||
"parameters": [], "phases": ["interact"],
|
||||
"steps": [_step("step-filter-1", "apply_native_filter"), _step("step-table-1", "extract_table")],
|
||||
"outputs": [], "artifact_plan": [], "checklist_coverage": [],
|
||||
"warnings": [], "blockers": [], "risk_summary": {},
|
||||
}
|
||||
# #endregion Test.ScenarioEditor.StepInputs.Graph
|
||||
|
||||
|
||||
# #region Test.ScenarioEditor.StepInputs.CanonicalPin [C:2] [TYPE Function] [SEMANTICS test,scenario,step,inputs,canonical]
|
||||
# @ingroup Test.ScenarioEditor.StepInputs
|
||||
# @BRIEF Parsed steps preserve canonical refs and explicit empty refs.
|
||||
def test_step_inputs_refs_are_canonical_list():
|
||||
absent = ScenarioStep.model_validate(_step("step-filter-1", "apply_native_filter"))
|
||||
assert absent.inputs == []
|
||||
assert absent.model_dump()["inputs"] == []
|
||||
present = ScenarioStep.model_validate({**_step("step-filter-1", "apply_native_filter"), "inputs": []})
|
||||
assert present.model_dump()["inputs"] == []
|
||||
# @BRIEF Legacy refs remain list[Ref], while action arguments use a distinct optional field.
|
||||
def test_inputs_refs_remain_typed_and_action_inputs_are_distinct():
|
||||
graph = _canonical_graph()
|
||||
graph["steps"][0]["inputs"] = [{"name": "step.previous", "kind": "step_output"}]
|
||||
scenario = DashboardTestScenario.model_validate(graph)
|
||||
assert scenario.steps[0].inputs == [Ref(name="step.previous", kind="step_output")]
|
||||
assert scenario.steps[0].action_inputs is None
|
||||
|
||||
# #endregion Test.ScenarioEditor.StepInputs.CanonicalPin
|
||||
|
||||
|
||||
# @BRIEF Canonical bytes for an old graph remain unchanged after schema recovery.
|
||||
def test_legacy_canonical_bytes_remain_unchanged():
|
||||
graph = _canonical_graph()
|
||||
graph["steps"][0]["inputs"] = []
|
||||
expected = DashboardTestScenario.model_validate(graph).canonical_bytes()
|
||||
round_tripped = DashboardTestScenario.model_validate(
|
||||
DashboardTestScenario.model_validate(graph).model_dump(mode="json")
|
||||
).canonical_bytes()
|
||||
assert round_tripped == expected
|
||||
|
||||
|
||||
# #region Test.ScenarioEditor.StepInputs.Apply [C:2] [TYPE Function] [SEMANTICS test,scenario,editor,step,inputs]
|
||||
# @ingroup Test.ScenarioEditor.StepInputs
|
||||
# @BRIEF set_step_inputs pins the validated mapping on the target step and leaves other steps untouched.
|
||||
def test_apply_ops_pins_validated_inputs():
|
||||
ops = _OPS_ADAPTER.validate_python([{
|
||||
"op": "set_step_inputs", "logical_step_id": "step-filter-1", "action_inputs": _FILTER_INPUTS,
|
||||
}])
|
||||
assert isinstance(ops[0], SetStepInputsOp)
|
||||
result = apply_ops(_canonical_graph(), ops)
|
||||
assert result["steps"][0]["action_inputs"] == _FILTER_INPUTS
|
||||
assert result["steps"][0]["inputs"] == []
|
||||
assert result["steps"][1]["action_inputs"] is None
|
||||
# #endregion Test.ScenarioEditor.StepInputs.Apply
|
||||
|
||||
|
||||
# #region Test.ScenarioEditor.StepInputs.RejectMetric [C:2] [TYPE Function] [SEMANTICS test,scenario,editor,step,inputs,literal]
|
||||
# @ingroup Test.ScenarioEditor.StepInputs
|
||||
# @BRIEF Raw numeric business truth is refused before it can be pinned on a step.
|
||||
def test_apply_ops_rejects_business_literal():
|
||||
ops = _OPS_ADAPTER.validate_python([{
|
||||
"op": "set_step_inputs", "logical_step_id": "step-filter-1", "action_inputs": {"expected_total": 1234.5},
|
||||
}])
|
||||
with pytest.raises(ValueError, match="EMBEDDED_METRIC_LITERAL"):
|
||||
apply_ops(_canonical_graph(), ops)
|
||||
|
||||
|
||||
def test_apply_ops_rejects_unsupported_step_action_inputs():
|
||||
ops = _OPS_ADAPTER.validate_python([{
|
||||
"op": "set_step_inputs", "logical_step_id": "step-filter-1", "action_inputs": {"selector": "#x"},
|
||||
}])
|
||||
with pytest.raises(ValueError, match="UNKNOWN_STEP_INPUT"):
|
||||
apply_ops(_canonical_graph(), ops)
|
||||
# #endregion Test.ScenarioEditor.StepInputs.RejectMetric
|
||||
|
||||
|
||||
# #region Test.ScenarioEditor.StepInputs.PerActionSchema [C:2] [TYPE Function] [SEMANTICS test,scenario,step,inputs,schema]
|
||||
# @ingroup Test.ScenarioEditor.StepInputs
|
||||
# @BRIEF Every bounded action accepts its structural inputs, and unsupported actions have no schema.
|
||||
def test_per_action_schema_accepts_structural_inputs():
|
||||
cases = {
|
||||
"navigate_tab": {"tab": "Overview"},
|
||||
"apply_native_filter": {"filter_id": "region", "search_text": "EU", "mode": "clear"},
|
||||
"apply_table_filter": {"column": "revenue", "values": ["x"]},
|
||||
"click": {"selector": "#tab-1", "effect_class": "read_only"},
|
||||
"scroll_to": {"selector": "#table", "baseline_ref": "baseline:scroll"},
|
||||
"assert_dom": {"selector": "#t", "element_count": 5, "sticky": True},
|
||||
"inspect_filter_options": {"filter_id": "region", "limit": 25},
|
||||
"wait_for_selector": {"selector": "#spinner", "timeout_ms": 5000, "state": "hidden"},
|
||||
"extract_table": {"chart_id": 128, "max_rows": 100, "max_columns": 10},
|
||||
}
|
||||
for action, inputs in cases.items():
|
||||
assert validate_step_inputs(action, inputs).valid, action
|
||||
|
||||
|
||||
def test_unsupported_action_and_unknown_key_are_rejected():
|
||||
assert [f.code for f in validate_step_inputs("open_dashboard", {"selector": "#x"}).findings] == [
|
||||
"UNSUPPORTED_STEP_INPUT_ACTION"
|
||||
]
|
||||
assert "REQUIRED_STEP_INPUT" in [f.code for f in validate_step_inputs("navigate_tab", {}).findings]
|
||||
# #endregion Test.ScenarioEditor.StepInputs.PerActionSchema
|
||||
|
||||
|
||||
# #region Test.ScenarioEditor.StepInputs.ValidatorFinding [C:2] [TYPE Function] [SEMANTICS test,scenario,validator,step,inputs,literal]
|
||||
# @ingroup Test.ScenarioEditor.StepInputs
|
||||
# @BRIEF The validator rejects embedded metric truth while accepting structural bounds and baseline tolerance.
|
||||
def test_validator_flags_embedded_metric_literal():
|
||||
graph = _canonical_graph()
|
||||
graph["steps"][0]["action_inputs"] = {"selector": "#metric", "expected_total": 98765.43}
|
||||
codes = [f.code for f in validate_scenario(DashboardTestScenario.model_validate(graph)).errors]
|
||||
assert "EMBEDDED_METRIC_LITERAL" in codes
|
||||
|
||||
|
||||
def test_validator_accepts_structural_bounds_and_baseline_tolerance():
|
||||
graph = _canonical_graph()
|
||||
graph["steps"][0]["action_inputs"] = {
|
||||
"filter_id": "region", "mode": "set", "tolerance": 0.01, "baseline_ref": "baseline:filter-state",
|
||||
}
|
||||
assert [f.code for f in validate_scenario(DashboardTestScenario.model_validate(graph)).errors] == []
|
||||
# #endregion Test.ScenarioEditor.StepInputs.ValidatorFinding
|
||||
|
||||
|
||||
# #region Test.ScenarioEditor.StepInputs.McpParity [C:3] [TYPE Function] [SEMANTICS test,mcp,authoring,step,inputs,parity]
|
||||
# @ingroup Test.ScenarioEditor.StepInputs
|
||||
# @BRIEF The MCP propose_graph_revision boundary accepts the same typed operation and derives the same graph.
|
||||
# @INVARIANT MCP operations are passed verbatim into the shared EditOperation union; no MCP-only or editor-only type exists.
|
||||
def test_mcp_propose_graph_revision_parity(seeded_registry):
|
||||
revision = seeded_registry.query(ScenarioRevision).filter_by(revision_id=_BASE_REVISION_ID).one()
|
||||
revision.graph_snapshot = _canonical_graph()
|
||||
seeded_registry.commit()
|
||||
|
||||
operation = {"op": "set_step_inputs", "logical_step_id": "step-filter-1", "action_inputs": _FILTER_INPUTS}
|
||||
request = GraphRevisionInput(
|
||||
workspace_id="ws-step-inputs", idempotency_key="idem-step-inputs-1", expected_cas_version=0,
|
||||
request_text="pin filter inputs", operations=[operation],
|
||||
)
|
||||
assert request.operations == [operation]
|
||||
|
||||
result = agent_propose(
|
||||
seeded_registry, _SCENARIO_ID, _BASE_REVISION_ID, request.request_text, request.operations,
|
||||
created_by="agent-bot", agent_action_id="agent-action-step-inputs",
|
||||
)
|
||||
seeded_registry.commit()
|
||||
proposal = seeded_registry.query(ScenarioEditProposal).filter_by(proposal_id=result["proposal_id"]).one()
|
||||
assert proposal.proposed_graph["steps"][0]["action_inputs"] == _FILTER_INPUTS
|
||||
assert proposal.operations == [operation]
|
||||
|
||||
|
||||
def test_mcp_propose_graph_revision_rejects_business_literal(seeded_registry):
|
||||
revision = seeded_registry.query(ScenarioRevision).filter_by(revision_id=_BASE_REVISION_ID).one()
|
||||
revision.graph_snapshot = _canonical_graph()
|
||||
seeded_registry.commit()
|
||||
|
||||
with pytest.raises(ValueError, match="EMBEDDED_METRIC_LITERAL"):
|
||||
agent_propose(
|
||||
seeded_registry, _SCENARIO_ID, _BASE_REVISION_ID, "pin raw truth",
|
||||
[{"op": "set_step_inputs", "logical_step_id": "step-filter-1", "action_inputs": {"expected_total": 42}}],
|
||||
created_by="agent-bot",
|
||||
)
|
||||
# #endregion Test.ScenarioEditor.StepInputs.McpParity
|
||||
# #endregion Test.ScenarioEditor.StepInputs
|
||||
Reference in New Issue
Block a user