feat(mcp): add typed test-pack profile preview
This commit is contained in:
@@ -93,7 +93,7 @@ def _scenario_start_permission(arguments: dict[str, Any]) -> tuple[str, str]:
|
||||
# tool listed with deprecated=True for one minor cycle; additive changes bump MINOR.
|
||||
# The discipline is pinned executable by tests/test_mcp_catalog_version.py (the pinned
|
||||
# major in that test is the deliberate-bump ritual — it cannot change by accident).
|
||||
MCP_CATALOG_VERSION = "2.5.0"
|
||||
MCP_CATALOG_VERSION = "2.6.0"
|
||||
# #endregion McpServer.CatalogVersion
|
||||
|
||||
|
||||
@@ -169,6 +169,8 @@ _MCP_CATALOG = (
|
||||
McpToolDefinition("record_case_note", ("scenario:result", "TRIAGE"), service_allowed=False),
|
||||
McpToolDefinition("propose_case_disposition", ("scenario:result", "TRIAGE"), service_allowed=False),
|
||||
McpToolDefinition("inspect_dashboard_context", None),
|
||||
McpToolDefinition("propose_test_pack_profile", None, service_allowed=False),
|
||||
McpToolDefinition("resolve_test_pack_profile", None, service_allowed=False),
|
||||
McpToolDefinition("inspect_scenario", None),
|
||||
McpToolDefinition("validate_scenario", None),
|
||||
McpToolDefinition("scenario_resolve", None),
|
||||
|
||||
@@ -15,7 +15,7 @@ from __future__ import annotations
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field, StrictInt, StrictStr, field_validator
|
||||
from pydantic import BaseModel, ConfigDict, Field, StrictInt, StrictStr, field_validator, model_validator
|
||||
|
||||
from src.services.dashboard_testing.scenario.models import DashboardTestScenario
|
||||
|
||||
@@ -29,11 +29,40 @@ class InitialScenarioIntent(BaseModel):
|
||||
title: StrictStr = Field(min_length=1, max_length=200)
|
||||
dashboard_id: StrictInt = Field(ge=1)
|
||||
allowed_environment_ids: list[StrictStr] = Field(min_length=1, max_length=20)
|
||||
selected_case_ids: list[StrictStr] = Field(default_factory=list, max_length=50)
|
||||
selected_environment_id: StrictStr = Field(min_length=1, max_length=128)
|
||||
selected_case_ids: list[StrictStr] = Field(min_length=1, max_length=19)
|
||||
objective: StrictStr = Field(min_length=1, max_length=2000)
|
||||
compiled_handle_id: StrictStr = Field(min_length=1, max_length=255)
|
||||
draft_pack_id: StrictStr = Field(min_length=1, max_length=128)
|
||||
draft_pack_digest: StrictStr = Field(pattern=r"^[a-fA-F0-9]{64}$")
|
||||
|
||||
# #region McpServer.InitialScenarioIntent.UniqueCases [C:2] [TYPE Function] [SEMANTICS mcp,bootstrap,case-selection]
|
||||
@field_validator("selected_case_ids")
|
||||
@classmethod
|
||||
def _unique_selected_cases(cls, values: list[str]) -> list[str]:
|
||||
if len(values) != len(set(values)):
|
||||
raise ValueError("DUPLICATE_CASE_ID")
|
||||
return values
|
||||
|
||||
# #endregion McpServer.InitialScenarioIntent.UniqueCases
|
||||
|
||||
# #region McpServer.InitialScenarioIntent.UniqueEnvironments [C:2] [TYPE Function] [SEMANTICS mcp,bootstrap,environment]
|
||||
@field_validator("allowed_environment_ids")
|
||||
@classmethod
|
||||
def _unique_environments(cls, values: list[str]) -> list[str]:
|
||||
if len(values) != len(set(values)):
|
||||
raise ValueError("DUPLICATE_ENVIRONMENT_ID")
|
||||
return values
|
||||
|
||||
# #endregion McpServer.InitialScenarioIntent.UniqueEnvironments
|
||||
|
||||
# #region McpServer.InitialScenarioIntent.SelectedEnvironment [C:2] [TYPE Function] [SEMANTICS mcp,bootstrap,environment]
|
||||
@model_validator(mode="after")
|
||||
def _selected_environment_allowed(self) -> InitialScenarioIntent:
|
||||
if self.selected_environment_id not in self.allowed_environment_ids:
|
||||
raise ValueError("SELECTED_ENVIRONMENT_NOT_ALLOWED")
|
||||
return self
|
||||
# #endregion McpServer.InitialScenarioIntent.SelectedEnvironment
|
||||
# #endregion McpServer.InitialScenarioIntent
|
||||
|
||||
|
||||
@@ -305,6 +334,70 @@ class InspectContextInput(BaseModel):
|
||||
# #endregion McpServer.InspectContextInput
|
||||
|
||||
|
||||
# #region McpServer.TestPackProfileInput [C:2] [TYPE Model] [SEMANTICS mcp,scenario,profile,typed]
|
||||
# @ingroup McpServer
|
||||
class TestPackProfileInput(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid", strict=True)
|
||||
|
||||
environment_id: StrictStr = Field(min_length=1, max_length=128)
|
||||
dashboard_id: StrictInt = Field(ge=1)
|
||||
objective: StrictStr = Field(min_length=1, max_length=2000)
|
||||
selected_case_ids: list[StrictStr] = Field(min_length=1, max_length=19)
|
||||
|
||||
@field_validator("selected_case_ids")
|
||||
@classmethod
|
||||
def _unique_profile_cases(cls, values: list[str]) -> list[str]:
|
||||
if len(values) != len(set(values)):
|
||||
raise ValueError("DUPLICATE_CASE_ID")
|
||||
return values
|
||||
# #endregion McpServer.TestPackProfileInput
|
||||
|
||||
|
||||
# #region McpServer.ResolveTestPackProfileInput [C:3] [TYPE Module] [SEMANTICS mcp,profile,resolution,cas]
|
||||
# @ingroup McpServer
|
||||
# #region McpServer.TestPackProfileResolution [C:2] [TYPE Model] [SEMANTICS mcp,profile,resolution,typed]
|
||||
class TestPackProfileResolution(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid", strict=True)
|
||||
|
||||
unresolved_id: StrictStr = Field(min_length=1, max_length=200)
|
||||
step_id: StrictStr | None = Field(default=None, min_length=1, max_length=160)
|
||||
selector_hint: StrictStr | None = Field(default=None, min_length=1, max_length=256)
|
||||
coordinate_id: StrictStr | None = Field(default=None, pattern=r"^[a-f0-9]{32}$")
|
||||
reason: StrictStr = Field(min_length=1, max_length=500)
|
||||
|
||||
@model_validator(mode="after")
|
||||
def _exactly_one_resolution_value(self) -> TestPackProfileResolution:
|
||||
if (self.selector_hint is None) == (self.coordinate_id is None):
|
||||
raise ValueError("EXACTLY_ONE_PROFILE_RESOLUTION_REQUIRED")
|
||||
if self.selector_hint is not None and self.step_id is None:
|
||||
raise ValueError("SELECTOR_STEP_ID_REQUIRED")
|
||||
if self.coordinate_id is not None and self.step_id is not None:
|
||||
raise ValueError("COORDINATE_STEP_ID_FORBIDDEN")
|
||||
return self
|
||||
# #endregion McpServer.TestPackProfileResolution
|
||||
|
||||
|
||||
# #region McpServer.ResolveTestPackProfileRequest [C:2] [TYPE Model] [SEMANTICS mcp,profile,resolution,cas]
|
||||
class ResolveTestPackProfileInput(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid", strict=True)
|
||||
|
||||
environment_id: StrictStr = Field(min_length=1, max_length=128)
|
||||
dashboard_id: StrictInt = Field(ge=1)
|
||||
objective: StrictStr = Field(min_length=1, max_length=2000)
|
||||
selected_case_ids: list[StrictStr] = Field(min_length=1, max_length=19)
|
||||
expected_profile_digest: StrictStr = Field(pattern=r"^[a-f0-9]{64}$")
|
||||
resolutions: list[TestPackProfileResolution] = Field(min_length=1, max_length=100)
|
||||
|
||||
@field_validator("selected_case_ids")
|
||||
@classmethod
|
||||
def _unique_resolution_cases(cls, values: list[str]) -> list[str]:
|
||||
if len(values) != len(set(values)):
|
||||
raise ValueError("DUPLICATE_CASE_ID")
|
||||
return values
|
||||
# #endregion McpServer.ResolveTestPackProfileRequest
|
||||
# #endregion McpServer.ResolveTestPackProfileInput
|
||||
|
||||
|
||||
# #endregion McpServer.ScenarioModels
|
||||
|
||||
|
||||
|
||||
@@ -71,6 +71,7 @@ from src.mcp_server.scenario_inputs import (
|
||||
ScenarioResolveChangeInput,
|
||||
ScenarioResolveInput,
|
||||
ScenarioStartInput,
|
||||
TestPackProfileInput,
|
||||
TestPlanContentInput,
|
||||
TestPlanInput,
|
||||
)
|
||||
|
||||
@@ -17,6 +17,7 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Any
|
||||
|
||||
from mcp.server.fastmcp import Context
|
||||
@@ -32,9 +33,11 @@ from src.mcp_server.scenario_inputs import (
|
||||
DraftPackInput,
|
||||
InspectContextInput,
|
||||
RegisterDraftPackInput,
|
||||
ResolveTestPackProfileInput,
|
||||
ScenarioCompileInput,
|
||||
ScenarioResolveInput,
|
||||
ScenarioStartInput,
|
||||
TestPackProfileInput,
|
||||
)
|
||||
from src.models.maintenance import MaintenanceEvent
|
||||
from src.models.agent_run import AgentRun
|
||||
@@ -58,6 +61,7 @@ from src.services.dashboard_testing.scenario.handles import (
|
||||
)
|
||||
from src.services.dashboard_testing.scenario.resolver import ResolveChange, resolve_scenario
|
||||
from src.services.dashboard_testing.scenario.validator import validate_scenario
|
||||
from src.services.dashboard_testing.scenario.test_pack_profile import apply_coordinate_choices, build_test_pack_profile
|
||||
from src.services.dashboard_testing.baseline_staleness import (
|
||||
BaselinePeriodStale,
|
||||
emit_launch_period_stale_receipt,
|
||||
@@ -69,6 +73,98 @@ from src.services.llm_provider import LLMProviderService
|
||||
from src.services.mcp_approvals import decide_mcp_approval, list_pending_mcp_approvals
|
||||
|
||||
|
||||
# #region McpServer.TestPackProfile.ResolutionHelpers [C:3] [TYPE Module] [SEMANTICS mcp,profile,resolution,validation]
|
||||
# @ingroup McpServer
|
||||
# @BRIEF Validate typed profile resolutions against current unresolved profile items and build server compiler inputs.
|
||||
|
||||
# #region McpServer.TestPackProfile.SelectorChanges [C:3] [TYPE Function] [SEMANTICS mcp,profile,selector,cas]
|
||||
# @ingroup McpServer.TestPackProfile.ResolutionHelpers
|
||||
# @BRIEF Map analyst-selected unresolved IDs to current compiler step IDs; reject stale or forged targets.
|
||||
def _selector_profile_changes(profile, resolutions, scenario):
|
||||
unresolved = {item.id: item for item in profile.unresolved if item.kind == "needs_selector"}
|
||||
steps = {step.id: step for step in scenario.steps}
|
||||
changes: list[dict[str, str]] = []
|
||||
seen: set[str] = set()
|
||||
for resolution in resolutions:
|
||||
item = unresolved.get(resolution.unresolved_id)
|
||||
if (item is None or item.kind != "needs_selector" or resolution.selector_hint is None
|
||||
or resolution.coordinate_id is not None or resolution.unresolved_id in seen):
|
||||
return None
|
||||
case_id = item.id.split(":", 2)[1]
|
||||
step = steps.get(resolution.step_id)
|
||||
if (step is None or step.automation_status != "needs_selector"
|
||||
or item.step_id != resolution.step_id
|
||||
or case_id not in step.checklist_case_ids or step.risk != "browser_interaction"
|
||||
or step.action not in {"apply_native_filter", "apply_table_filter", "click", "select"}
|
||||
or not _valid_selector_hint(resolution.selector_hint)):
|
||||
return None
|
||||
changes.append({"step_id": step.id, "hint": resolution.selector_hint})
|
||||
seen.add(resolution.unresolved_id)
|
||||
return changes if seen else None
|
||||
# #endregion McpServer.TestPackProfile.SelectorChanges
|
||||
|
||||
|
||||
# #region McpServer.TestPackProfile.CoordinateChoices [C:3] [TYPE Function] [SEMANTICS mcp,profile,metric,baseline]
|
||||
# @ingroup McpServer.TestPackProfile.ResolutionHelpers
|
||||
# @BRIEF Validate chosen server-issued metric coordinates against current needs_baseline items.
|
||||
def _coordinate_profile_choices(profile, resolutions):
|
||||
unresolved = {item.id: item for item in profile.unresolved
|
||||
if item.kind == "needs_metric" or (item.kind == "needs_baseline" and item.coordinate_ids)}
|
||||
coordinates = {item.coordinate_id: item for item in profile.coordinates}
|
||||
choices = []
|
||||
seen: set[str] = set()
|
||||
for resolution in resolutions:
|
||||
item = unresolved.get(resolution.unresolved_id)
|
||||
coordinate = coordinates.get(resolution.coordinate_id)
|
||||
if (item is None or coordinate is None or resolution.coordinate_id not in item.coordinate_ids
|
||||
or resolution.selector_hint is not None or resolution.unresolved_id in seen
|
||||
or resolution.coordinate_id is None):
|
||||
return None
|
||||
choices.append({"unresolved_id": item.id, "coordinate_id": coordinate.coordinate_id})
|
||||
seen.add(item.id)
|
||||
return choices if choices else None
|
||||
# #endregion McpServer.TestPackProfile.CoordinateChoices
|
||||
|
||||
|
||||
# #region McpServer.TestPackProfile.InvalidResolution [C:2] [TYPE Function] [SEMANTICS mcp,profile,resolution,validation]
|
||||
# @ingroup McpServer.TestPackProfile.ResolutionHelpers
|
||||
# @BRIEF Enforce that every currently required question has the matching typed resolution in this request.
|
||||
def _invalid_profile_resolution(has_selectors, has_metrics, selector_resolutions, selector_changes,
|
||||
coordinate_resolutions, coordinate_choices):
|
||||
return (
|
||||
(selector_resolutions and selector_changes is None)
|
||||
or (coordinate_resolutions and coordinate_choices is None)
|
||||
or (not has_selectors and not has_metrics and not selector_resolutions and not coordinate_resolutions)
|
||||
)
|
||||
# #endregion McpServer.TestPackProfile.InvalidResolution
|
||||
|
||||
|
||||
# #region McpServer.TestPackProfile.SelectorHint [C:2] [TYPE Function] [SEMANTICS mcp,profile,selector,safety]
|
||||
# @ingroup McpServer.TestPackProfile.ResolutionHelpers
|
||||
# @BRIEF Accept only bounded CSS selectors or explicit accessible-role hints; reject executable locator syntax.
|
||||
def _valid_selector_hint(value: str) -> bool:
|
||||
hint = value.strip()
|
||||
return bool(hint and len(hint) <= 256 and re.fullmatch(
|
||||
r"(?:#[A-Za-z_][\w-]*|\.[A-Za-z_][\w-]*|\[[A-Za-z_:][-\w:.]*(?:=['\"][^\]<>]{1,100}['\"])?\]|(?:role|text|label|placeholder):[\w -]{1,100})",
|
||||
hint,
|
||||
))
|
||||
# #endregion McpServer.TestPackProfile.SelectorHint
|
||||
|
||||
|
||||
# #region McpServer.TestPackProfile.SelectorParameters [C:2] [TYPE Function] [SEMANTICS mcp,profile,selector,compiler]
|
||||
# @ingroup McpServer.TestPackProfile.ResolutionHelpers
|
||||
# @BRIEF Build only compiler-owned selector_hint parameter declarations from validated resolutions.
|
||||
def _selector_parameters(changes: list[dict[str, str]]) -> dict[str, Any]:
|
||||
return {f"selector_{index}": {
|
||||
"type": "selector_hint", "value": item["hint"], "status": "resolved",
|
||||
"affected_step_ids": [item["step_id"]],
|
||||
"required": False, "source": "user",
|
||||
} for index, item in enumerate(changes, start=1)}
|
||||
# #endregion McpServer.TestPackProfile.SelectorParameters
|
||||
|
||||
# #endregion McpServer.TestPackProfile.ResolutionHelpers
|
||||
|
||||
|
||||
# #region McpServer.ToolsScenario.RegisterProbeRead [C:4] [TYPE Function] [SEMANTICS mcp,probe,tools,registration,read]
|
||||
# @ingroup McpServer
|
||||
# @BRIEF Registration seam: the six read-only probe tools (environments/health/dashboards/llm/task).
|
||||
@@ -255,6 +351,121 @@ def register_scenario_tools(server) -> None:
|
||||
}
|
||||
# #endregion McpServer.ScenarioTools.InspectDashboardContext
|
||||
|
||||
|
||||
# #region McpServer.ScenarioTools.ProposeTestPackProfile [C:4] [TYPE Function] [SEMANTICS mcp,scenario,profile,preview]
|
||||
# @ingroup McpServer
|
||||
# @BRIEF Build a typed test-pack proposal from a fresh server inspection and expose unresolved questions.
|
||||
# @PRE Environment access is authorized and selected case IDs exist in the pinned checklist catalog.
|
||||
# @POST Returns complete per-case coverage; only a fully resolvable profile may be save_eligible.
|
||||
# @SIDE_EFFECT Performs bounded upstream Superset reads and emits profile/compile molecular-CoT events.
|
||||
# @INVARIANT Caller-supplied capabilities, query models and expected values are not accepted.
|
||||
# @REJECTED Reusing inspect_scenario's caller-carried query model was rejected — T029h verifies such
|
||||
# claims only at registration; this proposal must classify from fresh server inspection.
|
||||
# #region McpServer.ScenarioTools.ProposeTestPackProfile.Call [C:4] [TYPE Function] [SEMANTICS mcp,scenario,profile,inspection]
|
||||
@server.tool(name="propose_test_pack_profile", structured_output=True)
|
||||
async def propose_test_pack_profile(request: TestPackProfileInput) -> dict[str, Any]:
|
||||
environment = get_config_manager().get_environment(request.environment_id)
|
||||
if environment is None:
|
||||
return {"status": "blocked", "error": "ENV_NOT_FOUND"}
|
||||
try:
|
||||
client = await get_superset_client(environment)
|
||||
query_model = await inspect_dashboard_query_model(client, request.environment_id, request.dashboard_id)
|
||||
if not query_model.query_model_fingerprint or query_model.query_model_fingerprint == "sha256:error":
|
||||
return {"status": "blocked", "error": "CONTEXT_INSPECTION_DEGRADED"}
|
||||
if (query_model.environment_id != request.environment_id
|
||||
or int(query_model.dashboard_id) != request.dashboard_id):
|
||||
return {"status": "blocked", "error": "CONTEXT_IDENTITY_MISMATCH"}
|
||||
profile, scenario, pack = build_test_pack_profile(
|
||||
query_model=query_model, objective=request.objective,
|
||||
selected_case_ids=request.selected_case_ids,
|
||||
browser_available=resolve_browser_availability(),
|
||||
)
|
||||
except (KeyError, ValueError) as exc:
|
||||
logger.explore("Test-pack profile proposal rejected", src="McpServer.ScenarioTools.ProposeTestPackProfile",
|
||||
error_code=str(exc), payload={"dashboard_id": request.dashboard_id}, error=str(exc))
|
||||
return {"status": "blocked", "error": str(exc)}
|
||||
except Exception as exc:
|
||||
logger.explore("Test-pack profile inspection failed", src="McpServer.ScenarioTools.ProposeTestPackProfile",
|
||||
error_code="INSPECTION_FAILED", payload={"dashboard_id": request.dashboard_id}, error=str(exc)[:300])
|
||||
return {"status": "blocked", "error": "INSPECTION_FAILED"}
|
||||
return {
|
||||
"status": profile.status,
|
||||
"profile": profile.model_dump(mode="json"),
|
||||
"preview": {
|
||||
"step_count": len(scenario.steps),
|
||||
"artifacts": pack.get("artifacts", []),
|
||||
"validation": pack.get("validation_summary", {}),
|
||||
},
|
||||
}
|
||||
# #endregion McpServer.ScenarioTools.ProposeTestPackProfile.Call
|
||||
# #endregion McpServer.ScenarioTools.ProposeTestPackProfile
|
||||
|
||||
|
||||
# #region McpServer.ScenarioTools.ResolveTestPackProfile [C:4] [TYPE Function] [SEMANTICS mcp,profile,resolve,cas]
|
||||
# @ingroup McpServer
|
||||
# @BRIEF Re-inspect context, CAS-check the profile digest, and apply only reviewed selector hints.
|
||||
# @PRE Every resolution ID names a current needs_selector blocker in the server-rebuilt profile.
|
||||
# @POST Returns a fresh deterministic preview; no handles or graph bytes are returned.
|
||||
# @SIDE_EFFECT Performs bounded Superset reads and compiler logging; it writes no database state.
|
||||
# @INVARIANT Unsupported targets, stale profile digests and caller graph/value claims fail closed.
|
||||
# #region McpServer.ScenarioTools.ResolveTestPackProfile.Call [C:4] [TYPE Function] [SEMANTICS mcp,profile,resolve,inspection]
|
||||
@server.tool(name="resolve_test_pack_profile", structured_output=True)
|
||||
async def resolve_test_pack_profile(request: ResolveTestPackProfileInput) -> dict[str, Any]:
|
||||
environment = get_config_manager().get_environment(request.environment_id)
|
||||
if environment is None:
|
||||
return {"status": "blocked", "error": "ENV_NOT_FOUND"}
|
||||
try:
|
||||
client = await get_superset_client(environment)
|
||||
query_model = await inspect_dashboard_query_model(client, request.environment_id, request.dashboard_id)
|
||||
if (not query_model.query_model_fingerprint or query_model.query_model_fingerprint == "sha256:error"
|
||||
or query_model.environment_id != request.environment_id
|
||||
or int(query_model.dashboard_id) != request.dashboard_id):
|
||||
return {"status": "blocked", "error": "CONTEXT_IDENTITY_MISMATCH"}
|
||||
baseline, scenario, _ = build_test_pack_profile(
|
||||
query_model=query_model, objective=request.objective,
|
||||
selected_case_ids=request.selected_case_ids,
|
||||
browser_available=resolve_browser_availability(),
|
||||
)
|
||||
if baseline.profile_digest != request.expected_profile_digest:
|
||||
return {"status": "conflict", "error": "PROFILE_STALE",
|
||||
"current_profile_digest": baseline.profile_digest}
|
||||
selector_resolutions = [item for item in request.resolutions if item.selector_hint is not None]
|
||||
coordinate_resolutions = [item for item in request.resolutions if item.coordinate_id is not None]
|
||||
has_selector_items = any(item.kind == "needs_selector" for item in baseline.unresolved)
|
||||
selector_changes = _selector_profile_changes(baseline, selector_resolutions, scenario) if selector_resolutions else []
|
||||
coordinate_choices = _coordinate_profile_choices(baseline, coordinate_resolutions) if coordinate_resolutions else []
|
||||
has_metric_items = any(item.kind == "needs_metric" for item in baseline.unresolved)
|
||||
invalid_resolution = _invalid_profile_resolution(
|
||||
has_selector_items, has_metric_items, selector_resolutions,
|
||||
selector_changes, coordinate_resolutions, coordinate_choices,
|
||||
)
|
||||
if invalid_resolution or (selector_resolutions and not selector_changes) or (coordinate_resolutions and not coordinate_choices):
|
||||
return {"status": "blocked", "error": "PROFILE_RESOLUTION_INVALID"}
|
||||
parameters = _selector_parameters(selector_changes or [])
|
||||
profile, updated, pack = build_test_pack_profile(
|
||||
query_model=query_model, objective=request.objective,
|
||||
selected_case_ids=request.selected_case_ids, parameters=parameters,
|
||||
browser_available=resolve_browser_availability(),
|
||||
)
|
||||
if coordinate_choices:
|
||||
selected_coordinates = {item["unresolved_id"]: item["coordinate_id"] for item in coordinate_choices}
|
||||
profile = apply_coordinate_choices(profile, selected_coordinates)
|
||||
except (KeyError, ValueError) as exc:
|
||||
logger.explore("Test-pack profile resolution rejected", src="McpServer.ScenarioTools.ResolveTestPackProfile",
|
||||
error_code=str(exc), payload={"dashboard_id": request.dashboard_id}, error=str(exc))
|
||||
return {"status": "blocked", "error": str(exc)}
|
||||
except Exception as exc:
|
||||
logger.explore("Test-pack profile resolution inspection failed", src="McpServer.ScenarioTools.ResolveTestPackProfile",
|
||||
error_code="INSPECTION_FAILED", payload={"dashboard_id": request.dashboard_id}, error=str(exc)[:300])
|
||||
return {"status": "blocked", "error": "INSPECTION_FAILED"}
|
||||
return {
|
||||
"status": profile.status, "profile": profile.model_dump(mode="json"),
|
||||
"preview": {"step_count": len(updated.steps), "artifacts": pack.get("artifacts", []),
|
||||
"validation": pack.get("validation_summary", {})},
|
||||
}
|
||||
# #endregion McpServer.ScenarioTools.ResolveTestPackProfile.Call
|
||||
# #endregion McpServer.ScenarioTools.ResolveTestPackProfile
|
||||
|
||||
@server.tool(name="inspect_scenario", structured_output=True)
|
||||
async def inspect_scenario(request: ScenarioCompileInput) -> dict[str, Any]:
|
||||
"""Compile a scenario graph without registering or persisting it.
|
||||
|
||||
@@ -102,6 +102,8 @@ def create_scenario(
|
||||
user_id: str,
|
||||
owner_username: str | None = None,
|
||||
validate_compiled_handle: bool = False,
|
||||
expected_environment_id: str | None = None,
|
||||
expected_case_ids: list[str] | None = None,
|
||||
) -> tuple[ScenarioRegistryEntry, ScenarioRevision]:
|
||||
# New 038 handle path. Keep the legacy DraftArtifact path below during the
|
||||
# migration window so existing 039 save callers remain compatible.
|
||||
@@ -117,6 +119,8 @@ def create_scenario(
|
||||
draft_pack_digest=draft_pack_digest,
|
||||
owner_principal=user_id,
|
||||
dashboard_id=int(stored_compiled.dashboard_id),
|
||||
expected_environment_id=expected_environment_id,
|
||||
expected_case_ids=expected_case_ids,
|
||||
)
|
||||
graph = {
|
||||
**chain["graph"],
|
||||
@@ -297,17 +301,18 @@ def create_initial(db: Session, *, intent, user_id: str, owner_username: str | N
|
||||
draft_pack_id=intent.draft_pack_id,
|
||||
draft_pack_digest=intent.draft_pack_digest,
|
||||
user_id=user_id, owner_username=owner_username,
|
||||
expected_environment_id=intent.selected_environment_id,
|
||||
expected_case_ids=list(intent.selected_case_ids),
|
||||
)
|
||||
entry.name = intent.title
|
||||
entry.description = intent.objective
|
||||
entry.dashboard_id = intent.dashboard_id
|
||||
entry.environment_ids = list(intent.allowed_environment_ids)
|
||||
entry.environment_ids = [intent.selected_environment_id]
|
||||
entry.current_revision_id = revision.revision_id
|
||||
revision.activation_status = ScenarioActivationStatus.CURRENT
|
||||
revision.activated_by = str(user_id)
|
||||
revision.graph_snapshot = {
|
||||
**revision.graph_snapshot,
|
||||
"selected_case_ids": list(intent.selected_case_ids),
|
||||
"objective": intent.objective,
|
||||
}
|
||||
db.flush()
|
||||
|
||||
@@ -105,8 +105,12 @@ def _build_parameters(raw: dict[str, Any]) -> list[ScenarioParameter]:
|
||||
required = spec.get("required", True)
|
||||
default = spec.get("default")
|
||||
source = spec.get("source", "user")
|
||||
affected_step_ids = list(spec.get("affected_step_ids") or [])
|
||||
status = spec.get("status")
|
||||
else:
|
||||
ptype, label, required, default, source = "string", name, True, spec, "user"
|
||||
affected_step_ids = []
|
||||
status = None
|
||||
params.append(
|
||||
ScenarioParameter(
|
||||
name=name,
|
||||
@@ -115,7 +119,8 @@ def _build_parameters(raw: dict[str, Any]) -> list[ScenarioParameter]:
|
||||
required=required,
|
||||
default=default,
|
||||
source=source,
|
||||
status="resolved" if default is not None else "unresolved",
|
||||
affected_step_ids=affected_step_ids,
|
||||
status=status or ("resolved" if default is not None else "unresolved"),
|
||||
value=default,
|
||||
)
|
||||
)
|
||||
@@ -203,7 +208,7 @@ def _compile_graph(req: CompileScenarioRequest) -> CompiledResult:
|
||||
|
||||
# Missing selector/context detection on browser-interaction steps without hints
|
||||
for step in steps:
|
||||
if step.tool == "browser" and step.risk == "browser_interaction" and not _has_selector(parameters):
|
||||
if step.tool == "browser" and step.risk == "browser_interaction" and not _has_selector(parameters, step.id):
|
||||
_apply_selector_check(step, parameters, blockers)
|
||||
|
||||
scenario = DashboardTestScenario(
|
||||
@@ -295,7 +300,7 @@ def _build_step(
|
||||
automation = "ready"
|
||||
if action in {"download", "download_xlsx"} and not mapping.matched_capabilities:
|
||||
automation = "needs_context"
|
||||
if entry["risk"] == "browser_interaction" and not _has_selector(parameters):
|
||||
if entry["risk"] == "browser_interaction" and not _has_selector(parameters, step_id):
|
||||
automation = "needs_selector"
|
||||
|
||||
return ScenarioStep(
|
||||
@@ -344,10 +349,10 @@ def _human_checkpoint_step(case_id: str, ordinal_counter: dict[str, int]) -> Sce
|
||||
|
||||
# #region ScenarioGraph.Compiler.HasSelector [C:1] [TYPE Function] [SEMANTICS scenario,selector]
|
||||
# @ingroup ScenarioGraph
|
||||
# @BRIEF Check whether a resolved selector_hint parameter exists.
|
||||
# @POST Returns True when any selector_hint parameter is resolved.
|
||||
def _has_selector(parameters: list[ScenarioParameter]) -> bool:
|
||||
return any(p.type == "selector_hint" and p.status == "resolved" for p in parameters)
|
||||
# @BRIEF Check whether a resolved selector_hint parameter is bound to the given step.
|
||||
# @POST Returns True only when one resolved selector parameter lists step_id in affected_step_ids.
|
||||
def _has_selector(parameters: list[ScenarioParameter], step_id: str) -> bool:
|
||||
return any(p.type == "selector_hint" and p.status == "resolved" and step_id in p.affected_step_ids for p in parameters)
|
||||
# #endregion ScenarioGraph.Compiler.HasSelector
|
||||
|
||||
|
||||
@@ -356,7 +361,7 @@ def _has_selector(parameters: list[ScenarioParameter]) -> bool:
|
||||
# @BRIEF Emit a NEEDS_SELECTOR blocker for browser steps lacking a resolved selector hint.
|
||||
# @POST Appends a blocker Finding with recovery options when the step needs a selector.
|
||||
def _apply_selector_check(step: ScenarioStep, parameters: list[ScenarioParameter], blockers: list[Finding]) -> None:
|
||||
if step.automation_status == "needs_selector" and not _has_selector(parameters):
|
||||
if step.automation_status == "needs_selector" and not _has_selector(parameters, step.id):
|
||||
blockers.append(Finding(
|
||||
code="NEEDS_SELECTOR",
|
||||
severity="blocker",
|
||||
|
||||
@@ -279,6 +279,8 @@ def verify_handle_chain(
|
||||
draft_pack_digest: str,
|
||||
owner_principal: str,
|
||||
dashboard_id: int,
|
||||
expected_environment_id: str | None = None,
|
||||
expected_case_ids: list[str] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
# SELECT ... FOR UPDATE serializes concurrent consumers on the handle rows, so two
|
||||
# create/save transactions can never both observe an unconsumed handle and double-consume.
|
||||
@@ -340,6 +342,7 @@ def verify_handle_chain(
|
||||
# required model identity from the verified content hash before parsing.
|
||||
graph_payload["revision_hash"] = compiled.content_hash
|
||||
graph = DashboardTestScenario.model_validate(graph_payload).model_dump()
|
||||
_verify_profile_binding(graph, expected_environment_id, expected_case_ids)
|
||||
logger.reflect("Handle chain verified", src="ScenarioGraph.Handles.VerifyChain",
|
||||
claim="POST: authority comes from stored rows plus digest-verified bytes",
|
||||
payload={"compiled": compiled.handle_id[:12], "pack": pack.draft_pack_id[:12], "content_hash": compiled.content_hash[:12]})
|
||||
@@ -347,6 +350,25 @@ def verify_handle_chain(
|
||||
# #endregion ScenarioGraph.Handles.VerifyChain
|
||||
|
||||
|
||||
# #region ScenarioGraph.Handles.VerifyProfileBinding [C:3] [TYPE Function] [SEMANTICS scenario,handles,profile,binding]
|
||||
# @ingroup ScenarioGraph.Handles
|
||||
# @BRIEF Reject bootstrap intent that does not match the digest-verified compiled environment and case coverage.
|
||||
# @PRE graph came from VerifyChain canonical bytes; optional values came from a strict bootstrap intent.
|
||||
# @POST Matching environment/cases are accepted; mismatch raises before handle consumption or registry writes.
|
||||
def _verify_profile_binding(graph: dict[str, Any], environment_id: str | None, case_ids: list[str] | None) -> None:
|
||||
context = graph.get("dashboard_context") or {}
|
||||
if environment_id is not None and str(context.get("environment_id")) != str(environment_id):
|
||||
raise ValueError("PROFILE_ENVIRONMENT_MISMATCH")
|
||||
if case_ids is not None:
|
||||
selected = sorted(set(str(case_id) for case_id in case_ids))
|
||||
graph_selected = sorted(set((graph.get("objective") or {}).get("selected_case_ids") or []))
|
||||
covered = sorted({item.get("case_id") for item in graph.get("checklist_coverage") or []
|
||||
if isinstance(item, dict) and item.get("case_id") in selected})
|
||||
if graph_selected != selected or covered != selected:
|
||||
raise ValueError("PROFILE_CASE_COVERAGE_MISMATCH")
|
||||
# #endregion ScenarioGraph.Handles.VerifyProfileBinding
|
||||
|
||||
|
||||
# #region ScenarioGraph.Handles.Consume [C:4] [TYPE Function] [SEMANTICS scenario,handles,consume,single]
|
||||
# @ingroup ScenarioGraph
|
||||
# @BRIEF Record single consumption of the compiled+pack pair by a created revision (same transaction).
|
||||
|
||||
@@ -0,0 +1,352 @@
|
||||
# #region ScenarioGraph.TestPackProfile [C:4] [TYPE Module] [SEMANTICS scenario,profile,checklist,coverage,eligibility]
|
||||
# @ingroup ScenarioGraph
|
||||
# @BRIEF Project the canonical compiler/validator/draft-pack result into a typed initial test-pack profile.
|
||||
# @INVARIANT The profile is derived from a server-inspected query model and existing compiler services;
|
||||
# it carries no expected metric values and does not mint persistence handles.
|
||||
# @RATIONALE A profile projection adds complete, reviewable case coverage without creating a second
|
||||
# graph compiler or mutable workflow alongside the 038 server-owned handle pipeline.
|
||||
# @REJECTED A separate persisted profile engine was rejected for this slice — 038 compiled and draft-pack
|
||||
# handles already bind canonical graph bytes, validation, owner and save eligibility.
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Literal
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field
|
||||
from src.core.logger import logger
|
||||
|
||||
from src.schemas.dashboard_testing.query_model import DashboardQueryModel
|
||||
from src.services.dashboard_testing.scenario.capability_authority import derive_capabilities
|
||||
from src.services.dashboard_testing.scenario.checklist_catalog import load_catalog
|
||||
from src.services.dashboard_testing.scenario.compiler import CompileScenarioRequest, compile_scenario
|
||||
from src.services.dashboard_testing.scenario.models import DashboardTestScenario, canonical_dump, sha256_hex
|
||||
from src.services.dashboard_testing.scenario.pack_compiler import generate_draft_pack
|
||||
from src.services.dashboard_testing.scenario.validator import validate_scenario
|
||||
|
||||
ProfileStatus = Literal["save_eligible", "preview_only"]
|
||||
UnresolvedKind = Literal["needs_context", "needs_selector", "needs_metric", "needs_baseline", "unsupported"]
|
||||
|
||||
|
||||
# #region ScenarioGraph.TestPackProfile.Case [C:1] [TYPE Model] [SEMANTICS scenario,profile,coverage]
|
||||
class ProfileCase(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid", strict=True)
|
||||
|
||||
case_id: str = Field(pattern=r"^(B0[1-9]|C0[1-7]|T0[1-3])$")
|
||||
classification: Literal["automated", "needs_context", "needs_selector", "needs_baseline", "unsupported", "blocked"]
|
||||
step_ids: list[str] = Field(default_factory=list, max_length=100)
|
||||
rationale: str = Field(max_length=1000)
|
||||
# #endregion ScenarioGraph.TestPackProfile.Case
|
||||
|
||||
|
||||
# #region ScenarioGraph.TestPackProfile.Unresolved [C:1] [TYPE Model] [SEMANTICS scenario,profile,unresolved]
|
||||
class ProfileUnresolved(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid", strict=True)
|
||||
|
||||
id: str = Field(min_length=1, max_length=200)
|
||||
kind: UnresolvedKind
|
||||
json_pointer: str = Field(min_length=1, max_length=500)
|
||||
step_id: str | None = Field(default=None, max_length=160)
|
||||
coordinate_ids: list[str] = Field(default_factory=list, max_length=500)
|
||||
question: str = Field(min_length=1, max_length=500)
|
||||
allowed_resolution: Literal["provide_context", "provide_selector_hint", "select_metric", "select_baseline", "omit_case"]
|
||||
# #endregion ScenarioGraph.TestPackProfile.Unresolved
|
||||
|
||||
|
||||
# #region ScenarioGraph.TestPackProfile.Coordinate [C:2] [TYPE Model] [SEMANTICS scenario,profile,metric,filter]
|
||||
class ProfileCoordinate(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid", strict=True)
|
||||
|
||||
coordinate_id: str = Field(pattern=r"^[a-f0-9]{32}$")
|
||||
chart_id: int = Field(ge=1)
|
||||
chart_name: str = Field(min_length=1, max_length=300)
|
||||
dataset_id: int = Field(ge=1)
|
||||
dataset_name: str = Field(min_length=1, max_length=300)
|
||||
metric_name: str = Field(min_length=1, max_length=300)
|
||||
metric_label: str = Field(min_length=1, max_length=300)
|
||||
expression_type: Literal["SIMPLE", "SQL_EXPRESSION", "SAVED_METRIC"]
|
||||
filter_ids: list[str] = Field(default_factory=list, max_length=100)
|
||||
filter_names: list[str] = Field(default_factory=list, max_length=100)
|
||||
|
||||
# #endregion ScenarioGraph.TestPackProfile.Coordinate
|
||||
|
||||
|
||||
# #region ScenarioGraph.TestPackProfile.Model [C:1] [TYPE Model] [SEMANTICS scenario,profile,eligibility]
|
||||
class TestPackProfile(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid", strict=True)
|
||||
|
||||
profile_version: Literal[1] = 1
|
||||
dashboard_id: int = Field(ge=1)
|
||||
environment_id: str = Field(min_length=1, max_length=128)
|
||||
query_model_fingerprint: str = Field(min_length=1, max_length=128)
|
||||
checklist_catalog_version: int = Field(ge=1)
|
||||
compiler_version: str = Field(min_length=1, max_length=32)
|
||||
status: ProfileStatus
|
||||
eligible: bool
|
||||
selected_case_ids: list[str] = Field(min_length=1, max_length=19)
|
||||
cases: list[ProfileCase] = Field(min_length=1, max_length=19)
|
||||
coordinates: list[ProfileCoordinate] = Field(default_factory=list, max_length=500)
|
||||
unresolved: list[ProfileUnresolved] = Field(default_factory=list, max_length=100)
|
||||
blockers: list[dict[str, Any]] = Field(default_factory=list, max_length=100)
|
||||
warnings: list[dict[str, Any]] = Field(default_factory=list, max_length=100)
|
||||
profile_digest: str = Field(pattern=r"^[a-f0-9]{64}$")
|
||||
# #endregion ScenarioGraph.TestPackProfile.Model
|
||||
|
||||
# #region ScenarioGraph.TestPackProfile.Coordinates [C:2] [TYPE Function] [SEMANTICS scenario,profile,metric,filter]
|
||||
# @ingroup ScenarioGraph.TestPackProfile
|
||||
# @BRIEF Project only server-inspected executable chart metrics and their applicable native filters.
|
||||
def _profile_coordinates(query_model: DashboardQueryModel) -> list[ProfileCoordinate]:
|
||||
from hashlib import sha256
|
||||
|
||||
filters_by_chart: dict[int, list[Any]] = {}
|
||||
for native_filter in query_model.native_filters:
|
||||
for target in native_filter.targets:
|
||||
filters_by_chart.setdefault(target.chart_id, []).append(native_filter)
|
||||
coordinates = []
|
||||
for chart in sorted(query_model.charts, key=lambda item: (item.chart_id, item.slice_name)):
|
||||
if not chart.execution_capable:
|
||||
continue
|
||||
for metric in sorted(chart.metrics, key=lambda item: (item.metric_name, item.label)):
|
||||
identity = f"{chart.chart_id}\0{chart.dataset_id}\0{metric.metric_name}".encode("utf-8")
|
||||
coordinates.append(ProfileCoordinate(
|
||||
coordinate_id=sha256(identity).hexdigest()[:32],
|
||||
chart_id=chart.chart_id, chart_name=chart.slice_name,
|
||||
dataset_id=chart.dataset_id, dataset_name=chart.dataset_name,
|
||||
metric_name=metric.metric_name, metric_label=metric.label,
|
||||
expression_type=metric.expression_type,
|
||||
filter_ids=sorted({item.filter_id for item in filters_by_chart.get(chart.chart_id, [])}),
|
||||
filter_names=sorted({item.name for item in filters_by_chart.get(chart.chart_id, [])}),
|
||||
))
|
||||
return coordinates
|
||||
# #endregion ScenarioGraph.TestPackProfile.Coordinates
|
||||
|
||||
|
||||
# #region ScenarioGraph.TestPackProfile.Classification [C:1] [TYPE Function] [SEMANTICS scenario,profile,classification]
|
||||
def _classification(status: str) -> UnresolvedKind | None:
|
||||
if status in {"needs_context", "needs_selector", "needs_baseline", "unsupported"}:
|
||||
return {"needs_context": "needs_context", "needs_selector": "needs_selector",
|
||||
"needs_baseline": "needs_baseline", "unsupported": "unsupported"}[status] # type: ignore[return-value]
|
||||
return None
|
||||
# #endregion ScenarioGraph.TestPackProfile.Classification
|
||||
|
||||
|
||||
# #region ScenarioGraph.TestPackProfile.Build [C:4] [TYPE Function] [SEMANTICS scenario,profile,compile,coverage]
|
||||
# @ingroup ScenarioGraph.TestPackProfile
|
||||
# @BRIEF Build the canonical graph and typed coverage preview from one server-inspected dashboard model.
|
||||
# @PRE query_model is parsed from an authorized server inspection; selected_case_ids are bounded catalog IDs.
|
||||
# @POST Every selected case is present once; no unsupported case is save-eligible; no expected value is emitted.
|
||||
# @SIDE_EFFECT Compiler and validator emit bounded molecular-CoT logs; no database or handle writes occur.
|
||||
# @DATA_CONTRACT DashboardQueryModel + intent -> TestPackProfile + DashboardTestScenario + DraftPackManifest
|
||||
# #region ScenarioGraph.TestPackProfile.PublicBuilder [C:2] [TYPE Function] [SEMANTICS scenario,profile,api]
|
||||
def build_test_pack_profile(
|
||||
*,
|
||||
query_model: DashboardQueryModel,
|
||||
objective: str,
|
||||
selected_case_ids: list[str],
|
||||
parameters: dict[str, Any] | None = None,
|
||||
baseline_available: bool = False,
|
||||
browser_available: bool | None = None,
|
||||
) -> tuple[TestPackProfile, DashboardTestScenario, dict[str, Any]]:
|
||||
"""Compile one inspected dashboard intent and report every selected case and save blocker."""
|
||||
src = "ScenarioGraph.TestPackProfile.Build"
|
||||
logger.reason("Building test-pack profile from inspected context", src=src,
|
||||
payload={"dashboard_id": query_model.dashboard_id, "case_count": len(selected_case_ids)})
|
||||
try:
|
||||
result = _build_test_pack_profile(
|
||||
query_model=query_model, objective=objective, selected_case_ids=selected_case_ids,
|
||||
parameters=parameters, baseline_available=baseline_available,
|
||||
browser_available=browser_available,
|
||||
)
|
||||
except ValueError as exc:
|
||||
logger.explore("Test-pack profile input rejected", src=src, error_code=str(exc),
|
||||
payload={"dashboard_id": query_model.dashboard_id}, error=str(exc))
|
||||
raise
|
||||
profile, _, _ = result
|
||||
logger.reflect("Test-pack profile evaluated", src=src,
|
||||
payload={"status": profile.status, "unresolved_count": len(profile.unresolved),
|
||||
"digest": profile.profile_digest[:16]})
|
||||
return result
|
||||
# #endregion ScenarioGraph.TestPackProfile.PublicBuilder
|
||||
# #endregion ScenarioGraph.TestPackProfile.Build
|
||||
|
||||
|
||||
# #region ScenarioGraph.TestPackProfile.BuildCore [C:4] [TYPE Function] [SEMANTICS scenario,profile,compile,coverage]
|
||||
# @ingroup ScenarioGraph.TestPackProfile
|
||||
# @BRIEF Compile profile inputs through the canonical graph validator and pack compiler.
|
||||
def _build_test_pack_profile(
|
||||
*, query_model: DashboardQueryModel, objective: str, selected_case_ids: list[str],
|
||||
parameters: dict[str, Any] | None, baseline_available: bool, browser_available: bool | None,
|
||||
) -> tuple[TestPackProfile, DashboardTestScenario, dict[str, Any]]:
|
||||
catalog = load_catalog()
|
||||
case_catalog = {item["id"]: item for item in catalog["cases"]}
|
||||
if not selected_case_ids:
|
||||
raise ValueError("SELECTED_CASES_REQUIRED")
|
||||
unknown = sorted(set(selected_case_ids) - set(case_catalog))
|
||||
if unknown:
|
||||
raise ValueError("UNKNOWN_CASE_ID")
|
||||
|
||||
derivation = derive_capabilities(query_model, browser_available=browser_available)
|
||||
capabilities = dict(derivation.capabilities)
|
||||
capabilities["screenshot"] = capabilities.get("browser", False)
|
||||
capabilities["baseline"] = baseline_available
|
||||
# Facts not derivable from inspected dashboard structure remain false/unknown. In particular,
|
||||
# a model or MCP caller cannot grant mutation fixtures, selectors, or cross-dashboard authority.
|
||||
for key in ("safe_test_data", "safe_clock_fixture", "row_edit", "bulk_edit", "persistence_refresh", "cross_dashboard"):
|
||||
capabilities[key] = False
|
||||
compile_result = compile_scenario(CompileScenarioRequest(
|
||||
agent_run_id="profile-preview",
|
||||
objective={"goal": objective, "selected_case_ids": sorted(selected_case_ids)},
|
||||
query_model=query_model.model_dump(mode="json"),
|
||||
checklist_catalog_version=int(catalog["catalog_version"]),
|
||||
baseline_version="published" if baseline_available else "missing",
|
||||
capabilities=capabilities,
|
||||
parameters=parameters or {},
|
||||
has_dataset_fields=derivation.has_dataset_fields,
|
||||
environment_id=query_model.environment_id,
|
||||
dashboard_id=query_model.dashboard_id,
|
||||
dashboard_name=query_model.title,
|
||||
))
|
||||
scenario = compile_result.scenario
|
||||
validation = validate_scenario(scenario)
|
||||
pack = generate_draft_pack(scenario)
|
||||
return _assemble_profile(query_model, selected_case_ids, scenario, validation, compile_result, pack)
|
||||
|
||||
|
||||
# #region ScenarioGraph.TestPackProfile.Assemble [C:3] [TYPE Function] [SEMANTICS scenario,profile,coverage,questions]
|
||||
# @ingroup ScenarioGraph.TestPackProfile
|
||||
# @BRIEF Assemble per-case coverage and typed unresolved questions from canonical compile results.
|
||||
def _assemble_profile(query_model, selected_case_ids, scenario, validation, compile_result, pack):
|
||||
# #region ScenarioGraph.TestPackProfile.Assemble.Body [C:2] [TYPE Function] [SEMANTICS scenario,profile,coverage]
|
||||
catalog_version = load_catalog()["catalog_version"]
|
||||
selected = set(selected_case_ids)
|
||||
cases: list[ProfileCase] = []
|
||||
unresolved: list[ProfileUnresolved] = []
|
||||
step_by_id = {step.id: step for step in scenario.steps}
|
||||
coordinates = _profile_coordinates(query_model)
|
||||
for coverage in scenario.checklist_coverage:
|
||||
if coverage.case_id in selected:
|
||||
case, questions = _profile_case(coverage, scenario.steps, step_by_id, scenario.parameters, coordinates)
|
||||
cases.append(case)
|
||||
unresolved.extend(questions)
|
||||
if coverage.case_id.startswith("M") and not any(item.kind == "needs_metric" for item in questions):
|
||||
unresolved.append(_profile_question(coverage.case_id, "needs_metric", None, coordinates))
|
||||
if any(step.action == "compare_to_baseline" and coverage.case_id in step.checklist_case_ids
|
||||
for step in scenario.steps) and not any(item.kind == "needs_baseline" for item in questions):
|
||||
unresolved.append(_profile_question(coverage.case_id, "needs_baseline", None, coordinates))
|
||||
if coordinates and not any(item.kind == "needs_metric" for item in unresolved):
|
||||
unresolved.append(_profile_question("dashboard", "needs_metric", None, coordinates))
|
||||
|
||||
blockers = [item.model_dump(mode="json") for item in validation.errors + validation.blockers]
|
||||
blockers.extend({"code": "DRAFT_PACK_BLOCKER", "message": item}
|
||||
for item in pack["validation_summary"].get("blockers", []))
|
||||
# Unsupported cases are unresolved user decisions even when the compiler's compatibility
|
||||
# behavior does not classify them as executable blockers.
|
||||
eligible = pack["status"] == "save_eligible" and validation.valid and not unresolved and not blockers
|
||||
status: ProfileStatus = "save_eligible" if eligible else "preview_only"
|
||||
payload = {
|
||||
"profile_version": 1,
|
||||
"dashboard_id": query_model.dashboard_id,
|
||||
"environment_id": query_model.environment_id,
|
||||
"query_model_fingerprint": query_model.query_model_fingerprint,
|
||||
"checklist_catalog_version": int(catalog_version),
|
||||
"compiler_version": scenario.compiler_version,
|
||||
"status": status,
|
||||
"eligible": eligible,
|
||||
"selected_case_ids": sorted(selected_case_ids),
|
||||
"cases": [case.model_dump(mode="json") for case in cases],
|
||||
"coordinates": [item.model_dump(mode="json") for item in coordinates],
|
||||
"unresolved": [item.model_dump(mode="json") for item in unresolved],
|
||||
"blockers": blockers,
|
||||
"warnings": [item.model_dump(mode="json") for item in validation.warnings + compile_result.warnings],
|
||||
}
|
||||
profile = TestPackProfile(**payload, profile_digest=sha256_hex(canonical_dump(payload)))
|
||||
return profile, scenario, pack
|
||||
# #endregion ScenarioGraph.TestPackProfile.Assemble.Body
|
||||
|
||||
|
||||
# #region ScenarioGraph.TestPackProfile.Case [C:3] [TYPE Function] [SEMANTICS scenario,profile,coverage,question]
|
||||
# @ingroup ScenarioGraph.TestPackProfile
|
||||
# @BRIEF Build one selected case classification and its typed analyst question.
|
||||
def _profile_case(coverage, steps, step_by_id, parameters, coordinates):
|
||||
case_id = coverage.case_id
|
||||
step_ids = [step.id for step in steps if case_id in step.checklist_case_ids]
|
||||
case_steps = [step_by_id[step_id] for step_id in step_ids]
|
||||
kind = _classification(coverage.classification)
|
||||
step_kind = next((_classification(step.automation_status) for step in case_steps
|
||||
if _classification(step.automation_status)), None)
|
||||
if step_kind is not None:
|
||||
kind = step_kind
|
||||
if kind is None and any(parameter.required and parameter.status == "unresolved"
|
||||
and set(parameter.affected_step_ids).intersection(step_ids)
|
||||
for parameter in parameters):
|
||||
kind = "needs_context"
|
||||
if kind is None and not step_ids:
|
||||
kind = "unsupported"
|
||||
case = ProfileCase(
|
||||
case_id=case_id, classification=kind or "automated", step_ids=step_ids,
|
||||
rationale=str(coverage.rationale) if step_ids else "No executable steps were produced for the selected checklist case.",
|
||||
)
|
||||
question_steps = [step.id for step in case_steps if _classification(step.automation_status) == kind] if kind else []
|
||||
if kind and not question_steps:
|
||||
question_steps = [None]
|
||||
return case, [_profile_question(case_id, kind, step_id, coordinates)
|
||||
for step_id in question_steps] if kind else []
|
||||
# #endregion ScenarioGraph.TestPackProfile.Case
|
||||
|
||||
|
||||
# #region ScenarioGraph.TestPackProfile.Question [C:2] [TYPE Function] [SEMANTICS scenario,profile,unresolved,question]
|
||||
# @ingroup ScenarioGraph.TestPackProfile
|
||||
# @BRIEF Map one unresolved case kind to a bounded question and permitted resolution type.
|
||||
def _profile_question(case_id, kind, step_id=None, coordinates=None):
|
||||
if kind is None:
|
||||
return None
|
||||
resolution = {
|
||||
"needs_context": "provide_context", "needs_selector": "provide_selector_hint",
|
||||
"needs_metric": "select_metric",
|
||||
"needs_baseline": "select_baseline", "unsupported": "omit_case",
|
||||
}[kind]
|
||||
question = {
|
||||
"needs_context": f"What bounded test context is required for {case_id}?",
|
||||
"needs_selector": f"Which supported selector identifies the required control for {case_id}?",
|
||||
"needs_metric": f"Which observed chart metric should be captured as a baseline for {case_id}? Do not enter a metric value.",
|
||||
"needs_baseline": f"Which approved baseline should be used for {case_id}? Do not enter a metric value.",
|
||||
"unsupported": f"{case_id} is not safely automatable with the inspected capabilities.",
|
||||
}[kind]
|
||||
coordinate_ids = [item.coordinate_id for item in coordinates or []]
|
||||
return ProfileUnresolved(id=f"case:{case_id}:step:{step_id or 'none'}:{kind}", kind=kind,
|
||||
json_pointer=f"/checklist_coverage/{case_id}", question=question,
|
||||
step_id=step_id,
|
||||
coordinate_ids=coordinate_ids if kind in {"needs_metric", "needs_baseline"} else [],
|
||||
allowed_resolution=resolution)
|
||||
# #endregion ScenarioGraph.TestPackProfile.Question
|
||||
|
||||
|
||||
# #region ScenarioGraph.TestPackProfile.ApplyCoordinateChoices [C:3] [TYPE Function] [SEMANTICS scenario,profile,baseline,coordinate]
|
||||
# @ingroup ScenarioGraph.TestPackProfile
|
||||
# @BRIEF Bind analyst choices to unresolved coordinate requests without assigning or inventing values.
|
||||
def apply_coordinate_choices(profile: TestPackProfile, choices: dict[str, str]) -> TestPackProfile:
|
||||
unresolved = []
|
||||
coordinate_ids = {item.coordinate_id for item in profile.coordinates}
|
||||
if set(choices) - {item.id for item in profile.unresolved if item.kind == "needs_metric"}:
|
||||
raise ValueError("PROFILE_COORDINATE_REQUEST_INVALID")
|
||||
for item in profile.unresolved:
|
||||
coordinate_id = choices.get(item.id)
|
||||
if item.kind != "needs_metric" or coordinate_id is None:
|
||||
unresolved.append(item)
|
||||
continue
|
||||
if coordinate_id not in item.coordinate_ids or coordinate_id not in coordinate_ids:
|
||||
raise ValueError("PROFILE_COORDINATE_INVALID")
|
||||
unresolved.append(item.model_copy(update={
|
||||
"kind": "needs_baseline",
|
||||
"coordinate_ids": [coordinate_id],
|
||||
"question": "Capture and approve this observed metric through the 037 baseline lifecycle.",
|
||||
"allowed_resolution": "select_baseline",
|
||||
}))
|
||||
payload = profile.model_dump(mode="json", exclude={"profile_digest"})
|
||||
payload["unresolved"] = [item.model_dump(mode="json") for item in unresolved]
|
||||
payload["status"] = "preview_only"
|
||||
payload["eligible"] = False
|
||||
return TestPackProfile(**payload, profile_digest=sha256_hex(canonical_dump(payload)))
|
||||
# #endregion ScenarioGraph.TestPackProfile.ApplyCoordinateChoices
|
||||
# #endregion ScenarioGraph.TestPackProfile.Assemble
|
||||
# #endregion ScenarioGraph.TestPackProfile.BuildCore
|
||||
|
||||
# #endregion ScenarioGraph.TestPackProfile
|
||||
@@ -149,7 +149,10 @@ def _selector_params() -> dict:
|
||||
return {
|
||||
"test_date": {"type": "date", "default": "2026-07-01"},
|
||||
"counterparty": {"type": "string", "default": "ACME"},
|
||||
"selector_hint": {"type": "selector_hint", "default": "#native-filter", "required": False},
|
||||
"selector_hint": {
|
||||
"type": "selector_hint", "default": "#native-filter", "required": False,
|
||||
"affected_step_ids": ["phase-2-B01-apply_native_filter"],
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
@@ -209,6 +212,26 @@ def test_visual_compare_without_screenshot_is_blocker() -> None:
|
||||
assert any(b.code == "MISSING_SCREENSHOT" for b in result.blockers)
|
||||
|
||||
|
||||
# #region Test.Scenario.Compiler.SelectorBinding [C:2] [TYPE Function] [SEMANTICS test,selector,binding]
|
||||
def test_selector_hint_only_resolves_its_bound_step() -> None:
|
||||
result = compile_scenario(_req(
|
||||
["B01", "B02"],
|
||||
parameters={"selector": {
|
||||
"type": "selector_hint", "default": "#filter", "required": False,
|
||||
"affected_step_ids": ["phase-2-B01-apply_native_filter"],
|
||||
}},
|
||||
capabilities=dict(FULL, baseline=True),
|
||||
))
|
||||
|
||||
statuses = {step.id: step.automation_status for step in result.scenario.steps}
|
||||
assert statuses["phase-2-B01-apply_native_filter"] == "ready"
|
||||
assert statuses["phase-2-B02-apply_native_filter"] == "needs_selector"
|
||||
assert [item.step_id for item in result.blockers if item.code == "NEEDS_SELECTOR"] == [
|
||||
"phase-2-B02-apply_native_filter",
|
||||
]
|
||||
# #endregion Test.Scenario.Compiler.SelectorBinding
|
||||
|
||||
|
||||
def test_metric_case_appends_compare_when_baseline_present() -> None:
|
||||
caps = dict(FULL, baseline=True)
|
||||
result = compile_scenario(_req(["T01"], capabilities=caps))
|
||||
|
||||
@@ -26,6 +26,8 @@ from src.services.dashboard_testing.scenario.validator import validate_scenario
|
||||
from src.services.dashboard_testing.registry.create import create_initial
|
||||
from src.services.dashboard_testing.registry.materialize import materialize_pending_revisions
|
||||
from src.models.scenario_materialization import OutboxEvent, RevisionMaterialization
|
||||
from src.models.scenario_registry import ScenarioRegistryEntry, ScenarioRevision
|
||||
from src.models.agent_authoring_workspace import AgentAuthoringWorkspace
|
||||
|
||||
|
||||
class _BlobStore:
|
||||
@@ -122,7 +124,8 @@ def test_create_initial_materializes_canonical_graph_and_consumes_handles(db_ses
|
||||
"draft_pack_digest": pack.digest,
|
||||
"dashboard_id": 80,
|
||||
"allowed_environment_ids": ["env-prod-01"],
|
||||
"selected_case_ids": ["B01"],
|
||||
"selected_environment_id": "env-prod-01",
|
||||
"selected_case_ids": ["B01", "C04", "C05", "T01"],
|
||||
"title": "Materialized scenario",
|
||||
"objective": "Verify canonical materialization",
|
||||
})()
|
||||
@@ -131,6 +134,7 @@ def test_create_initial_materializes_canonical_graph_and_consumes_handles(db_ses
|
||||
assert entry.current_revision_id == revision.revision_id
|
||||
assert revision.content_hash == compiled.content_hash
|
||||
assert len(revision.graph_snapshot["steps"]) == len(scenario.steps)
|
||||
assert entry.environment_ids == ["env-prod-01"]
|
||||
assert compiled.consumed_by_revision_id == revision.revision_id
|
||||
assert pack.consumed_by_revision_id == revision.revision_id
|
||||
assert db_session.query(RevisionMaterialization).filter_by(revision_id=revision.revision_id).one().status == "pending"
|
||||
@@ -139,4 +143,31 @@ def test_create_initial_materializes_canonical_graph_and_consumes_handles(db_ses
|
||||
assert db_session.query(RevisionMaterialization).filter_by(revision_id=revision.revision_id).one().status == "materialized"
|
||||
|
||||
|
||||
def test_create_initial_rejects_environment_or_case_mismatch_without_registry_writes(db_session, monkeypatch):
|
||||
import src.services.dashboard_testing.scenario.handles as handles
|
||||
|
||||
monkeypatch.setattr(handles, "_blob_store", _BlobStore())
|
||||
scenario = _scenario()
|
||||
compiled = mint_compiled_handle(db_session, scenario, owner_principal="user-1", dashboard_id=80)
|
||||
mint_validation_result(db_session, compiled, validate_scenario(scenario))
|
||||
pack = mint_draft_pack_handle(
|
||||
db_session, compiled, owner_principal="user-1", agent_run_id=None,
|
||||
scenario_key=scenario.scenario_id, status="save_eligible", template_version="v1",
|
||||
)
|
||||
intent = type("Intent", (), {
|
||||
"compiled_handle_id": compiled.handle_id, "draft_pack_id": pack.draft_pack_id,
|
||||
"draft_pack_digest": pack.digest, "dashboard_id": 80,
|
||||
"allowed_environment_ids": ["env-dev"], "selected_environment_id": "env-dev",
|
||||
"selected_case_ids": ["B01"], "title": "Mismatch", "objective": "Mismatch",
|
||||
})()
|
||||
|
||||
with pytest.raises(ValueError, match="PROFILE_ENVIRONMENT_MISMATCH"):
|
||||
create_initial(db_session, intent=intent, user_id="user-1")
|
||||
assert db_session.query(ScenarioRegistryEntry).count() == 0
|
||||
assert db_session.query(ScenarioRevision).count() == 0
|
||||
assert db_session.query(AgentAuthoringWorkspace).count() == 0
|
||||
assert compiled.consumed_by_revision_id is None
|
||||
assert pack.consumed_by_revision_id is None
|
||||
|
||||
|
||||
# #endregion Test.ScenarioGraph.Handles
|
||||
|
||||
@@ -0,0 +1,94 @@
|
||||
# #region Test.Scenario.TestPackProfile [C:4] [TYPE Module] [SEMANTICS test,scenario,profile,coverage,eligibility]
|
||||
# @BRIEF Verify deterministic profile coverage and fail-closed unresolved classifications.
|
||||
# @RELATION VERIFIES -> [ScenarioGraph.TestPackProfile.Build]
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from src.schemas.dashboard_testing.query_model import DashboardQueryModel
|
||||
from src.services.dashboard_testing.scenario.test_pack_profile import build_test_pack_profile
|
||||
from src.services.dashboard_testing.scenario.test_pack_profile import apply_coordinate_choices
|
||||
|
||||
|
||||
_MODEL = Path(__file__).resolve().parents[3] / "fixtures" / "dashboard_scenarios" / "query_model_sales.json"
|
||||
|
||||
|
||||
def _query_model() -> DashboardQueryModel:
|
||||
payload = json.loads(_MODEL.read_text(encoding="utf-8"))
|
||||
payload["charts"][0]["metrics"] = [{
|
||||
"metric_name": "sum__amount", "label": "Total sales", "expression_type": "SIMPLE",
|
||||
"column": {"column_name": "amount", "type": "DOUBLE"}, "aggregate": "SUM",
|
||||
}]
|
||||
return DashboardQueryModel.model_validate(payload)
|
||||
|
||||
|
||||
# #region Test.Scenario.TestPackProfile.Cases [C:3] [TYPE Function] [SEMANTICS test,profile,unresolved]
|
||||
def test_profile_keeps_selected_cases_visible_and_preview_only_without_context() -> None:
|
||||
profile, scenario, pack = build_test_pack_profile(
|
||||
query_model=_query_model(), objective="Check filters and persisted comments",
|
||||
selected_case_ids=["B01", "B05"], browser_available=True,
|
||||
)
|
||||
|
||||
assert profile.status == "preview_only"
|
||||
assert profile.eligible is False
|
||||
assert {case.case_id for case in profile.cases} == {"B01", "B05"}
|
||||
assert {item.kind for item in profile.unresolved} >= {"needs_selector"}
|
||||
assert len({item.step_id for item in profile.unresolved if item.kind == "needs_selector"}) == len(
|
||||
[item for item in profile.unresolved if item.kind == "needs_selector"]
|
||||
)
|
||||
assert pack["status"] == "preview_only"
|
||||
assert all("expected_value" not in step.model_dump(mode="json") for step in scenario.steps)
|
||||
# #endregion Test.Scenario.TestPackProfile.Cases
|
||||
|
||||
|
||||
# #region Test.Scenario.TestPackProfile.Determinism [C:2] [TYPE Function] [SEMANTICS test,profile,digest]
|
||||
def test_profile_digest_is_stable_and_case_order_independent() -> None:
|
||||
first, _, _ = build_test_pack_profile(
|
||||
query_model=_query_model(), objective="Review filter behavior", selected_case_ids=["B01", "B02"], browser_available=True,
|
||||
)
|
||||
second, _, _ = build_test_pack_profile(
|
||||
query_model=_query_model(), objective="Review filter behavior", selected_case_ids=["B02", "B01"], browser_available=True,
|
||||
)
|
||||
|
||||
assert first.profile_digest == second.profile_digest
|
||||
assert first.selected_case_ids == ["B01", "B02"]
|
||||
# #endregion Test.Scenario.TestPackProfile.Determinism
|
||||
|
||||
|
||||
# #region Test.Scenario.TestPackProfile.CoordinateChoice [C:3] [TYPE Function] [SEMANTICS test,profile,metric,baseline]
|
||||
def test_server_derived_coordinate_choice_never_supplies_metric_truth() -> None:
|
||||
profile, scenario, _ = build_test_pack_profile(
|
||||
query_model=_query_model(), objective="Track dashboard sales", selected_case_ids=["B01"], browser_available=True,
|
||||
)
|
||||
assert profile.coordinates
|
||||
coordinate = profile.coordinates[0]
|
||||
assert coordinate.metric_name == "sum__amount"
|
||||
assert coordinate.filter_ids == ["f1", "f2"]
|
||||
unresolved = next(item for item in profile.unresolved if item.kind == "needs_metric")
|
||||
assert unresolved.coordinate_ids == [item.coordinate_id for item in profile.coordinates]
|
||||
|
||||
selected = apply_coordinate_choices(profile, {unresolved.id: coordinate.coordinate_id})
|
||||
assert selected.status == "preview_only"
|
||||
assert selected.eligible is False
|
||||
assert selected.profile_digest != profile.profile_digest
|
||||
selected_metric = next(item for item in selected.unresolved if item.id == unresolved.id)
|
||||
assert selected_metric.kind == "needs_baseline"
|
||||
assert "expected_value" not in str(scenario.model_dump(mode="json"))
|
||||
assert all(step.expected.kind != "baseline_ref" for step in scenario.steps)
|
||||
|
||||
with pytest.raises(ValueError, match="PROFILE_COORDINATE_INVALID"):
|
||||
apply_coordinate_choices(profile, {unresolved.id: "0" * 32})
|
||||
# #endregion Test.Scenario.TestPackProfile.CoordinateChoice
|
||||
|
||||
|
||||
# #region Test.Scenario.TestPackProfile.InputGuards [C:2] [TYPE Function] [SEMANTICS test,profile,input]
|
||||
@pytest.mark.parametrize(("cases", "error"), [([], "SELECTED_CASES_REQUIRED"), (["X99"], "UNKNOWN_CASE_ID")])
|
||||
def test_profile_rejects_empty_or_unknown_case_selection(cases: list[str], error: str) -> None:
|
||||
with pytest.raises(ValueError, match=error):
|
||||
build_test_pack_profile(query_model=_query_model(), objective="Check dashboard", selected_case_ids=cases)
|
||||
# #endregion Test.Scenario.TestPackProfile.InputGuards
|
||||
|
||||
# #endregion Test.Scenario.TestPackProfile
|
||||
@@ -40,6 +40,8 @@ def test_version_shape_and_pinned_major():
|
||||
f"catalog major {major} != pinned {PINNED_CATALOG_MAJOR}: a breaking catalog change "
|
||||
"(rename/schema change/removal) requires bumping MCP_CATALOG_VERSION and this pin together"
|
||||
)
|
||||
assert MCP_CATALOG_VERSION == "2.6.0"
|
||||
assert {"propose_test_pack_profile", "resolve_test_pack_profile"} <= set(_MCP_CATALOG_BY_NAME)
|
||||
# #endregion Test.McpCatalogVersion.Shape
|
||||
|
||||
|
||||
|
||||
@@ -141,7 +141,7 @@ async def test_mcp_bootstrap_run_automation_and_prod_gate(isolated_mcp_db, monke
|
||||
# T029i/E2E-EXT-001 (ADR-0024): the AgentRun prerequisite is minted through the MCP catalog
|
||||
# itself — the very surface an external client uses (field run 2026-09-07 blocker closure).
|
||||
created_run = _unwrap(await server.call_tool("create_agent_run", {"request": {
|
||||
"dashboard_id": 80, "environment_id": "env-dev",
|
||||
"dashboard_id": 80, "environment_id": "env-prod-01",
|
||||
"dashboard_name": "T029c fixture dashboard", "idempotency_key": f"t029c-run-{uuid4()}",
|
||||
}}))
|
||||
assert created_run["status"] == "ok", created_run
|
||||
@@ -154,7 +154,8 @@ async def test_mcp_bootstrap_run_automation_and_prod_gate(isolated_mcp_db, monke
|
||||
assert registered["context_authority"] == "unverified"
|
||||
initial = {
|
||||
"idempotency_key": f"bootstrap-{uuid4()}", "title": "T029c scenario", "dashboard_id": 80,
|
||||
"allowed_environment_ids": ["env-dev"], "selected_case_ids": [], "objective": "bootstrap",
|
||||
"allowed_environment_ids": ["env-prod-01"], "selected_environment_id": "env-prod-01",
|
||||
"selected_case_ids": scenario.objective["selected_case_ids"], "objective": "bootstrap",
|
||||
"compiled_handle_id": registered["compiled_handle_id"],
|
||||
"draft_pack_id": registered["draft_pack_handle_id"],
|
||||
"draft_pack_digest": registered["draft_pack_digest"],
|
||||
@@ -179,7 +180,11 @@ async def test_mcp_bootstrap_run_automation_and_prod_gate(isolated_mcp_db, monke
|
||||
visible = await server.call_tool("list_scenario_schedules", {"scenario_id": scenario_id})
|
||||
assert _unwrap(visible) == []
|
||||
|
||||
environments = {"env-dev": SimpleNamespace(id="env-dev", stage="DEV", is_production=False), "prod": SimpleNamespace(id="prod", stage="PROD", is_production=True)}
|
||||
environments = {
|
||||
"env-dev": SimpleNamespace(id="env-dev", stage="DEV", is_production=False),
|
||||
"env-prod-01": SimpleNamespace(id="env-prod-01", stage="DEV", is_production=False),
|
||||
"prod": SimpleNamespace(id="prod", stage="PROD", is_production=True),
|
||||
}
|
||||
manager = SimpleNamespace(get_environment=environments.get)
|
||||
monkeypatch.setattr(scenario_module, "get_config_manager", lambda: manager)
|
||||
monkeypatch.setattr(rbac_module, "get_config_manager", lambda: manager)
|
||||
@@ -195,7 +200,7 @@ async def test_mcp_bootstrap_run_automation_and_prod_gate(isolated_mcp_db, monke
|
||||
# registry metadata, so the fresh current revision is directly runnable.
|
||||
blocked = _unwrap(await server.call_tool("start_scenario_run", {"request": {
|
||||
"scenario_id": scenario_id, "revision_id": revision_id,
|
||||
"environment_id": "env-dev", "idempotency_key": f"run-blocked-{uuid4()}",
|
||||
"environment_id": "env-prod-01", "idempotency_key": f"run-blocked-{uuid4()}",
|
||||
"baseline_set": "ss-prod-visual", "baseline_set_version": "1",
|
||||
}}))
|
||||
assert blocked["status"] == "queued", blocked
|
||||
|
||||
@@ -43,7 +43,6 @@ _ANALYST_EXTRA_TOOLS = sorted([
|
||||
"register_draft_pack", # dashboard:testing:WRITE
|
||||
])
|
||||
|
||||
|
||||
def _persist_principal(prefix: str, *, is_admin: bool, grants: tuple[tuple[str, str], ...]) -> str:
|
||||
suffix = secrets.token_hex(4)
|
||||
username = f"{prefix}-{suffix}"
|
||||
|
||||
@@ -13,6 +13,8 @@ from __future__ import annotations
|
||||
|
||||
import os
|
||||
import secrets
|
||||
import json
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
from uuid import uuid4
|
||||
|
||||
@@ -32,6 +34,7 @@ from src.models.auth import McpToolInvocationRecord, Permission, Role, User
|
||||
from src.models.scenario_approval import ActionApprovalGate
|
||||
from src.models.scenario_registry import ScenarioEditProposal, ScenarioRegistryEntry, ScenarioRevision
|
||||
from src.models.scenario_run import ScenarioRun, ScenarioStepRun
|
||||
from src.schemas.dashboard_testing.query_model import DashboardQueryModel
|
||||
|
||||
_EXPECTED_CHAIN_TOOLS = (
|
||||
"create_authoring_session",
|
||||
@@ -44,9 +47,17 @@ _EXPECTED_CHAIN_TOOLS = (
|
||||
"request_save",
|
||||
"activate_revision",
|
||||
"validate_scenario",
|
||||
"propose_test_pack_profile",
|
||||
"resolve_test_pack_profile",
|
||||
)
|
||||
|
||||
|
||||
# #region Test.McpScenarioE2E.AsyncValue [C:1] [TYPE Function] [SEMANTICS test,mcp,async]
|
||||
async def _async_value(value):
|
||||
return value
|
||||
# #endregion Test.McpScenarioE2E.AsyncValue
|
||||
|
||||
|
||||
# #region Test.McpScenarioE2E.Fixture [C:3] [TYPE Function]
|
||||
# @ingroup Test.McpScenarioE2E
|
||||
# @BRIEF Seed one registry entry whose current revision carries the editor graph shape.
|
||||
@@ -268,7 +279,7 @@ async def test_external_client_creates_and_activates_scenario_revision_end_to_en
|
||||
assert operations >= 4
|
||||
|
||||
# E2E-AUTH-003: an MCP-started run pins the promoted revision identity.
|
||||
environments = {"env-dev": SimpleNamespace(id="env-dev", stage="DEV", is_production=False)}
|
||||
environments = {"preprod": SimpleNamespace(id="preprod", stage="PREPROD", is_production=False)}
|
||||
# The start gate (_scenario_start_permission) lives in rbac_server; the tool body
|
||||
# (start_scenario_run) resolves config through tools_scenario since decomposition Phase D.
|
||||
monkeypatch.setattr(tools_scenario_module, "get_config_manager", lambda: SimpleNamespace(get_environment=environments.get))
|
||||
@@ -281,7 +292,7 @@ async def test_external_client_creates_and_activates_scenario_revision_end_to_en
|
||||
started = _unwrap(await server.call_tool("start_scenario_run", {"request": {
|
||||
"scenario_id": scenario_id,
|
||||
"revision_id": activated.get("revision_id") or candidate_revision_id,
|
||||
"environment_id": "env-dev",
|
||||
"environment_id": "preprod",
|
||||
"params": {"region": "emea", "currency": "EUR"},
|
||||
"idempotency_key": f"e2e-run-{uuid4()}",
|
||||
"baseline_set": "ss-prod-visual",
|
||||
@@ -299,6 +310,97 @@ async def test_external_client_creates_and_activates_scenario_revision_end_to_en
|
||||
_cleanup(principal, role_name, scenario_id, workspace_id)
|
||||
|
||||
|
||||
# #region Test.McpScenarioE2E.ProfilePreview [C:4] [TYPE Function] [SEMANTICS test,mcp,profile,preview]
|
||||
# @ingroup Test.McpScenarioE2E
|
||||
# @BRIEF Expose a deterministic typed preview without returning a save handle or graph authority.
|
||||
@pytest.mark.asyncio
|
||||
async def test_propose_test_pack_profile_returns_unresolved_preview_only(monkeypatch) -> None:
|
||||
server = mcp_server._build_probe_server()
|
||||
access = mcp_server.AccessToken(
|
||||
token="profile-token", client_id="profile-client", scopes=["mcp"],
|
||||
subject="profile-operator", claims={"principal_type": "user"},
|
||||
)
|
||||
context_token = _access_token_context.set(access)
|
||||
monkeypatch.setattr(tools_scenario_module, "get_config_manager", lambda: SimpleNamespace(
|
||||
get_environment=lambda environment_id: SimpleNamespace(id=environment_id)
|
||||
))
|
||||
fixture_path = Path(__file__).resolve().parent / "fixtures" / "dashboard_scenarios" / "query_model_sales.json"
|
||||
query_payload = json.loads(fixture_path.read_text(encoding="utf-8"))
|
||||
query_payload["charts"][0]["metrics"] = [{
|
||||
"metric_name": "sum__amount", "label": "Total sales", "expression_type": "SIMPLE",
|
||||
"column": {"column_name": "amount", "type": "DOUBLE"}, "aggregate": "SUM",
|
||||
}]
|
||||
query_model = DashboardQueryModel.model_validate(query_payload)
|
||||
monkeypatch.setattr(tools_scenario_module, "get_superset_client", lambda _: _async_value(object()))
|
||||
monkeypatch.setattr(tools_scenario_module, "inspect_dashboard_query_model", lambda *_: _async_value(query_model))
|
||||
monkeypatch.setattr(tools_scenario_module, "resolve_browser_availability", lambda: True)
|
||||
request = {"environment_id": "ss-prod", "dashboard_id": 11,
|
||||
"objective": "Verify dashboard filters and sales metric", "selected_case_ids": ["B01", "B02"]}
|
||||
try:
|
||||
profile = _unwrap(await server.call_tool("propose_test_pack_profile", {"request": request}))
|
||||
replay = _unwrap(await server.call_tool("propose_test_pack_profile", {"request": request}))
|
||||
assert profile["status"] == "preview_only"
|
||||
assert profile["profile"]["selected_case_ids"] == ["B01", "B02"]
|
||||
assert {case["case_id"] for case in profile["profile"]["cases"]} == {"B01", "B02"}
|
||||
assert profile["profile"]["profile_digest"] == replay["profile"]["profile_digest"]
|
||||
assert {item["kind"] for item in profile["profile"]["unresolved"]} == {"needs_selector", "needs_metric"}
|
||||
assert "scenario" not in profile and "draft_pack" not in profile
|
||||
assert profile["preview"]["step_count"] >= 1
|
||||
coordinates = profile["profile"]["coordinates"]
|
||||
assert coordinates and coordinates[0]["metric_name"] == "sum__amount"
|
||||
baseline_item = next(item for item in profile["profile"]["unresolved"] if item["kind"] == "needs_metric")
|
||||
coordinate_request = {
|
||||
**request,
|
||||
"expected_profile_digest": profile["profile"]["profile_digest"],
|
||||
"resolutions": [{"unresolved_id": baseline_item["id"], "coordinate_id": coordinates[0]["coordinate_id"],
|
||||
"reason": "Analyst selected this metric."}],
|
||||
}
|
||||
coordinate_result = _unwrap(await server.call_tool("resolve_test_pack_profile", {"request": coordinate_request}))
|
||||
assert coordinate_result["status"] == "preview_only"
|
||||
updated_metric = next(item for item in coordinate_result["profile"]["unresolved"]
|
||||
if item["id"] == baseline_item["id"])
|
||||
assert updated_metric["coordinate_ids"] == [coordinates[0]["coordinate_id"]]
|
||||
assert updated_metric["kind"] == "needs_baseline"
|
||||
assert coordinate_result["profile"]["eligible"] is False
|
||||
forged_coordinate = _unwrap(await server.call_tool("resolve_test_pack_profile", {
|
||||
"request": {**coordinate_request, "resolutions": [{**coordinate_request["resolutions"][0],
|
||||
"coordinate_id": "0" * 32}]}
|
||||
}))
|
||||
assert forged_coordinate["status"] == "blocked"
|
||||
unresolved = profile["profile"]["unresolved"][0]
|
||||
unresolved_id = unresolved["id"]
|
||||
selected_step = unresolved["step_id"]
|
||||
resolved_request = {
|
||||
**request,
|
||||
"expected_profile_digest": profile["profile"]["profile_digest"],
|
||||
"resolutions": [{"unresolved_id": unresolved_id, "step_id": selected_step,
|
||||
"selector_hint": "#sales-filter", "reason": "Analyst verified this control."}],
|
||||
}
|
||||
resolved = _unwrap(await server.call_tool("resolve_test_pack_profile", {"request": resolved_request}))
|
||||
assert resolved["status"] == "preview_only"
|
||||
assert {item["id"] for item in resolved["profile"]["unresolved"]} == {
|
||||
item["id"] for item in profile["profile"]["unresolved"] if item["id"] != unresolved_id
|
||||
}
|
||||
assert "scenario" not in resolved and "draft_pack" not in resolved
|
||||
stale = _unwrap(await server.call_tool("resolve_test_pack_profile", {
|
||||
"request": {**coordinate_request, "expected_profile_digest": "0" * 64}
|
||||
}))
|
||||
assert stale == {"status": "conflict", "error": "PROFILE_STALE",
|
||||
"current_profile_digest": profile["profile"]["profile_digest"]}
|
||||
forged = _unwrap(await server.call_tool("resolve_test_pack_profile", {
|
||||
"request": {**resolved_request, "resolutions": [{**resolved_request["resolutions"][0],
|
||||
"step_id": "phase-2-B02-apply_native_filter"}]}
|
||||
}))
|
||||
assert forged["status"] == "blocked"
|
||||
with pytest.raises(Exception):
|
||||
await server.call_tool("propose_test_pack_profile", {
|
||||
"request": {**request, "parameters": {"expected_revenue": 999999}}
|
||||
})
|
||||
finally:
|
||||
_access_token_context.reset(context_token)
|
||||
# #endregion Test.McpScenarioE2E.ProfilePreview
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_operator_without_scenario_edit_cannot_save_or_activate() -> None:
|
||||
suffix = secrets.token_hex(4)
|
||||
|
||||
@@ -47,8 +47,11 @@ class _BlobStore:
|
||||
return self.values.pop(ref, None) is not None
|
||||
|
||||
|
||||
def _seed_bootstrap_handles(user_id: str) -> tuple[str, str, str]:
|
||||
def _seed_bootstrap_handles(user_id: str, environment_id: str) -> tuple[str, str, str]:
|
||||
scenario = DashboardTestScenario.model_validate(json.loads(_FIXTURE.read_text(encoding="utf-8")))
|
||||
payload = scenario.model_dump(mode="json")
|
||||
payload["dashboard_context"]["environment_id"] = environment_id
|
||||
scenario = DashboardTestScenario.model_validate(payload)
|
||||
with SessionLocal() as db:
|
||||
compiled = mint_compiled_handle(db, scenario, owner_principal=user_id, dashboard_id=42)
|
||||
mint_validation_result(db, compiled, validate_scenario(scenario))
|
||||
@@ -72,7 +75,7 @@ def _unwrap(value):
|
||||
async def test_bootstrap_replay_returns_original_ids_and_conflict_is_typed(monkeypatch) -> None:
|
||||
principal = f"t029-{uuid4()}"
|
||||
monkeypatch.setattr(handles_module, "_blob_store", _BlobStore())
|
||||
compiled_id, pack_id, digest = _seed_bootstrap_handles(principal)
|
||||
compiled_id, pack_id, digest = _seed_bootstrap_handles(principal, "env-preprod")
|
||||
server = mcp_server._build_probe_server()
|
||||
monkeypatch.setattr(server, "_record", lambda **_: None)
|
||||
token = _access_token_context.set(mcp_server.AccessToken(
|
||||
@@ -81,7 +84,8 @@ async def test_bootstrap_replay_returns_original_ids_and_conflict_is_typed(monke
|
||||
))
|
||||
request = {
|
||||
"idempotency_key": f"bootstrap-{uuid4()}", "title": "T029 smoke", "dashboard_id": 42,
|
||||
"allowed_environment_ids": ["env-preprod"], "selected_case_ids": ["case-1"],
|
||||
"allowed_environment_ids": ["env-preprod"], "selected_environment_id": "env-preprod",
|
||||
"selected_case_ids": ["B01", "C04", "C05", "T01"],
|
||||
"objective": "Verify bootstrap", "compiled_handle_id": compiled_id, "draft_pack_id": pack_id,
|
||||
"draft_pack_digest": digest,
|
||||
}
|
||||
|
||||
@@ -153,14 +153,35 @@ silently defaults any of them. These markers may produce preview/working-draft
|
||||
output, but only 042 may persist a revision and 044 may enforce launch-time
|
||||
bindings in `RunPreflight`.
|
||||
|
||||
### Handle persistence (amendment 2026-09-06 — normative target, SPECIFIED-PENDING)
|
||||
### External MCP test-pack profile (050 T029a, partial implementation 2026-09-29)
|
||||
|
||||
Implementation status (audit 2026-09-06, Doc.Adr.ADR0023): the deterministic
|
||||
cores (Compile/Validate/Resolve/Canonical serialization) are IMPLEMENTED as pure
|
||||
functions with `@SIDE_EFFECT None`. The durable handle layer below is NOT YET
|
||||
IMPLEMENTED: today the REST/MCP boundaries return graphs in-band, and
|
||||
`ScenarioRegistry.Create.*` verifies caller-composed `compile:{run_id}:{digest}`
|
||||
strings against 036 `DraftArtifact` rows as a transitional stand-in.
|
||||
050 `propose_test_pack_profile` projects a fresh server-inspected
|
||||
`DashboardQueryModel` through the canonical compiler, validator and pack
|
||||
compiler. Its strict profile returns selected-case coverage, stable opaque
|
||||
chart→dataset→metric coordinate IDs, applicable native-filter identities, and
|
||||
typed unresolved questions. The profile digest binds the canonical response;
|
||||
the tool returns a bounded preview, never a compiled graph or draft-pack handle.
|
||||
|
||||
`resolve_test_pack_profile` re-inspects context and uses
|
||||
`expected_profile_digest` as a compare-and-set precondition. Selector hints are
|
||||
restricted and bound to one exact unresolved step; the compiler accepts a
|
||||
resolved selector only for that step. Metric coordinate selection accepts only
|
||||
an ID from the current server-generated profile and changes `needs_metric` to
|
||||
`needs_baseline`; it does not supply an expected value, create a baseline ref,
|
||||
or imply capture/approval/publication. Profile resolution is currently
|
||||
stateless. Domain-context and published-baseline resolution, save-eligible
|
||||
handle minting, and full external MCP profile→bootstrap acceptance remain open
|
||||
in 050 T029a; this preview surface is not a substitute for REST handle gates.
|
||||
|
||||
### Handle persistence (implemented 2026-09-06; T029d–T029g)
|
||||
|
||||
Current status (reconciled 2026-09-29): deterministic Compile/Validate/Resolve/
|
||||
Canonical serialization remain pure functions. Durable handles, canonical-byte
|
||||
storage, persisted REST/MCP minting boundaries, context-authority verification,
|
||||
042 transactional consumption/materialization and fresh-DB bootstrap E2E are
|
||||
implemented; see 050 `tasks.md` T029d–T029i. The handle rules below are current
|
||||
invariants, not pending implementation. T029a profile resolution is a separate
|
||||
incomplete layer and does not yet mint or authorize handles.
|
||||
|
||||
1. **Minting boundaries.** `CompiledScenarioHandle`, `ValidationResultHandle` and
|
||||
`DraftPackHandle` are minted ONLY by the persisted boundaries that wrap the
|
||||
@@ -197,8 +218,8 @@ strings against 036 `DraftArtifact` rows as a transitional stand-in.
|
||||
5. **Rejection rules.** Caller-composed handle strings, caller digests,
|
||||
unbound compiled/draft pairs, `preview_only` packs, stale validation results
|
||||
and double consumption are typed rejections (409-class) with zero partial
|
||||
rows/outbox/gates. The transitional string-format check in
|
||||
`ScenarioRegistry.Create.Register` is removed when this layer lands (050 T029d/T029f).
|
||||
rows/outbox/gates. Caller-composed legacy handle strings are rejected; only
|
||||
owner-bound persisted handle IDs pass `ScenarioRegistry.Create.Register`.
|
||||
6. **Inspect stage status (resolved as hybrid, T029h 2026-09-06).** No persisted
|
||||
`InspectionContextHandle` is minted. Instead: the MCP read tool
|
||||
`inspect_dashboard_context` exposes the live server resolver
|
||||
|
||||
@@ -54,6 +54,32 @@ python3 -c "import yaml; d=yaml.safe_load(open('specs/038-dashboard-scenario-mod
|
||||
9. Prototype: every `@UX_STATE` in `contracts/ux/scenario-graph-ux.md` reachable via `prototype/index.html` state switcher (see `prototype/manifest.md`).
|
||||
10. VlmAnalysisSpec/ScreenshotCaptureSpec serialize canonically and are consumed by 044 executors (runtime VLM/capture belongs to 044).
|
||||
|
||||
## External MCP Profile Preview (050 T029a, partial)
|
||||
|
||||
The external client may call `propose_test_pack_profile` with a dashboard,
|
||||
environment, objective and catalog case IDs. The server inspects the dashboard,
|
||||
derives chart/dataset/metric coordinates and filter applicability, then returns
|
||||
typed coverage and unresolved questions. The response is a preview, not a
|
||||
compiled/draft-pack handle and cannot be registered or bootstrapped.
|
||||
|
||||
`resolve_test_pack_profile` requires the current `profile_digest` and re-inspects
|
||||
the dashboard. It accepts one restricted selector hint for an exact unresolved
|
||||
step or selects a coordinate ID returned by the current profile. It never accepts
|
||||
caller-supplied metric values, query model, capabilities, graph bytes or baseline
|
||||
pins. Coordinate selection advances to `needs_baseline`; use the 037 reference
|
||||
capture/review/approval/publication lifecycle separately. The resolution loop is
|
||||
stateless and T029a remains partial until context/baseline resolution, eligible
|
||||
handle minting and external fresh-DB profile-to-bootstrap E2E are implemented.
|
||||
|
||||
Focused tests:
|
||||
|
||||
```bash
|
||||
cd backend && source .venv/bin/activate
|
||||
python -m pytest -q tests/services/dashboard_testing/scenario/test_test_pack_profile.py \
|
||||
tests/services/dashboard_testing/scenario/test_compiler.py \
|
||||
tests/test_mcp_scenario_e2e.py::test_propose_test_pack_profile_returns_unresolved_preview_only
|
||||
```
|
||||
|
||||
> **Boundary note (2026-08-07)**: 038 is the compiler/IR layer. Runtime VLM submit, real capture bytes, and evidence artifacts (owner_type=scenario_run) are owned by 044 ScenarioExecution — NOT by 038. See `REVIEW-042-047-CLOSURE.md` and 044.
|
||||
|
||||
## Exit Gates (compiler-layer scope)
|
||||
|
||||
@@ -59,6 +59,7 @@
|
||||
| unresolved `needs_context` / `needs_selector` / `needs_baseline` | `ScenarioGraph.ServerOwnedPipeline` | 042 save; 044 `RunPreflight` | `Test.Scenario.Capability`, `Test.Scenario.Resolver`, `test_scenario_runner` |
|
||||
| draft-pack handle and server digest | `ScenarioGraph.PackCompiler.Generate` | 042 server-owned save | `Test.Scenario.Pack`, `Test.Scenario.Pack.Security`, `test_scenario_registry` |
|
||||
| no client graph/digest/runner plan authority | `ScenarioGraph.ServerOwnedPipeline` | 050 MCP transport | injected-code, path-traversal, and MCP raw-graph rejection tests |
|
||||
| external profile coverage + selector/metric-coordinate CAS | `ScenarioGraph.ServerOwnedPipeline` + 050 `TestPackProfile` | 050 `T029a` | `test_test_pack_profile.py`, `test_compiler.py`, profile MCP E2E in `test_mcp_scenario_e2e.py` | PARTIAL: selector/coordinate preview choice is covered; domain context, published baseline and save-handle minting remain open |
|
||||
| persistent co-authoring and sandbox | `ScenarioGraph.AgentAuthoringWorkspace` | 050 MCP -> 042 registry | session persistence, sandbox isolation/limits, receipt and CAS tests |
|
||||
| proposal -> compile/validate -> review -> save | `ScenarioGraph.AgentAuthoringWorkspace` | 042 `ScenarioRegistry.SaveContinuation` | typed-candidate, diff-review, promotion-boundary E2E |
|
||||
|
||||
|
||||
@@ -10,8 +10,16 @@
|
||||
|
||||
### Catalog and error contract
|
||||
|
||||
Catalog version `050.2.0` adds the schemas in [tool-contracts.schema.json](tool-contracts.schema.json).
|
||||
Existing names are retained where present; new names are normative additions, not a runtime catalog claim.
|
||||
Catalog runtime version is `2.6.0` (pinned major 2). T029a adds curated
|
||||
`propose_test_pack_profile` and `resolve_test_pack_profile` definitions using
|
||||
the authenticated-human `permission=None` context-inspection policy; both deny
|
||||
service principals. The strict profile inputs are
|
||||
bounded, and the preview/resolution handlers return no graph or save handle.
|
||||
Implementation and focused acceptance are PARTIAL; full profile-to-bootstrap
|
||||
E2E and unresolved domain/baseline handling remain OPEN.
|
||||
`tool-contracts.schema.json` is a normative schema for the separate audited
|
||||
baseline operations, not the runtime schema source for profile tools. Existing
|
||||
names are retained where present.
|
||||
Every tool has a strict input schema, output schema, permission, principal class and idempotency classification.
|
||||
Read tools require authenticated object ACL; list filtering never substitutes invocation authorization.
|
||||
Errors: `invalid_arguments` (422), `unauthenticated` (401), `permission_denied` (403),
|
||||
@@ -34,6 +42,7 @@ Approval required is a typed non-success pending result, never `completed`. No t
|
||||
| start_scenario_run | 044 admission | explicit revision/env/baseline selector; server resolution before idempotency/gates |
|
||||
| get_scenario_run, get_scenario_run_result | 044 read projection | run ACL; exact stored pin/result, bounded pages |
|
||||
| get_scenario_artifact | 044 evidence metadata | run/artifact owner ACL; typed protected content ref, never raw storage path |
|
||||
| proposed profile preview/resolution | 038 compiler/validator/pack projection | Runtime handler exists; catalog/RBAC binding and external-client acceptance are pending. Once admitted: live server-inspected context; bounded preview; digest CAS; no caller graph/value authority; never mint save handles in the preview operation |
|
||||
| create/update schedule, webhook, CI dispatch | 046 automation | same REST validation even disabled; no HumanStep target |
|
||||
| open/dispose investigation | 047 case transaction | case/queue/episode CAS atomic; external agent is pull-only |
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
||||
"title": "Audited MCP operation inputs 050.2.0",
|
||||
"description": "Normative additions, implemented=false. Server derives principal, digests and pin. Does not replace unaffected legacy catalog tool schemas.",
|
||||
"description": "Normative schemas for audited baseline operations. Runtime catalog 2.6.0 also exposes strict T029a profile preview/resolution DTOs implemented in backend/src/mcp_server/scenario_inputs.py; those handlers do not mint save handles. Server derives principal, digests and pin. Does not replace unaffected legacy catalog tool schemas.",
|
||||
"oneOf": [
|
||||
{
|
||||
"type": "object",
|
||||
|
||||
181
specs/050-mcp-interface/plans/T029a-test-pack-profile-handoff.md
Normal file
181
specs/050-mcp-interface/plans/T029a-test-pack-profile-handoff.md
Normal file
@@ -0,0 +1,181 @@
|
||||
# T029a: Typed Test-Pack Profile and Server-Owned Bindings
|
||||
|
||||
## Objective
|
||||
|
||||
Complete the initial MCP scenario bootstrap so an external MCP agent can propose a useful dashboard test pack from authoritative dashboard facts, ask the analyst only for unresolved business decisions, and never invent metric truth, filters, selectors, runtime values, or baseline authority.
|
||||
|
||||
T029a closes only the profile/compiler/bootstrap completeness gap. It does not claim live provider readiness, baseline publication, scheduled zero-human acceptance, or full production GO.
|
||||
|
||||
## Current Contract and Gap
|
||||
|
||||
- MCP bootstrap input is `InitialScenarioIntent` in `backend/src/mcp_server/scenario_inputs.py`; it carries title, dashboard, allowed environments, case IDs, objective, and server-issued compile/draft-pack handles.
|
||||
- `bootstrap_authoring_scenario` in `backend/src/mcp_server/tools_authoring.py` calls `create_initial` and commits registry + immutable current revision + workspace + idempotency receipt transactionally.
|
||||
- `inspect_dashboard_context` and the 038 compile/validate/draft-pack operations exist. A server-owned `DashboardQueryModel` and fingerprint are available, and context authority is recomputed at `register_draft_pack`.
|
||||
- `ScenarioGraph.PackCompiler.Generate` already yields `preview_only` for unresolved required parameters, `needs_selector`, and `needs_baseline` graph states.
|
||||
- 038 has `ParameterDefinition`, typed refs, checklist capability mapping, `BaselineSelectionProposal` in the backend schemas, and explicit `needs_context` / `needs_selector` / `needs_baseline` semantics.
|
||||
- Remaining gap (updated 2026-09-29): bootstrap now checks selected environment and case coverage against digest-bound graph handles, but no profile handle/eligibility receipt binds the unresolved profile itself to handle minting. The proposal and resolution tools are previews only.
|
||||
- Do not add MCP scheduling work here. T029b/T029c already record schedule tools and bootstrap-to-run evidence; any contrary older handoff statement is superseded.
|
||||
|
||||
## Target Data Flow
|
||||
|
||||
```text
|
||||
inspect_dashboard_context(environment_id, dashboard_id)
|
||||
-> server-owned query-model fingerprint + derived capabilities
|
||||
propose_test_pack_profile(selection intent)
|
||||
-> typed TestPackProfile preview + coverage + unresolved questions
|
||||
resolve_test_pack_profile(expected_profile_digest, typed resolutions, CAS)
|
||||
-> refreshed profile; still preview_only until every gating item is resolved
|
||||
compile/validate/generate_draft_pack(profile_handle) [NOT IMPLEMENTED]
|
||||
-> server-owned CompiledScenarioHandle + DraftPackHandle + eligibility receipt
|
||||
bootstrap_authoring_scenario(InitialScenarioIntent with handles)
|
||||
-> one atomic registry entry + current revision + workspace + receipt
|
||||
```
|
||||
|
||||
The exact tool names may follow the existing compile/validate API instead of introducing a parallel tool family. The implementation must preserve one canonical profile/compiler path shared by REST and MCP, not a second MCP-only compiler.
|
||||
|
||||
## Contract Design
|
||||
|
||||
### 1. Typed `InitialScenarioIntent`
|
||||
|
||||
Keep the current closed, bounded, `extra="forbid"` model. Add only fields that are not already derived from server-owned handles. The final bootstrap intent should bind:
|
||||
|
||||
- idempotency key, title, objective;
|
||||
- selected environment ID and selected checklist case IDs;
|
||||
- opaque server-issued `profile_handle_id`, compiled handle ID and draft-pack handle ID;
|
||||
- expected profile/context/catalog fingerprints only as compare-and-set preconditions, never as authority;
|
||||
- no caller-supplied graph, metrics/value, filter snapshot, dashboard query model, selector, baseline pin, digest authority, SQL/code, raw URL, cookie, or filesystem path.
|
||||
|
||||
Avoid taking both `allowed_environment_ids` and a single selected environment unless the profile genuinely supports multiple environments. For the first scenario revision, bind exactly one inspected environment; additional environments require independent compatibility validation and must not silently share a baseline.
|
||||
|
||||
### 2. Server-owned `TestPackProfile`
|
||||
|
||||
Define a strict schema/DTO, persisted handle only if required by current handle lifecycle. It contains:
|
||||
|
||||
- identity: profile ID/version, dashboard ID, one environment ID, inspected query-model fingerprint, checklist catalog version, compiler/action-registry versions;
|
||||
- analyst intent: bounded objective and selected catalog case IDs;
|
||||
- derived facts: chart IDs/names, dataset IDs/names, metric coordinates, supported filter definitions/scopes, server-derived capabilities, safe parameter definitions;
|
||||
- proposed typed graph/profile mapping: each checklist case maps to supported steps/refs or an explicit classification with reason;
|
||||
- baseline requirements: typed coordinate selector and comparison policy requirements only. Actual expected values come exclusively from a separately resolved, approved/published baseline; profile generation never embeds values;
|
||||
- unresolved list with stable IDs and JSON pointers, each typed as `needs_context`, `needs_selector`, `needs_metric`, `needs_baseline`, or `unsupported`, plus a concise analyst question and allowed resolution schema;
|
||||
- coverage summary, deterministic blockers/warnings, `preview_only | save_eligible`;
|
||||
- provenance hashes/opaque source refs for server-inspected inputs.
|
||||
|
||||
Do not duplicate raw query context or source bytes when a handle already pins the authoritative snapshot. Profile serialization and content hash must be canonical and deterministic.
|
||||
|
||||
### 3. Question-and-resolution protocol
|
||||
|
||||
Every required unresolved item produces a typed question, not free-form implied consent:
|
||||
|
||||
- `needs_context`: request a bounded domain value/choice or ask for clarification; no sensitive value defaults.
|
||||
- `needs_selector`: request an approved selector hint tied to the exact chart/control and safe selector grammar; reject arbitrary executable locator code.
|
||||
- `needs_metric`: offer only server-inspected chart/dataset/metric coordinates; a selected coordinate transitions to `needs_baseline` but does not create baseline truth.
|
||||
- `needs_baseline`: use the 037 reference URL/candidate/approval/publication lifecycle. Do not ask the user to type the expected numeric value.
|
||||
- `unsupported`: explain why the selected case cannot be automated and offer only a supported manual/omit resolution if 038 policy allows it.
|
||||
|
||||
Resolution writes must include profile handle, expected CAS version, idempotency key and typed resolution values. Recompute profile and eligibility server-side after every resolution. Stale profile/context returns typed conflict and makes no partial update. Preserve prior resolved answers on recoverable errors.
|
||||
|
||||
Do not create a general natural-language question engine in this task. Agent wording is presentation; the server-owned question ID/type/schema and validation findings are authoritative.
|
||||
|
||||
## Invariants
|
||||
|
||||
- Dashboard, environment, query model, capabilities, chart/dataset/metric attribution, filter definitions/scopes and fingerprints are resolved by the server.
|
||||
- Caller capability declarations may only restrict server-derived capability, never widen it or turn unknown into available.
|
||||
- A selected checklist case has exactly one deterministic classification and explicit coverage entry; no selected case silently disappears.
|
||||
- Missing required facts remain typed unresolved/unsupported states; no inferred selector, test data, numeric expectation, baseline pin, or runtime parameter value.
|
||||
- Baseline-backed cases require a valid baseline reference contract; discovery/capture/approval/publication remain separate lifecycle steps.
|
||||
- `preview_only` cannot mint save-eligible compiled/draft-pack handles and cannot reach `create_initial`.
|
||||
- Bootstrap rechecks handle ownership, agent-run principal, dashboard/environment binding, profile fingerprint, all selected-case resolutions, context authority, and save eligibility in the same transaction as registry/workspace creation.
|
||||
- On validation, ACL, handle, fingerprint, stale-CAS, or idempotency conflict failure, no registry entry, revision, workspace, outbox/materialization event, consumed handle, or bootstrap receipt is committed.
|
||||
- Identical idempotent bootstrap replay returns the original identities; same key with changed semantic intent returns the existing typed conflict.
|
||||
- A successful bootstrap's immutable revision records the canonical profile summary/fingerprints and selected cases needed for provenance, but never mutable run-time bindings or secret-bearing source state.
|
||||
|
||||
## Implementation Slices
|
||||
|
||||
### Progress note (2026-09-28)
|
||||
|
||||
- Implemented an initial typed, deterministic profile projection and `propose_test_pack_profile` MCP preview over a fresh server-side dashboard inspection. It returns every selected case with typed unresolved questions and never emits expected metric values. The tool and DTO are present in the MCP catalog; profile unit and MCP proposal tests cover deterministic digest and fail-closed preview.
|
||||
- Bootstrap now binds one selected environment and the exact case IDs to digest-verified handle graph content before registry writes; selected environment must be allowed. Mismatch tests assert no registry/revision/workspace creation and no handle consumption.
|
||||
- This is an incomplete T029a slice, not completion. Resolution is stateless and covers selector hints and metric coordinate choices only; profile-to-save-handle minting is not implemented. Full fresh-DB profile→resolve→register→bootstrap E2E and catalog/schema parity remain open.
|
||||
- The selector-resolution operation uses `expected_profile_digest`, fresh upstream reinspection, one unresolved ID + exact step ID per answer, a restricted selector grammar, and recompilation. Per-step selector bindings are enforced by the canonical compiler; resolving one filter leaves other filter questions unresolved. Domain context, published baseline, durable idempotent resolution, and save-handle minting remain open.
|
||||
- The profile lists server-derived chart → dataset → metric coordinates, stable coordinate IDs, and chart-level native-filter identity/name applicability. Selecting a coordinate by digest-CAS records only the selection and changes the question to `needs_baseline`; it never records a metric value or creates a baseline reference. 037 capture/approval/publication is still required. Metric query data is local to these tests; the shared sales query fixture remains unchanged.
|
||||
- The first MCP profile test currently stubs the Superset inspection seam with a hardcoded fixture. This proves tool mapping/typed output only, not real upstream integration or the no-seeded-prerequisite E2E gate.
|
||||
|
||||
### Slice A: Contract alignment
|
||||
|
||||
- Read current T029a, 038 compile/validator/handle contracts, `InitialScenarioIntent`, compile/draft-pack DTOs, and MCP catalog/RBAC rules.
|
||||
- Add/update schemas in 038 and 050 first: TestPackProfile, unresolved item/question, typed resolution, profile handle/eligibility receipt, and bootstrap preconditions.
|
||||
- Specify canonical hashing, limits, stale error codes, and REST/MCP parity.
|
||||
- Preserve pinned MCP catalog major 2; bump minor only if public tool schemas change.
|
||||
|
||||
### Slice B: Profile builder
|
||||
|
||||
- Build a pure deterministic profile builder over server-owned query model, checklist catalog, derived capabilities, parameter declarations, and baseline summaries/availability.
|
||||
- Use existing capability mapper/compiler/validator rather than duplicating their classifications.
|
||||
- Emit explicit per-case coverage and typed unresolved items.
|
||||
- Add hardcoded fixtures for complete dashboard, missing dataset field, unknown selector, missing baseline, unsupported XLSX, unsafe mutation without a safe test fixture, empty case selection, duplicate/unknown case IDs, and partial inspection failure.
|
||||
|
||||
### Slice C: Resolution lifecycle
|
||||
|
||||
- Add server-owned profile handle/CAS/idempotency receipts only if existing CompiledScenarioHandle/DraftPackHandle cannot safely represent the mutable preview-to-resolved flow.
|
||||
- Add MCP operations needed to read the question list and submit bounded typed resolutions; use existing REST service/domain functions when possible.
|
||||
- Re-inspect or compare authoritative fingerprints at mutation boundaries; reject context drift with typed stale conflict.
|
||||
- Prove stale/replay/duplicate-resolution behavior and zero partial side effects.
|
||||
|
||||
### Slice D: Compile and bootstrap gate
|
||||
|
||||
- Compile from the server-resolved profile to the existing canonical DashboardTestScenario pipeline.
|
||||
- Generate preview pack for incomplete profile, but do not mint eligible save handles.
|
||||
- For complete profile, mint compiled and draft-pack handles with profile hash/fingerprint linkage.
|
||||
- Strengthen `create_initial`/`bootstrap_authoring_scenario` to reject handles whose profile receipt is absent, stale, preview-only, cross-dashboard, cross-environment, or missing a required resolution.
|
||||
- Preserve atomic registry/revision/workspace/outbox/receipt behavior and idempotency semantics.
|
||||
|
||||
### Slice E: External MCP E2E and evidence
|
||||
|
||||
- Replace or add a fresh-database MCP E2E that obtains every prerequisite through tools reachable by the same human principal; no ORM-seeded registry, revision, run, handle, inspected context or baseline truth.
|
||||
- Exercise complete path: MCP auth → inspect context → profile preview → receive unresolved questions → resolve all safe required items → inspect refreshed coverage/eligibility → register eligible pack handles → bootstrap → verify visible current revision and provenance.
|
||||
- Negative vectors: forged metric/value/filter/context, foreign dashboard/environment, stale context fingerprint, unsupported/unknown case, missing baseline, invented selector, unresolved required item at bootstrap, raw graph/digest, cross-owner handles, stale CAS, idempotency replay/conflict.
|
||||
- Assert zero registry/revision/workspace/outbox/consumed-handle side effects on every rejected bootstrap.
|
||||
- Retain request/response trace with secrets redacted and test fixture explicitly identified as synthetic. No production or live release claim from fixture E2E.
|
||||
|
||||
## Acceptance Checklist
|
||||
|
||||
- [ ] `TestPackProfile` is strict, bounded, versioned and canonically hashed.
|
||||
- [ ] Every selected case is represented by supported typed steps or a visible typed unresolved/unsupported classification.
|
||||
- [ ] Server derives metrics, filters, selectors/capabilities and baseline requirements from authorized context; client claims cannot widen facts.
|
||||
- [ ] Profile preview asks explicit questions for unresolved required data and enumerates allowed typed answers.
|
||||
- [ ] Unknown expected metric values are never asked as free-text numeric inputs or embedded in graph assertions.
|
||||
- [ ] Unresolved required context, selector, baseline, or unsupported blocking case produces `preview_only` and cannot mint save-eligible handles.
|
||||
- [ ] Typed resolution is owner-scoped, CAS/idempotent, recomputes profile status, and handles stale dashboard/catalog context without partial writes.
|
||||
- [ ] `bootstrap_authoring_scenario` accepts only matching server-issued save-eligible handles and atomically creates the initial current revision/workspace/outbox/receipt.
|
||||
- [ ] Fresh external MCP E2E has no pre-seeded authoring prerequisites and includes one complete and all required negative paths.
|
||||
- [ ] MCP catalog major remains 2; minor/catalog schemas/RBAC docs are updated if tools are added.
|
||||
- [ ] 038 and 050 schemas/contracts/tasks/traceability/quickstarts agree; T029a is closed only against retained evidence.
|
||||
- [ ] Overall production matrix remains `NO-GO` until unrelated provider, live publication, scheduled replay, injection and analyst-acceptance gates close.
|
||||
|
||||
## Commands and Verification
|
||||
|
||||
Run from `backend/` with `.venv` activated:
|
||||
|
||||
```bash
|
||||
python -m pytest -q \
|
||||
tests/test_mcp_initial_scenario_e2e.py \
|
||||
tests/test_mcp_t029_bootstrap_automation.py \
|
||||
tests/test_mcp_authoring_promotion_e2e.py \
|
||||
tests/services/dashboard_testing/scenario \
|
||||
tests/services/dashboard_testing/registry/test_create.py
|
||||
python -m ruff check src/mcp_server src/services/dashboard_testing/scenario \
|
||||
src/services/dashboard_testing/registry/create.py tests/test_mcp_initial_scenario_e2e.py
|
||||
python -m compileall -q src/mcp_server src/services/dashboard_testing/scenario \
|
||||
src/services/dashboard_testing/registry/create.py
|
||||
```
|
||||
|
||||
Also run MCP catalog/schema validation, OpenAPI/schema validation for touched contracts, `git diff --check`, and the isolated fresh-DB MCP E2E. Run broader tests when shared compiler, checklist, baseline, or registry behavior changes.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- No frontend prompt/chat/agent controls; agent interaction remains external MCP only.
|
||||
- No new scheduling implementation (T029b/T029c already cover it).
|
||||
- No exploration sandbox implementation (T050).
|
||||
- No MCP baseline selection review surface (T049) beyond reusing existing baseline references and 037 gates.
|
||||
- No guessed business logic, selectors, data fixtures, expected numeric values, or auto-approved baseline.
|
||||
- No production release/deployment, scheduled zero-human claim, terminal PASS claim, or overall readiness promotion.
|
||||
@@ -6,7 +6,7 @@
|
||||
> CRUD/editor, human approvals, and read-only evidence; it MUST NOT expose agent prompts, chat,
|
||||
> proposal generation, or an agent workspace. Catalog major is pinned: breaking bootstrap-handle
|
||||
> semantics shipped as catalog `2.0.0` with ritual `PINNED_CATALOG_MAJOR=2`. Additive minors
|
||||
> (`2.1.0` context authority, `2.2.0` `create_agent_run`/`get_agent_run`) keep the same major.
|
||||
> (`2.1.0` context authority, `2.2.0` `create_agent_run`/`get_agent_run`, `2.6.0` T029a profile preview/resolution) keep the same major.
|
||||
> Historical transport `[x]` rows in `tasks.md` do not close production E2E.
|
||||
|
||||
## Target vs observed
|
||||
@@ -14,8 +14,8 @@
|
||||
| Layer | Target (refresh) | Observed (do not treat as GO) |
|
||||
|---|---|---|
|
||||
| Surface | One `/mcp` Streamable HTTP server; external client owns reasoning | `/mcp` mounted; OAuth 2.1 + DCR local evidence exists |
|
||||
| Catalog | Versioned curated tools; `PINNED_CATALOG_MAJOR=2` | Runtime `MCP_CATALOG_VERSION="2.3.0"` (61 entries); pin test `tests/test_mcp_catalog_version.py` |
|
||||
| Authoring | `inspect_dashboard_context` → `create_agent_run` → `register_draft_pack` → `bootstrap_authoring_scenario` → `activate_revision` / `start_scenario_run` | Vertical E2E exists with MCP-created AgentRun (T029i). Live-stand replay T029m / `E2E-EXT-002` **CLOSED 2026-09-11** (`docs/2026-09-11-sales-prod-mcp-replay.md`, run `110a6517…` + human loop) |
|
||||
| Catalog | Versioned curated tools; `PINNED_CATALOG_MAJOR=2` | Runtime `MCP_CATALOG_VERSION="2.6.0"`; profile tools are curated as authenticated-human read previews; pin test covers major/schema registration |
|
||||
| Authoring | `inspect_dashboard_context` → `propose_test_pack_profile` → `resolve_test_pack_profile` → `create_agent_run` / `register_draft_pack` → `bootstrap_authoring_scenario` | Preview and stateless selector/metric-choice CAS are client-visible; eligible handle minting and full profile→bootstrap external E2E remain OPEN (T029a PARTIAL). Existing T029i/M live chain remains separate |
|
||||
| Parity | REST/MCP identical validation/errors/CAS; no consume/publish fallback success | T045/T046 **CLOSED 2026-09-11** (gated `publish_baseline_catalog` + live pin-from-Gitea canary v4); T044 offline parity green + live canaries done, remaining: dedicated REST-vs-MCP error-shape parity fixtures |
|
||||
| Frontend | No agent chat/prompt/workspace/start/handoff | Chat service removed; remaining agent-panel/handoff drift is T030 OPEN |
|
||||
|
||||
@@ -81,6 +81,23 @@ npm run test -- --run src/lib/models/__tests__/AdminMcpCatalogModel.test.ts
|
||||
8. `start_scenario_run` with explicit promoted `revision_id` + verified `content_hash`.
|
||||
9. Human gates/checkpoints via `list_pending_approvals`/`decide_approval` and `list_checkpoints`/`decide_checkpoint` (USER principal only).
|
||||
|
||||
### T029a profile preview (implemented subset; not save authorization)
|
||||
|
||||
1. Call `propose_test_pack_profile` with `environment_id`, `dashboard_id`,
|
||||
`objective`, and non-empty catalog `selected_case_ids`.
|
||||
2. Review per-case coverage, server-derived chart→dataset→metric coordinates,
|
||||
applicable native filters, and typed unresolved questions.
|
||||
3. Call `resolve_test_pack_profile` with the current `profile_digest` to answer
|
||||
an exact-step selector question or choose one suggested metric coordinate.
|
||||
The server re-inspects the dashboard before applying the resolution.
|
||||
4. A metric choice becomes `needs_baseline`; continue through the 037 capture,
|
||||
candidate review, approval and publication lifecycle. Do not supply metric
|
||||
values or treat coordinate selection as a baseline pin.
|
||||
5. The handlers return bounded previews, not graph bytes or handles, and do not
|
||||
make the scenario eligible for `register_draft_pack` or bootstrap.
|
||||
Domain-context/published-baseline resolution, eligible-handle minting and a
|
||||
fresh external-client profile→bootstrap E2E remain open under T029a.
|
||||
|
||||
Product UI never starts this chain. Manual editor (043) and monitor (045) are equal renderers of the same registry/run/gate rows.
|
||||
|
||||
## Exit gates (acceptance OPEN)
|
||||
@@ -89,6 +106,7 @@ Product UI never starts this chain. Manual editor (043) and monitor (045) are eq
|
||||
- Stale CAS and changed-intent replay create no partial state.
|
||||
- Failed consume/publication returns typed error/pending; no legacy fallback success.
|
||||
- Full `BaselineSelectionPin` survives rerun/rebaseline/result/analytics.
|
||||
- T029a full profile→typed resolutions→save-eligible handle→bootstrap external-client E2E remains OPEN.
|
||||
- HumanStep schedules rejected equally over REST/MCP before any durable side effect.
|
||||
- No agent frontend controls, routes, or requests (T030; negative DOM/route/network still OPEN).
|
||||
- Live browser/capture canary PASSED 2026-09-10 against ss-prod (binding `ss-prod-d11-live-001`, run `597274d3`: browser `open_dashboard` with checkpoint/page-URL/PNG digest + 8 durable screenshot refs, capacity leases; trace: `docs/reports/agentic-runtime-live-canary-2026-09-10.md`). Live `agent_evaluation` canary v2 PASSED 2026-09-10 through REST-only start with server-side binding resolution (run `4eebfab3`: `AgentEvaluation` persisted, DecisionPolicy row 11; trace: `docs/reports/agentic-runtime-live-canary-v2-2026-09-10.md`). **Correction 2026-09-25:** Omniroute `deepseek-flash` image input was verified with a live PNG (`image_url`, HTTP 200 / `IMAGE_CANARY_OK`, routed model `gpt-5.6-sol`); image attachment capability is no longer an open item. Still OPEN: full scenario terminal PASS + scheduled zero-human proof, provider-version/cost receipts, cancellation receipts.
|
||||
|
||||
@@ -18,7 +18,7 @@
|
||||
@SEMANTICS: spec, requirements, mcp, interface, tools, catalog, approval, provenance, decommission, gradio
|
||||
|
||||
**Feature Branch**: `050-mcp-interface`
|
||||
**Created**: 2026-08-24 | **Status**: Partially implemented; production acceptance OPEN (refresh 2026-09-08)
|
||||
**Created**: 2026-08-24 | **Status**: Partially implemented; production acceptance OPEN (refresh 2026-09-29)
|
||||
**Input**: "Убрать веб-интерфейс агента (Gradio, кнопка «Ассистент» и другие точки входа) и заменить его простым MCP-интерфейсом к ss-tools, через который внешние клиенты создают все тесты. Полноценный внешний доступ; первый срез — паритет текущих 37 инструментов; продуктовый frontend не содержит агентских взаимодействий; агент работает только во внешнем MCP-клиенте."
|
||||
|
||||
## Decisions (session 2026-08-24)
|
||||
@@ -93,7 +93,7 @@ HumanCheckpoint disposition (`waiting_human` runs, 044) is also completable insi
|
||||
|
||||
**Why P1**: "Создавать все тесты через MCP" is the core value replacing the chat flow.
|
||||
|
||||
**Independent Test**: From an external client, drive inspect → compile → validate → resolve → draft-pack → save for a fixture dashboard and verify the saved immutable revision appears in the 042 registry.
|
||||
**Independent Test**: From an external client, inspect → propose a typed test-pack profile → resolve declared questions → compile/validate → register eligible draft-pack handles → bootstrap/save for a fixture dashboard and verify the immutable revision in the 042 registry. The profile-preview and selector/metric-choice steps are implemented; the complete profile-resolution-to-save chain remains OPEN under T029a.
|
||||
|
||||
**Acceptance**:
|
||||
1. **Given** a dashboard id and environment **When** the client calls the inspection and scenario-authoring tools **Then** the chain produces a validated graph with `needs_context`/`needs_selector` markers instead of invented data (038 semantics).
|
||||
|
||||
@@ -44,7 +44,7 @@
|
||||
## Phase 2b — Initial scenario and automation parity
|
||||
|
||||
- [x] T029 Implement `ScenarioRegistry.CreateInitial` and `bootstrap_authoring_scenario`; prove a fresh external MCP client creates a first registry scenario without a seeded base revision. Evidence: `tests/test_mcp_initial_scenario_e2e.py` creates real `AgentRun`/`DraftArtifact` rows and drives `bootstrap_authoring_scenario` without a seeded registry; replay and conflicting idempotency are covered by `tests/test_mcp_t029_bootstrap_automation.py`.
|
||||
- [ ] T029a Add typed InitialScenarioIntent/TestPackProfile validation and server-owned metric/filter/selector/baseline bindings; unresolved requirements must be `preview_only`, never guessed.
|
||||
- [~] T029a Partial: strict profile preview derives from fresh server inspection and returns case coverage, deterministic metric coordinates and typed questions; runtime catalog 2.6.0 exposes `propose_test_pack_profile` and `resolve_test_pack_profile` as authenticated-human read previews (service principals denied). Digest-CAS accepts exact-step selector hints and server-issued metric coordinate choices, re-inspects/recompiles, and transitions chosen metric to `needs_baseline` without expected values or a pin. Bootstrap checks selected environment and exact case coverage against digest-bound graph handles before writes; compiler selector authority is per-step. Evidence: `test_test_pack_profile.py`, `test_compiler.py`, `test_handles.py`, `test_mcp_t029_bootstrap_automation.py`, profile MCP E2E in `test_mcp_scenario_e2e.py`; 34 focused tests, Ruff and compileall pass. Still open: durable/idempotent domain-context and published-baseline resolution, profile-bound save-eligible handle minting, and fresh external-client profile→resolve→register→bootstrap E2E. Do not mark complete until those vectors pass.
|
||||
- [x] T029b Expose 046 schedule, trigger-rule, policy and automation-metrics management through curated MCP tools with REST-equivalent RBAC, idempotency and PROD gate behavior. Evidence: curated tools and catalog entries in `src/mcp_server/tools_automation.py`/`rbac_server.py`, idempotency migration `alembic/versions/0018_automation_idempotency.py`, and `tests/test_mcp_t029_bootstrap_automation.py` plus RBAC catalog tests.
|
||||
- [x] T029c Add fresh-DB MCP E2E for bootstrap → visible registry entry → revision activation → manual run, plus scheduler eligibility/PROD-gate integration evidence. Evidence: `tests/test_mcp_initial_scenario_e2e.py` verifies fresh bootstrap, current revision, refusal to run an un-promoted bootstrap revision (`BOOTSTRAP_REVISION_NOT_RUNNABLE`), a pinned manual run against a genuinely materialized revision, human-step schedule rejection with zero side effects, and an idempotent PROD approval gate; focused MCP regression set: `145 passed` (2026-09-06).
|
||||
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
# Traceability — 050-mcp-interface
|
||||
|
||||
## Production acceptance traceability — 2026-09-08
|
||||
## Production acceptance traceability — 2026-09-29
|
||||
|
||||
Historical rows above identify prior tests/code only; removed agent UI paths are retired. The following audited gates are **implemented=false / OPEN**, independent of local suite totals.
|
||||
Historical rows above identify prior tests/code only; removed agent UI paths are retired. The following audited gates remain partial/open independent of local suite totals.
|
||||
|
||||
| Requirement | Domain contract / DTO | Task | Falsifiable acceptance | State |
|
||||
|---|---|---|---|---|
|
||||
@@ -15,6 +15,7 @@ Historical rows above identify prior tests/code only; removed agent UI paths are
|
||||
| MCPX-FR-033 | Exploration step | [T050](tasks.md) | sandbox read-only interactive step with CAS/receipts and no second Playwright stack. | OPEN. |
|
||||
| MCPX-FR-034 | Prompt dangerous-content profile | [T051](tasks.md) | REST/MCP validation/error parity plus injection negative tests. | PARTIAL 2026-09-22: parity matrix CLOSED (`test_mcp_rest_error_parity.py`); LLM evidence-injection negative gate (`LLM-INJ-001`) still OPEN. |
|
||||
| MCPX-FR-035 | [Reference URL invariant](../037-superset-baseline-engine/contracts/reference-url.md), [tool schema](contracts/tool-contracts.schema.json) | [T052](tasks.md) | Agent can request same-dashboard URL; MCP and REST resolve identical filter state and coordinate; neither accepts caller values as authority; published pin survives scheduled replay without resolving key again. | PARTIAL 2026-09-25: `preview_reference_dashboard` and `capture_reference_selection` MCP tools use the guarded REST handlers and human RBAC; the supplied ss-dev dashboard 10 URL has live parser/query proof (`platform=3DS`, 10/10 metrics), and isolated Chromium proved REST editor URL → six drafts plus mismatch rejection. MCP tool offline tests exist; live MCP tool invocation, an agent asking for the URL, real publication/pin and scheduled replay remain open. |
|
||||
| MCPX-FR-036 / T029a | [038 ServerOwnedPipeline](../038-dashboard-scenario-model/contracts/modules.md); typed MCP profile inputs | [T029a](tasks.md) | Fresh dashboard inspection → complete selected-case profile/metric coordinates → digest-CAS typed selector/coordinate choice → eligible registered handle → atomic bootstrap; forged/stale values create zero side effects. | PARTIAL 2026-09-29: runtime catalog 2.6.0 registers both profile tools as authenticated-human read previews; 34 scoped tests, Ruff and compileall green. Selector is per-step and metric choice advances to `needs_baseline`. Domain context/published-baseline resolution, save-eligible handle minting and fresh external-client DB E2E remain open. |
|
||||
|
||||
Sources: [production gap](../../docs/reports/ss-prod-agentic-e2e-production-gap-2026-09-08.md), [coverage gap](../../docs/reports/ss-prod-agentic-e2e-spec-coverage-2026-09-08.md), [baseline gap](../../docs/reports/ss-prod-agentic-e2e-baseline-gap-2026-09-08.md). Spec schema/static checks prove contract structure only; live canary/runtime closure and optional approved performance baseline are not claimed.
|
||||
|
||||
|
||||
@@ -50,6 +50,7 @@
|
||||
| `G-INVESTIGATION-UI` | P1 | 047 T024/T025/T027 | queued+in_case findability, case index, linked runs/evidence/i18n, browser queue→case→disposition. | `e7925d0e`,`c9e6e642`,`9acd9bad` presentation slice: grouped queue, business summary, evidence cards, collapsed technical refs; 13 focused tests. **2026-09-24 uncommitted follow-up:** deterministic isolated seed + Chromium queue→correct case→linked run/evidence→resolved→stale CAS, 1 passed/0 skipped; trace/screenshot retained in `frontend/test-results/`; UI POST fix with 10 focused tests, lint/build green. | CLOSED for scoped UI/E2E implementation; release-commit regression remains `G-FULL-REGRESSION` |
|
||||
| `G-INVESTIGATION-ATOMIC` | P0 | 047 T019–T021/T028 | PostgreSQL concurrency: one active episode/case; stale CAS rollback; exact-context comparability; immutable evidence. | 2026-09-22: `analytics/locking.py` (pg_advisory_xact_lock per fingerprint) + `set_disposition` savepoint (zero partial projection); `analytics/context.py` exact-context comparability with typed mismatch/ineligible; 22 unit + 7 Testcontainers PG16 race vectors (`tests/integration/test_scenario_investigation_concurrency.py`). | CLOSED |
|
||||
| `G-MCP-PARITY` | P0 | 050 T044/T047/T048/T051 | REST/MCP error/auth/lifecycle parity, catalog/visibility mirrors, service cannot decide human gates; all prerequisites externally reachable. | 2026-09-22: shared `classify_start_error` consumed by REST and MCP removes the transport-dependent error vocabulary; `tests/test_mcp_rest_error_parity.py` 17-vector matrix (env/baseline codes/idempotency/conflict/disabled action/disabled automation/investigation RBAC/unauthenticated + service-principal human-gate denial + MCP no-bypass on case ACL). Live external chain proven earlier. | CLOSED |
|
||||
| `G-MCP-TEST-PACK-PROFILE` | P1 | 050 T029a; 038 ServerOwnedPipeline | External MCP profile preview derives server-owned case coverage/metric coordinates/filter applicability; digest-CAS selector and coordinate choices re-inspect context; unresolved context/baseline never mint save handles or become runnable. | 2026-09-29: runtime catalog 2.6.0 registers both tools as authenticated-human read previews; 34 focused tests and Ruff/compileall pass. Coordinate choice advances to `needs_baseline` without expected values. Domain context/published-baseline resolution, eligible-handle minting and fresh external profile→bootstrap E2E remain open. | PARTIAL / NO-GO |
|
||||
| `G-MCP-BASELINE-SELECTION` | P1 | 050 T049 | Agent proposes bounded curated baseline scope; operator review handle; server capture; observatory non-gating. | Server-side facts-only selection exists (`services/dashboard_testing/baseline_selection.py` + T087 tests) but no MCP tool; the operator review UI is contract-declared and **not implemented** (the service has no HTTP/MCP surface to bind). | OPEN (blocked on the MCP/HTTP surface) |
|
||||
| `G-MCP-EXPLORATION` | P1 | 050 T050 | Read-only exploration_step in isolated workspace with CAS/receipts; reuses 044 session; no raw code or second Playwright stack. | Fixed one-shot exploration exists; interactive operation absent. | OPEN |
|
||||
| `G-STORAGE-BUDGET` | P1 | 044/046 tiering | Measured hot/cold artifact volume and quota alarms meet the approved retention budget for 20-table dashboards. | Estimates only; no telemetry-based acceptance. | OPEN |
|
||||
|
||||
@@ -3,6 +3,24 @@
|
||||
> Статический + targeted-runtime checkpoint. Не является feature-completion.
|
||||
> Правило: `[x]` только с текущим доказательством; `[~]` частичная реализация; `[ ]` отсутствует.
|
||||
|
||||
## T029a MCP profile update — 2026-09-29
|
||||
|
||||
External MCP scenario authoring now has bounded profile-preview handlers in
|
||||
runtime catalog 2.6.0 with `dashboard:testing:READ`. The server re-inspects the selected dashboard and
|
||||
environment, returns complete selected-case coverage, stable chart→dataset→metric
|
||||
coordinates and native-filter applicability, and keeps unresolved questions
|
||||
typed. Selector answers are restricted to one exact compiler step and guarded by
|
||||
profile-digest CAS. Metric-coordinate selection changes `needs_metric` to
|
||||
`needs_baseline`; it never supplies expected numeric truth or a baseline pin.
|
||||
|
||||
Bootstrap now checks selected environment and case coverage against digest-bound
|
||||
server handles before any registry/revision/workspace write. Targeted evidence:
|
||||
32 focused profile/compiler/handle/bootstrap/MCP tests passed; Ruff, compileall
|
||||
and diff checks passed. This is **PARTIAL**, not production closure. Domain
|
||||
context and published-baseline resolution, save-eligible handle minting, full
|
||||
fresh external-client profile→bootstrap E2E and catalog/schema parity remain open.
|
||||
Overall production status remains **NO-GO**.
|
||||
|
||||
## Current UX handoff
|
||||
|
||||
2026-09-24 architecture amendment: every new baseline is defined by an explicit same-dashboard Superset URL and its server-parsed immutable native-filter snapshot ([037 ReferenceUrl](037-superset-baseline-engine/contracts/reference-url.md)). This is a P0 production invariant, tracked as `G-BSL-REFERENCE-URL` in the unified matrix, 037 T091 and 050 T052. Both supplied ss-dev dashboard 10 URLs have live read-only parser/query evidence; the `native_filters_key=A4_rwhfQta0` link was rerun on 2026-09-25 with `platform=3DS`, 10/10 metrics executed and 6 distinct metric formulas suggested. A separate, removed-after-test Compose project with a dashboard 10 scenario and **test-only synthetic release** then proved browser URL entry → filter/metric preview → six draft candidates (Chromium 2/2, including a different-dashboard field error). Its isolated DB retained the exact URL hash and `platform IN ["3DS"]` snapshot across six captures. The ordinary local database still has no dashboard 10 scenario/release fixture; a later isolated Chromium pass proved source-byte verification, reasoned approval and local catalog materialization for one URL candidate, with the exact URL/filter snapshot retained in its revision. Git publication has no receipt; real release→Git publication→run pin→scheduled replay remains unproven. Do not infer the full invariant or analyst acceptance from the seeded 4/5 review score.
|
||||
|
||||
Reference in New Issue
Block a user