feat(mcp): add typed test-pack profile preview

This commit is contained in:
2026-09-29 14:15:17 +03:00
parent e99306eb54
commit f73169b371
28 changed files with 1272 additions and 45 deletions

View File

@@ -93,7 +93,7 @@ def _scenario_start_permission(arguments: dict[str, Any]) -> tuple[str, str]:
# tool listed with deprecated=True for one minor cycle; additive changes bump MINOR.
# The discipline is pinned executable by tests/test_mcp_catalog_version.py (the pinned
# major in that test is the deliberate-bump ritual — it cannot change by accident).
MCP_CATALOG_VERSION = "2.5.0"
MCP_CATALOG_VERSION = "2.6.0"
# #endregion McpServer.CatalogVersion
@@ -169,6 +169,8 @@ _MCP_CATALOG = (
McpToolDefinition("record_case_note", ("scenario:result", "TRIAGE"), service_allowed=False),
McpToolDefinition("propose_case_disposition", ("scenario:result", "TRIAGE"), service_allowed=False),
McpToolDefinition("inspect_dashboard_context", None),
McpToolDefinition("propose_test_pack_profile", None, service_allowed=False),
McpToolDefinition("resolve_test_pack_profile", None, service_allowed=False),
McpToolDefinition("inspect_scenario", None),
McpToolDefinition("validate_scenario", None),
McpToolDefinition("scenario_resolve", None),

View File

@@ -15,7 +15,7 @@ from __future__ import annotations
import json
from typing import Any
from pydantic import BaseModel, ConfigDict, Field, StrictInt, StrictStr, field_validator
from pydantic import BaseModel, ConfigDict, Field, StrictInt, StrictStr, field_validator, model_validator
from src.services.dashboard_testing.scenario.models import DashboardTestScenario
@@ -29,11 +29,40 @@ class InitialScenarioIntent(BaseModel):
title: StrictStr = Field(min_length=1, max_length=200)
dashboard_id: StrictInt = Field(ge=1)
allowed_environment_ids: list[StrictStr] = Field(min_length=1, max_length=20)
selected_case_ids: list[StrictStr] = Field(default_factory=list, max_length=50)
selected_environment_id: StrictStr = Field(min_length=1, max_length=128)
selected_case_ids: list[StrictStr] = Field(min_length=1, max_length=19)
objective: StrictStr = Field(min_length=1, max_length=2000)
compiled_handle_id: StrictStr = Field(min_length=1, max_length=255)
draft_pack_id: StrictStr = Field(min_length=1, max_length=128)
draft_pack_digest: StrictStr = Field(pattern=r"^[a-fA-F0-9]{64}$")
# #region McpServer.InitialScenarioIntent.UniqueCases [C:2] [TYPE Function] [SEMANTICS mcp,bootstrap,case-selection]
@field_validator("selected_case_ids")
@classmethod
def _unique_selected_cases(cls, values: list[str]) -> list[str]:
if len(values) != len(set(values)):
raise ValueError("DUPLICATE_CASE_ID")
return values
# #endregion McpServer.InitialScenarioIntent.UniqueCases
# #region McpServer.InitialScenarioIntent.UniqueEnvironments [C:2] [TYPE Function] [SEMANTICS mcp,bootstrap,environment]
@field_validator("allowed_environment_ids")
@classmethod
def _unique_environments(cls, values: list[str]) -> list[str]:
if len(values) != len(set(values)):
raise ValueError("DUPLICATE_ENVIRONMENT_ID")
return values
# #endregion McpServer.InitialScenarioIntent.UniqueEnvironments
# #region McpServer.InitialScenarioIntent.SelectedEnvironment [C:2] [TYPE Function] [SEMANTICS mcp,bootstrap,environment]
@model_validator(mode="after")
def _selected_environment_allowed(self) -> InitialScenarioIntent:
if self.selected_environment_id not in self.allowed_environment_ids:
raise ValueError("SELECTED_ENVIRONMENT_NOT_ALLOWED")
return self
# #endregion McpServer.InitialScenarioIntent.SelectedEnvironment
# #endregion McpServer.InitialScenarioIntent
@@ -305,6 +334,70 @@ class InspectContextInput(BaseModel):
# #endregion McpServer.InspectContextInput
# #region McpServer.TestPackProfileInput [C:2] [TYPE Model] [SEMANTICS mcp,scenario,profile,typed]
# @ingroup McpServer
class TestPackProfileInput(BaseModel):
model_config = ConfigDict(extra="forbid", strict=True)
environment_id: StrictStr = Field(min_length=1, max_length=128)
dashboard_id: StrictInt = Field(ge=1)
objective: StrictStr = Field(min_length=1, max_length=2000)
selected_case_ids: list[StrictStr] = Field(min_length=1, max_length=19)
@field_validator("selected_case_ids")
@classmethod
def _unique_profile_cases(cls, values: list[str]) -> list[str]:
if len(values) != len(set(values)):
raise ValueError("DUPLICATE_CASE_ID")
return values
# #endregion McpServer.TestPackProfileInput
# #region McpServer.ResolveTestPackProfileInput [C:3] [TYPE Module] [SEMANTICS mcp,profile,resolution,cas]
# @ingroup McpServer
# #region McpServer.TestPackProfileResolution [C:2] [TYPE Model] [SEMANTICS mcp,profile,resolution,typed]
class TestPackProfileResolution(BaseModel):
model_config = ConfigDict(extra="forbid", strict=True)
unresolved_id: StrictStr = Field(min_length=1, max_length=200)
step_id: StrictStr | None = Field(default=None, min_length=1, max_length=160)
selector_hint: StrictStr | None = Field(default=None, min_length=1, max_length=256)
coordinate_id: StrictStr | None = Field(default=None, pattern=r"^[a-f0-9]{32}$")
reason: StrictStr = Field(min_length=1, max_length=500)
@model_validator(mode="after")
def _exactly_one_resolution_value(self) -> TestPackProfileResolution:
if (self.selector_hint is None) == (self.coordinate_id is None):
raise ValueError("EXACTLY_ONE_PROFILE_RESOLUTION_REQUIRED")
if self.selector_hint is not None and self.step_id is None:
raise ValueError("SELECTOR_STEP_ID_REQUIRED")
if self.coordinate_id is not None and self.step_id is not None:
raise ValueError("COORDINATE_STEP_ID_FORBIDDEN")
return self
# #endregion McpServer.TestPackProfileResolution
# #region McpServer.ResolveTestPackProfileRequest [C:2] [TYPE Model] [SEMANTICS mcp,profile,resolution,cas]
class ResolveTestPackProfileInput(BaseModel):
model_config = ConfigDict(extra="forbid", strict=True)
environment_id: StrictStr = Field(min_length=1, max_length=128)
dashboard_id: StrictInt = Field(ge=1)
objective: StrictStr = Field(min_length=1, max_length=2000)
selected_case_ids: list[StrictStr] = Field(min_length=1, max_length=19)
expected_profile_digest: StrictStr = Field(pattern=r"^[a-f0-9]{64}$")
resolutions: list[TestPackProfileResolution] = Field(min_length=1, max_length=100)
@field_validator("selected_case_ids")
@classmethod
def _unique_resolution_cases(cls, values: list[str]) -> list[str]:
if len(values) != len(set(values)):
raise ValueError("DUPLICATE_CASE_ID")
return values
# #endregion McpServer.ResolveTestPackProfileRequest
# #endregion McpServer.ResolveTestPackProfileInput
# #endregion McpServer.ScenarioModels

View File

@@ -71,6 +71,7 @@ from src.mcp_server.scenario_inputs import (
ScenarioResolveChangeInput,
ScenarioResolveInput,
ScenarioStartInput,
TestPackProfileInput,
TestPlanContentInput,
TestPlanInput,
)

View File

@@ -17,6 +17,7 @@
from __future__ import annotations
import re
from typing import Any
from mcp.server.fastmcp import Context
@@ -32,9 +33,11 @@ from src.mcp_server.scenario_inputs import (
DraftPackInput,
InspectContextInput,
RegisterDraftPackInput,
ResolveTestPackProfileInput,
ScenarioCompileInput,
ScenarioResolveInput,
ScenarioStartInput,
TestPackProfileInput,
)
from src.models.maintenance import MaintenanceEvent
from src.models.agent_run import AgentRun
@@ -58,6 +61,7 @@ from src.services.dashboard_testing.scenario.handles import (
)
from src.services.dashboard_testing.scenario.resolver import ResolveChange, resolve_scenario
from src.services.dashboard_testing.scenario.validator import validate_scenario
from src.services.dashboard_testing.scenario.test_pack_profile import apply_coordinate_choices, build_test_pack_profile
from src.services.dashboard_testing.baseline_staleness import (
BaselinePeriodStale,
emit_launch_period_stale_receipt,
@@ -69,6 +73,98 @@ from src.services.llm_provider import LLMProviderService
from src.services.mcp_approvals import decide_mcp_approval, list_pending_mcp_approvals
# #region McpServer.TestPackProfile.ResolutionHelpers [C:3] [TYPE Module] [SEMANTICS mcp,profile,resolution,validation]
# @ingroup McpServer
# @BRIEF Validate typed profile resolutions against current unresolved profile items and build server compiler inputs.
# #region McpServer.TestPackProfile.SelectorChanges [C:3] [TYPE Function] [SEMANTICS mcp,profile,selector,cas]
# @ingroup McpServer.TestPackProfile.ResolutionHelpers
# @BRIEF Map analyst-selected unresolved IDs to current compiler step IDs; reject stale or forged targets.
def _selector_profile_changes(profile, resolutions, scenario):
unresolved = {item.id: item for item in profile.unresolved if item.kind == "needs_selector"}
steps = {step.id: step for step in scenario.steps}
changes: list[dict[str, str]] = []
seen: set[str] = set()
for resolution in resolutions:
item = unresolved.get(resolution.unresolved_id)
if (item is None or item.kind != "needs_selector" or resolution.selector_hint is None
or resolution.coordinate_id is not None or resolution.unresolved_id in seen):
return None
case_id = item.id.split(":", 2)[1]
step = steps.get(resolution.step_id)
if (step is None or step.automation_status != "needs_selector"
or item.step_id != resolution.step_id
or case_id not in step.checklist_case_ids or step.risk != "browser_interaction"
or step.action not in {"apply_native_filter", "apply_table_filter", "click", "select"}
or not _valid_selector_hint(resolution.selector_hint)):
return None
changes.append({"step_id": step.id, "hint": resolution.selector_hint})
seen.add(resolution.unresolved_id)
return changes if seen else None
# #endregion McpServer.TestPackProfile.SelectorChanges
# #region McpServer.TestPackProfile.CoordinateChoices [C:3] [TYPE Function] [SEMANTICS mcp,profile,metric,baseline]
# @ingroup McpServer.TestPackProfile.ResolutionHelpers
# @BRIEF Validate chosen server-issued metric coordinates against current needs_baseline items.
def _coordinate_profile_choices(profile, resolutions):
unresolved = {item.id: item for item in profile.unresolved
if item.kind == "needs_metric" or (item.kind == "needs_baseline" and item.coordinate_ids)}
coordinates = {item.coordinate_id: item for item in profile.coordinates}
choices = []
seen: set[str] = set()
for resolution in resolutions:
item = unresolved.get(resolution.unresolved_id)
coordinate = coordinates.get(resolution.coordinate_id)
if (item is None or coordinate is None or resolution.coordinate_id not in item.coordinate_ids
or resolution.selector_hint is not None or resolution.unresolved_id in seen
or resolution.coordinate_id is None):
return None
choices.append({"unresolved_id": item.id, "coordinate_id": coordinate.coordinate_id})
seen.add(item.id)
return choices if choices else None
# #endregion McpServer.TestPackProfile.CoordinateChoices
# #region McpServer.TestPackProfile.InvalidResolution [C:2] [TYPE Function] [SEMANTICS mcp,profile,resolution,validation]
# @ingroup McpServer.TestPackProfile.ResolutionHelpers
# @BRIEF Enforce that every currently required question has the matching typed resolution in this request.
def _invalid_profile_resolution(has_selectors, has_metrics, selector_resolutions, selector_changes,
coordinate_resolutions, coordinate_choices):
return (
(selector_resolutions and selector_changes is None)
or (coordinate_resolutions and coordinate_choices is None)
or (not has_selectors and not has_metrics and not selector_resolutions and not coordinate_resolutions)
)
# #endregion McpServer.TestPackProfile.InvalidResolution
# #region McpServer.TestPackProfile.SelectorHint [C:2] [TYPE Function] [SEMANTICS mcp,profile,selector,safety]
# @ingroup McpServer.TestPackProfile.ResolutionHelpers
# @BRIEF Accept only bounded CSS selectors or explicit accessible-role hints; reject executable locator syntax.
def _valid_selector_hint(value: str) -> bool:
hint = value.strip()
return bool(hint and len(hint) <= 256 and re.fullmatch(
r"(?:#[A-Za-z_][\w-]*|\.[A-Za-z_][\w-]*|\[[A-Za-z_:][-\w:.]*(?:=['\"][^\]<>]{1,100}['\"])?\]|(?:role|text|label|placeholder):[\w -]{1,100})",
hint,
))
# #endregion McpServer.TestPackProfile.SelectorHint
# #region McpServer.TestPackProfile.SelectorParameters [C:2] [TYPE Function] [SEMANTICS mcp,profile,selector,compiler]
# @ingroup McpServer.TestPackProfile.ResolutionHelpers
# @BRIEF Build only compiler-owned selector_hint parameter declarations from validated resolutions.
def _selector_parameters(changes: list[dict[str, str]]) -> dict[str, Any]:
return {f"selector_{index}": {
"type": "selector_hint", "value": item["hint"], "status": "resolved",
"affected_step_ids": [item["step_id"]],
"required": False, "source": "user",
} for index, item in enumerate(changes, start=1)}
# #endregion McpServer.TestPackProfile.SelectorParameters
# #endregion McpServer.TestPackProfile.ResolutionHelpers
# #region McpServer.ToolsScenario.RegisterProbeRead [C:4] [TYPE Function] [SEMANTICS mcp,probe,tools,registration,read]
# @ingroup McpServer
# @BRIEF Registration seam: the six read-only probe tools (environments/health/dashboards/llm/task).
@@ -255,6 +351,121 @@ def register_scenario_tools(server) -> None:
}
# #endregion McpServer.ScenarioTools.InspectDashboardContext
# #region McpServer.ScenarioTools.ProposeTestPackProfile [C:4] [TYPE Function] [SEMANTICS mcp,scenario,profile,preview]
# @ingroup McpServer
# @BRIEF Build a typed test-pack proposal from a fresh server inspection and expose unresolved questions.
# @PRE Environment access is authorized and selected case IDs exist in the pinned checklist catalog.
# @POST Returns complete per-case coverage; only a fully resolvable profile may be save_eligible.
# @SIDE_EFFECT Performs bounded upstream Superset reads and emits profile/compile molecular-CoT events.
# @INVARIANT Caller-supplied capabilities, query models and expected values are not accepted.
# @REJECTED Reusing inspect_scenario's caller-carried query model was rejected — T029h verifies such
# claims only at registration; this proposal must classify from fresh server inspection.
# #region McpServer.ScenarioTools.ProposeTestPackProfile.Call [C:4] [TYPE Function] [SEMANTICS mcp,scenario,profile,inspection]
@server.tool(name="propose_test_pack_profile", structured_output=True)
async def propose_test_pack_profile(request: TestPackProfileInput) -> dict[str, Any]:
environment = get_config_manager().get_environment(request.environment_id)
if environment is None:
return {"status": "blocked", "error": "ENV_NOT_FOUND"}
try:
client = await get_superset_client(environment)
query_model = await inspect_dashboard_query_model(client, request.environment_id, request.dashboard_id)
if not query_model.query_model_fingerprint or query_model.query_model_fingerprint == "sha256:error":
return {"status": "blocked", "error": "CONTEXT_INSPECTION_DEGRADED"}
if (query_model.environment_id != request.environment_id
or int(query_model.dashboard_id) != request.dashboard_id):
return {"status": "blocked", "error": "CONTEXT_IDENTITY_MISMATCH"}
profile, scenario, pack = build_test_pack_profile(
query_model=query_model, objective=request.objective,
selected_case_ids=request.selected_case_ids,
browser_available=resolve_browser_availability(),
)
except (KeyError, ValueError) as exc:
logger.explore("Test-pack profile proposal rejected", src="McpServer.ScenarioTools.ProposeTestPackProfile",
error_code=str(exc), payload={"dashboard_id": request.dashboard_id}, error=str(exc))
return {"status": "blocked", "error": str(exc)}
except Exception as exc:
logger.explore("Test-pack profile inspection failed", src="McpServer.ScenarioTools.ProposeTestPackProfile",
error_code="INSPECTION_FAILED", payload={"dashboard_id": request.dashboard_id}, error=str(exc)[:300])
return {"status": "blocked", "error": "INSPECTION_FAILED"}
return {
"status": profile.status,
"profile": profile.model_dump(mode="json"),
"preview": {
"step_count": len(scenario.steps),
"artifacts": pack.get("artifacts", []),
"validation": pack.get("validation_summary", {}),
},
}
# #endregion McpServer.ScenarioTools.ProposeTestPackProfile.Call
# #endregion McpServer.ScenarioTools.ProposeTestPackProfile
# #region McpServer.ScenarioTools.ResolveTestPackProfile [C:4] [TYPE Function] [SEMANTICS mcp,profile,resolve,cas]
# @ingroup McpServer
# @BRIEF Re-inspect context, CAS-check the profile digest, and apply only reviewed selector hints.
# @PRE Every resolution ID names a current needs_selector blocker in the server-rebuilt profile.
# @POST Returns a fresh deterministic preview; no handles or graph bytes are returned.
# @SIDE_EFFECT Performs bounded Superset reads and compiler logging; it writes no database state.
# @INVARIANT Unsupported targets, stale profile digests and caller graph/value claims fail closed.
# #region McpServer.ScenarioTools.ResolveTestPackProfile.Call [C:4] [TYPE Function] [SEMANTICS mcp,profile,resolve,inspection]
@server.tool(name="resolve_test_pack_profile", structured_output=True)
async def resolve_test_pack_profile(request: ResolveTestPackProfileInput) -> dict[str, Any]:
environment = get_config_manager().get_environment(request.environment_id)
if environment is None:
return {"status": "blocked", "error": "ENV_NOT_FOUND"}
try:
client = await get_superset_client(environment)
query_model = await inspect_dashboard_query_model(client, request.environment_id, request.dashboard_id)
if (not query_model.query_model_fingerprint or query_model.query_model_fingerprint == "sha256:error"
or query_model.environment_id != request.environment_id
or int(query_model.dashboard_id) != request.dashboard_id):
return {"status": "blocked", "error": "CONTEXT_IDENTITY_MISMATCH"}
baseline, scenario, _ = build_test_pack_profile(
query_model=query_model, objective=request.objective,
selected_case_ids=request.selected_case_ids,
browser_available=resolve_browser_availability(),
)
if baseline.profile_digest != request.expected_profile_digest:
return {"status": "conflict", "error": "PROFILE_STALE",
"current_profile_digest": baseline.profile_digest}
selector_resolutions = [item for item in request.resolutions if item.selector_hint is not None]
coordinate_resolutions = [item for item in request.resolutions if item.coordinate_id is not None]
has_selector_items = any(item.kind == "needs_selector" for item in baseline.unresolved)
selector_changes = _selector_profile_changes(baseline, selector_resolutions, scenario) if selector_resolutions else []
coordinate_choices = _coordinate_profile_choices(baseline, coordinate_resolutions) if coordinate_resolutions else []
has_metric_items = any(item.kind == "needs_metric" for item in baseline.unresolved)
invalid_resolution = _invalid_profile_resolution(
has_selector_items, has_metric_items, selector_resolutions,
selector_changes, coordinate_resolutions, coordinate_choices,
)
if invalid_resolution or (selector_resolutions and not selector_changes) or (coordinate_resolutions and not coordinate_choices):
return {"status": "blocked", "error": "PROFILE_RESOLUTION_INVALID"}
parameters = _selector_parameters(selector_changes or [])
profile, updated, pack = build_test_pack_profile(
query_model=query_model, objective=request.objective,
selected_case_ids=request.selected_case_ids, parameters=parameters,
browser_available=resolve_browser_availability(),
)
if coordinate_choices:
selected_coordinates = {item["unresolved_id"]: item["coordinate_id"] for item in coordinate_choices}
profile = apply_coordinate_choices(profile, selected_coordinates)
except (KeyError, ValueError) as exc:
logger.explore("Test-pack profile resolution rejected", src="McpServer.ScenarioTools.ResolveTestPackProfile",
error_code=str(exc), payload={"dashboard_id": request.dashboard_id}, error=str(exc))
return {"status": "blocked", "error": str(exc)}
except Exception as exc:
logger.explore("Test-pack profile resolution inspection failed", src="McpServer.ScenarioTools.ResolveTestPackProfile",
error_code="INSPECTION_FAILED", payload={"dashboard_id": request.dashboard_id}, error=str(exc)[:300])
return {"status": "blocked", "error": "INSPECTION_FAILED"}
return {
"status": profile.status, "profile": profile.model_dump(mode="json"),
"preview": {"step_count": len(updated.steps), "artifacts": pack.get("artifacts", []),
"validation": pack.get("validation_summary", {})},
}
# #endregion McpServer.ScenarioTools.ResolveTestPackProfile.Call
# #endregion McpServer.ScenarioTools.ResolveTestPackProfile
@server.tool(name="inspect_scenario", structured_output=True)
async def inspect_scenario(request: ScenarioCompileInput) -> dict[str, Any]:
"""Compile a scenario graph without registering or persisting it.

View File

@@ -102,6 +102,8 @@ def create_scenario(
user_id: str,
owner_username: str | None = None,
validate_compiled_handle: bool = False,
expected_environment_id: str | None = None,
expected_case_ids: list[str] | None = None,
) -> tuple[ScenarioRegistryEntry, ScenarioRevision]:
# New 038 handle path. Keep the legacy DraftArtifact path below during the
# migration window so existing 039 save callers remain compatible.
@@ -117,6 +119,8 @@ def create_scenario(
draft_pack_digest=draft_pack_digest,
owner_principal=user_id,
dashboard_id=int(stored_compiled.dashboard_id),
expected_environment_id=expected_environment_id,
expected_case_ids=expected_case_ids,
)
graph = {
**chain["graph"],
@@ -297,17 +301,18 @@ def create_initial(db: Session, *, intent, user_id: str, owner_username: str | N
draft_pack_id=intent.draft_pack_id,
draft_pack_digest=intent.draft_pack_digest,
user_id=user_id, owner_username=owner_username,
expected_environment_id=intent.selected_environment_id,
expected_case_ids=list(intent.selected_case_ids),
)
entry.name = intent.title
entry.description = intent.objective
entry.dashboard_id = intent.dashboard_id
entry.environment_ids = list(intent.allowed_environment_ids)
entry.environment_ids = [intent.selected_environment_id]
entry.current_revision_id = revision.revision_id
revision.activation_status = ScenarioActivationStatus.CURRENT
revision.activated_by = str(user_id)
revision.graph_snapshot = {
**revision.graph_snapshot,
"selected_case_ids": list(intent.selected_case_ids),
"objective": intent.objective,
}
db.flush()

View File

@@ -105,8 +105,12 @@ def _build_parameters(raw: dict[str, Any]) -> list[ScenarioParameter]:
required = spec.get("required", True)
default = spec.get("default")
source = spec.get("source", "user")
affected_step_ids = list(spec.get("affected_step_ids") or [])
status = spec.get("status")
else:
ptype, label, required, default, source = "string", name, True, spec, "user"
affected_step_ids = []
status = None
params.append(
ScenarioParameter(
name=name,
@@ -115,7 +119,8 @@ def _build_parameters(raw: dict[str, Any]) -> list[ScenarioParameter]:
required=required,
default=default,
source=source,
status="resolved" if default is not None else "unresolved",
affected_step_ids=affected_step_ids,
status=status or ("resolved" if default is not None else "unresolved"),
value=default,
)
)
@@ -203,7 +208,7 @@ def _compile_graph(req: CompileScenarioRequest) -> CompiledResult:
# Missing selector/context detection on browser-interaction steps without hints
for step in steps:
if step.tool == "browser" and step.risk == "browser_interaction" and not _has_selector(parameters):
if step.tool == "browser" and step.risk == "browser_interaction" and not _has_selector(parameters, step.id):
_apply_selector_check(step, parameters, blockers)
scenario = DashboardTestScenario(
@@ -295,7 +300,7 @@ def _build_step(
automation = "ready"
if action in {"download", "download_xlsx"} and not mapping.matched_capabilities:
automation = "needs_context"
if entry["risk"] == "browser_interaction" and not _has_selector(parameters):
if entry["risk"] == "browser_interaction" and not _has_selector(parameters, step_id):
automation = "needs_selector"
return ScenarioStep(
@@ -344,10 +349,10 @@ def _human_checkpoint_step(case_id: str, ordinal_counter: dict[str, int]) -> Sce
# #region ScenarioGraph.Compiler.HasSelector [C:1] [TYPE Function] [SEMANTICS scenario,selector]
# @ingroup ScenarioGraph
# @BRIEF Check whether a resolved selector_hint parameter exists.
# @POST Returns True when any selector_hint parameter is resolved.
def _has_selector(parameters: list[ScenarioParameter]) -> bool:
return any(p.type == "selector_hint" and p.status == "resolved" for p in parameters)
# @BRIEF Check whether a resolved selector_hint parameter is bound to the given step.
# @POST Returns True only when one resolved selector parameter lists step_id in affected_step_ids.
def _has_selector(parameters: list[ScenarioParameter], step_id: str) -> bool:
return any(p.type == "selector_hint" and p.status == "resolved" and step_id in p.affected_step_ids for p in parameters)
# #endregion ScenarioGraph.Compiler.HasSelector
@@ -356,7 +361,7 @@ def _has_selector(parameters: list[ScenarioParameter]) -> bool:
# @BRIEF Emit a NEEDS_SELECTOR blocker for browser steps lacking a resolved selector hint.
# @POST Appends a blocker Finding with recovery options when the step needs a selector.
def _apply_selector_check(step: ScenarioStep, parameters: list[ScenarioParameter], blockers: list[Finding]) -> None:
if step.automation_status == "needs_selector" and not _has_selector(parameters):
if step.automation_status == "needs_selector" and not _has_selector(parameters, step.id):
blockers.append(Finding(
code="NEEDS_SELECTOR",
severity="blocker",

View File

@@ -279,6 +279,8 @@ def verify_handle_chain(
draft_pack_digest: str,
owner_principal: str,
dashboard_id: int,
expected_environment_id: str | None = None,
expected_case_ids: list[str] | None = None,
) -> dict[str, Any]:
# SELECT ... FOR UPDATE serializes concurrent consumers on the handle rows, so two
# create/save transactions can never both observe an unconsumed handle and double-consume.
@@ -340,6 +342,7 @@ def verify_handle_chain(
# required model identity from the verified content hash before parsing.
graph_payload["revision_hash"] = compiled.content_hash
graph = DashboardTestScenario.model_validate(graph_payload).model_dump()
_verify_profile_binding(graph, expected_environment_id, expected_case_ids)
logger.reflect("Handle chain verified", src="ScenarioGraph.Handles.VerifyChain",
claim="POST: authority comes from stored rows plus digest-verified bytes",
payload={"compiled": compiled.handle_id[:12], "pack": pack.draft_pack_id[:12], "content_hash": compiled.content_hash[:12]})
@@ -347,6 +350,25 @@ def verify_handle_chain(
# #endregion ScenarioGraph.Handles.VerifyChain
# #region ScenarioGraph.Handles.VerifyProfileBinding [C:3] [TYPE Function] [SEMANTICS scenario,handles,profile,binding]
# @ingroup ScenarioGraph.Handles
# @BRIEF Reject bootstrap intent that does not match the digest-verified compiled environment and case coverage.
# @PRE graph came from VerifyChain canonical bytes; optional values came from a strict bootstrap intent.
# @POST Matching environment/cases are accepted; mismatch raises before handle consumption or registry writes.
def _verify_profile_binding(graph: dict[str, Any], environment_id: str | None, case_ids: list[str] | None) -> None:
context = graph.get("dashboard_context") or {}
if environment_id is not None and str(context.get("environment_id")) != str(environment_id):
raise ValueError("PROFILE_ENVIRONMENT_MISMATCH")
if case_ids is not None:
selected = sorted(set(str(case_id) for case_id in case_ids))
graph_selected = sorted(set((graph.get("objective") or {}).get("selected_case_ids") or []))
covered = sorted({item.get("case_id") for item in graph.get("checklist_coverage") or []
if isinstance(item, dict) and item.get("case_id") in selected})
if graph_selected != selected or covered != selected:
raise ValueError("PROFILE_CASE_COVERAGE_MISMATCH")
# #endregion ScenarioGraph.Handles.VerifyProfileBinding
# #region ScenarioGraph.Handles.Consume [C:4] [TYPE Function] [SEMANTICS scenario,handles,consume,single]
# @ingroup ScenarioGraph
# @BRIEF Record single consumption of the compiled+pack pair by a created revision (same transaction).

View File

@@ -0,0 +1,352 @@
# #region ScenarioGraph.TestPackProfile [C:4] [TYPE Module] [SEMANTICS scenario,profile,checklist,coverage,eligibility]
# @ingroup ScenarioGraph
# @BRIEF Project the canonical compiler/validator/draft-pack result into a typed initial test-pack profile.
# @INVARIANT The profile is derived from a server-inspected query model and existing compiler services;
# it carries no expected metric values and does not mint persistence handles.
# @RATIONALE A profile projection adds complete, reviewable case coverage without creating a second
# graph compiler or mutable workflow alongside the 038 server-owned handle pipeline.
# @REJECTED A separate persisted profile engine was rejected for this slice — 038 compiled and draft-pack
# handles already bind canonical graph bytes, validation, owner and save eligibility.
from __future__ import annotations
from typing import Any, Literal
from pydantic import BaseModel, ConfigDict, Field
from src.core.logger import logger
from src.schemas.dashboard_testing.query_model import DashboardQueryModel
from src.services.dashboard_testing.scenario.capability_authority import derive_capabilities
from src.services.dashboard_testing.scenario.checklist_catalog import load_catalog
from src.services.dashboard_testing.scenario.compiler import CompileScenarioRequest, compile_scenario
from src.services.dashboard_testing.scenario.models import DashboardTestScenario, canonical_dump, sha256_hex
from src.services.dashboard_testing.scenario.pack_compiler import generate_draft_pack
from src.services.dashboard_testing.scenario.validator import validate_scenario
ProfileStatus = Literal["save_eligible", "preview_only"]
UnresolvedKind = Literal["needs_context", "needs_selector", "needs_metric", "needs_baseline", "unsupported"]
# #region ScenarioGraph.TestPackProfile.Case [C:1] [TYPE Model] [SEMANTICS scenario,profile,coverage]
class ProfileCase(BaseModel):
model_config = ConfigDict(extra="forbid", strict=True)
case_id: str = Field(pattern=r"^(B0[1-9]|C0[1-7]|T0[1-3])$")
classification: Literal["automated", "needs_context", "needs_selector", "needs_baseline", "unsupported", "blocked"]
step_ids: list[str] = Field(default_factory=list, max_length=100)
rationale: str = Field(max_length=1000)
# #endregion ScenarioGraph.TestPackProfile.Case
# #region ScenarioGraph.TestPackProfile.Unresolved [C:1] [TYPE Model] [SEMANTICS scenario,profile,unresolved]
class ProfileUnresolved(BaseModel):
model_config = ConfigDict(extra="forbid", strict=True)
id: str = Field(min_length=1, max_length=200)
kind: UnresolvedKind
json_pointer: str = Field(min_length=1, max_length=500)
step_id: str | None = Field(default=None, max_length=160)
coordinate_ids: list[str] = Field(default_factory=list, max_length=500)
question: str = Field(min_length=1, max_length=500)
allowed_resolution: Literal["provide_context", "provide_selector_hint", "select_metric", "select_baseline", "omit_case"]
# #endregion ScenarioGraph.TestPackProfile.Unresolved
# #region ScenarioGraph.TestPackProfile.Coordinate [C:2] [TYPE Model] [SEMANTICS scenario,profile,metric,filter]
class ProfileCoordinate(BaseModel):
model_config = ConfigDict(extra="forbid", strict=True)
coordinate_id: str = Field(pattern=r"^[a-f0-9]{32}$")
chart_id: int = Field(ge=1)
chart_name: str = Field(min_length=1, max_length=300)
dataset_id: int = Field(ge=1)
dataset_name: str = Field(min_length=1, max_length=300)
metric_name: str = Field(min_length=1, max_length=300)
metric_label: str = Field(min_length=1, max_length=300)
expression_type: Literal["SIMPLE", "SQL_EXPRESSION", "SAVED_METRIC"]
filter_ids: list[str] = Field(default_factory=list, max_length=100)
filter_names: list[str] = Field(default_factory=list, max_length=100)
# #endregion ScenarioGraph.TestPackProfile.Coordinate
# #region ScenarioGraph.TestPackProfile.Model [C:1] [TYPE Model] [SEMANTICS scenario,profile,eligibility]
class TestPackProfile(BaseModel):
model_config = ConfigDict(extra="forbid", strict=True)
profile_version: Literal[1] = 1
dashboard_id: int = Field(ge=1)
environment_id: str = Field(min_length=1, max_length=128)
query_model_fingerprint: str = Field(min_length=1, max_length=128)
checklist_catalog_version: int = Field(ge=1)
compiler_version: str = Field(min_length=1, max_length=32)
status: ProfileStatus
eligible: bool
selected_case_ids: list[str] = Field(min_length=1, max_length=19)
cases: list[ProfileCase] = Field(min_length=1, max_length=19)
coordinates: list[ProfileCoordinate] = Field(default_factory=list, max_length=500)
unresolved: list[ProfileUnresolved] = Field(default_factory=list, max_length=100)
blockers: list[dict[str, Any]] = Field(default_factory=list, max_length=100)
warnings: list[dict[str, Any]] = Field(default_factory=list, max_length=100)
profile_digest: str = Field(pattern=r"^[a-f0-9]{64}$")
# #endregion ScenarioGraph.TestPackProfile.Model
# #region ScenarioGraph.TestPackProfile.Coordinates [C:2] [TYPE Function] [SEMANTICS scenario,profile,metric,filter]
# @ingroup ScenarioGraph.TestPackProfile
# @BRIEF Project only server-inspected executable chart metrics and their applicable native filters.
def _profile_coordinates(query_model: DashboardQueryModel) -> list[ProfileCoordinate]:
from hashlib import sha256
filters_by_chart: dict[int, list[Any]] = {}
for native_filter in query_model.native_filters:
for target in native_filter.targets:
filters_by_chart.setdefault(target.chart_id, []).append(native_filter)
coordinates = []
for chart in sorted(query_model.charts, key=lambda item: (item.chart_id, item.slice_name)):
if not chart.execution_capable:
continue
for metric in sorted(chart.metrics, key=lambda item: (item.metric_name, item.label)):
identity = f"{chart.chart_id}\0{chart.dataset_id}\0{metric.metric_name}".encode("utf-8")
coordinates.append(ProfileCoordinate(
coordinate_id=sha256(identity).hexdigest()[:32],
chart_id=chart.chart_id, chart_name=chart.slice_name,
dataset_id=chart.dataset_id, dataset_name=chart.dataset_name,
metric_name=metric.metric_name, metric_label=metric.label,
expression_type=metric.expression_type,
filter_ids=sorted({item.filter_id for item in filters_by_chart.get(chart.chart_id, [])}),
filter_names=sorted({item.name for item in filters_by_chart.get(chart.chart_id, [])}),
))
return coordinates
# #endregion ScenarioGraph.TestPackProfile.Coordinates
# #region ScenarioGraph.TestPackProfile.Classification [C:1] [TYPE Function] [SEMANTICS scenario,profile,classification]
def _classification(status: str) -> UnresolvedKind | None:
if status in {"needs_context", "needs_selector", "needs_baseline", "unsupported"}:
return {"needs_context": "needs_context", "needs_selector": "needs_selector",
"needs_baseline": "needs_baseline", "unsupported": "unsupported"}[status] # type: ignore[return-value]
return None
# #endregion ScenarioGraph.TestPackProfile.Classification
# #region ScenarioGraph.TestPackProfile.Build [C:4] [TYPE Function] [SEMANTICS scenario,profile,compile,coverage]
# @ingroup ScenarioGraph.TestPackProfile
# @BRIEF Build the canonical graph and typed coverage preview from one server-inspected dashboard model.
# @PRE query_model is parsed from an authorized server inspection; selected_case_ids are bounded catalog IDs.
# @POST Every selected case is present once; no unsupported case is save-eligible; no expected value is emitted.
# @SIDE_EFFECT Compiler and validator emit bounded molecular-CoT logs; no database or handle writes occur.
# @DATA_CONTRACT DashboardQueryModel + intent -> TestPackProfile + DashboardTestScenario + DraftPackManifest
# #region ScenarioGraph.TestPackProfile.PublicBuilder [C:2] [TYPE Function] [SEMANTICS scenario,profile,api]
def build_test_pack_profile(
*,
query_model: DashboardQueryModel,
objective: str,
selected_case_ids: list[str],
parameters: dict[str, Any] | None = None,
baseline_available: bool = False,
browser_available: bool | None = None,
) -> tuple[TestPackProfile, DashboardTestScenario, dict[str, Any]]:
"""Compile one inspected dashboard intent and report every selected case and save blocker."""
src = "ScenarioGraph.TestPackProfile.Build"
logger.reason("Building test-pack profile from inspected context", src=src,
payload={"dashboard_id": query_model.dashboard_id, "case_count": len(selected_case_ids)})
try:
result = _build_test_pack_profile(
query_model=query_model, objective=objective, selected_case_ids=selected_case_ids,
parameters=parameters, baseline_available=baseline_available,
browser_available=browser_available,
)
except ValueError as exc:
logger.explore("Test-pack profile input rejected", src=src, error_code=str(exc),
payload={"dashboard_id": query_model.dashboard_id}, error=str(exc))
raise
profile, _, _ = result
logger.reflect("Test-pack profile evaluated", src=src,
payload={"status": profile.status, "unresolved_count": len(profile.unresolved),
"digest": profile.profile_digest[:16]})
return result
# #endregion ScenarioGraph.TestPackProfile.PublicBuilder
# #endregion ScenarioGraph.TestPackProfile.Build
# #region ScenarioGraph.TestPackProfile.BuildCore [C:4] [TYPE Function] [SEMANTICS scenario,profile,compile,coverage]
# @ingroup ScenarioGraph.TestPackProfile
# @BRIEF Compile profile inputs through the canonical graph validator and pack compiler.
def _build_test_pack_profile(
*, query_model: DashboardQueryModel, objective: str, selected_case_ids: list[str],
parameters: dict[str, Any] | None, baseline_available: bool, browser_available: bool | None,
) -> tuple[TestPackProfile, DashboardTestScenario, dict[str, Any]]:
catalog = load_catalog()
case_catalog = {item["id"]: item for item in catalog["cases"]}
if not selected_case_ids:
raise ValueError("SELECTED_CASES_REQUIRED")
unknown = sorted(set(selected_case_ids) - set(case_catalog))
if unknown:
raise ValueError("UNKNOWN_CASE_ID")
derivation = derive_capabilities(query_model, browser_available=browser_available)
capabilities = dict(derivation.capabilities)
capabilities["screenshot"] = capabilities.get("browser", False)
capabilities["baseline"] = baseline_available
# Facts not derivable from inspected dashboard structure remain false/unknown. In particular,
# a model or MCP caller cannot grant mutation fixtures, selectors, or cross-dashboard authority.
for key in ("safe_test_data", "safe_clock_fixture", "row_edit", "bulk_edit", "persistence_refresh", "cross_dashboard"):
capabilities[key] = False
compile_result = compile_scenario(CompileScenarioRequest(
agent_run_id="profile-preview",
objective={"goal": objective, "selected_case_ids": sorted(selected_case_ids)},
query_model=query_model.model_dump(mode="json"),
checklist_catalog_version=int(catalog["catalog_version"]),
baseline_version="published" if baseline_available else "missing",
capabilities=capabilities,
parameters=parameters or {},
has_dataset_fields=derivation.has_dataset_fields,
environment_id=query_model.environment_id,
dashboard_id=query_model.dashboard_id,
dashboard_name=query_model.title,
))
scenario = compile_result.scenario
validation = validate_scenario(scenario)
pack = generate_draft_pack(scenario)
return _assemble_profile(query_model, selected_case_ids, scenario, validation, compile_result, pack)
# #region ScenarioGraph.TestPackProfile.Assemble [C:3] [TYPE Function] [SEMANTICS scenario,profile,coverage,questions]
# @ingroup ScenarioGraph.TestPackProfile
# @BRIEF Assemble per-case coverage and typed unresolved questions from canonical compile results.
def _assemble_profile(query_model, selected_case_ids, scenario, validation, compile_result, pack):
# #region ScenarioGraph.TestPackProfile.Assemble.Body [C:2] [TYPE Function] [SEMANTICS scenario,profile,coverage]
catalog_version = load_catalog()["catalog_version"]
selected = set(selected_case_ids)
cases: list[ProfileCase] = []
unresolved: list[ProfileUnresolved] = []
step_by_id = {step.id: step for step in scenario.steps}
coordinates = _profile_coordinates(query_model)
for coverage in scenario.checklist_coverage:
if coverage.case_id in selected:
case, questions = _profile_case(coverage, scenario.steps, step_by_id, scenario.parameters, coordinates)
cases.append(case)
unresolved.extend(questions)
if coverage.case_id.startswith("M") and not any(item.kind == "needs_metric" for item in questions):
unresolved.append(_profile_question(coverage.case_id, "needs_metric", None, coordinates))
if any(step.action == "compare_to_baseline" and coverage.case_id in step.checklist_case_ids
for step in scenario.steps) and not any(item.kind == "needs_baseline" for item in questions):
unresolved.append(_profile_question(coverage.case_id, "needs_baseline", None, coordinates))
if coordinates and not any(item.kind == "needs_metric" for item in unresolved):
unresolved.append(_profile_question("dashboard", "needs_metric", None, coordinates))
blockers = [item.model_dump(mode="json") for item in validation.errors + validation.blockers]
blockers.extend({"code": "DRAFT_PACK_BLOCKER", "message": item}
for item in pack["validation_summary"].get("blockers", []))
# Unsupported cases are unresolved user decisions even when the compiler's compatibility
# behavior does not classify them as executable blockers.
eligible = pack["status"] == "save_eligible" and validation.valid and not unresolved and not blockers
status: ProfileStatus = "save_eligible" if eligible else "preview_only"
payload = {
"profile_version": 1,
"dashboard_id": query_model.dashboard_id,
"environment_id": query_model.environment_id,
"query_model_fingerprint": query_model.query_model_fingerprint,
"checklist_catalog_version": int(catalog_version),
"compiler_version": scenario.compiler_version,
"status": status,
"eligible": eligible,
"selected_case_ids": sorted(selected_case_ids),
"cases": [case.model_dump(mode="json") for case in cases],
"coordinates": [item.model_dump(mode="json") for item in coordinates],
"unresolved": [item.model_dump(mode="json") for item in unresolved],
"blockers": blockers,
"warnings": [item.model_dump(mode="json") for item in validation.warnings + compile_result.warnings],
}
profile = TestPackProfile(**payload, profile_digest=sha256_hex(canonical_dump(payload)))
return profile, scenario, pack
# #endregion ScenarioGraph.TestPackProfile.Assemble.Body
# #region ScenarioGraph.TestPackProfile.Case [C:3] [TYPE Function] [SEMANTICS scenario,profile,coverage,question]
# @ingroup ScenarioGraph.TestPackProfile
# @BRIEF Build one selected case classification and its typed analyst question.
def _profile_case(coverage, steps, step_by_id, parameters, coordinates):
case_id = coverage.case_id
step_ids = [step.id for step in steps if case_id in step.checklist_case_ids]
case_steps = [step_by_id[step_id] for step_id in step_ids]
kind = _classification(coverage.classification)
step_kind = next((_classification(step.automation_status) for step in case_steps
if _classification(step.automation_status)), None)
if step_kind is not None:
kind = step_kind
if kind is None and any(parameter.required and parameter.status == "unresolved"
and set(parameter.affected_step_ids).intersection(step_ids)
for parameter in parameters):
kind = "needs_context"
if kind is None and not step_ids:
kind = "unsupported"
case = ProfileCase(
case_id=case_id, classification=kind or "automated", step_ids=step_ids,
rationale=str(coverage.rationale) if step_ids else "No executable steps were produced for the selected checklist case.",
)
question_steps = [step.id for step in case_steps if _classification(step.automation_status) == kind] if kind else []
if kind and not question_steps:
question_steps = [None]
return case, [_profile_question(case_id, kind, step_id, coordinates)
for step_id in question_steps] if kind else []
# #endregion ScenarioGraph.TestPackProfile.Case
# #region ScenarioGraph.TestPackProfile.Question [C:2] [TYPE Function] [SEMANTICS scenario,profile,unresolved,question]
# @ingroup ScenarioGraph.TestPackProfile
# @BRIEF Map one unresolved case kind to a bounded question and permitted resolution type.
def _profile_question(case_id, kind, step_id=None, coordinates=None):
if kind is None:
return None
resolution = {
"needs_context": "provide_context", "needs_selector": "provide_selector_hint",
"needs_metric": "select_metric",
"needs_baseline": "select_baseline", "unsupported": "omit_case",
}[kind]
question = {
"needs_context": f"What bounded test context is required for {case_id}?",
"needs_selector": f"Which supported selector identifies the required control for {case_id}?",
"needs_metric": f"Which observed chart metric should be captured as a baseline for {case_id}? Do not enter a metric value.",
"needs_baseline": f"Which approved baseline should be used for {case_id}? Do not enter a metric value.",
"unsupported": f"{case_id} is not safely automatable with the inspected capabilities.",
}[kind]
coordinate_ids = [item.coordinate_id for item in coordinates or []]
return ProfileUnresolved(id=f"case:{case_id}:step:{step_id or 'none'}:{kind}", kind=kind,
json_pointer=f"/checklist_coverage/{case_id}", question=question,
step_id=step_id,
coordinate_ids=coordinate_ids if kind in {"needs_metric", "needs_baseline"} else [],
allowed_resolution=resolution)
# #endregion ScenarioGraph.TestPackProfile.Question
# #region ScenarioGraph.TestPackProfile.ApplyCoordinateChoices [C:3] [TYPE Function] [SEMANTICS scenario,profile,baseline,coordinate]
# @ingroup ScenarioGraph.TestPackProfile
# @BRIEF Bind analyst choices to unresolved coordinate requests without assigning or inventing values.
def apply_coordinate_choices(profile: TestPackProfile, choices: dict[str, str]) -> TestPackProfile:
unresolved = []
coordinate_ids = {item.coordinate_id for item in profile.coordinates}
if set(choices) - {item.id for item in profile.unresolved if item.kind == "needs_metric"}:
raise ValueError("PROFILE_COORDINATE_REQUEST_INVALID")
for item in profile.unresolved:
coordinate_id = choices.get(item.id)
if item.kind != "needs_metric" or coordinate_id is None:
unresolved.append(item)
continue
if coordinate_id not in item.coordinate_ids or coordinate_id not in coordinate_ids:
raise ValueError("PROFILE_COORDINATE_INVALID")
unresolved.append(item.model_copy(update={
"kind": "needs_baseline",
"coordinate_ids": [coordinate_id],
"question": "Capture and approve this observed metric through the 037 baseline lifecycle.",
"allowed_resolution": "select_baseline",
}))
payload = profile.model_dump(mode="json", exclude={"profile_digest"})
payload["unresolved"] = [item.model_dump(mode="json") for item in unresolved]
payload["status"] = "preview_only"
payload["eligible"] = False
return TestPackProfile(**payload, profile_digest=sha256_hex(canonical_dump(payload)))
# #endregion ScenarioGraph.TestPackProfile.ApplyCoordinateChoices
# #endregion ScenarioGraph.TestPackProfile.Assemble
# #endregion ScenarioGraph.TestPackProfile.BuildCore
# #endregion ScenarioGraph.TestPackProfile

View File

@@ -149,7 +149,10 @@ def _selector_params() -> dict:
return {
"test_date": {"type": "date", "default": "2026-07-01"},
"counterparty": {"type": "string", "default": "ACME"},
"selector_hint": {"type": "selector_hint", "default": "#native-filter", "required": False},
"selector_hint": {
"type": "selector_hint", "default": "#native-filter", "required": False,
"affected_step_ids": ["phase-2-B01-apply_native_filter"],
},
}
@@ -209,6 +212,26 @@ def test_visual_compare_without_screenshot_is_blocker() -> None:
assert any(b.code == "MISSING_SCREENSHOT" for b in result.blockers)
# #region Test.Scenario.Compiler.SelectorBinding [C:2] [TYPE Function] [SEMANTICS test,selector,binding]
def test_selector_hint_only_resolves_its_bound_step() -> None:
result = compile_scenario(_req(
["B01", "B02"],
parameters={"selector": {
"type": "selector_hint", "default": "#filter", "required": False,
"affected_step_ids": ["phase-2-B01-apply_native_filter"],
}},
capabilities=dict(FULL, baseline=True),
))
statuses = {step.id: step.automation_status for step in result.scenario.steps}
assert statuses["phase-2-B01-apply_native_filter"] == "ready"
assert statuses["phase-2-B02-apply_native_filter"] == "needs_selector"
assert [item.step_id for item in result.blockers if item.code == "NEEDS_SELECTOR"] == [
"phase-2-B02-apply_native_filter",
]
# #endregion Test.Scenario.Compiler.SelectorBinding
def test_metric_case_appends_compare_when_baseline_present() -> None:
caps = dict(FULL, baseline=True)
result = compile_scenario(_req(["T01"], capabilities=caps))

View File

@@ -26,6 +26,8 @@ from src.services.dashboard_testing.scenario.validator import validate_scenario
from src.services.dashboard_testing.registry.create import create_initial
from src.services.dashboard_testing.registry.materialize import materialize_pending_revisions
from src.models.scenario_materialization import OutboxEvent, RevisionMaterialization
from src.models.scenario_registry import ScenarioRegistryEntry, ScenarioRevision
from src.models.agent_authoring_workspace import AgentAuthoringWorkspace
class _BlobStore:
@@ -122,7 +124,8 @@ def test_create_initial_materializes_canonical_graph_and_consumes_handles(db_ses
"draft_pack_digest": pack.digest,
"dashboard_id": 80,
"allowed_environment_ids": ["env-prod-01"],
"selected_case_ids": ["B01"],
"selected_environment_id": "env-prod-01",
"selected_case_ids": ["B01", "C04", "C05", "T01"],
"title": "Materialized scenario",
"objective": "Verify canonical materialization",
})()
@@ -131,6 +134,7 @@ def test_create_initial_materializes_canonical_graph_and_consumes_handles(db_ses
assert entry.current_revision_id == revision.revision_id
assert revision.content_hash == compiled.content_hash
assert len(revision.graph_snapshot["steps"]) == len(scenario.steps)
assert entry.environment_ids == ["env-prod-01"]
assert compiled.consumed_by_revision_id == revision.revision_id
assert pack.consumed_by_revision_id == revision.revision_id
assert db_session.query(RevisionMaterialization).filter_by(revision_id=revision.revision_id).one().status == "pending"
@@ -139,4 +143,31 @@ def test_create_initial_materializes_canonical_graph_and_consumes_handles(db_ses
assert db_session.query(RevisionMaterialization).filter_by(revision_id=revision.revision_id).one().status == "materialized"
def test_create_initial_rejects_environment_or_case_mismatch_without_registry_writes(db_session, monkeypatch):
import src.services.dashboard_testing.scenario.handles as handles
monkeypatch.setattr(handles, "_blob_store", _BlobStore())
scenario = _scenario()
compiled = mint_compiled_handle(db_session, scenario, owner_principal="user-1", dashboard_id=80)
mint_validation_result(db_session, compiled, validate_scenario(scenario))
pack = mint_draft_pack_handle(
db_session, compiled, owner_principal="user-1", agent_run_id=None,
scenario_key=scenario.scenario_id, status="save_eligible", template_version="v1",
)
intent = type("Intent", (), {
"compiled_handle_id": compiled.handle_id, "draft_pack_id": pack.draft_pack_id,
"draft_pack_digest": pack.digest, "dashboard_id": 80,
"allowed_environment_ids": ["env-dev"], "selected_environment_id": "env-dev",
"selected_case_ids": ["B01"], "title": "Mismatch", "objective": "Mismatch",
})()
with pytest.raises(ValueError, match="PROFILE_ENVIRONMENT_MISMATCH"):
create_initial(db_session, intent=intent, user_id="user-1")
assert db_session.query(ScenarioRegistryEntry).count() == 0
assert db_session.query(ScenarioRevision).count() == 0
assert db_session.query(AgentAuthoringWorkspace).count() == 0
assert compiled.consumed_by_revision_id is None
assert pack.consumed_by_revision_id is None
# #endregion Test.ScenarioGraph.Handles

View File

@@ -0,0 +1,94 @@
# #region Test.Scenario.TestPackProfile [C:4] [TYPE Module] [SEMANTICS test,scenario,profile,coverage,eligibility]
# @BRIEF Verify deterministic profile coverage and fail-closed unresolved classifications.
# @RELATION VERIFIES -> [ScenarioGraph.TestPackProfile.Build]
from __future__ import annotations
import json
from pathlib import Path
import pytest
from src.schemas.dashboard_testing.query_model import DashboardQueryModel
from src.services.dashboard_testing.scenario.test_pack_profile import build_test_pack_profile
from src.services.dashboard_testing.scenario.test_pack_profile import apply_coordinate_choices
_MODEL = Path(__file__).resolve().parents[3] / "fixtures" / "dashboard_scenarios" / "query_model_sales.json"
def _query_model() -> DashboardQueryModel:
payload = json.loads(_MODEL.read_text(encoding="utf-8"))
payload["charts"][0]["metrics"] = [{
"metric_name": "sum__amount", "label": "Total sales", "expression_type": "SIMPLE",
"column": {"column_name": "amount", "type": "DOUBLE"}, "aggregate": "SUM",
}]
return DashboardQueryModel.model_validate(payload)
# #region Test.Scenario.TestPackProfile.Cases [C:3] [TYPE Function] [SEMANTICS test,profile,unresolved]
def test_profile_keeps_selected_cases_visible_and_preview_only_without_context() -> None:
profile, scenario, pack = build_test_pack_profile(
query_model=_query_model(), objective="Check filters and persisted comments",
selected_case_ids=["B01", "B05"], browser_available=True,
)
assert profile.status == "preview_only"
assert profile.eligible is False
assert {case.case_id for case in profile.cases} == {"B01", "B05"}
assert {item.kind for item in profile.unresolved} >= {"needs_selector"}
assert len({item.step_id for item in profile.unresolved if item.kind == "needs_selector"}) == len(
[item for item in profile.unresolved if item.kind == "needs_selector"]
)
assert pack["status"] == "preview_only"
assert all("expected_value" not in step.model_dump(mode="json") for step in scenario.steps)
# #endregion Test.Scenario.TestPackProfile.Cases
# #region Test.Scenario.TestPackProfile.Determinism [C:2] [TYPE Function] [SEMANTICS test,profile,digest]
def test_profile_digest_is_stable_and_case_order_independent() -> None:
first, _, _ = build_test_pack_profile(
query_model=_query_model(), objective="Review filter behavior", selected_case_ids=["B01", "B02"], browser_available=True,
)
second, _, _ = build_test_pack_profile(
query_model=_query_model(), objective="Review filter behavior", selected_case_ids=["B02", "B01"], browser_available=True,
)
assert first.profile_digest == second.profile_digest
assert first.selected_case_ids == ["B01", "B02"]
# #endregion Test.Scenario.TestPackProfile.Determinism
# #region Test.Scenario.TestPackProfile.CoordinateChoice [C:3] [TYPE Function] [SEMANTICS test,profile,metric,baseline]
def test_server_derived_coordinate_choice_never_supplies_metric_truth() -> None:
profile, scenario, _ = build_test_pack_profile(
query_model=_query_model(), objective="Track dashboard sales", selected_case_ids=["B01"], browser_available=True,
)
assert profile.coordinates
coordinate = profile.coordinates[0]
assert coordinate.metric_name == "sum__amount"
assert coordinate.filter_ids == ["f1", "f2"]
unresolved = next(item for item in profile.unresolved if item.kind == "needs_metric")
assert unresolved.coordinate_ids == [item.coordinate_id for item in profile.coordinates]
selected = apply_coordinate_choices(profile, {unresolved.id: coordinate.coordinate_id})
assert selected.status == "preview_only"
assert selected.eligible is False
assert selected.profile_digest != profile.profile_digest
selected_metric = next(item for item in selected.unresolved if item.id == unresolved.id)
assert selected_metric.kind == "needs_baseline"
assert "expected_value" not in str(scenario.model_dump(mode="json"))
assert all(step.expected.kind != "baseline_ref" for step in scenario.steps)
with pytest.raises(ValueError, match="PROFILE_COORDINATE_INVALID"):
apply_coordinate_choices(profile, {unresolved.id: "0" * 32})
# #endregion Test.Scenario.TestPackProfile.CoordinateChoice
# #region Test.Scenario.TestPackProfile.InputGuards [C:2] [TYPE Function] [SEMANTICS test,profile,input]
@pytest.mark.parametrize(("cases", "error"), [([], "SELECTED_CASES_REQUIRED"), (["X99"], "UNKNOWN_CASE_ID")])
def test_profile_rejects_empty_or_unknown_case_selection(cases: list[str], error: str) -> None:
with pytest.raises(ValueError, match=error):
build_test_pack_profile(query_model=_query_model(), objective="Check dashboard", selected_case_ids=cases)
# #endregion Test.Scenario.TestPackProfile.InputGuards
# #endregion Test.Scenario.TestPackProfile

View File

@@ -40,6 +40,8 @@ def test_version_shape_and_pinned_major():
f"catalog major {major} != pinned {PINNED_CATALOG_MAJOR}: a breaking catalog change "
"(rename/schema change/removal) requires bumping MCP_CATALOG_VERSION and this pin together"
)
assert MCP_CATALOG_VERSION == "2.6.0"
assert {"propose_test_pack_profile", "resolve_test_pack_profile"} <= set(_MCP_CATALOG_BY_NAME)
# #endregion Test.McpCatalogVersion.Shape

View File

@@ -141,7 +141,7 @@ async def test_mcp_bootstrap_run_automation_and_prod_gate(isolated_mcp_db, monke
# T029i/E2E-EXT-001 (ADR-0024): the AgentRun prerequisite is minted through the MCP catalog
# itself — the very surface an external client uses (field run 2026-09-07 blocker closure).
created_run = _unwrap(await server.call_tool("create_agent_run", {"request": {
"dashboard_id": 80, "environment_id": "env-dev",
"dashboard_id": 80, "environment_id": "env-prod-01",
"dashboard_name": "T029c fixture dashboard", "idempotency_key": f"t029c-run-{uuid4()}",
}}))
assert created_run["status"] == "ok", created_run
@@ -154,7 +154,8 @@ async def test_mcp_bootstrap_run_automation_and_prod_gate(isolated_mcp_db, monke
assert registered["context_authority"] == "unverified"
initial = {
"idempotency_key": f"bootstrap-{uuid4()}", "title": "T029c scenario", "dashboard_id": 80,
"allowed_environment_ids": ["env-dev"], "selected_case_ids": [], "objective": "bootstrap",
"allowed_environment_ids": ["env-prod-01"], "selected_environment_id": "env-prod-01",
"selected_case_ids": scenario.objective["selected_case_ids"], "objective": "bootstrap",
"compiled_handle_id": registered["compiled_handle_id"],
"draft_pack_id": registered["draft_pack_handle_id"],
"draft_pack_digest": registered["draft_pack_digest"],
@@ -179,7 +180,11 @@ async def test_mcp_bootstrap_run_automation_and_prod_gate(isolated_mcp_db, monke
visible = await server.call_tool("list_scenario_schedules", {"scenario_id": scenario_id})
assert _unwrap(visible) == []
environments = {"env-dev": SimpleNamespace(id="env-dev", stage="DEV", is_production=False), "prod": SimpleNamespace(id="prod", stage="PROD", is_production=True)}
environments = {
"env-dev": SimpleNamespace(id="env-dev", stage="DEV", is_production=False),
"env-prod-01": SimpleNamespace(id="env-prod-01", stage="DEV", is_production=False),
"prod": SimpleNamespace(id="prod", stage="PROD", is_production=True),
}
manager = SimpleNamespace(get_environment=environments.get)
monkeypatch.setattr(scenario_module, "get_config_manager", lambda: manager)
monkeypatch.setattr(rbac_module, "get_config_manager", lambda: manager)
@@ -195,7 +200,7 @@ async def test_mcp_bootstrap_run_automation_and_prod_gate(isolated_mcp_db, monke
# registry metadata, so the fresh current revision is directly runnable.
blocked = _unwrap(await server.call_tool("start_scenario_run", {"request": {
"scenario_id": scenario_id, "revision_id": revision_id,
"environment_id": "env-dev", "idempotency_key": f"run-blocked-{uuid4()}",
"environment_id": "env-prod-01", "idempotency_key": f"run-blocked-{uuid4()}",
"baseline_set": "ss-prod-visual", "baseline_set_version": "1",
}}))
assert blocked["status"] == "queued", blocked

View File

@@ -43,7 +43,6 @@ _ANALYST_EXTRA_TOOLS = sorted([
"register_draft_pack", # dashboard:testing:WRITE
])
def _persist_principal(prefix: str, *, is_admin: bool, grants: tuple[tuple[str, str], ...]) -> str:
suffix = secrets.token_hex(4)
username = f"{prefix}-{suffix}"

View File

@@ -13,6 +13,8 @@ from __future__ import annotations
import os
import secrets
import json
from pathlib import Path
from types import SimpleNamespace
from uuid import uuid4
@@ -32,6 +34,7 @@ from src.models.auth import McpToolInvocationRecord, Permission, Role, User
from src.models.scenario_approval import ActionApprovalGate
from src.models.scenario_registry import ScenarioEditProposal, ScenarioRegistryEntry, ScenarioRevision
from src.models.scenario_run import ScenarioRun, ScenarioStepRun
from src.schemas.dashboard_testing.query_model import DashboardQueryModel
_EXPECTED_CHAIN_TOOLS = (
"create_authoring_session",
@@ -44,9 +47,17 @@ _EXPECTED_CHAIN_TOOLS = (
"request_save",
"activate_revision",
"validate_scenario",
"propose_test_pack_profile",
"resolve_test_pack_profile",
)
# #region Test.McpScenarioE2E.AsyncValue [C:1] [TYPE Function] [SEMANTICS test,mcp,async]
async def _async_value(value):
return value
# #endregion Test.McpScenarioE2E.AsyncValue
# #region Test.McpScenarioE2E.Fixture [C:3] [TYPE Function]
# @ingroup Test.McpScenarioE2E
# @BRIEF Seed one registry entry whose current revision carries the editor graph shape.
@@ -268,7 +279,7 @@ async def test_external_client_creates_and_activates_scenario_revision_end_to_en
assert operations >= 4
# E2E-AUTH-003: an MCP-started run pins the promoted revision identity.
environments = {"env-dev": SimpleNamespace(id="env-dev", stage="DEV", is_production=False)}
environments = {"preprod": SimpleNamespace(id="preprod", stage="PREPROD", is_production=False)}
# The start gate (_scenario_start_permission) lives in rbac_server; the tool body
# (start_scenario_run) resolves config through tools_scenario since decomposition Phase D.
monkeypatch.setattr(tools_scenario_module, "get_config_manager", lambda: SimpleNamespace(get_environment=environments.get))
@@ -281,7 +292,7 @@ async def test_external_client_creates_and_activates_scenario_revision_end_to_en
started = _unwrap(await server.call_tool("start_scenario_run", {"request": {
"scenario_id": scenario_id,
"revision_id": activated.get("revision_id") or candidate_revision_id,
"environment_id": "env-dev",
"environment_id": "preprod",
"params": {"region": "emea", "currency": "EUR"},
"idempotency_key": f"e2e-run-{uuid4()}",
"baseline_set": "ss-prod-visual",
@@ -299,6 +310,97 @@ async def test_external_client_creates_and_activates_scenario_revision_end_to_en
_cleanup(principal, role_name, scenario_id, workspace_id)
# #region Test.McpScenarioE2E.ProfilePreview [C:4] [TYPE Function] [SEMANTICS test,mcp,profile,preview]
# @ingroup Test.McpScenarioE2E
# @BRIEF Expose a deterministic typed preview without returning a save handle or graph authority.
@pytest.mark.asyncio
async def test_propose_test_pack_profile_returns_unresolved_preview_only(monkeypatch) -> None:
server = mcp_server._build_probe_server()
access = mcp_server.AccessToken(
token="profile-token", client_id="profile-client", scopes=["mcp"],
subject="profile-operator", claims={"principal_type": "user"},
)
context_token = _access_token_context.set(access)
monkeypatch.setattr(tools_scenario_module, "get_config_manager", lambda: SimpleNamespace(
get_environment=lambda environment_id: SimpleNamespace(id=environment_id)
))
fixture_path = Path(__file__).resolve().parent / "fixtures" / "dashboard_scenarios" / "query_model_sales.json"
query_payload = json.loads(fixture_path.read_text(encoding="utf-8"))
query_payload["charts"][0]["metrics"] = [{
"metric_name": "sum__amount", "label": "Total sales", "expression_type": "SIMPLE",
"column": {"column_name": "amount", "type": "DOUBLE"}, "aggregate": "SUM",
}]
query_model = DashboardQueryModel.model_validate(query_payload)
monkeypatch.setattr(tools_scenario_module, "get_superset_client", lambda _: _async_value(object()))
monkeypatch.setattr(tools_scenario_module, "inspect_dashboard_query_model", lambda *_: _async_value(query_model))
monkeypatch.setattr(tools_scenario_module, "resolve_browser_availability", lambda: True)
request = {"environment_id": "ss-prod", "dashboard_id": 11,
"objective": "Verify dashboard filters and sales metric", "selected_case_ids": ["B01", "B02"]}
try:
profile = _unwrap(await server.call_tool("propose_test_pack_profile", {"request": request}))
replay = _unwrap(await server.call_tool("propose_test_pack_profile", {"request": request}))
assert profile["status"] == "preview_only"
assert profile["profile"]["selected_case_ids"] == ["B01", "B02"]
assert {case["case_id"] for case in profile["profile"]["cases"]} == {"B01", "B02"}
assert profile["profile"]["profile_digest"] == replay["profile"]["profile_digest"]
assert {item["kind"] for item in profile["profile"]["unresolved"]} == {"needs_selector", "needs_metric"}
assert "scenario" not in profile and "draft_pack" not in profile
assert profile["preview"]["step_count"] >= 1
coordinates = profile["profile"]["coordinates"]
assert coordinates and coordinates[0]["metric_name"] == "sum__amount"
baseline_item = next(item for item in profile["profile"]["unresolved"] if item["kind"] == "needs_metric")
coordinate_request = {
**request,
"expected_profile_digest": profile["profile"]["profile_digest"],
"resolutions": [{"unresolved_id": baseline_item["id"], "coordinate_id": coordinates[0]["coordinate_id"],
"reason": "Analyst selected this metric."}],
}
coordinate_result = _unwrap(await server.call_tool("resolve_test_pack_profile", {"request": coordinate_request}))
assert coordinate_result["status"] == "preview_only"
updated_metric = next(item for item in coordinate_result["profile"]["unresolved"]
if item["id"] == baseline_item["id"])
assert updated_metric["coordinate_ids"] == [coordinates[0]["coordinate_id"]]
assert updated_metric["kind"] == "needs_baseline"
assert coordinate_result["profile"]["eligible"] is False
forged_coordinate = _unwrap(await server.call_tool("resolve_test_pack_profile", {
"request": {**coordinate_request, "resolutions": [{**coordinate_request["resolutions"][0],
"coordinate_id": "0" * 32}]}
}))
assert forged_coordinate["status"] == "blocked"
unresolved = profile["profile"]["unresolved"][0]
unresolved_id = unresolved["id"]
selected_step = unresolved["step_id"]
resolved_request = {
**request,
"expected_profile_digest": profile["profile"]["profile_digest"],
"resolutions": [{"unresolved_id": unresolved_id, "step_id": selected_step,
"selector_hint": "#sales-filter", "reason": "Analyst verified this control."}],
}
resolved = _unwrap(await server.call_tool("resolve_test_pack_profile", {"request": resolved_request}))
assert resolved["status"] == "preview_only"
assert {item["id"] for item in resolved["profile"]["unresolved"]} == {
item["id"] for item in profile["profile"]["unresolved"] if item["id"] != unresolved_id
}
assert "scenario" not in resolved and "draft_pack" not in resolved
stale = _unwrap(await server.call_tool("resolve_test_pack_profile", {
"request": {**coordinate_request, "expected_profile_digest": "0" * 64}
}))
assert stale == {"status": "conflict", "error": "PROFILE_STALE",
"current_profile_digest": profile["profile"]["profile_digest"]}
forged = _unwrap(await server.call_tool("resolve_test_pack_profile", {
"request": {**resolved_request, "resolutions": [{**resolved_request["resolutions"][0],
"step_id": "phase-2-B02-apply_native_filter"}]}
}))
assert forged["status"] == "blocked"
with pytest.raises(Exception):
await server.call_tool("propose_test_pack_profile", {
"request": {**request, "parameters": {"expected_revenue": 999999}}
})
finally:
_access_token_context.reset(context_token)
# #endregion Test.McpScenarioE2E.ProfilePreview
@pytest.mark.asyncio
async def test_operator_without_scenario_edit_cannot_save_or_activate() -> None:
suffix = secrets.token_hex(4)

View File

@@ -47,8 +47,11 @@ class _BlobStore:
return self.values.pop(ref, None) is not None
def _seed_bootstrap_handles(user_id: str) -> tuple[str, str, str]:
def _seed_bootstrap_handles(user_id: str, environment_id: str) -> tuple[str, str, str]:
scenario = DashboardTestScenario.model_validate(json.loads(_FIXTURE.read_text(encoding="utf-8")))
payload = scenario.model_dump(mode="json")
payload["dashboard_context"]["environment_id"] = environment_id
scenario = DashboardTestScenario.model_validate(payload)
with SessionLocal() as db:
compiled = mint_compiled_handle(db, scenario, owner_principal=user_id, dashboard_id=42)
mint_validation_result(db, compiled, validate_scenario(scenario))
@@ -72,7 +75,7 @@ def _unwrap(value):
async def test_bootstrap_replay_returns_original_ids_and_conflict_is_typed(monkeypatch) -> None:
principal = f"t029-{uuid4()}"
monkeypatch.setattr(handles_module, "_blob_store", _BlobStore())
compiled_id, pack_id, digest = _seed_bootstrap_handles(principal)
compiled_id, pack_id, digest = _seed_bootstrap_handles(principal, "env-preprod")
server = mcp_server._build_probe_server()
monkeypatch.setattr(server, "_record", lambda **_: None)
token = _access_token_context.set(mcp_server.AccessToken(
@@ -81,7 +84,8 @@ async def test_bootstrap_replay_returns_original_ids_and_conflict_is_typed(monke
))
request = {
"idempotency_key": f"bootstrap-{uuid4()}", "title": "T029 smoke", "dashboard_id": 42,
"allowed_environment_ids": ["env-preprod"], "selected_case_ids": ["case-1"],
"allowed_environment_ids": ["env-preprod"], "selected_environment_id": "env-preprod",
"selected_case_ids": ["B01", "C04", "C05", "T01"],
"objective": "Verify bootstrap", "compiled_handle_id": compiled_id, "draft_pack_id": pack_id,
"draft_pack_digest": digest,
}

View File

@@ -153,14 +153,35 @@ silently defaults any of them. These markers may produce preview/working-draft
output, but only 042 may persist a revision and 044 may enforce launch-time
bindings in `RunPreflight`.
### Handle persistence (amendment 2026-09-06 — normative target, SPECIFIED-PENDING)
### External MCP test-pack profile (050 T029a, partial implementation 2026-09-29)
Implementation status (audit 2026-09-06, Doc.Adr.ADR0023): the deterministic
cores (Compile/Validate/Resolve/Canonical serialization) are IMPLEMENTED as pure
functions with `@SIDE_EFFECT None`. The durable handle layer below is NOT YET
IMPLEMENTED: today the REST/MCP boundaries return graphs in-band, and
`ScenarioRegistry.Create.*` verifies caller-composed `compile:{run_id}:{digest}`
strings against 036 `DraftArtifact` rows as a transitional stand-in.
050 `propose_test_pack_profile` projects a fresh server-inspected
`DashboardQueryModel` through the canonical compiler, validator and pack
compiler. Its strict profile returns selected-case coverage, stable opaque
chart→dataset→metric coordinate IDs, applicable native-filter identities, and
typed unresolved questions. The profile digest binds the canonical response;
the tool returns a bounded preview, never a compiled graph or draft-pack handle.
`resolve_test_pack_profile` re-inspects context and uses
`expected_profile_digest` as a compare-and-set precondition. Selector hints are
restricted and bound to one exact unresolved step; the compiler accepts a
resolved selector only for that step. Metric coordinate selection accepts only
an ID from the current server-generated profile and changes `needs_metric` to
`needs_baseline`; it does not supply an expected value, create a baseline ref,
or imply capture/approval/publication. Profile resolution is currently
stateless. Domain-context and published-baseline resolution, save-eligible
handle minting, and full external MCP profile→bootstrap acceptance remain open
in 050 T029a; this preview surface is not a substitute for REST handle gates.
### Handle persistence (implemented 2026-09-06; T029d–T029g)
Current status (reconciled 2026-09-29): deterministic Compile/Validate/Resolve/
Canonical serialization remain pure functions. Durable handles, canonical-byte
storage, persisted REST/MCP minting boundaries, context-authority verification,
042 transactional consumption/materialization and fresh-DB bootstrap E2E are
implemented; see 050 `tasks.md` T029d–T029i. The handle rules below are current
invariants, not pending implementation. T029a profile resolution is a separate
incomplete layer and does not yet mint or authorize handles.
1. **Minting boundaries.** `CompiledScenarioHandle`, `ValidationResultHandle` and
`DraftPackHandle` are minted ONLY by the persisted boundaries that wrap the
@@ -197,8 +218,8 @@ strings against 036 `DraftArtifact` rows as a transitional stand-in.
5. **Rejection rules.** Caller-composed handle strings, caller digests,
unbound compiled/draft pairs, `preview_only` packs, stale validation results
and double consumption are typed rejections (409-class) with zero partial
rows/outbox/gates. The transitional string-format check in
`ScenarioRegistry.Create.Register` is removed when this layer lands (050 T029d/T029f).
rows/outbox/gates. Caller-composed legacy handle strings are rejected; only
owner-bound persisted handle IDs pass `ScenarioRegistry.Create.Register`.
6. **Inspect stage status (resolved as hybrid, T029h 2026-09-06).** No persisted
`InspectionContextHandle` is minted. Instead: the MCP read tool
`inspect_dashboard_context` exposes the live server resolver

View File

@@ -54,6 +54,32 @@ python3 -c "import yaml; d=yaml.safe_load(open('specs/038-dashboard-scenario-mod
9. Prototype: every `@UX_STATE` in `contracts/ux/scenario-graph-ux.md` reachable via `prototype/index.html` state switcher (see `prototype/manifest.md`).
10. VlmAnalysisSpec/ScreenshotCaptureSpec serialize canonically and are consumed by 044 executors (runtime VLM/capture belongs to 044).
## External MCP Profile Preview (050 T029a, partial)
The external client may call `propose_test_pack_profile` with a dashboard,
environment, objective and catalog case IDs. The server inspects the dashboard,
derives chart/dataset/metric coordinates and filter applicability, then returns
typed coverage and unresolved questions. The response is a preview, not a
compiled/draft-pack handle and cannot be registered or bootstrapped.
`resolve_test_pack_profile` requires the current `profile_digest` and re-inspects
the dashboard. It accepts one restricted selector hint for an exact unresolved
step or selects a coordinate ID returned by the current profile. It never accepts
caller-supplied metric values, query model, capabilities, graph bytes or baseline
pins. Coordinate selection advances to `needs_baseline`; use the 037 reference
capture/review/approval/publication lifecycle separately. The resolution loop is
stateless and T029a remains partial until context/baseline resolution, eligible
handle minting and external fresh-DB profile-to-bootstrap E2E are implemented.
Focused tests:
```bash
cd backend && source .venv/bin/activate
python -m pytest -q tests/services/dashboard_testing/scenario/test_test_pack_profile.py \
tests/services/dashboard_testing/scenario/test_compiler.py \
tests/test_mcp_scenario_e2e.py::test_propose_test_pack_profile_returns_unresolved_preview_only
```
> **Boundary note (2026-08-07)**: 038 is the compiler/IR layer. Runtime VLM submit, real capture bytes, and evidence artifacts (owner_type=scenario_run) are owned by 044 ScenarioExecution — NOT by 038. See `REVIEW-042-047-CLOSURE.md` and 044.
## Exit Gates (compiler-layer scope)

View File

@@ -59,6 +59,7 @@
| unresolved `needs_context` / `needs_selector` / `needs_baseline` | `ScenarioGraph.ServerOwnedPipeline` | 042 save; 044 `RunPreflight` | `Test.Scenario.Capability`, `Test.Scenario.Resolver`, `test_scenario_runner` |
| draft-pack handle and server digest | `ScenarioGraph.PackCompiler.Generate` | 042 server-owned save | `Test.Scenario.Pack`, `Test.Scenario.Pack.Security`, `test_scenario_registry` |
| no client graph/digest/runner plan authority | `ScenarioGraph.ServerOwnedPipeline` | 050 MCP transport | injected-code, path-traversal, and MCP raw-graph rejection tests |
| external profile coverage + selector/metric-coordinate CAS | `ScenarioGraph.ServerOwnedPipeline` + 050 `TestPackProfile` | 050 `T029a` | `test_test_pack_profile.py`, `test_compiler.py`, profile MCP E2E in `test_mcp_scenario_e2e.py` | PARTIAL: selector/coordinate preview choice is covered; domain context, published baseline and save-handle minting remain open |
| persistent co-authoring and sandbox | `ScenarioGraph.AgentAuthoringWorkspace` | 050 MCP -> 042 registry | session persistence, sandbox isolation/limits, receipt and CAS tests |
| proposal -> compile/validate -> review -> save | `ScenarioGraph.AgentAuthoringWorkspace` | 042 `ScenarioRegistry.SaveContinuation` | typed-candidate, diff-review, promotion-boundary E2E |

View File

@@ -10,8 +10,16 @@
### Catalog and error contract
Catalog version `050.2.0` adds the schemas in [tool-contracts.schema.json](tool-contracts.schema.json).
Existing names are retained where present; new names are normative additions, not a runtime catalog claim.
Catalog runtime version is `2.6.0` (pinned major 2). T029a adds curated
`propose_test_pack_profile` and `resolve_test_pack_profile` definitions using
the authenticated-human `permission=None` context-inspection policy; both deny
service principals. The strict profile inputs are
bounded, and the preview/resolution handlers return no graph or save handle.
Implementation and focused acceptance are PARTIAL; full profile-to-bootstrap
E2E and unresolved domain/baseline handling remain OPEN.
`tool-contracts.schema.json` is a normative schema for the separate audited
baseline operations, not the runtime schema source for profile tools. Existing
names are retained where present.
Every tool has a strict input schema, output schema, permission, principal class and idempotency classification.
Read tools require authenticated object ACL; list filtering never substitutes invocation authorization.
Errors: `invalid_arguments` (422), `unauthenticated` (401), `permission_denied` (403),
@@ -34,6 +42,7 @@ Approval required is a typed non-success pending result, never `completed`. No t
| start_scenario_run | 044 admission | explicit revision/env/baseline selector; server resolution before idempotency/gates |
| get_scenario_run, get_scenario_run_result | 044 read projection | run ACL; exact stored pin/result, bounded pages |
| get_scenario_artifact | 044 evidence metadata | run/artifact owner ACL; typed protected content ref, never raw storage path |
| proposed profile preview/resolution | 038 compiler/validator/pack projection | Runtime handler exists; catalog/RBAC binding and external-client acceptance are pending. Once admitted: live server-inspected context; bounded preview; digest CAS; no caller graph/value authority; never mint save handles in the preview operation |
| create/update schedule, webhook, CI dispatch | 046 automation | same REST validation even disabled; no HumanStep target |
| open/dispose investigation | 047 case transaction | case/queue/episode CAS atomic; external agent is pull-only |

View File

@@ -1,7 +1,7 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"title": "Audited MCP operation inputs 050.2.0",
"description": "Normative additions, implemented=false. Server derives principal, digests and pin. Does not replace unaffected legacy catalog tool schemas.",
"description": "Normative schemas for audited baseline operations. Runtime catalog 2.6.0 also exposes strict T029a profile preview/resolution DTOs implemented in backend/src/mcp_server/scenario_inputs.py; those handlers do not mint save handles. Server derives principal, digests and pin. Does not replace unaffected legacy catalog tool schemas.",
"oneOf": [
{
"type": "object",

View File

@@ -0,0 +1,181 @@
# T029a: Typed Test-Pack Profile and Server-Owned Bindings
## Objective
Complete the initial MCP scenario bootstrap so an external MCP agent can propose a useful dashboard test pack from authoritative dashboard facts, ask the analyst only for unresolved business decisions, and never invent metric truth, filters, selectors, runtime values, or baseline authority.
T029a closes only the profile/compiler/bootstrap completeness gap. It does not claim live provider readiness, baseline publication, scheduled zero-human acceptance, or full production GO.
## Current Contract and Gap
- MCP bootstrap input is `InitialScenarioIntent` in `backend/src/mcp_server/scenario_inputs.py`; it carries title, dashboard, allowed environments, case IDs, objective, and server-issued compile/draft-pack handles.
- `bootstrap_authoring_scenario` in `backend/src/mcp_server/tools_authoring.py` calls `create_initial` and commits registry + immutable current revision + workspace + idempotency receipt transactionally.
- `inspect_dashboard_context` and the 038 compile/validate/draft-pack operations exist. A server-owned `DashboardQueryModel` and fingerprint are available, and context authority is recomputed at `register_draft_pack`.
- `ScenarioGraph.PackCompiler.Generate` already yields `preview_only` for unresolved required parameters, `needs_selector`, and `needs_baseline` graph states.
- 038 has `ParameterDefinition`, typed refs, checklist capability mapping, `BaselineSelectionProposal` in the backend schemas, and explicit `needs_context` / `needs_selector` / `needs_baseline` semantics.
- Remaining gap (updated 2026-09-29): bootstrap now checks selected environment and case coverage against digest-bound graph handles, but no profile handle/eligibility receipt binds the unresolved profile itself to handle minting. The proposal and resolution tools are previews only.
- Do not add MCP scheduling work here. T029b/T029c already record schedule tools and bootstrap-to-run evidence; any contrary older handoff statement is superseded.
## Target Data Flow
```text
inspect_dashboard_context(environment_id, dashboard_id)
-> server-owned query-model fingerprint + derived capabilities
propose_test_pack_profile(selection intent)
-> typed TestPackProfile preview + coverage + unresolved questions
resolve_test_pack_profile(expected_profile_digest, typed resolutions, CAS)
-> refreshed profile; still preview_only until every gating item is resolved
compile/validate/generate_draft_pack(profile_handle) [NOT IMPLEMENTED]
-> server-owned CompiledScenarioHandle + DraftPackHandle + eligibility receipt
bootstrap_authoring_scenario(InitialScenarioIntent with handles)
-> one atomic registry entry + current revision + workspace + receipt
```
The exact tool names may follow the existing compile/validate API instead of introducing a parallel tool family. The implementation must preserve one canonical profile/compiler path shared by REST and MCP, not a second MCP-only compiler.
## Contract Design
### 1. Typed `InitialScenarioIntent`
Keep the current closed, bounded, `extra="forbid"` model. Add only fields that are not already derived from server-owned handles. The final bootstrap intent should bind:
- idempotency key, title, objective;
- selected environment ID and selected checklist case IDs;
- opaque server-issued `profile_handle_id`, compiled handle ID and draft-pack handle ID;
- expected profile/context/catalog fingerprints only as compare-and-set preconditions, never as authority;
- no caller-supplied graph, metrics/value, filter snapshot, dashboard query model, selector, baseline pin, digest authority, SQL/code, raw URL, cookie, or filesystem path.
Avoid taking both `allowed_environment_ids` and a single selected environment unless the profile genuinely supports multiple environments. For the first scenario revision, bind exactly one inspected environment; additional environments require independent compatibility validation and must not silently share a baseline.
### 2. Server-owned `TestPackProfile`
Define a strict schema/DTO, persisted handle only if required by current handle lifecycle. It contains:
- identity: profile ID/version, dashboard ID, one environment ID, inspected query-model fingerprint, checklist catalog version, compiler/action-registry versions;
- analyst intent: bounded objective and selected catalog case IDs;
- derived facts: chart IDs/names, dataset IDs/names, metric coordinates, supported filter definitions/scopes, server-derived capabilities, safe parameter definitions;
- proposed typed graph/profile mapping: each checklist case maps to supported steps/refs or an explicit classification with reason;
- baseline requirements: typed coordinate selector and comparison policy requirements only. Actual expected values come exclusively from a separately resolved, approved/published baseline; profile generation never embeds values;
- unresolved list with stable IDs and JSON pointers, each typed as `needs_context`, `needs_selector`, `needs_metric`, `needs_baseline`, or `unsupported`, plus a concise analyst question and allowed resolution schema;
- coverage summary, deterministic blockers/warnings, `preview_only | save_eligible`;
- provenance hashes/opaque source refs for server-inspected inputs.
Do not duplicate raw query context or source bytes when a handle already pins the authoritative snapshot. Profile serialization and content hash must be canonical and deterministic.
### 3. Question-and-resolution protocol
Every required unresolved item produces a typed question, not free-form implied consent:
- `needs_context`: request a bounded domain value/choice or ask for clarification; no sensitive value defaults.
- `needs_selector`: request an approved selector hint tied to the exact chart/control and safe selector grammar; reject arbitrary executable locator code.
- `needs_metric`: offer only server-inspected chart/dataset/metric coordinates; a selected coordinate transitions to `needs_baseline` but does not create baseline truth.
- `needs_baseline`: use the 037 reference URL/candidate/approval/publication lifecycle. Do not ask the user to type the expected numeric value.
- `unsupported`: explain why the selected case cannot be automated and offer only a supported manual/omit resolution if 038 policy allows it.
Resolution writes must include profile handle, expected CAS version, idempotency key and typed resolution values. Recompute profile and eligibility server-side after every resolution. Stale profile/context returns typed conflict and makes no partial update. Preserve prior resolved answers on recoverable errors.
Do not create a general natural-language question engine in this task. Agent wording is presentation; the server-owned question ID/type/schema and validation findings are authoritative.
## Invariants
- Dashboard, environment, query model, capabilities, chart/dataset/metric attribution, filter definitions/scopes and fingerprints are resolved by the server.
- Caller capability declarations may only restrict server-derived capability, never widen it or turn unknown into available.
- A selected checklist case has exactly one deterministic classification and explicit coverage entry; no selected case silently disappears.
- Missing required facts remain typed unresolved/unsupported states; no inferred selector, test data, numeric expectation, baseline pin, or runtime parameter value.
- Baseline-backed cases require a valid baseline reference contract; discovery/capture/approval/publication remain separate lifecycle steps.
- `preview_only` cannot mint save-eligible compiled/draft-pack handles and cannot reach `create_initial`.
- Bootstrap rechecks handle ownership, agent-run principal, dashboard/environment binding, profile fingerprint, all selected-case resolutions, context authority, and save eligibility in the same transaction as registry/workspace creation.
- On validation, ACL, handle, fingerprint, stale-CAS, or idempotency conflict failure, no registry entry, revision, workspace, outbox/materialization event, consumed handle, or bootstrap receipt is committed.
- Identical idempotent bootstrap replay returns the original identities; same key with changed semantic intent returns the existing typed conflict.
- A successful bootstrap's immutable revision records the canonical profile summary/fingerprints and selected cases needed for provenance, but never mutable run-time bindings or secret-bearing source state.
## Implementation Slices
### Progress note (2026-09-28)
- Implemented an initial typed, deterministic profile projection and `propose_test_pack_profile` MCP preview over a fresh server-side dashboard inspection. It returns every selected case with typed unresolved questions and never emits expected metric values. The tool and DTO are present in the MCP catalog; profile unit and MCP proposal tests cover deterministic digest and fail-closed preview.
- Bootstrap now binds one selected environment and the exact case IDs to digest-verified handle graph content before registry writes; selected environment must be allowed. Mismatch tests assert no registry/revision/workspace creation and no handle consumption.
- This is an incomplete T029a slice, not completion. Resolution is stateless and covers selector hints and metric coordinate choices only; profile-to-save-handle minting is not implemented. Full fresh-DB profile→resolve→register→bootstrap E2E and catalog/schema parity remain open.
- The selector-resolution operation uses `expected_profile_digest`, fresh upstream reinspection, one unresolved ID + exact step ID per answer, a restricted selector grammar, and recompilation. Per-step selector bindings are enforced by the canonical compiler; resolving one filter leaves other filter questions unresolved. Domain context, published baseline, durable idempotent resolution, and save-handle minting remain open.
- The profile lists server-derived chart → dataset → metric coordinates, stable coordinate IDs, and chart-level native-filter identity/name applicability. Selecting a coordinate by digest-CAS records only the selection and changes the question to `needs_baseline`; it never records a metric value or creates a baseline reference. 037 capture/approval/publication is still required. Metric query data is local to these tests; the shared sales query fixture remains unchanged.
- The first MCP profile test currently stubs the Superset inspection seam with a hardcoded fixture. This proves tool mapping/typed output only, not real upstream integration or the no-seeded-prerequisite E2E gate.
### Slice A: Contract alignment
- Read current T029a, 038 compile/validator/handle contracts, `InitialScenarioIntent`, compile/draft-pack DTOs, and MCP catalog/RBAC rules.
- Add/update schemas in 038 and 050 first: TestPackProfile, unresolved item/question, typed resolution, profile handle/eligibility receipt, and bootstrap preconditions.
- Specify canonical hashing, limits, stale error codes, and REST/MCP parity.
- Preserve pinned MCP catalog major 2; bump minor only if public tool schemas change.
### Slice B: Profile builder
- Build a pure deterministic profile builder over server-owned query model, checklist catalog, derived capabilities, parameter declarations, and baseline summaries/availability.
- Use existing capability mapper/compiler/validator rather than duplicating their classifications.
- Emit explicit per-case coverage and typed unresolved items.
- Add hardcoded fixtures for complete dashboard, missing dataset field, unknown selector, missing baseline, unsupported XLSX, unsafe mutation without a safe test fixture, empty case selection, duplicate/unknown case IDs, and partial inspection failure.
### Slice C: Resolution lifecycle
- Add server-owned profile handle/CAS/idempotency receipts only if existing CompiledScenarioHandle/DraftPackHandle cannot safely represent the mutable preview-to-resolved flow.
- Add MCP operations needed to read the question list and submit bounded typed resolutions; use existing REST service/domain functions when possible.
- Re-inspect or compare authoritative fingerprints at mutation boundaries; reject context drift with typed stale conflict.
- Prove stale/replay/duplicate-resolution behavior and zero partial side effects.
### Slice D: Compile and bootstrap gate
- Compile from the server-resolved profile to the existing canonical DashboardTestScenario pipeline.
- Generate preview pack for incomplete profile, but do not mint eligible save handles.
- For complete profile, mint compiled and draft-pack handles with profile hash/fingerprint linkage.
- Strengthen `create_initial`/`bootstrap_authoring_scenario` to reject handles whose profile receipt is absent, stale, preview-only, cross-dashboard, cross-environment, or missing a required resolution.
- Preserve atomic registry/revision/workspace/outbox/receipt behavior and idempotency semantics.
### Slice E: External MCP E2E and evidence
- Replace or add a fresh-database MCP E2E that obtains every prerequisite through tools reachable by the same human principal; no ORM-seeded registry, revision, run, handle, inspected context or baseline truth.
- Exercise complete path: MCP auth → inspect context → profile preview → receive unresolved questions → resolve all safe required items → inspect refreshed coverage/eligibility → register eligible pack handles → bootstrap → verify visible current revision and provenance.
- Negative vectors: forged metric/value/filter/context, foreign dashboard/environment, stale context fingerprint, unsupported/unknown case, missing baseline, invented selector, unresolved required item at bootstrap, raw graph/digest, cross-owner handles, stale CAS, idempotency replay/conflict.
- Assert zero registry/revision/workspace/outbox/consumed-handle side effects on every rejected bootstrap.
- Retain request/response trace with secrets redacted and test fixture explicitly identified as synthetic. No production or live release claim from fixture E2E.
## Acceptance Checklist
- [ ] `TestPackProfile` is strict, bounded, versioned and canonically hashed.
- [ ] Every selected case is represented by supported typed steps or a visible typed unresolved/unsupported classification.
- [ ] Server derives metrics, filters, selectors/capabilities and baseline requirements from authorized context; client claims cannot widen facts.
- [ ] Profile preview asks explicit questions for unresolved required data and enumerates allowed typed answers.
- [ ] Unknown expected metric values are never asked as free-text numeric inputs or embedded in graph assertions.
- [ ] Unresolved required context, selector, baseline, or unsupported blocking case produces `preview_only` and cannot mint save-eligible handles.
- [ ] Typed resolution is owner-scoped, CAS/idempotent, recomputes profile status, and handles stale dashboard/catalog context without partial writes.
- [ ] `bootstrap_authoring_scenario` accepts only matching server-issued save-eligible handles and atomically creates the initial current revision/workspace/outbox/receipt.
- [ ] Fresh external MCP E2E has no pre-seeded authoring prerequisites and includes one complete and all required negative paths.
- [ ] MCP catalog major remains 2; minor/catalog schemas/RBAC docs are updated if tools are added.
- [ ] 038 and 050 schemas/contracts/tasks/traceability/quickstarts agree; T029a is closed only against retained evidence.
- [ ] Overall production matrix remains `NO-GO` until unrelated provider, live publication, scheduled replay, injection and analyst-acceptance gates close.
## Commands and Verification
Run from `backend/` with `.venv` activated:
```bash
python -m pytest -q \
tests/test_mcp_initial_scenario_e2e.py \
tests/test_mcp_t029_bootstrap_automation.py \
tests/test_mcp_authoring_promotion_e2e.py \
tests/services/dashboard_testing/scenario \
tests/services/dashboard_testing/registry/test_create.py
python -m ruff check src/mcp_server src/services/dashboard_testing/scenario \
src/services/dashboard_testing/registry/create.py tests/test_mcp_initial_scenario_e2e.py
python -m compileall -q src/mcp_server src/services/dashboard_testing/scenario \
src/services/dashboard_testing/registry/create.py
```
Also run MCP catalog/schema validation, OpenAPI/schema validation for touched contracts, `git diff --check`, and the isolated fresh-DB MCP E2E. Run broader tests when shared compiler, checklist, baseline, or registry behavior changes.
## Non-goals
- No frontend prompt/chat/agent controls; agent interaction remains external MCP only.
- No new scheduling implementation (T029b/T029c already cover it).
- No exploration sandbox implementation (T050).
- No MCP baseline selection review surface (T049) beyond reusing existing baseline references and 037 gates.
- No guessed business logic, selectors, data fixtures, expected numeric values, or auto-approved baseline.
- No production release/deployment, scheduled zero-human claim, terminal PASS claim, or overall readiness promotion.

View File

@@ -6,7 +6,7 @@
> CRUD/editor, human approvals, and read-only evidence; it MUST NOT expose agent prompts, chat,
> proposal generation, or an agent workspace. Catalog major is pinned: breaking bootstrap-handle
> semantics shipped as catalog `2.0.0` with ritual `PINNED_CATALOG_MAJOR=2`. Additive minors
> (`2.1.0` context authority, `2.2.0` `create_agent_run`/`get_agent_run`) keep the same major.
> (`2.1.0` context authority, `2.2.0` `create_agent_run`/`get_agent_run`, `2.6.0` T029a profile preview/resolution) keep the same major.
> Historical transport `[x]` rows in `tasks.md` do not close production E2E.
## Target vs observed
@@ -14,8 +14,8 @@
| Layer | Target (refresh) | Observed (do not treat as GO) |
|---|---|---|
| Surface | One `/mcp` Streamable HTTP server; external client owns reasoning | `/mcp` mounted; OAuth 2.1 + DCR local evidence exists |
| Catalog | Versioned curated tools; `PINNED_CATALOG_MAJOR=2` | Runtime `MCP_CATALOG_VERSION="2.3.0"` (61 entries); pin test `tests/test_mcp_catalog_version.py` |
| Authoring | `inspect_dashboard_context` → `create_agent_run` → `register_draft_pack` → `bootstrap_authoring_scenario` → `activate_revision` / `start_scenario_run` | Vertical E2E exists with MCP-created AgentRun (T029i). Live-stand replay T029m / `E2E-EXT-002` **CLOSED 2026-09-11** (`docs/2026-09-11-sales-prod-mcp-replay.md`, run `110a6517…` + human loop) |
| Catalog | Versioned curated tools; `PINNED_CATALOG_MAJOR=2` | Runtime `MCP_CATALOG_VERSION="2.6.0"`; profile tools are curated as authenticated-human read previews; pin test covers major/schema registration |
| Authoring | `inspect_dashboard_context` → `propose_test_pack_profile` → `resolve_test_pack_profile` → `create_agent_run` / `register_draft_pack` → `bootstrap_authoring_scenario` | Preview and stateless selector/metric-choice CAS are client-visible; eligible handle minting and full profile→bootstrap external E2E remain OPEN (T029a PARTIAL). Existing T029i/M live chain remains separate |
| Parity | REST/MCP identical validation/errors/CAS; no consume/publish fallback success | T045/T046 **CLOSED 2026-09-11** (gated `publish_baseline_catalog` + live pin-from-Gitea canary v4); T044 offline parity green + live canaries done, remaining: dedicated REST-vs-MCP error-shape parity fixtures |
| Frontend | No agent chat/prompt/workspace/start/handoff | Chat service removed; remaining agent-panel/handoff drift is T030 OPEN |
@@ -81,6 +81,23 @@ npm run test -- --run src/lib/models/__tests__/AdminMcpCatalogModel.test.ts
8. `start_scenario_run` with explicit promoted `revision_id` + verified `content_hash`.
9. Human gates/checkpoints via `list_pending_approvals`/`decide_approval` and `list_checkpoints`/`decide_checkpoint` (USER principal only).
### T029a profile preview (implemented subset; not save authorization)
1. Call `propose_test_pack_profile` with `environment_id`, `dashboard_id`,
`objective`, and non-empty catalog `selected_case_ids`.
2. Review per-case coverage, server-derived chart→dataset→metric coordinates,
applicable native filters, and typed unresolved questions.
3. Call `resolve_test_pack_profile` with the current `profile_digest` to answer
an exact-step selector question or choose one suggested metric coordinate.
The server re-inspects the dashboard before applying the resolution.
4. A metric choice becomes `needs_baseline`; continue through the 037 capture,
candidate review, approval and publication lifecycle. Do not supply metric
values or treat coordinate selection as a baseline pin.
5. The handlers return bounded previews, not graph bytes or handles, and do not
make the scenario eligible for `register_draft_pack` or bootstrap.
Domain-context/published-baseline resolution, eligible-handle minting and a
fresh external-client profile→bootstrap E2E remain open under T029a.
Product UI never starts this chain. Manual editor (043) and monitor (045) are equal renderers of the same registry/run/gate rows.
## Exit gates (acceptance OPEN)
@@ -89,6 +106,7 @@ Product UI never starts this chain. Manual editor (043) and monitor (045) are eq
- Stale CAS and changed-intent replay create no partial state.
- Failed consume/publication returns typed error/pending; no legacy fallback success.
- Full `BaselineSelectionPin` survives rerun/rebaseline/result/analytics.
- T029a full profile→typed resolutions→save-eligible handle→bootstrap external-client E2E remains OPEN.
- HumanStep schedules rejected equally over REST/MCP before any durable side effect.
- No agent frontend controls, routes, or requests (T030; negative DOM/route/network still OPEN).
- Live browser/capture canary PASSED 2026-09-10 against ss-prod (binding `ss-prod-d11-live-001`, run `597274d3`: browser `open_dashboard` with checkpoint/page-URL/PNG digest + 8 durable screenshot refs, capacity leases; trace: `docs/reports/agentic-runtime-live-canary-2026-09-10.md`). Live `agent_evaluation` canary v2 PASSED 2026-09-10 through REST-only start with server-side binding resolution (run `4eebfab3`: `AgentEvaluation` persisted, DecisionPolicy row 11; trace: `docs/reports/agentic-runtime-live-canary-v2-2026-09-10.md`). **Correction 2026-09-25:** Omniroute `deepseek-flash` image input was verified with a live PNG (`image_url`, HTTP 200 / `IMAGE_CANARY_OK`, routed model `gpt-5.6-sol`); image attachment capability is no longer an open item. Still OPEN: full scenario terminal PASS + scheduled zero-human proof, provider-version/cost receipts, cancellation receipts.

View File

@@ -18,7 +18,7 @@
@SEMANTICS: spec, requirements, mcp, interface, tools, catalog, approval, provenance, decommission, gradio
**Feature Branch**: `050-mcp-interface`
**Created**: 2026-08-24 | **Status**: Partially implemented; production acceptance OPEN (refresh 2026-09-08)
**Created**: 2026-08-24 | **Status**: Partially implemented; production acceptance OPEN (refresh 2026-09-29)
**Input**: "Убрать веб-интерфейс агента (Gradio, кнопка «Ассистент» и другие точки входа) и заменить его простым MCP-интерфейсом к ss-tools, через который внешние клиенты создают все тесты. Полноценный внешний доступ; первый срез — паритет текущих 37 инструментов; продуктовый frontend не содержит агентских взаимодействий; агент работает только во внешнем MCP-клиенте."
## Decisions (session 2026-08-24)
@@ -93,7 +93,7 @@ HumanCheckpoint disposition (`waiting_human` runs, 044) is also completable insi
**Why P1**: "Создавать все тесты через MCP" is the core value replacing the chat flow.
**Independent Test**: From an external client, drive inspect → compile → validate → resolve → draft-pack → save for a fixture dashboard and verify the saved immutable revision appears in the 042 registry.
**Independent Test**: From an external client, inspect → propose a typed test-pack profile → resolve declared questions → compile/validate → register eligible draft-pack handles → bootstrap/save for a fixture dashboard and verify the immutable revision in the 042 registry. The profile-preview and selector/metric-choice steps are implemented; the complete profile-resolution-to-save chain remains OPEN under T029a.
**Acceptance**:
1. **Given** a dashboard id and environment **When** the client calls the inspection and scenario-authoring tools **Then** the chain produces a validated graph with `needs_context`/`needs_selector` markers instead of invented data (038 semantics).

View File

@@ -44,7 +44,7 @@
## Phase 2b — Initial scenario and automation parity
- [x] T029 Implement `ScenarioRegistry.CreateInitial` and `bootstrap_authoring_scenario`; prove a fresh external MCP client creates a first registry scenario without a seeded base revision. Evidence: `tests/test_mcp_initial_scenario_e2e.py` creates real `AgentRun`/`DraftArtifact` rows and drives `bootstrap_authoring_scenario` without a seeded registry; replay and conflicting idempotency are covered by `tests/test_mcp_t029_bootstrap_automation.py`.
- [ ] T029a Add typed InitialScenarioIntent/TestPackProfile validation and server-owned metric/filter/selector/baseline bindings; unresolved requirements must be `preview_only`, never guessed.
- [~] T029a Partial: strict profile preview derives from fresh server inspection and returns case coverage, deterministic metric coordinates and typed questions; runtime catalog 2.6.0 exposes `propose_test_pack_profile` and `resolve_test_pack_profile` as authenticated-human read previews (service principals denied). Digest-CAS accepts exact-step selector hints and server-issued metric coordinate choices, re-inspects/recompiles, and transitions chosen metric to `needs_baseline` without expected values or a pin. Bootstrap checks selected environment and exact case coverage against digest-bound graph handles before writes; compiler selector authority is per-step. Evidence: `test_test_pack_profile.py`, `test_compiler.py`, `test_handles.py`, `test_mcp_t029_bootstrap_automation.py`, profile MCP E2E in `test_mcp_scenario_e2e.py`; 34 focused tests, Ruff and compileall pass. Still open: durable/idempotent domain-context and published-baseline resolution, profile-bound save-eligible handle minting, and fresh external-client profile→resolve→register→bootstrap E2E. Do not mark complete until those vectors pass.
- [x] T029b Expose 046 schedule, trigger-rule, policy and automation-metrics management through curated MCP tools with REST-equivalent RBAC, idempotency and PROD gate behavior. Evidence: curated tools and catalog entries in `src/mcp_server/tools_automation.py`/`rbac_server.py`, idempotency migration `alembic/versions/0018_automation_idempotency.py`, and `tests/test_mcp_t029_bootstrap_automation.py` plus RBAC catalog tests.
- [x] T029c Add fresh-DB MCP E2E for bootstrap → visible registry entry → revision activation → manual run, plus scheduler eligibility/PROD-gate integration evidence. Evidence: `tests/test_mcp_initial_scenario_e2e.py` verifies fresh bootstrap, current revision, refusal to run an un-promoted bootstrap revision (`BOOTSTRAP_REVISION_NOT_RUNNABLE`), a pinned manual run against a genuinely materialized revision, human-step schedule rejection with zero side effects, and an idempotent PROD approval gate; focused MCP regression set: `145 passed` (2026-09-06).

View File

@@ -1,8 +1,8 @@
# Traceability — 050-mcp-interface
## Production acceptance traceability — 2026-09-08
## Production acceptance traceability — 2026-09-29
Historical rows above identify prior tests/code only; removed agent UI paths are retired. The following audited gates are **implemented=false / OPEN**, independent of local suite totals.
Historical rows above identify prior tests/code only; removed agent UI paths are retired. The following audited gates remain partial/open independent of local suite totals.
| Requirement | Domain contract / DTO | Task | Falsifiable acceptance | State |
|---|---|---|---|---|
@@ -15,6 +15,7 @@ Historical rows above identify prior tests/code only; removed agent UI paths are
| MCPX-FR-033 | Exploration step | [T050](tasks.md) | sandbox read-only interactive step with CAS/receipts and no second Playwright stack. | OPEN. |
| MCPX-FR-034 | Prompt dangerous-content profile | [T051](tasks.md) | REST/MCP validation/error parity plus injection negative tests. | PARTIAL 2026-09-22: parity matrix CLOSED (`test_mcp_rest_error_parity.py`); LLM evidence-injection negative gate (`LLM-INJ-001`) still OPEN. |
| MCPX-FR-035 | [Reference URL invariant](../037-superset-baseline-engine/contracts/reference-url.md), [tool schema](contracts/tool-contracts.schema.json) | [T052](tasks.md) | Agent can request same-dashboard URL; MCP and REST resolve identical filter state and coordinate; neither accepts caller values as authority; published pin survives scheduled replay without resolving key again. | PARTIAL 2026-09-25: `preview_reference_dashboard` and `capture_reference_selection` MCP tools use the guarded REST handlers and human RBAC; the supplied ss-dev dashboard 10 URL has live parser/query proof (`platform=3DS`, 10/10 metrics), and isolated Chromium proved REST editor URL → six drafts plus mismatch rejection. MCP tool offline tests exist; live MCP tool invocation, an agent asking for the URL, real publication/pin and scheduled replay remain open. |
| MCPX-FR-036 / T029a | [038 ServerOwnedPipeline](../038-dashboard-scenario-model/contracts/modules.md); typed MCP profile inputs | [T029a](tasks.md) | Fresh dashboard inspection → complete selected-case profile/metric coordinates → digest-CAS typed selector/coordinate choice → eligible registered handle → atomic bootstrap; forged/stale values create zero side effects. | PARTIAL 2026-09-29: runtime catalog 2.6.0 registers both profile tools as authenticated-human read previews; 34 scoped tests, Ruff and compileall green. Selector is per-step and metric choice advances to `needs_baseline`. Domain context/published-baseline resolution, save-eligible handle minting and fresh external-client DB E2E remain open. |
Sources: [production gap](../../docs/reports/ss-prod-agentic-e2e-production-gap-2026-09-08.md), [coverage gap](../../docs/reports/ss-prod-agentic-e2e-spec-coverage-2026-09-08.md), [baseline gap](../../docs/reports/ss-prod-agentic-e2e-baseline-gap-2026-09-08.md). Spec schema/static checks prove contract structure only; live canary/runtime closure and optional approved performance baseline are not claimed.

View File

@@ -50,6 +50,7 @@
| `G-INVESTIGATION-UI` | P1 | 047 T024/T025/T027 | queued+in_case findability, case index, linked runs/evidence/i18n, browser queue→case→disposition. | `e7925d0e`,`c9e6e642`,`9acd9bad` presentation slice: grouped queue, business summary, evidence cards, collapsed technical refs; 13 focused tests. **2026-09-24 uncommitted follow-up:** deterministic isolated seed + Chromium queue→correct case→linked run/evidence→resolved→stale CAS, 1 passed/0 skipped; trace/screenshot retained in `frontend/test-results/`; UI POST fix with 10 focused tests, lint/build green. | CLOSED for scoped UI/E2E implementation; release-commit regression remains `G-FULL-REGRESSION` |
| `G-INVESTIGATION-ATOMIC` | P0 | 047 T019–T021/T028 | PostgreSQL concurrency: one active episode/case; stale CAS rollback; exact-context comparability; immutable evidence. | 2026-09-22: `analytics/locking.py` (pg_advisory_xact_lock per fingerprint) + `set_disposition` savepoint (zero partial projection); `analytics/context.py` exact-context comparability with typed mismatch/ineligible; 22 unit + 7 Testcontainers PG16 race vectors (`tests/integration/test_scenario_investigation_concurrency.py`). | CLOSED |
| `G-MCP-PARITY` | P0 | 050 T044/T047/T048/T051 | REST/MCP error/auth/lifecycle parity, catalog/visibility mirrors, service cannot decide human gates; all prerequisites externally reachable. | 2026-09-22: shared `classify_start_error` consumed by REST and MCP removes the transport-dependent error vocabulary; `tests/test_mcp_rest_error_parity.py` 17-vector matrix (env/baseline codes/idempotency/conflict/disabled action/disabled automation/investigation RBAC/unauthenticated + service-principal human-gate denial + MCP no-bypass on case ACL). Live external chain proven earlier. | CLOSED |
| `G-MCP-TEST-PACK-PROFILE` | P1 | 050 T029a; 038 ServerOwnedPipeline | External MCP profile preview derives server-owned case coverage/metric coordinates/filter applicability; digest-CAS selector and coordinate choices re-inspect context; unresolved context/baseline never mint save handles or become runnable. | 2026-09-29: runtime catalog 2.6.0 registers both tools as authenticated-human read previews; 34 focused tests and Ruff/compileall pass. Coordinate choice advances to `needs_baseline` without expected values. Domain context/published-baseline resolution, eligible-handle minting and fresh external profile→bootstrap E2E remain open. | PARTIAL / NO-GO |
| `G-MCP-BASELINE-SELECTION` | P1 | 050 T049 | Agent proposes bounded curated baseline scope; operator review handle; server capture; observatory non-gating. | Server-side facts-only selection exists (`services/dashboard_testing/baseline_selection.py` + T087 tests) but no MCP tool; the operator review UI is contract-declared and **not implemented** (the service has no HTTP/MCP surface to bind). | OPEN (blocked on the MCP/HTTP surface) |
| `G-MCP-EXPLORATION` | P1 | 050 T050 | Read-only exploration_step in isolated workspace with CAS/receipts; reuses 044 session; no raw code or second Playwright stack. | Fixed one-shot exploration exists; interactive operation absent. | OPEN |
| `G-STORAGE-BUDGET` | P1 | 044/046 tiering | Measured hot/cold artifact volume and quota alarms meet the approved retention budget for 20-table dashboards. | Estimates only; no telemetry-based acceptance. | OPEN |

View File

@@ -3,6 +3,24 @@
> Статический + targeted-runtime checkpoint. Не является feature-completion.
> Правило: `[x]` только с текущим доказательством; `[~]` частичная реализация; `[ ]` отсутствует.
## T029a MCP profile update — 2026-09-29
External MCP scenario authoring now has bounded profile-preview handlers in
runtime catalog 2.6.0 with `dashboard:testing:READ`. The server re-inspects the selected dashboard and
environment, returns complete selected-case coverage, stable chart→dataset→metric
coordinates and native-filter applicability, and keeps unresolved questions
typed. Selector answers are restricted to one exact compiler step and guarded by
profile-digest CAS. Metric-coordinate selection changes `needs_metric` to
`needs_baseline`; it never supplies expected numeric truth or a baseline pin.
Bootstrap now checks selected environment and case coverage against digest-bound
server handles before any registry/revision/workspace write. Targeted evidence:
32 focused profile/compiler/handle/bootstrap/MCP tests passed; Ruff, compileall
and diff checks passed. This is **PARTIAL**, not production closure. Domain
context and published-baseline resolution, save-eligible handle minting, full
fresh external-client profile→bootstrap E2E and catalog/schema parity remain open.
Overall production status remains **NO-GO**.
## Current UX handoff
2026-09-24 architecture amendment: every new baseline is defined by an explicit same-dashboard Superset URL and its server-parsed immutable native-filter snapshot ([037 ReferenceUrl](037-superset-baseline-engine/contracts/reference-url.md)). This is a P0 production invariant, tracked as `G-BSL-REFERENCE-URL` in the unified matrix, 037 T091 and 050 T052. Both supplied ss-dev dashboard 10 URLs have live read-only parser/query evidence; the `native_filters_key=A4_rwhfQta0` link was rerun on 2026-09-25 with `platform=3DS`, 10/10 metrics executed and 6 distinct metric formulas suggested. A separate, removed-after-test Compose project with a dashboard 10 scenario and **test-only synthetic release** then proved browser URL entry → filter/metric preview → six draft candidates (Chromium 2/2, including a different-dashboard field error). Its isolated DB retained the exact URL hash and `platform IN ["3DS"]` snapshot across six captures. The ordinary local database still has no dashboard 10 scenario/release fixture; a later isolated Chromium pass proved source-byte verification, reasoned approval and local catalog materialization for one URL candidate, with the exact URL/filter snapshot retained in its revision. Git publication has no receipt; real release→Git publication→run pin→scheduled replay remains unproven. Do not infer the full invariant or analyst acceptance from the seeded 4/5 review score.