Files
ss-tools/backend/tests/test_mcp_initial_scenario_e2e.py

302 lines
18 KiB
Python

# #region Test.McpInitialScenarioE2E [C:5] [TYPE Module] [SEMANTICS test,mcp,scenario,bootstrap,automation,approval]
# @BRIEF T029c vertical proof for bootstrap, registry visibility, pinned starts, automation guards, and PROD gates.
# T029i/E2E-EXT-001 (ADR-0024): fully external chain — the AgentRun prerequisite is minted through the
# MCP `create_agent_run` catalog tool, so every prerequisite of register_draft_pack is obtained through
# the MCP surface itself with ZERO non-MCP seeding (the prior raw-ORM/service-boundary fixture is gone;
# the field run of 2026-09-07 proved why fixture-minted prerequisites mask unreachable products).
# @RELATION VERIFIES -> [McpServer.BootstrapAuthoringScenario]
# @RELATION VERIFIES -> [McpServer.ToolsAgentRun]
# @RELATION VERIFIES -> [ScenarioExecution.Runner.Start]
# @RELATION VERIFIES -> [ScenarioExecution.RunnerPlan.Derive]
# @RELATION VERIFIES -> [ScenarioExecution.Approval.CreateGate]
# @TEST_INVARIANT Bootstrap creates the first current revision from server-owned draft artifacts on a fresh registry.
# @TEST_INVARIANT A provenance-only bootstrap revision is refused by start_scenario_run before any run/gate row exists.
# @TEST_INVARIANT Human-step automation is rejected before ScenarioSchedule, ScenarioRun, or gate side effects.
# @TEST_INVARIANT Identical PROD MCP retries reuse exactly one durable approval gate.
# @TEST_INVARIANT MCPX-FR-027 external reachability -> VERIFIED_BY: create_agent_run over tools/call, no seed
from __future__ import annotations
import hashlib
import json
from pathlib import Path
from types import SimpleNamespace
from uuid import uuid4
import pytest
from sqlalchemy import create_engine, event
from sqlalchemy.orm import Session, sessionmaker
from src.mcp_server import server as mcp_server
import src.mcp_server.rbac_server as rbac_module
import src.mcp_server.tools_agent_run as agent_run_module
import src.mcp_server.tools_automation as automation_module
import src.mcp_server.tools_authoring as authoring_module
import src.mcp_server.tools_scenario as scenario_module
from src.mcp_server.server import _access_token_context
from src.models.agent_run import AgentRun, DraftArtifact
from src.models.mapping import Base
from src.models.scenario_approval import ActionApprovalGate
from src.models.scenario_automation import ScenarioSchedule
from src.models.scenario_registry import ScenarioRegistryEntry, ScenarioRevision
from src.models.scenario_run import ScenarioRun
from src.models.auth import Permission, Role, User
from src.core.auth.security import get_password_hash
from src.services.dashboard_testing.scenario.models import DashboardTestScenario
from src.services.dashboard_testing.scenario.templates import ACTION_REGISTRY_VERSION, action_registry_fingerprint, resolve_action_descriptor
_FIXTURE = Path(__file__).resolve().parent / "fixtures" / "dashboard_scenarios" / "scenario_valid.json"
# #region Test.McpInitialScenarioE2E.Fixture [C:4] [TYPE Function] [SEMANTICS test,mcp,sqlite,artifacts]
# @BRIEF Build an isolated shared SQLite database wired into every MCP seam used by the vertical
# (authoring/scenario/automation/rbac/agent-run) with real DraftStorage-backed artifacts.
@pytest.fixture
def isolated_mcp_db(tmp_path, monkeypatch):
engine = create_engine(f"sqlite:///{tmp_path / 'mcp.db'}", connect_args={"check_same_thread": False})
event.listen(engine, "connect", lambda connection, _: connection.execute("PRAGMA foreign_keys=ON"))
Base.metadata.create_all(engine)
factory = sessionmaker(bind=engine)
monkeypatch.setattr(authoring_module, "SessionLocal", factory)
monkeypatch.setattr(scenario_module, "SessionLocal", factory)
monkeypatch.setattr(automation_module, "SessionLocal", factory)
monkeypatch.setattr(rbac_module, "SessionLocal", factory)
monkeypatch.setattr(agent_run_module, "SessionLocal", factory)
monkeypatch.setenv("DRAFT_STORAGE_ROOT", str(tmp_path / "drafts"))
monkeypatch.setenv("HANDLE_STORAGE_ROOT", str(tmp_path / "handles"))
import src.services.agent_runs.artifacts as artifacts
import src.services.dashboard_testing.scenario.handles as handles
artifacts._draft_storage = None
handles._blob_store = None
try:
yield factory
finally:
artifacts._draft_storage = None
handles._blob_store = None
engine.dispose()
def _scenario() -> DashboardTestScenario:
"""The fixture graph (dashboard 80). The AgentRun prerequisite is NOT built here — it is minted
through the MCP catalog inside the test body (T029i/E2E-EXT-001, ADR-0024 test-honesty rule)."""
return DashboardTestScenario.model_validate(json.loads(_FIXTURE.read_text(encoding="utf-8")))
def _published_catalog_snapshot() -> dict:
"""Published 037 catalog snapshot for the fail-closed BaselineSelectionPin resolver (Front 5-R):
the fixture graph carries baseline refs, so a runnable start needs an explicit selector plus
published catalog bytes (same shared fixture as test_mcp_scenario_e2e)."""
refresh = json.loads(
(
Path(__file__).resolve().parents[2]
/ "specs" / "044-dashboard-scenario-execution" / "fixtures" / "production-contract-refresh.json"
).read_text(encoding="utf-8")
)
pin = refresh["baseline_pin"]
return {
"baseline_set_id": pin["baseline_set_id"],
"baseline_set_version": pin["baseline_set_version"],
"release_id": pin["release_id"],
"baseline_family": pin["baseline_family"],
"catalog_digest": pin["catalog_digest"],
"catalog_revision": refresh["catalog_revision"],
}
def _seed_principal(factory: sessionmaker, user_id: str) -> None:
role = Role(name=f"role-{user_id}", is_admin=False, permissions=[
Permission(resource="scenario", action="RUN"), Permission(resource="scenario", action="RUN_PROD"),
Permission(resource="scenario:automation", action="MANAGE"),
Permission(resource="dashboard:testing", action="WRITE"),
Permission(resource="dashboard:testing", action="EXECUTE"), # create_agent_run (T029i)
])
with factory() as db:
db.add(User(username=user_id, password_hash=get_password_hash("test"), is_active=True, roles=[role]))
db.commit()
def _unwrap(value):
if isinstance(value, tuple):
return _unwrap(value[1])
if isinstance(value, dict) and set(value) == {"result"}:
return _unwrap(value["result"])
return value
# #endregion Test.McpInitialScenarioE2E.Fixture
# #region Test.McpInitialScenarioE2E.Vertical [C:5] [TYPE Function] [SEMANTICS test,mcp,scenario,vertical]
# @BRIEF Exercise the complete T029c MCP flow without seeded registry rows and without any non-MCP
# prerequisite minting: create_agent_run → register_draft_pack → bootstrap → start/PROD/schedules.
@pytest.mark.asyncio
async def test_mcp_bootstrap_run_automation_and_prod_gate(isolated_mcp_db, monkeypatch):
user_id = f"t029c-{uuid4()}"
scenario = _scenario()
_seed_principal(isolated_mcp_db, user_id)
access = mcp_server.AccessToken(token="t", client_id="c", scopes=["mcp"], subject=user_id, claims={"principal_type": "user"})
token = _access_token_context.set(access)
server = mcp_server._build_probe_server()
# T029i/E2E-EXT-001 (ADR-0024): the AgentRun prerequisite is minted through the MCP catalog
# itself — the very surface an external client uses (field run 2026-09-07 blocker closure).
created_run = _unwrap(await server.call_tool("create_agent_run", {"request": {
"dashboard_id": 80, "environment_id": "env-dev",
"dashboard_name": "T029c fixture dashboard", "idempotency_key": f"t029c-run-{uuid4()}",
}}))
assert created_run["status"] == "ok", created_run
run_id = created_run["run_id"]
registered = _unwrap(await server.call_tool("register_draft_pack", {
"request": {"agent_run_id": run_id, "scenario": scenario.model_dump(mode="json")}
}))
assert registered["status"] == "save_eligible", registered
# T029h: fixture env (env-prod-01) is not configured in tests → fail-open marker.
assert registered["context_authority"] == "unverified"
initial = {
"idempotency_key": f"bootstrap-{uuid4()}", "title": "T029c scenario", "dashboard_id": 80,
"allowed_environment_ids": ["env-dev"], "selected_case_ids": [], "objective": "bootstrap",
"compiled_handle_id": registered["compiled_handle_id"],
"draft_pack_id": registered["draft_pack_handle_id"],
"draft_pack_digest": registered["draft_pack_digest"],
}
try:
boot = _unwrap(await server.call_tool("bootstrap_authoring_scenario", {"request": initial}))
scenario_id, revision_id = boot["scenario_id"], boot["revision_id"]
with isolated_mcp_db() as db:
entry = db.get(ScenarioRegistryEntry, scenario_id)
revision = db.get(ScenarioRevision, revision_id)
assert entry and entry.current_revision_id == revision_id
assert revision and revision.activation_status == "current"
# T029h: the fail-open register marker materializes as a server-owned graph field.
assert revision.graph_snapshot.get("context_authority") == "unverified"
# T029m: compiled steps carry no per-step identity, so the registry materialization
# stamps the dashboard_context target identity onto every step (live-binding admission).
stamped_steps = revision.graph_snapshot.get("steps") or []
assert stamped_steps and all(
step.get("environment_id") == "env-prod-01" and step.get("dashboard_id") == 80
for step in stamped_steps
)
visible = await server.call_tool("list_scenario_schedules", {"scenario_id": scenario_id})
assert _unwrap(visible) == []
environments = {"env-dev": SimpleNamespace(id="env-dev", stage="DEV", is_production=False), "prod": SimpleNamespace(id="prod", stage="PROD", is_production=True)}
manager = SimpleNamespace(get_environment=environments.get)
monkeypatch.setattr(scenario_module, "get_config_manager", lambda: manager)
monkeypatch.setattr(rbac_module, "get_config_manager", lambda: manager)
# Front 5-R: the bootstrap graph carries baseline refs, so the fail-closed resolver needs
# an explicit set/version selector and published catalog bytes before the ScenarioRun row.
catalog_snapshot = _published_catalog_snapshot()
monkeypatch.setattr(
"src.services.dashboard_testing.execution.start_run.load_published_catalog",
lambda injected=None: injected if injected is not None else catalog_snapshot,
)
# T029e: handle-first bootstrap materializes the canonical graph and server-owned action
# registry metadata, so the fresh current revision is directly runnable.
blocked = _unwrap(await server.call_tool("start_scenario_run", {"request": {
"scenario_id": scenario_id, "revision_id": revision_id,
"environment_id": "env-dev", "idempotency_key": f"run-blocked-{uuid4()}",
"baseline_set": "ss-prod-visual", "baseline_set_version": "1",
}}))
assert blocked["status"] == "queued", blocked
with isolated_mcp_db() as db:
assert db.query(ScenarioRun).filter(ScenarioRun.scenario_id == scenario_id).count() == 1
# T029h: the unverified context marker blocks a PROD start before any run/gate side effect.
prod_blocked = _unwrap(await server.call_tool("start_scenario_run", {"request": {
"scenario_id": scenario_id, "revision_id": revision_id,
"environment_id": "prod", "idempotency_key": f"prod-ctx-{uuid4()}",
"baseline_set": "ss-prod-visual", "baseline_set_version": "1",
}}))
assert prod_blocked["status"] == "blocked", prod_blocked
assert prod_blocked["error"] == "CONTEXT_AUTHORITY_REQUIRED_FOR_PROD"
with isolated_mcp_db() as db:
assert db.query(ScenarioRun).filter(ScenarioRun.scenario_id == scenario_id).count() == 1
assert db.query(ActionApprovalGate).filter(ActionApprovalGate.owner_type == "scenario_run").count() == 0
# A second materialized revision proves the normal promoted shape independently.
with isolated_mcp_db() as db:
executable_scenario_id = str(uuid4())
executable_revision = ScenarioRevision(
revision_id=str(uuid4()), scenario_id=executable_scenario_id, content_hash="e" * 64,
graph_snapshot={
"action_registry_version": ACTION_REGISTRY_VERSION,
"action_registry_hash": action_registry_fingerprint(),
"steps": [{
"logical_step_id": "open", "tool": "browser", "action": "open_dashboard",
"action_descriptor": resolve_action_descriptor(
tool="browser", action="open_dashboard",
registry_version=ACTION_REGISTRY_VERSION, registry_hash=action_registry_fingerprint(),
).snapshot(),
}],
"dependencies": [], "environment_ids": ["env-dev"],
},
execution_template_hash="", template_version="v1", schema_version=1,
compatibility_family="default", change_summary={}, created_by=user_id,
activation_status="current",
)
db.add(ScenarioRegistryEntry(
scenario_id=executable_scenario_id, scenario_key="t029c-exec", name="T029c executable",
dashboard_id=80, environment_ids=["env-dev"], owner_id=user_id, owner_username=user_id,
lifecycle_status="DRAFT", validation_status="valid",
current_revision_id=executable_revision.revision_id,
))
db.add(executable_revision)
executable_revision_id = executable_revision.revision_id
db.commit()
started = _unwrap(await server.call_tool("start_scenario_run", {"request": {
"scenario_id": executable_scenario_id, "revision_id": executable_revision_id,
"environment_id": "env-dev", "idempotency_key": f"run-{uuid4()}",
}}))
assert started["status"] == "queued", started
with isolated_mcp_db() as db:
row = db.get(ScenarioRun, started["run_id"])
assert row and row.scenario_revision_id == executable_revision_id
assert row.scenario_content_hash == "e" * 64
human = {"action_registry_version": ACTION_REGISTRY_VERSION, "action_registry_hash": action_registry_fingerprint(), "steps": [{"logical_step_id": "human", "tool": "human", "action": "human_checkpoint", "action_descriptor": resolve_action_descriptor(tool="human", action="human_checkpoint", registry_version=ACTION_REGISTRY_VERSION, registry_hash=action_registry_fingerprint()).snapshot()}], "dependencies": [], "environment_ids": ["env-dev"]}
human_scenario_id = str(uuid4())
human_revision = ScenarioRevision(revision_id=str(uuid4()), scenario_id=human_scenario_id, content_hash="h" * 64, graph_snapshot=human, execution_template_hash="", template_version="v1", schema_version=1, compatibility_family="default", change_summary={}, created_by=user_id, activation_status="current")
db.add(ScenarioRegistryEntry(scenario_id=human_scenario_id, scenario_key="t029c-human", name="T029c human", dashboard_id=80, environment_ids=["env-dev"], owner_id=user_id, owner_username=user_id, lifecycle_status="DRAFT", validation_status="valid", current_revision_id=human_revision.revision_id))
db.add(human_revision)
db.flush()
before = (db.query(ScenarioSchedule).count(), db.query(ScenarioRun).count(), db.query(ActionApprovalGate).count())
schedule_request = {"scenario_id": human_scenario_id, "idempotency_key": f"schedule-{uuid4()}", "environment_id": "env-dev", "revision_id": human_revision.revision_id, "revision_policy": "pinned", "cron_expr": "0 * * * *"}
db.commit()
with pytest.raises(Exception, match="AUTOMATION_INELIGIBLE_HUMAN_STEP"):
await server.call_tool("upsert_scenario_schedule", {"request": schedule_request})
with isolated_mcp_db() as db:
assert (db.query(ScenarioSchedule).count(), db.query(ScenarioRun).count(), db.query(ActionApprovalGate).count()) == before
db.query(ScenarioRegistryEntry).filter_by(scenario_id=human_scenario_id).delete()
db.query(ScenarioRevision).filter_by(scenario_id=human_scenario_id).delete()
db.commit()
prod_request = {"request": {"scenario_id": executable_scenario_id, "revision_id": executable_revision_id, "environment_id": "prod", "idempotency_key": f"prod-{uuid4()}"}}
first = _unwrap(await server.call_tool("start_scenario_run", prod_request))
retry = _unwrap(await server.call_tool("start_scenario_run", prod_request))
assert first["status"] == retry["status"] == "pending_approval"
with isolated_mcp_db() as db:
assert db.query(ActionApprovalGate).filter(ActionApprovalGate.owner_type == "scenario_run").count() == 1
assert db.query(ScenarioRun).filter(ScenarioRun.status == "pending_approval").count() == 1
finally:
_access_token_context.reset(token)
with isolated_mcp_db() as db:
db.query(ActionApprovalGate).delete()
db.query(ScenarioRun).delete()
db.query(ScenarioSchedule).delete()
db.query(ScenarioRevision).delete()
db.query(ScenarioRegistryEntry).delete()
db.query(DraftArtifact).delete()
db.query(AgentRun).delete()
user = db.query(User).filter(User.username == user_id).first()
if user is not None:
db.delete(user)
role = db.query(Role).filter(Role.name == f"role-{user_id}").first()
if role is not None:
db.delete(role)
db.commit()
# #endregion Test.McpInitialScenarioE2E.Vertical
# #endregion Test.McpInitialScenarioE2E