FR-019 (was a false [x]): MCP list_checkpoints + decide_checkpoint over the
044 CAS lifecycle (server-resolved pending checkpoint, decision_version CAS,
continue_after_human_decision, typed conflict/not_found); catalog scenario:RUN
and service_allowed=False — automation has no path to checkpoints.
tests/test_mcp_checkpoints.py: 3 passed.
FR-013/SC-007: client_credentials grant for machine clients — confidential DCR
(one-time client_secret, sha256-only via migration 0017_oauth_client_secret
with idempotent guard), signed identity-only service-principal tokens
(principal_type=service, aud=mcp, no refresh), McpTokenVerifier service
short-circuit, AS metadata grants/auth-methods, INSTALL.md §MCP-client docs.
tests/test_mcp_client_flow_http.py: full scripted-client flows over real HTTP
(machine: discovery→DCR→client_credentials→/mcp initialize→tools/list→call;
user: DCR→PKCE S256 authorize via Bearer web session→exchange→/mcp live-RBAC
listing). 2 passed; oauth suite 10 passed.
PRODUCTION DEFECTS fixed en route: call_tool argument-inspection gates parsed
only the FLAT shape while FastMCP delivers {"request": {...}} —
start_scenario_run was uncallable over MCP and the PROD-SQL terminal denial was
bypassable by the wrapped shape. Both gates now unwrap via _gate_arguments;
regression pinned by flat x wrapped PROD matrix (6 combos).
E2E-AUTH-001..003 release-gate rows closed with evidence: T028 chain gained
propose_test_plan (single-trace 001); T023 extended with post-activation
start_scenario_run on the canonical runner-shaped fixture graph, asserting the
queued run pins promoted revision_id + content_hash (003); 002 already proven.
Record honesty: tasks.md T025/T008/T008b/T032 downgraded at review time;
T025+T008 re-closed with executable evidence; T032/T008b/T005a remain open in
the round-2 queue (recorded in WORKSTATE checkpoint with the full P1/P2 list:
MCL intent repair, dispatch/sandbox EXPLORE traces, INV_6 tombstones,
SC-005 remnants, FR-010 versioning, depth/rate limits).
Evidence: combined MCP slice 95 passed; alembic single head 0017; ruff and
compileall clean; anchors balanced in all touched files.
223 lines
10 KiB
Python
223 lines
10 KiB
Python
# #region Test.McpCheckpoints [C:4] [TYPE Module] [SEMANTICS test,mcp,scenario,checkpoint,human,fr019]
|
|
# @BRIEF FR-019/T025 contract tests: human observation checkpoints over MCP with the 044 CAS
|
|
# semantics, and no automation path (service principals are catalog-denied).
|
|
# @RELATION VERIFIES -> [McpServer.OperationsTools.CheckpointTools]
|
|
# @RELATION BINDS_TO -> [Models.ScenarioExecution.HumanCheckpoint]
|
|
# @TEST_INVARIANT automation_no_path -> service principal sees and calls neither checkpoint tool.
|
|
# @TEST_EDGE stale_version -> CAS conflict envelope without consuming the checkpoint.
|
|
# @TEST_EDGE no_pending -> typed checkpoint_not_found.
|
|
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
import secrets
|
|
from datetime import UTC, datetime
|
|
from uuid import uuid4
|
|
|
|
os.environ.setdefault("AUTH_SECRET_KEY", "test-secret-key-for-mcp")
|
|
os.environ.setdefault("DATABASE_URL", "sqlite:////tmp/ss_tools_mcp_checkpoints_test.db")
|
|
|
|
import pytest
|
|
|
|
from src.mcp_server import server as mcp_server
|
|
from src.mcp_server.server import _access_token_context, _MCP_CATALOG_BY_NAME
|
|
from src.core.auth.security import get_password_hash
|
|
from src.core.database import SessionLocal
|
|
from src.models.auth import McpToolInvocationRecord, Permission, Role, User
|
|
from src.models.scenario_checkpoint import HumanCheckpoint
|
|
from src.models.scenario_registry import ScenarioRegistryEntry, ScenarioRevision
|
|
from src.models.scenario_run import ScenarioRun, ScenarioStepRun
|
|
|
|
|
|
def _unwrap(result):
|
|
return result[1] if isinstance(result, tuple) else result
|
|
|
|
|
|
def _seed_run() -> tuple[str, str, str]:
|
|
scenario_id = str(uuid4())
|
|
revision_id = str(uuid4())
|
|
run_id = str(uuid4())
|
|
now = datetime.now(UTC)
|
|
with SessionLocal() as db:
|
|
db.add(ScenarioRegistryEntry(
|
|
scenario_id=scenario_id, scenario_key=f"cp-{scenario_id[:8]}", name="checkpoint fixture",
|
|
dashboard_id=7, owner_id="owner", owner_username="owner",
|
|
current_revision_id=revision_id,
|
|
))
|
|
db.add(ScenarioRevision(
|
|
revision_id=revision_id, scenario_id=scenario_id, content_hash="a" * 64,
|
|
graph_snapshot={"schema_version": 1, "phases": ["setup"], "steps": [], "parameters": {}, "assertions": {}, "dependencies": []},
|
|
created_by="owner", activation_status="current", created_at=now,
|
|
))
|
|
db.flush()
|
|
db.add(ScenarioRun(
|
|
id=run_id, scenario_id=scenario_id, scenario_revision_id=revision_id,
|
|
scenario_content_hash="a" * 64, environment_id="env-dev", status="waiting_human",
|
|
phase="verify", parameter_bindings={}, target_snapshot={"environment_id": "env-dev"},
|
|
trigger_source="manual", idempotency_key=f"cp-{uuid4()}", runner_plan={}, created_at=now,
|
|
))
|
|
db.flush()
|
|
db.add(ScenarioStepRun(
|
|
id=str(uuid4()), run_id=run_id, logical_step_id="step-visual", step_position=0,
|
|
status="waiting_human", inputs_snapshot={}, outputs={}, artifact_refs=[], step_outcome={},
|
|
))
|
|
db.flush()
|
|
db.add(HumanCheckpoint(
|
|
id=str(uuid4()), run_id=run_id, logical_step_id="step-visual", status="pending",
|
|
decision_version=1, evidence_refs=["draft:run:ev1"], created_at=now,
|
|
))
|
|
db.add(HumanCheckpoint(
|
|
id=str(uuid4()), run_id=run_id, logical_step_id="step-earlier", status="decided",
|
|
decision_version=2, disposition="pass", evidence_refs=[], created_at=now, decided_at=now,
|
|
))
|
|
db.commit()
|
|
return scenario_id, revision_id, run_id
|
|
|
|
|
|
def _cleanup(scenario_id: str, run_id: str, principal: str | None, role_name: str | None) -> None:
|
|
with SessionLocal() as db:
|
|
db.query(HumanCheckpoint).filter(HumanCheckpoint.run_id == run_id).delete()
|
|
db.query(ScenarioRun).filter(ScenarioRun.id == run_id).delete()
|
|
db.query(ScenarioRevision).filter(ScenarioRevision.scenario_id == scenario_id).delete()
|
|
db.query(ScenarioRegistryEntry).filter(ScenarioRegistryEntry.scenario_id == scenario_id).delete()
|
|
if principal:
|
|
db.query(McpToolInvocationRecord).filter(McpToolInvocationRecord.subject == principal).delete()
|
|
user = db.query(User).filter(User.username == principal).first()
|
|
if user is not None:
|
|
db.delete(user)
|
|
db.flush()
|
|
if role_name:
|
|
role = db.query(Role).filter(Role.name == role_name).first()
|
|
if role is not None:
|
|
db.delete(role)
|
|
db.commit()
|
|
|
|
|
|
def _seed_runner_principal() -> tuple[str, str]:
|
|
suffix = secrets.token_hex(4)
|
|
username = f"cp-operator-{suffix}"
|
|
role_name = f"ScenarioRunner-{suffix}"
|
|
role = Role(name=role_name, is_admin=False, permissions=[Permission(resource="scenario", action="RUN")])
|
|
user = User(username=username, password_hash=get_password_hash("pw"), is_active=True, roles=[role])
|
|
with SessionLocal() as db:
|
|
db.add_all([user])
|
|
db.commit()
|
|
return username, role_name
|
|
|
|
|
|
def _set_principal(server, username: str):
|
|
access = mcp_server.AccessToken(
|
|
token="cp-token", client_id="cp-client", scopes=["mcp"],
|
|
subject=username, claims={"principal_type": "user"},
|
|
)
|
|
return _access_token_context.set(access)
|
|
|
|
|
|
# #region Test.McpCheckpoints.Catalog [C:3] [TYPE Function]
|
|
# @ingroup Test.McpCheckpoints
|
|
# @BRIEF Checkpoint tools are catalogued human-only with the scenario RUN permission.
|
|
def test_checkpoint_catalog_declarations() -> None:
|
|
for name in ("list_checkpoints", "decide_checkpoint"):
|
|
definition = _MCP_CATALOG_BY_NAME[name]
|
|
assert definition.permission == ("scenario", "RUN")
|
|
assert definition.service_allowed is False
|
|
assert _MCP_CATALOG_BY_NAME["decide_checkpoint"].risk_level == "guarded"
|
|
|
|
|
|
def test_service_principal_has_no_checkpoint_path() -> None:
|
|
server = mcp_server._build_probe_server()
|
|
token = _access_token_context.set(mcp_server.AccessToken(
|
|
token="svc", client_id="svc", scopes=["mcp", "mcp:read"],
|
|
subject="service", claims={"principal_type": "service"},
|
|
))
|
|
try:
|
|
assert server._can_use_tool("list_checkpoints") is False
|
|
assert server._can_use_tool("decide_checkpoint") is False
|
|
finally:
|
|
_access_token_context.reset(token)
|
|
# #endregion Test.McpCheckpoints.Catalog
|
|
|
|
|
|
# #region Test.McpCheckpoints.Flow [C:4] [TYPE Function]
|
|
# @ingroup Test.McpCheckpoints
|
|
# @BRIEF list/decide mirror the REST /human/decision contract with CAS conflicts typed.
|
|
def test_list_and_decide_checkpoint_flow(monkeypatch) -> None:
|
|
from src.services.dashboard_testing.execution import runner as runner_module
|
|
|
|
continued: list[tuple[str, str]] = []
|
|
monkeypatch.setattr(
|
|
runner_module,
|
|
"continue_after_human_decision",
|
|
lambda db, run_id, worker_id="": continued.append((run_id, worker_id)),
|
|
)
|
|
scenario_id, revision_id, run_id = _seed_run()
|
|
principal, role_name = _seed_runner_principal()
|
|
server = mcp_server._build_probe_server()
|
|
token = _set_principal(server, principal)
|
|
try:
|
|
import asyncio
|
|
|
|
loop = asyncio.new_event_loop()
|
|
listed = _unwrap(loop.run_until_complete(server.call_tool("list_checkpoints", {"run_id": run_id})))
|
|
assert listed["status"] == "ok"
|
|
pending = [cp for cp in listed["checkpoints"] if cp["status"] == "pending"]
|
|
assert len(pending) == 1
|
|
assert pending[0]["logical_step_id"] == "step-visual"
|
|
assert pending[0]["decision_version"] == 1
|
|
assert len(listed["checkpoints"]) == 2
|
|
|
|
decision = _unwrap(loop.run_until_complete(server.call_tool("decide_checkpoint", {"request": {
|
|
"run_id": run_id,
|
|
"disposition": "confirm",
|
|
"expected_version": 1,
|
|
"comment": "looks stable",
|
|
}})))
|
|
assert decision["status"] == "ok", decision
|
|
assert decision["disposition"] == "confirm"
|
|
assert decision["decision_version"] == 2
|
|
assert decision["checkpoint_status"] == "decided"
|
|
assert len(decision["decided_by"]) > 0
|
|
assert continued and continued[0][0] == run_id and continued[0][1].startswith("mcp-human-decision-")
|
|
with SessionLocal() as db:
|
|
step = db.query(ScenarioStepRun).filter(
|
|
ScenarioStepRun.run_id == run_id, ScenarioStepRun.logical_step_id == "step-visual"
|
|
).one()
|
|
assert step.status == "passed"
|
|
assert step.step_outcome["disposition"] == "confirm"
|
|
run = db.get(ScenarioRun, run_id)
|
|
assert run.status == "queued" and run.phase == "executing"
|
|
|
|
stale = _unwrap(loop.run_until_complete(server.call_tool("decide_checkpoint", {"request": {
|
|
"run_id": run_id, "disposition": "pass", "expected_version": 1,
|
|
}})))
|
|
assert stale["status"] == "not_found", stale
|
|
|
|
# Re-arm a pending checkpoint at version 5; an expected_version=1 decision must CAS-conflict.
|
|
with SessionLocal() as db:
|
|
db.add(HumanCheckpoint(
|
|
id=str(uuid4()), run_id=run_id, logical_step_id="step-rearmed", status="pending",
|
|
decision_version=5, evidence_refs=[], created_at=datetime.now(UTC),
|
|
))
|
|
db.commit()
|
|
conflict = _unwrap(loop.run_until_complete(server.call_tool("decide_checkpoint", {"request": {
|
|
"run_id": run_id, "disposition": "pass", "expected_version": 1,
|
|
}})))
|
|
assert conflict["status"] == "conflict", conflict
|
|
assert conflict["error"] == "stale checkpoint decision"
|
|
with SessionLocal() as db:
|
|
rearmed = db.query(HumanCheckpoint).filter(
|
|
HumanCheckpoint.run_id == run_id, HumanCheckpoint.logical_step_id == "step-rearmed"
|
|
).one()
|
|
assert rearmed.status == "pending" and rearmed.decision_version == 5
|
|
|
|
missing_run = _unwrap(loop.run_until_complete(server.call_tool("decide_checkpoint", {"request": {
|
|
"run_id": str(uuid4()), "disposition": "pass", "expected_version": 1,
|
|
}})))
|
|
assert missing_run == {"status": "not_found", "error": "checkpoint_not_found"}
|
|
finally:
|
|
_access_token_context.reset(token)
|
|
_cleanup(scenario_id, run_id, principal, role_name)
|
|
# #endregion Test.McpCheckpoints.Flow
|
|
|
|
# #endregion Test.McpCheckpoints
|