diff --git a/backend/src/api/routes/dashboard_testing/scenario_live_bindings.py b/backend/src/api/routes/dashboard_testing/scenario_live_bindings.py index 7d66b1efa..3a6a99348 100644 --- a/backend/src/api/routes/dashboard_testing/scenario_live_bindings.py +++ b/backend/src/api/routes/dashboard_testing/scenario_live_bindings.py @@ -124,7 +124,14 @@ def upsert_live_binding( query_model = DashboardQueryModel.model_validate(request.query_model_snapshot) except ValidationError: raise HTTPException(status_code=422, detail="LIVE_BINDING_QUERY_MODEL_INVALID") from None - if query_model.environment_id != binding.environment_id or query_model.dashboard_id != binding.dashboard_id: + if ( + query_model.environment_id != binding.environment_id + or query_model.dashboard_id != binding.dashboard_id + or query_model.query_model_fingerprint != binding.query_model_fingerprint + ): + # The runtime matcher requires the query-model fingerprint to equal the binding's + # (LiveBinding.Runtime._runtime_matches); accepting a mismatched pair would persist an + # enabled binding that can never dispatch a live Superset step. raise HTTPException(status_code=422, detail="LIVE_BINDING_QUERY_MODEL_MISMATCH") record = ScenarioLiveExecutionBindingConfig( enabled=request.enabled, diff --git a/backend/src/scripts/publish_catalog.py b/backend/src/scripts/publish_catalog.py index c60293ddd..f6b8776bc 100644 --- a/backend/src/scripts/publish_catalog.py +++ b/backend/src/scripts/publish_catalog.py @@ -172,6 +172,10 @@ def main(argv: list[str] | None = None) -> int: except PublishError as exc: print(json.dumps({"result": "failed", "code": exc.code, "detail": exc.detail})) return 1 + except httpx.HTTPError as exc: + # Transport failures (ConnectError/ReadTimeout/...) are not OSError; keep the typed envelope. + print(json.dumps({"result": "failed", "code": "PUBLISH_UNAVAILABLE", "detail": f"{type(exc).__name__}: {exc}"})) + return 1 except OSError as exc: print(json.dumps({"result": "failed", "code": "PUBLISH_CATALOG_UNREADABLE", "detail": str(exc)})) return 1 diff --git a/backend/src/services/dashboard_testing/execution/evaluation_adapter.py b/backend/src/services/dashboard_testing/execution/evaluation_adapter.py index dead91447..69ce6c01d 100644 --- a/backend/src/services/dashboard_testing/execution/evaluation_adapter.py +++ b/backend/src/services/dashboard_testing/execution/evaluation_adapter.py @@ -163,8 +163,10 @@ def _normalize_provider_response(raw: dict[str, Any], spec: AgentEvaluationSpec) payload = dict(raw) payload.pop("spec_id", None) - payload["evaluation_id"] = payload.get("evaluation_id") or str(_uuid.uuid4()) - payload["operation_id"] = payload.get("operation_id") or str(_uuid.uuid4()) + # Server-owned identity: a provider-supplied evaluation_id/operation_id must never survive + # (it could force an EVALUATION_ALREADY_EXISTS self-collision or forge operation linkage). + payload["evaluation_id"] = str(_uuid.uuid4()) + payload["operation_id"] = str(_uuid.uuid4()) payload["provider_id"] = spec.provider_id payload["provider_version"] = spec.provider_version payload["model_id"] = spec.model_id @@ -273,8 +275,8 @@ def _stamp_provenance( "started_at": payload.get("started_at") or now, "finished_at": payload.get("finished_at") or now, }) - payload.setdefault("evaluation_id", str(uuid.uuid4())) - payload.setdefault("operation_id", str(uuid.uuid4())) + payload["evaluation_id"] = str(uuid.uuid4()) + payload["operation_id"] = str(uuid.uuid4()) return payload # #endregion ScenarioExecution.EvaluationAdapter.Provenance diff --git a/backend/src/services/dashboard_testing/execution/evaluation_images.py b/backend/src/services/dashboard_testing/execution/evaluation_images.py index 809aae432..e26ca60ac 100644 --- a/backend/src/services/dashboard_testing/execution/evaluation_images.py +++ b/backend/src/services/dashboard_testing/execution/evaluation_images.py @@ -14,6 +14,7 @@ from __future__ import annotations from hashlib import sha256 from typing import Any +from .artifacts import is_valid_sha256 from .mime_sniff import sniff_mime IMAGE_CONTENT_TYPES = frozenset({"image/jpeg", "image/png", "image/webp"}) @@ -53,9 +54,10 @@ def image_payloads_from_manifest( continue if not isinstance(data, (bytes, bytearray)) or not data: continue - # Defense in depth: the loaded bytes must match the manifest digest before reaching the model. + # Defense in depth: the loaded bytes must match a valid manifest digest before reaching the + # model; a missing/malformed digest is rejected, never silently trusted. expected_digest = item.get("sha256") - if isinstance(expected_digest, str) and sha256(bytes(data)).hexdigest() != expected_digest.lower(): + if not is_valid_sha256(expected_digest) or sha256(bytes(data)).hexdigest() != str(expected_digest).lower(): continue content_type = sniff_mime(bytes(data)) if content_type not in IMAGE_CONTENT_TYPES: diff --git a/backend/src/services/dashboard_testing/execution/mime_sniff.py b/backend/src/services/dashboard_testing/execution/mime_sniff.py index 6904a864b..e9c32315c 100644 --- a/backend/src/services/dashboard_testing/execution/mime_sniff.py +++ b/backend/src/services/dashboard_testing/execution/mime_sniff.py @@ -4,8 +4,9 @@ # @RELATION CALLED_BY -> [ScenarioExecution.ArtifactContent] # @RELATION CALLED_BY -> [ScenarioExecution.ScreenshotProvider] # @RELATION CALLED_BY -> [ScenarioExecution.BrowserProvider] -# @INVARIANT Detection is byte-sniffing only (no path, name, or provider hint) so stored MIME can -# never drift from the actual bytes (e.g. a WebP archive mislabeled image/jpeg). +# @INVARIANT Detection is byte-sniffing only (no path, name, or provider hint). Callers that must +# store a non-null content type apply a nominal fallback ONLY for unrecognized bytes; the +# artifact-delivery path re-sniffs and rejects a bytes/label mismatch (production-chain). # @RATIONALE Extracted from ArtifactContent so evidence providers can stamp per-ref MIME from the # captured bytes without depending on the HTTP artifact-delivery service layer. # @REJECTED Trusting the provider's nominal format (screenshot→jpeg, browser→png) was rejected: the diff --git a/backend/src/services/dashboard_testing/execution/start_run.py b/backend/src/services/dashboard_testing/execution/start_run.py index da2b34249..903f41254 100644 --- a/backend/src/services/dashboard_testing/execution/start_run.py +++ b/backend/src/services/dashboard_testing/execution/start_run.py @@ -151,11 +151,14 @@ def _resolve_configured_live_binding( dashboard_ids = { step.get("dashboard_id") for step in (plan.get("steps") or []) - if isinstance(step, dict) and isinstance(step.get("dashboard_id"), int) + if isinstance(step, dict) + and isinstance(step.get("dashboard_id"), int) + and not isinstance(step.get("dashboard_id"), bool) } - # Compiled (MCP/bootstrap) graphs carry no per-step dashboard identity; the registry entry's - # dashboard is server-owned, so it is a lawful binding-match key (T029m live replay defect). - if scenario_dashboard_id is not None: + # Compiled (MCP/bootstrap) graphs carry no per-step dashboard identity; only then is the + # registry entry's server-owned dashboard a lawful binding-match key (T029m live replay defect). + # When the plan DOES declare step identity, that pinned set stays authoritative (no widening). + if not dashboard_ids and scenario_dashboard_id is not None: dashboard_ids.add(int(scenario_dashboard_id)) if not dashboard_ids: return None diff --git a/backend/src/services/dashboard_testing/registry/create.py b/backend/src/services/dashboard_testing/registry/create.py index 1886f31f6..9ee8c62fa 100644 --- a/backend/src/services/dashboard_testing/registry/create.py +++ b/backend/src/services/dashboard_testing/registry/create.py @@ -124,10 +124,12 @@ def create_scenario( "action_registry_hash": action_registry_fingerprint(), } # T029m live-replay defect: compiled steps carry no per-step target identity, so the - # runner's live-binding admission (target match) cannot fire. The verified - # dashboard_context is server-owned (context_authority evaluated at the register - # boundary), so stamping its identity onto every step is authoritative, and steps that - # already declare identity (legacy/authored graphs) are never overwritten. + # runner's live-binding admission (target match) cannot fire. Identity is taken from the + # scenario's declared dashboard_context and materialized server-side; for a VERIFIED pack + # that context is the server-owned live model (context_authority evaluated at the register + # boundary), while a fail-open `unverified`/legacy pack keeps its client-declared target — + # PROD is separately blocked for unverified contexts (CONTEXT_AUTHORITY_REQUIRED_FOR_PROD). + # Steps that already declare identity (authored graphs) are never overwritten. _context = graph.get("dashboard_context") or {} _identity = { "environment_id": _context.get("environment_id"), diff --git a/backend/tests/api/test_scenario_live_bindings_api.py b/backend/tests/api/test_scenario_live_bindings_api.py index 0ea13486e..8aadaa599 100644 --- a/backend/tests/api/test_scenario_live_bindings_api.py +++ b/backend/tests/api/test_scenario_live_bindings_api.py @@ -149,6 +149,12 @@ def test_put_rejects_invalid_and_mismatched_query_model(): json=_put_body(query_model_snapshot={**_QUERY_MODEL, "environment_id": "other-env"}), ) assert mismatched.status_code == 422 + + fingerprint_mismatch = env.client.put( + f"/api/scenario-live-bindings/{_BINDING_REF}", + json=_put_body(query_model_snapshot={**_QUERY_MODEL, "query_model_fingerprint": "z" * 64}), + ) + assert fingerprint_mismatch.status_code == 422 assert env.config_manager.get_config().settings.scenario_live_execution_bindings == [] diff --git a/backend/tests/scripts/test_publish_catalog.py b/backend/tests/scripts/test_publish_catalog.py index 20c7fe587..409c39594 100644 --- a/backend/tests/scripts/test_publish_catalog.py +++ b/backend/tests/scripts/test_publish_catalog.py @@ -161,5 +161,27 @@ def test_main_publishes_end_to_end_with_fake_client(monkeypatch, tmp_path): assert code == 0 assert seen["kwargs"]["base_url"] == "http://gitea.test" assert seen["kwargs"]["headers"] == {"Authorization": "token token-value"} +def test_main_types_transport_errors(monkeypatch, tmp_path, capsys): + import httpx + + import src.scripts.publish_catalog as module + + catalog_file = tmp_path / "catalog.json" + catalog_file.write_bytes(_RAW) + monkeypatch.setenv("PUBLISHED_CATALOG_GITEA_URL", "http://gitea.test") + monkeypatch.setenv("PUBLISHED_CATALOG_GITEA_TOKEN", "token-value") + monkeypatch.setenv("PUBLISHED_CATALOG_GITEA_REPO", "o/r") + + class _BoomClient(_FakeClient): + def __init__(self, **kwargs) -> None: + super().__init__(_FakeResponse(404), _FakeResponse(201, {})) + + def get(self, url, params=None): + raise httpx.ConnectError("boom") + + monkeypatch.setattr(module.httpx, "Client", lambda **kwargs: _BoomClient(**kwargs)) + code = main(["--catalog", str(catalog_file), "--baseline-set", "ss-prod-visual", "--version", "1"]) + assert code == 1 + assert "PUBLISH_UNAVAILABLE" in capsys.readouterr().out # #endregion Test.Tooling.PublishCatalog.Main # #endregion Test.Tooling.PublishCatalog diff --git a/backend/tests/services/dashboard_testing/execution/test_agent_evaluation.py b/backend/tests/services/dashboard_testing/execution/test_agent_evaluation.py index 9e6b39b5b..d6e0fc6b5 100644 --- a/backend/tests/services/dashboard_testing/execution/test_agent_evaluation.py +++ b/backend/tests/services/dashboard_testing/execution/test_agent_evaluation.py @@ -331,6 +331,30 @@ def test_adapter_drops_provider_supplied_baseline_pin(): # #endregion Test.ScenarioExecution.EvaluationAdapter.Identity +# #region Test.ScenarioExecution.EvaluationAdapter.ServerIdentity [C:2] [TYPE Function] +# @BRIEF Provider-supplied evaluation_id/operation_id are replaced by server-generated ids. +# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter.Normalize: provider text never authors +# evaluation/operation identity. -> VERIFIED_BY: test_adapter_replaces_provider_identity_ids +def test_adapter_replaces_provider_identity_ids(): + record = _run_adapter_with_response({ + "status": "succeeded", + "verdict": "inconclusive", + "confidence": 0.2, + "findings": [{ + "criterion_id": "crit-visual", + "evidence_artifact_ids": ["draft:run-norm:" + "e" * 64], + "message": "identity claim", + }], + "reason_codes": [], + "evaluation_id": "provider-chosen-eval", + "operation_id": "provider-chosen-op", + "usage": {"input_tokens": 1, "output_tokens": 1}, + }) + assert record["evaluation_id"] != "provider-chosen-eval" + assert record["operation_id"] != "provider-chosen-op" +# #endregion Test.ScenarioExecution.EvaluationAdapter.ServerIdentity + + # #region Test.ScenarioExecution.EvaluationAdapter.Empty [C:2] [TYPE Function] # @BRIEF Missing prior-step evidence is fail-closed; the adapter never synthesizes PASS. # @TEST_INVARIANT ScenarioExecution.EvaluationAdapter: empty completed fails closed. -> VERIFIED_BY: adapter_empty_completed_fail_closed @@ -559,5 +583,9 @@ def test_image_payloads_skip_digest_mismatch(): store = _RetrievingStore({"ref-a": _PNG_BYTES}) tampered = {**_image_manifest_item("ref-a", _PNG_BYTES), "sha256": "0" * 64} assert image_payloads_from_manifest([tampered], store, max_images=8) == [] + + missing_digest = {**_image_manifest_item("ref-a", _PNG_BYTES)} + missing_digest.pop("sha256") + assert image_payloads_from_manifest([missing_digest], store, max_images=8) == [] # #endregion Test.ScenarioExecution.EvaluationAdapter.Multimodal # #endregion Test.ScenarioExecution.AgentEvaluation \ No newline at end of file diff --git a/backend/tests/services/dashboard_testing/registry/test_start_run_server_authority.py b/backend/tests/services/dashboard_testing/registry/test_start_run_server_authority.py index 81dce8942..e4cabffb7 100644 --- a/backend/tests/services/dashboard_testing/registry/test_start_run_server_authority.py +++ b/backend/tests/services/dashboard_testing/registry/test_start_run_server_authority.py @@ -257,6 +257,22 @@ def test_resolution_uses_registry_dashboard_for_identity_less_steps(): config_manager=manager, scenario_dashboard_id=99, ) assert foreign is None + + # No widening: when the plan declares its own step identity, the registry dashboard is NOT unioned. + mismatched = _resolve_configured_live_binding( + environment_id=_BINDING_ENV, plan={"steps": [{"logical_step_id": "s", "dashboard_id": 5}]}, + principal_fingerprint=hashlib.sha256(_BINDING_ACTOR.encode()).hexdigest(), + config_manager=manager, scenario_dashboard_id=_BINDING_DASHBOARD, + ) + assert mismatched is None + + # A bool is not a dashboard id (True is an int subclass). + bool_step = _resolve_configured_live_binding( + environment_id=_BINDING_ENV, plan={"steps": [{"logical_step_id": "s", "dashboard_id": True}]}, + principal_fingerprint=hashlib.sha256(_BINDING_ACTOR.encode()).hexdigest(), + config_manager=manager, scenario_dashboard_id=_BINDING_DASHBOARD, + ) + assert bool_step is not None # falls back to the registry dashboard, since the bool step is invalid # #endregion Test.ScenarioExecution.StartRunServerAuthority.BindingMatrix diff --git a/backend/tests/test_mcp_initial_scenario_e2e.py b/backend/tests/test_mcp_initial_scenario_e2e.py index c0dc92614..0303e5c05 100644 --- a/backend/tests/test_mcp_initial_scenario_e2e.py +++ b/backend/tests/test_mcp_initial_scenario_e2e.py @@ -168,6 +168,13 @@ async def test_mcp_bootstrap_run_automation_and_prod_gate(isolated_mcp_db, monke assert revision and revision.activation_status == "current" # T029h: the fail-open register marker materializes as a server-owned graph field. assert revision.graph_snapshot.get("context_authority") == "unverified" + # T029m: compiled steps carry no per-step identity, so the registry materialization + # stamps the dashboard_context target identity onto every step (live-binding admission). + stamped_steps = revision.graph_snapshot.get("steps") or [] + assert stamped_steps and all( + step.get("environment_id") == "env-prod-01" and step.get("dashboard_id") == 80 + for step in stamped_steps + ) visible = await server.call_tool("list_scenario_schedules", {"scenario_id": scenario_id}) assert _unwrap(visible) == [] diff --git a/docs/2026-09-11-sales-prod-mcp-replay.md b/docs/2026-09-11-sales-prod-mcp-replay.md index 0e130e4df..030b494ba 100644 --- a/docs/2026-09-11-sales-prod-mcp-replay.md +++ b/docs/2026-09-11-sales-prod-mcp-replay.md @@ -75,3 +75,33 @@ remains blocked by the Gitea PAT (published catalog unavailable); `050 T045` and `E2E-EXT-002` (T029m) → **CLOSED** by this trace. Remaining MCP-surface work is limited to the PAT-blocked baseline-pinned variant (T045/T046-pin) and `Expected.threshold` (038 schema authority, separate architect decision). + +## Human-checkpoint loop (T029m formulation completeness) + +`live_mcp_replay.py`'s B01 graph contains no HumanStep, so the task's +`list_checkpoints`/`decide_checkpoint` element was exercised by a companion replay +`specs/044-dashboard-scenario-execution/prototype/live_mcp_human_loop.py` (same scripted client): + +| Stage | Result | +|---|---| +| `inspect_scenario` B05 with live derived capabilities | one step: `phase-5-B05-human_checkpoint-1` (`tool=human`) — the script refuses any graph with a non-human step, so no mutation case is started | +| validate → draft-pack → `register_draft_pack` | valid, save_eligible, `context_authority=verified` | +| bootstrap → `start_scenario_run` (ss-prod, manual) | `pending_approval` → approved → run `6dcb9b8b-4f96-4491-9480-a3c74a67ee42` | +| run state | `waiting_human` | +| MCP `list_checkpoints(run_id)` | pending checkpoint `35039c17-6469-4cdf-90a1-753a357c0e08`, `decision_version=1` | +| MCP `decide_checkpoint(confirm, expected_version=1)` | ok; checkpoint `decided`, `decision_version=2` (CAS consumed) | +| terminal | **passed** — the immutable 044 mapping `confirm → passed` names the human's conformance observation, not a synthesized PASS | + +This closes the aligned-label human loop live; combined with the B01 chain above, every element of +the T029m task text is now backed by a live trace. + +## Review residuals (recorded, not defects) + +- The compile input declared `screenshot=true` as a caller capability; `screenshot` is not a + server-derived dashboard fact, so `merge_capabilities` passes it through verbatim. The + derived-wins axis that FR-028 protects (verifiable facts never downgraded to `human_checkpoint`) + is intact and `capability_authority.status=derived`; the caller-key nuance belongs to the 038/T029k + authority-boundary backlog, not to this replay. +- `live_mcp_replay.py` now asserts the honest typed non-pass terminal (a `passed` terminal on the + B01 chain raises `UNEXPECTED_SYNTHETIC_PASS`) and asserts the idempotent retry reuses the same + `run_id`. diff --git a/specs/044-dashboard-scenario-execution/prototype/live_mcp_human_loop.py b/specs/044-dashboard-scenario-execution/prototype/live_mcp_human_loop.py new file mode 100644 index 000000000..54d205fdb --- /dev/null +++ b/specs/044-dashboard-scenario-execution/prototype/live_mcp_human_loop.py @@ -0,0 +1,208 @@ +"""T029m human-checkpoint loop — live external MCP replay of the waiting_human decision cycle. + +Completes the T029m acceptance (`... → list_checkpoints/decide_checkpoint human loop with aligned +labels`) that the B01 browser chain cannot exercise: a compiled HumanCheckpoint case drives the run +to `waiting_human`, then the external client lists the checkpoint and disposes it via the MCP CAS +tool, and the run continues to a terminal state. + +Reuses the proven scripted client from live_mcp_replay.py (same OAuth/PKCE + SSE transport); no new +client machinery. Fail-closed: no mutation case is ever started (only a graph whose steps are all +tool=human is accepted). +""" +import json +import os +import sys +import time +import uuid + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) + +from live_mcp_replay import ( # noqa: E402 + CHECKLIST_CATALOG_VERSION, + DASHBOARD_ID, + DASHBOARD_NAME, + ENVIRONMENT_ID, + POLL_TIMEOUT_SECONDS, + _TERMINAL_STATUSES, + _step_report, + LiveMcpReplay, + ReplayError, +) + +HUMAN_CASE_CANDIDATES = ("B05", "B06", "B07", "B08", "B09", "C01", "C03") +WAITING_HUMAN = "waiting_human" + + +# #region ScenarioExecution.LiveMcpHumanLoop.Chain [C:4] [TYPE Function] [SEMANTICS scenario,execution,mcp,human-checkpoint,replay] +# @ingroup ScenarioExecution +# @BRIEF Drive one HumanCheckpoint case to waiting_human, list_checkpoints, decide_checkpoint, finish. +# @PRE Live stand ready; the MCP admin principal has scenario RUN/RUN_PROD. +# @POST Returns the full evidence dict; every refusal raises ReplayError (never synthesized success). +# @INVARIANT Only a compiled graph whose every step is tool=human is started — a mutation-capable graph +# is rejected by the script before any run/gate exists. +# @REJECTED Auto-confirming without an explicit disposition was rejected — the loop proves the human +# decision surface, and `confirm` is the operator's aligned-label choice for conformance. +def run_human_loop() -> dict: + replay = LiveMcpReplay() + evidence: dict = {"base": replay.http.base_url, "environment_id": ENVIRONMENT_ID, "dashboard_id": DASHBOARD_ID} + + user_jwt = replay.login() + evidence["login"] = "ok" + replay.oauth_pkce_token(user_jwt) + replay.mcp("initialize", {"protocolVersion": "2025-06-18", "capabilities": {}, + "clientInfo": {"name": "t029m-human-loop", "version": "1.0"}}) + replay.mcp("notifications/initialized", notify=True) + evidence["tools_listed"] = len(replay.mcp("tools/list", {}).get("tools", [])) + + context = replay.call("inspect_dashboard_context", { + "request": {"environment_id": ENVIRONMENT_ID, "dashboard_id": DASHBOARD_ID}, + }) + if context.get("status") != "ok": + raise ReplayError(f"inspect_dashboard_context blocked: {json.dumps(context)[:300]}") + evidence["query_model_fingerprint"] = context["query_model_fingerprint"] + + agent_run = replay.call("create_agent_run", {"request": { + "dashboard_id": DASHBOARD_ID, "environment_id": ENVIRONMENT_ID, + "dashboard_name": DASHBOARD_NAME, "idempotency_key": f"t029m-human-agent-{uuid.uuid4().hex[:12]}", + }}) + if agent_run.get("status") != "ok": + raise ReplayError(f"create_agent_run blocked: {json.dumps(agent_run)[:300]}") + agent_run_id = agent_run["run_id"] + evidence["agent_run_id"] = agent_run_id + + capabilities = {**context["derived_capabilities"]["capabilities"], "screenshot": False, "baseline": False} + scenario, case_id = _compile_human_case(replay, context, agent_run_id, capabilities) + evidence["case_id"] = case_id + evidence["compiled_steps"] = [ + {"id": step["id"], "tool": step["tool"], "action": step["action"], "risk": step["risk"]} + for step in scenario["steps"] + ] + + validation = replay.call("validate_scenario", {"scenario": scenario}) + evidence["validation"] = {"valid": validation["valid"], "coverage": validation["coverage"]} + if not validation["valid"]: + raise ReplayError(f"human-loop scenario invalid: {json.dumps(validation['errors'])[:300]}") + draft = replay.call("generate_draft_pack", {"request": {"scenario": scenario}}) + if draft["status"] != "save_eligible": + raise ReplayError(f"human-loop draft pack not save-eligible: {json.dumps(draft)[:300]}") + registered = replay.call("register_draft_pack", {"request": {"agent_run_id": agent_run_id, "scenario": scenario}}) + if registered.get("status") != "save_eligible": + raise ReplayError(f"register_draft_pack blocked: {json.dumps(registered)[:300]}") + evidence["register"] = {"status": registered.get("status"), "context_authority": registered.get("context_authority")} + + boot = replay.call("bootstrap_authoring_scenario", {"request": { + "idempotency_key": f"t029m-human-bootstrap-{uuid.uuid4().hex[:12]}", + "title": "T029m human loop replay", "dashboard_id": DASHBOARD_ID, + "allowed_environment_ids": [ENVIRONMENT_ID], "selected_case_ids": [case_id], + "objective": "T029m live MCP human-checkpoint loop replay", + "compiled_handle_id": registered["compiled_handle_id"], + "draft_pack_id": registered["draft_pack_handle_id"], + "draft_pack_digest": registered["draft_pack_digest"], + }}) + evidence["bootstrap"] = {"scenario_id": boot["scenario_id"], "revision_id": boot["revision_id"]} + + started = replay.call("start_scenario_run", {"request": { + "scenario_id": boot["scenario_id"], "revision_id": boot["revision_id"], + "environment_id": ENVIRONMENT_ID, "idempotency_key": f"t029m-human-run-{uuid.uuid4().hex[:12]}", + }}) + evidence["start"] = {"status": started.get("status"), "run_id": started.get("run_id"), "error": started.get("error")} + if started.get("status") != "pending_approval": + raise ReplayError(f"human plan did not reach the PROD gate: {json.dumps(started)[:300]}") + run_id = started["run_id"] + + approve = replay.http.post(f"/api/scenario-runs/{run_id}/approval/decision", + json={"decision": "approve", "comment": "T029m human-loop replay"}) + if approve.status_code != 200: + raise ReplayError(f"PROD gate approval failed: {approve.status_code} {approve.text[:300]}") + evidence["approval"] = approve.json().get("status") + + status = _poll_for(replay, run_id, {WAITING_HUMAN}) + evidence["waiting_human"] = status + if status != WAITING_HUMAN: + raise ReplayError(f"run did not reach {WAITING_HUMAN}: {status}") + + checkpoints = replay.call("list_checkpoints", {"run_id": run_id}) + pending = [c for c in checkpoints.get("checkpoints", []) if c.get("status") == "pending"] + if checkpoints.get("status") != "ok" or not pending: + raise ReplayError(f"no pending checkpoint via MCP: {json.dumps(checkpoints)[:300]}") + target = pending[0] + evidence["checkpoint"] = {"id": target["id"], "logical_step_id": target["logical_step_id"], + "decision_version": target["decision_version"]} + + decided = replay.call("decide_checkpoint", {"request": { + "run_id": run_id, "disposition": "confirm", + "expected_version": int(target["decision_version"]), + "comment": "T029m live human-loop replay — conformance confirmed", + }}) + evidence["decision"] = {"status": decided.get("status"), "checkpoint_status": decided.get("checkpoint_status"), + "disposition": decided.get("disposition"), "decision_version": decided.get("decision_version")} + if decided.get("status") != "ok": + raise ReplayError(f"decide_checkpoint refused: {json.dumps(decided)[:300]}") + + terminal = _poll_for(replay, run_id, _TERMINAL_STATUSES) + detail = replay.http.get(f"/api/scenario-runs/{run_id}").json() + evidence["terminal"] = {"status": terminal, "phase": detail.get("phase"), "error_code": detail.get("error_code")} + evidence["steps"] = _step_report(run_id) + return evidence +# #endregion ScenarioExecution.LiveMcpHumanLoop.Chain + + +# #region ScenarioExecution.LiveMcpHumanLoop.CompileHuman [C:3] [TYPE Function] [SEMANTICS mcp,compile,human-checkpoint,guard] +# @ingroup ScenarioExecution +# @BRIEF Compile the first candidate case whose graph is entirely tool=human; refuse mutation graphs. +def _compile_human_case(replay: LiveMcpReplay, context: dict, agent_run_id: str, capabilities: dict) -> tuple[dict, str]: + for case_id in HUMAN_CASE_CANDIDATES: + compiled = replay.call("inspect_scenario", {"request": { + "agent_run_id": agent_run_id, + "objective": {"goal": f"t029m human loop {case_id}", "selected_case_ids": [case_id], + "rationale": "external human-checkpoint loop replay"}, + "query_model": context["query_model"], + "checklist_catalog_version": CHECKLIST_CATALOG_VERSION, + "baseline_version": "unpinned-live-replay", + "capabilities": capabilities, + "parameters": { + "row_key": {"type": "string", "required": False, "default": "replay-row"}, + "comment": {"type": "string", "required": False, "default": "replay"}, + "selector_hint": {"type": "selector_hint", "required": False, "default": "live-replay"}, + }, + "has_dataset_fields": bool(context["derived_capabilities"]["has_dataset_fields"]), + "environment_id": ENVIRONMENT_ID, "dashboard_id": DASHBOARD_ID, "dashboard_name": DASHBOARD_NAME, + }}) + steps = compiled.get("scenario", {}).get("steps") or [] + if steps and all(step.get("tool") == "human" for step in steps): + return compiled["scenario"], case_id + raise ReplayError("no candidate case compiled to a pure human-checkpoint graph") + + +def _poll_for(replay: LiveMcpReplay, run_id: str, targets: frozenset | set) -> str: + deadline = time.monotonic() + POLL_TIMEOUT_SECONDS + status = "queued" + while time.monotonic() < deadline: + status = replay.http.get(f"/api/scenario-runs/{run_id}").json().get("status", "unknown") + if status in targets: + return status + if status in _TERMINAL_STATUSES: + return status + time.sleep(5) + return status +# #endregion ScenarioExecution.LiveMcpHumanLoop.CompileHuman + + +# #region ScenarioExecution.LiveMcpHumanLoop.Main [C:2] [TYPE Function] [SEMANTICS replay,entrypoint,report] +# @BRIEF Entry point: run the human-checkpoint loop and dump the evidence JSON (fail-closed). +def main() -> None: + try: + evidence = run_human_loop() + evidence["result"] = "ok" + except ReplayError as exc: + evidence = {"result": "blocked", "error": str(exc)} + print("[t029m-human] RESULT") + print(json.dumps(evidence, indent=1, ensure_ascii=False, default=str)) + result_path = os.environ.get("SS_REPLAY_RESULT", "/tmp/kilo/live_mcp_human_loop_result.json") + with open(result_path, "w") as handle: + json.dump(evidence, handle, indent=1, default=str) + + +if __name__ == "__main__": + main() +# #endregion ScenarioExecution.LiveMcpHumanLoop.Main diff --git a/specs/044-dashboard-scenario-execution/prototype/live_mcp_replay.py b/specs/044-dashboard-scenario-execution/prototype/live_mcp_replay.py index eb8ac2932..adc4fed1d 100644 --- a/specs/044-dashboard-scenario-execution/prototype/live_mcp_replay.py +++ b/specs/044-dashboard-scenario-execution/prototype/live_mcp_replay.py @@ -23,6 +23,7 @@ tests/test_mcp_initial_scenario_e2e.py — only the transport is live (httpx + S import base64 from hashlib import sha256 import json +import os import secrets import sys import time @@ -31,22 +32,24 @@ import uuid from dotenv import load_dotenv -load_dotenv("/home/busya/dev/ss-tools/backend/.env") -sys.path.insert(0, "/home/busya/dev/ss-tools/backend") +_REPO_ROOT = os.path.dirname(os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))) +load_dotenv(os.path.join(_REPO_ROOT, "backend", ".env")) +sys.path.insert(0, os.path.join(_REPO_ROOT, "backend")) import httpx # noqa: E402 from src.core.database import SessionLocal # noqa: E402 from src.models.scenario_run import ScenarioStepRun # noqa: E402 -BASE = "http://127.0.0.1:8000" +BASE = os.environ.get("SS_REPLAY_BASE", "http://127.0.0.1:8000") ENVIRONMENT_ID = "ss-prod" DASHBOARD_ID = 11 DASHBOARD_NAME = "Sales Dashboard" -ACTOR = "admin" -ACTOR_PASSWORD = "admin" +ACTOR = os.environ.get("SS_REPLAY_USER", "admin") +ACTOR_PASSWORD = os.environ.get("SS_REPLAY_PASSWORD", "admin") CHECKLIST_CATALOG_VERSION = 1 POLL_TIMEOUT_SECONDS = 300 +_TERMINAL_STATUSES = frozenset({"passed", "failed", "blocked", "inconclusive", "cancelled"}) # #region ScenarioExecution.LiveMcpReplay.Client [C:5] [TYPE Class] [SEMANTICS mcp,oauth,pkce,sse,client] @@ -283,9 +286,13 @@ def run_chain() -> dict: "environment_id": ENVIRONMENT_ID, "idempotency_key": start_key, }}) evidence["start"] = {"status": started.get("status"), "run_id": started.get("run_id"), - "retry_status": retry.get("status"), "error": started.get("error")} + "retry_status": retry.get("status"), "retry_run_id": retry.get("run_id"), + "error": started.get("error")} if started.get("status") != "pending_approval" or retry.get("status") != "pending_approval": raise ReplayError(f"PROD start did not gate: {json.dumps({'first': started, 'retry': retry})[:400]}") + # Idempotency proof: the retry must reuse the same durable run/gate, not create a second one. + if started.get("run_id") != retry.get("run_id"): + raise ReplayError(f"idempotent retry created a different run: {started.get('run_id')} != {retry.get('run_id')}") run_id = started["run_id"] approve = replay.http.post(f"/api/scenario-runs/{run_id}/approval/decision", @@ -300,10 +307,18 @@ def run_chain() -> dict: while time.monotonic() < deadline: detail = replay.http.get(f"/api/scenario-runs/{run_id}").json() status = detail.get("status", "unknown") - if status in {"passed", "failed", "blocked", "inconclusive", "cancelled"}: + if status in _TERMINAL_STATUSES: break time.sleep(5) - evidence["terminal"] = {"status": status, "phase": detail.get("phase"), "error_code": detail.get("error_code")} + if status not in _TERMINAL_STATUSES: + raise ReplayError(f"run did not reach a terminal status within {POLL_TIMEOUT_SECONDS}s (last={status})") + # Fail-closed evidence guard: the compiled B01 graph's browser action (apply_native_filter) is not + # supported by the isolated transport, so this chain's honest terminal is a typed non-pass. A + # `passed` terminal would mean the unsupported step was silently synthesized into success. + if status == "passed": + raise ReplayError("UNEXPECTED_SYNTHETIC_PASS: the typed non-pass terminal became passed") + evidence["terminal"] = {"status": status, "phase": detail.get("phase"), "error_code": detail.get("error_code"), + "synthetic_pass_guard": "non_pass_confirmed"} evidence["steps"] = _step_report(run_id) return evidence @@ -333,7 +348,9 @@ def main() -> None: evidence = {"result": "blocked", "error": str(exc)} print("[t029m] RESULT") print(json.dumps(evidence, indent=1, ensure_ascii=False, default=str)) - json.dump(evidence, open("/tmp/kilo/live_mcp_replay_result.json", "w"), indent=1, default=str) + result_path = os.environ.get("SS_REPLAY_RESULT", "/tmp/kilo/live_mcp_replay_result.json") + with open(result_path, "w") as handle: + json.dump(evidence, handle, indent=1, default=str) if __name__ == "__main__": diff --git a/specs/044-dashboard-scenario-execution/quickstart.md b/specs/044-dashboard-scenario-execution/quickstart.md index a6fc44183..8c2b33383 100644 --- a/specs/044-dashboard-scenario-execution/quickstart.md +++ b/specs/044-dashboard-scenario-execution/quickstart.md @@ -26,7 +26,7 @@ | Visual chain | Compiler emits browser→capture→comparison→evaluation→policy (Slice E) | One-action templates; live Playwright canary PASSED 2026-09-10 (browser `open_dashboard` + screenshot capture vs ss-prod dashboard 11; comparison/evaluation steps not in the canary graph) | | Live providers | Registered browser/screenshot + multimodal LLM | Binding `ss-prod-d11-live-001` resolved **server-side at REST start** (no binding in the request body); `/api/ready`: browser/screenshot ready, bindings 1; canary v2 live trace: `docs/reports/agentic-runtime-live-canary-v2-2026-09-10.md` | -**Open live gaps (2026-09-10, for the next agent):** multimodal image attachment (prompts are text-only → visual verdicts stay inconclusive); `baseline_pin` in the persisted `AgentEvaluation` record is `{}` for baseline-backed plans; screenshot/browser evidence MIME is hardcoded (`image/jpeg`/`image/png`) and would mismatch a WebP archive; live baseline pin is blocked on a valid Gitea PAT. +**Open live gaps (2026-09-10, for the next agent):** ~~multimodal image attachment (prompts are text-only → visual verdicts stay inconclusive); `baseline_pin` in the persisted `AgentEvaluation` record is `{}` for baseline-backed plans; screenshot/browser evidence MIME is hardcoded (`image/jpeg`/`image/png`) and would mismatch a WebP archive~~ → all three CLOSED 2026-09-10/11 (commit `792bb125`): multimodal evidence attachment (`evaluation_images.py` + `_evaluation_messages`), server-authoritative `baseline_pin` stamping (`baseline_resolver.stamp_baseline_pin`, live canary v3 `9e6f59c0`: `verdict=pass`, visual explanation), per-ref MIME sniffing (`mime_sniff.py`). Still open: live baseline pin is blocked on a valid Gitea PAT (publisher ready: `backend/src/scripts/publish_catalog.py`). The former agent-driven product-UI flow (dashboard → `/agent` workspace → agent-generated scenario) is SUPERSEDED. 044 is a headless execution engine; agent interaction is external MCP (050). Product UI (045) is read-only evidence plus human approvals. diff --git a/specs/044-dashboard-scenario-execution/traceability.md b/specs/044-dashboard-scenario-execution/traceability.md index 58fcafb70..b9b30f89f 100644 --- a/specs/044-dashboard-scenario-execution/traceability.md +++ b/specs/044-dashboard-scenario-execution/traceability.md @@ -84,3 +84,17 @@ Historical rows above identify prior tests/code only; removed agent UI paths are | SCEX-FR-028; external-MCP-only UI | manual editor/review; read-only evidence | [production tasks](tasks.md) | No frontend agent prompt/chat/assistant editing/proposal generation/workspace/start/handoff routes or requests; human approval remains usable. | OPEN | Sources: [production gap](../../docs/reports/ss-prod-agentic-e2e-production-gap-2026-09-08.md), [coverage gap](../../docs/reports/ss-prod-agentic-e2e-spec-coverage-2026-09-08.md), [baseline gap](../../docs/reports/ss-prod-agentic-e2e-baseline-gap-2026-09-08.md). Spec schema/static checks prove contract structure only; live canary/runtime closure and optional approved performance baseline are not claimed. + +## Live-binding admin surface + catalog publisher — evidence 2026-09-11 + +Two deployment/operator seams added by commits `792bb125` + `d04927bd`; both are server-owned +surfaces outside the 050 MCP catalog (no spec row existed before this note): + +| Seam | Contract | Surface | Evidence | +|---|---|---|---| +| `Api.ScenarioLiveBindings` | `ScenarioExecution.LiveBinding` + `Core.ConfigManager` | admin REST `GET/PUT/PATCH /api/scenario-live-bindings` (`admin:settings`); validates `LiveExecutionBinding.from_snapshot` + `DashboardQueryModel` (env/dashboard/fingerprint) and writes `settings.scenario_live_execution_bindings` atomically | `backend/tests/api/test_scenario_live_bindings_api.py` (7); first live exercise: re-registered `ss-prod-d11-live-001` with the correct principal, then the T029m run adopted it server-side | +| `Tooling.PublishCatalog` | T045/D7 published-catalog source (`PUBLISHED_CATALOG_*`) | CLI `backend/src/scripts/publish_catalog.py` — pre-validates catalog bytes, then Gitea contents PUT (typed auth/conflict/unavailable) | `backend/tests/scripts/test_publish_catalog.py` (7, offline); publication itself blocked on a valid Gitea PAT | + +Cross-reference: the live external MCP replay that exercised the binding path end-to-end (and exposed +the identity-less-compiled-step defect) is `docs/2026-09-11-sales-prod-mcp-replay.md` (050 T029m / +`E2E-EXT-002`, CLOSED). diff --git a/specs/050-mcp-interface/spec.md b/specs/050-mcp-interface/spec.md index 206c7d580..c81158b7b 100644 --- a/specs/050-mcp-interface/spec.md +++ b/specs/050-mcp-interface/spec.md @@ -302,7 +302,8 @@ raw-ORM insert, which masked the gap (test-honesty rule, ADR-0024 §3). chain with zero non-MCP seeding and the reachability pin is hard-green (`E2E-EXT-001` CLOSED); MCPX-FR-028 landed as `ScenarioGraph.CapabilityAuthority` (T029k, `CAP-001` CLOSED); MCPX-FR-029 landed as outcome-based labels/descriptions -(T029l, `DISP-001` CLOSED). Remaining: live-stand replay `E2E-EXT-002` (T029m). +(T029l, `DISP-001` CLOSED). Remaining: none for Phase 2d — the live-stand replay `E2E-EXT-002` +(T029m) CLOSED 2026-09-11 (see the gate row below). | Stage | Input | Authoritative owner | Output handle | Digest | Persistence | Failure / no-side-effect rule | Test id | |---|---|---|---|---|---|---|---| @@ -423,7 +424,7 @@ Required evidence rows produced by the live external run against ss-prod | Evidence row | Required trace | Evidence required | Status | |---|---|---|---| | `E2E-EXT-001` | `create_agent_run` -> `register_draft_pack` -> `bootstrap_authoring_scenario` -> `start_scenario_run` | External MCP-only chain on a fresh DB with ZERO non-MCP prerequisite seeding; the strict-xfail pin `tests/test_mcp_agent_run_reachability.py` flipped to green and unmarked (MCPX-FR-027, T029i) | `[x] CLOSED 2026-09-07` — `tests/test_mcp_initial_scenario_e2e.py` converted (AgentRun minted via MCP `create_agent_run` tools/call; no service/ORM seeding remains); reachability pin unmarked+green; `tests/test_mcp_agent_run_tools.py` (5) covers create/replay/read/RBAC; slice `70 passed`, full suite `11357 passed` | -| `E2E-EXT-002` | live stand replay of the 2026-09-07 sales run | `inspect_dashboard_context` -> derived-capability compile -> register -> bootstrap -> queued/gated run on ss-prod with retained evidence (T029m) | `[ ] OPEN` | +| `E2E-EXT-002` | live stand replay of the 2026-09-07 sales run | `inspect_dashboard_context` -> derived-capability compile -> `create_agent_run` -> register -> bootstrap -> queued/gated run on ss-prod + `list_checkpoints`/`decide_checkpoint` human loop, with retained evidence (T029m) | `[x] CLOSED 2026-09-11` — committed replay client `specs/044-dashboard-scenario-execution/prototype/live_mcp_replay.py` replayed the full external chain on the live stand (run `110a6517…`: inspect derived capabilities → create_agent_run → compile B01 → register `context_authority=verified` → bootstrap → PROD gate `pending_approval` (identical retry = one durable gate) → approval → live `capture_screenshot` passed (8 refs) + typed `BROWSER_ACTION_NOT_SUPPORTED` → honest `inconclusive`); the human loop was exercised live by `live_mcp_human_loop.py` (compiled B05 HumanCheckpoint → `waiting_human` → MCP `list_checkpoints` (v1) → `decide_checkpoint confirm` (v2 CAS) → terminal `passed`). Two fail-closed defects found and fixed (binding resolution for identity-less compiled steps; per-step target-identity stamping at bootstrap). Trace: `docs/2026-09-11-sales-prod-mcp-replay.md` | | `CAP-001` | `inspect_dashboard_context` -> compile capabilities | Server-derived capability map proven: verifiably-available dataset fields/native filters classify `automated`, not `human_checkpoint`; unsafe-mutation cases stay `human_checkpoint` (MCPX-FR-028, 038 amendment, T029k) | `[x] CLOSED 2026-09-07` — `ScenarioGraph.CapabilityAuthority` (derivation/derived-wins merge/boundary choke point) wired into MCP `inspect_scenario`+`inspect_dashboard_context` and REST `api_compile_scenario`; CAP-001 classification-fix test through `map_all` on the sales-shape fixture (B/T automated, C04–C06 unsupported, mutation cases legitimately human) + tool/REST parity pins; slice `67 passed` | | `DISP-001` | disposition label/vocabulary audit | RU/EN labels and MCP tool descriptions name the persisted outcome (`confirm`→passed); no defect-confirming wording or destructive styling on a passing choice; mapping pinned by tests (MCPX-FR-029, 044/045 amendment, T029l) | `[x] CLOSED 2026-09-07` — ru/en labels renamed to outcome-based wording; confirm buttons `bg-destructive`→`bg-primary` (HumanCheckpointPanel + WaitingForMeView); `decide_checkpoint` docstring carries the immutable outcome table; vitest DISP-001 style/label/dispatch pin; frontend `3507 passed`, lint 0 errors, build OK | | `TEST-001` | vertical-test honesty remediation | `test_mcp_initial_scenario_e2e.py::_pack()` creates the AgentRun through the production `create_agent_run` service boundary (no raw-ORM prerequisite seeding); reachability requirement pinned as `strict=True` xfail; typed zero-side-effect denial pinned (ADR-0024 §3, T029j) | `[x] CLOSED 2026-09-07` — `pytest tests/test_mcp_initial_scenario_e2e.py tests/test_mcp_agent_run_reachability.py` green (1 passed + 1 xfailed(strict) + denial pin); metadata records the open T029i gap. *(Superseded the same day by T029i closure: the strict-xfail flip ritual executed as designed — pin unmarked and hardened, vertical converted to the fully external chain; denial pin retained.)* | diff --git a/specs/050-mcp-interface/tasks.md b/specs/050-mcp-interface/tasks.md index 6b44ff302..9cdceed5d 100644 --- a/specs/050-mcp-interface/tasks.md +++ b/specs/050-mcp-interface/tasks.md @@ -70,7 +70,7 @@ - [x] T029j Test-honesty remediation (ADR-0024 §3, binding rule): vertical/E2E tests obtain every prerequisite through a boundary the principal under test can reach; raw-ORM seeding of a chain prerequisite in a test claiming external reachability is forbidden. Executed: `test_mcp_initial_scenario_e2e.py::_pack()` replaced the raw `AgentRun(...)` insert with the production `create_agent_run` service boundary with honest test metadata; new `tests/test_mcp_agent_run_reachability.py` pins (a) the external-reachability REQUIREMENT as `strict=True` xfail (flips the suite red the moment `create_agent_run` lands without spec follow-through) and (b) the current typed zero-side-effect denial (`DRAFT_PACK_ACCESS_DENIED`, no handle rows). Evidence: see `TEST-001` row in spec.md release gates (targeted pytest green 2026-09-07). *(Update 2026-09-07, T029i closure: строгий xfail-пин сработал как спроектирован — конвертирован в обычный requirement-тест (unmarked, green) вместе с конвертацией вертикали в fully external chain; denial-pin сохранён.)* - [x] T029k Context-authority-derived capability map (MCPX-FR-028; 038 amendment 2026-09-07): server-owned derivation of `capabilities`/`has_dataset_fields` from the authoritative `DashboardQueryModel` (the same live inspection `context_authority` binds), environment policy and provider readiness — exposed through `inspect_dashboard_context` (derived-capability section) and consumed by the persisted compile boundaries (MCP `scenario_compile` + REST `api_compile_scenario`); caller-declared capabilities remain accepted only as an explicit subset narrowing, never as an authority that downgrades verifiable facts into `human_checkpoint`; unresolved facts stay `needs_context`/`needs_selector`/`needs_baseline`; genuinely unsafe PROD mutation contexts keep `human_checkpoint`; HumanStep revisions remain automation-ineligible (SCEX-FR-004a stands). Evidence: `CAP-001` row + capability-derivation unit/E2E tests. Доказательство (2026-09-07): новый `src/services/dashboard_testing/scenario/capability_authority.py` (236 LOC, `ScenarioGraph.CapabilityAuthority`): `derive_capabilities` — только верифицируемые факты (native_filters; text_filter по STRING-фильтру; time_rollover по DATE/TIME/TIME_GRAIN; table_filter+pagination по executable table-viz; xlsx_export из capabilities-флага; dataset_field_read+has_dataset_fields по accessible dataset с колонками; browser — только при readiness-факте из T040-снимка composition-root), `NEVER_DERIVED` = row_edit/bulk_edit/persistence_refresh/safe_test_data/safe_clock_fixture/cross_dashboard (unsafe-mutation автоматизация метаданными невозможна — module @INVARIANT); `merge_capabilities` — derived-wins в ОБЕ стороны (declared-false не понижает проверяемую правду; declared-true не фабрикует) + overrides-аудит; legacy/unparseable payload → дословный caller-declared passthrough (register-time context_authority остаётся жёстким гейтом). Wiring: MCP `inspect_scenario` (+additive `capability_authority` секция), `inspect_dashboard_context` (+`derived_capabilities`), REST `api_compile_scenario` (parity через общий `build_capability_authority` choke point + additive секция). Тесты: фикстура `query_model_sales.json` (форма sales-стенда); unit truth-table `tests/services/dashboard_testing/scenario/test_capability_authority.py` (7, включая **CAP-001 classification-fix через `map_all`**: полевого shape декларации → B01–B04/T01–T03 `automated`, C04–C06 `unsupported`, B05–B09/C01–C03/C02/C07 легитимно `human_checkpoint`); tool-level `tests/test_mcp_capability_authority.py` (3: derived-wins overrides `["text_filter","xlsx_export"]`, legacy caller_declared, derived_capabilities в inspect-ответе); REST-parity pin `test_scenario_routes.py::test_capability_authority_derived_wins`. Браузер-readiness seam запинен monkeypatch (детерминизм против глобального composition-root состояния в full-suite). Gate: 5-файловый срез `67 passed`; полный suite `11357 passed`. - [x] T029l Disposition vocabulary clarity (MCPX-FR-029; 044/045 amendment 2026-09-07): the 044 lifecycle mapping (`confirm`→`passed`, `false_positive`/`inconclusive`→`inconclusive`, aliases `pass`/`fail`) is immutable and stays; human-facing wording aligns to the persisted outcome — RU labels «Подтвердить соответствие» / «Проблема не подтверждена» / «Недостаточно данных» (+ EN parity), confirm-button restyled from `bg-destructive` to a positive token, `decide_checkpoint`/`decide_approval` MCP tool descriptions and 045 monitor docs carry the outcome table, vitest pins updated (`WaitingForMeView.test.ts`, `HumanCheckpointPanel` tests). Evidence: `DISP-001` row. Доказательство (2026-09-07): `waiting_disposition_confirm/false_positive` в ru/en `dashboard-testing.json` именуют персистентный исход (confirm→passed: «Подтвердить соответствие»/"Confirm conformance"; false_positive: «Проблема не подтверждена»/"Issue not confirmed"; inconclusive без изменений); confirm-кнопки `HumanCheckpointPanel.svelte` и `WaitingForMeView.svelte` переведены `bg-destructive` → `bg-primary` (прецедент approve-кнопки ApprovalDecisionPanel) + @RATIONALE в контрактах компонентов; API-вокабуляр не переименован (continuity аудита/CAS); MCP `decide_checkpoint` docstring (видим в tools/list) несёт immutable outcome-mapping и правило «confirm = проверка пройдена, никогда не подтверждение дефекта»; vitest-пины обновлены (`RunMonitorViews.test.ts` — новый DISP-001 pin лейбла+стиля+dispatch "confirm", `WaitingForMeView.test.ts`, `run.ux.test.ts`). Lifecycle-маппинг не менялся — backend-тесты decision-пути зелёные без правок. Gate: frontend `3507 passed` (206 файлов), lint `0 errors / 364 warnings` (baseline), `npm run build` OK. -- [x] T029m Live-stand replay of the field run: after T029i/T029k/T029l, re-run the 2026-09-07 sales scenario externally against the live stand — `inspect_dashboard_context` → derived-capability compile/validate → `create_agent_run` → `register_draft_pack` → `bootstrap_authoring_scenario` → `start_scenario_run` (PROD gate observed, no unexpected high-risk approvals) → `list_checkpoints`/`decide_checkpoint` human loop with aligned labels; retain the run report beside `docs/2026-09-07-sales-prod-mcp-run.md`. Evidence: `E2E-EXT-002` row. **Status (2026-09-11): CLOSED.** Committed replay client `specs/044-dashboard-scenario-execution/prototype/live_mcp_replay.py` (transport-only adaptation of the proven scripted OAuth/PKCE + tool-chain flows, no new client machinery) replayed the FULL external chain on the live stand with zero non-MCP seeding: inspect (derived capabilities) → `create_agent_run` `a6168c7d…` → compile B01 (capability_authority `derived`) → validate → draft-pack `save_eligible` → `register_draft_pack` (**context_authority verified**) → bootstrap (scenario `e2687f03…`, current revision) → PROD start `pending_approval` (identical retry = one durable gate) → approval → live execution to an honest typed terminal (`capture_screenshot` **passed** with 8 durable refs; `apply_native_filter` typed `BROWSER_ACTION_NOT_SUPPORTED` — no synthesized PASS). Full trace + two fail-closed defects the replay exposed and fixed (binding resolution for identity-less compiled steps; per-step target-identity stamping at bootstrap materialization) + D5 binding-admin live exercise: `docs/2026-09-11-sales-prod-mcp-replay.md`. The baseline-pinned variant remains blocked on the Gitea PAT (050 T045 / T046-pin). +- [x] T029m Live-stand replay of the field run: after T029i/T029k/T029l, re-run the 2026-09-07 sales scenario externally against the live stand — `inspect_dashboard_context` → derived-capability compile/validate → `create_agent_run` → `register_draft_pack` → `bootstrap_authoring_scenario` → `start_scenario_run` (PROD gate observed, no unexpected high-risk approvals) → `list_checkpoints`/`decide_checkpoint` human loop with aligned labels; retain the run report beside `docs/2026-09-07-sales-prod-mcp-run.md`. Evidence: `E2E-EXT-002` row. **Status (2026-09-11): CLOSED.** Committed replay client `specs/044-dashboard-scenario-execution/prototype/live_mcp_replay.py` (transport-only adaptation of the proven scripted OAuth/PKCE + tool-chain flows, no new client machinery) replayed the FULL external chain on the live stand with zero non-MCP seeding: inspect (derived capabilities) → `create_agent_run` `a6168c7d…` → compile B01 (capability_authority `derived`) → validate → draft-pack `save_eligible` → `register_draft_pack` (**context_authority verified**) → bootstrap (scenario `e2687f03…`, current revision) → PROD start `pending_approval` (identical retry = one durable gate) → approval → live execution to an honest typed terminal (`capture_screenshot` **passed** with 8 durable refs; `apply_native_filter` typed `BROWSER_ACTION_NOT_SUPPORTED` — no synthesized PASS). Full trace + two fail-closed defects the replay exposed and fixed (binding resolution for identity-less compiled steps; per-step target-identity stamping at bootstrap materialization) + D5 binding-admin live exercise: `docs/2026-09-11-sales-prod-mcp-replay.md`. The human-checkpoint loop was exercised live by the companion `live_mcp_human_loop.py` (compiled B05 HumanCheckpoint → PROD gate → `waiting_human` → MCP `list_checkpoints` decision_version 1 → `decide_checkpoint confirm` decision_version 2 CAS → terminal `passed`), so every element of this task's formulation is covered. The baseline-pinned variant remains blocked on the Gitea PAT (050 T045 / T046-pin). ## Phase 3 — Frontend decommission (flag-driven) @@ -93,7 +93,7 @@ Historical [x] rows above retain only their dated local/transport evidence; they Contract: [Contract-complete public parity](contracts/modules.md). - [ ] T044 [P0/P1/P2] REST/MCP lifecycle/read/auth errors and disabled automation validation are identical; service principal cannot decide human gate. Implement at the existing 050 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md. **Status (2026-09-11):** offline parity tests green; live REST-only canary v2 exercised start→gate→approve→dispatch (`4eebfab3`); the live MCP-external replay (T029m CLOSED, `docs/2026-09-11-sales-prod-mcp-replay.md`) exercised the same lifecycle through `/mcp` with the identical typed start/gate/terminal outcomes — remaining: dedicated REST-vs-MCP error-shape parity fixtures for this matrix. -- [ ] T045 [P0/P1/P2] consume/publish failures return typed errors/pending state with no legacy fallback; every prerequisite is externally MCP-reachable. Implement at the existing 050 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md. **Status (2026-09-10):** the caller-side published-catalog source returns typed `None`/fail-closed (commit `bbbd4ccf`); Gitea publication path BLOCKED on a valid PAT. No live publish failure canary yet. +- [ ] T045 [P0/P1/P2] consume/publish failures return typed errors/pending state with no legacy fallback; every prerequisite is externally MCP-reachable. Implement at the existing 050 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md. **Status (2026-09-11):** the caller-side published-catalog source returns typed `None`/fail-closed (commit `bbbd4ccf`); the publisher helper landed 2026-09-11 (`backend/src/scripts/publish_catalog.py`, pre-validate + Gitea contents PUT, typed auth/conflict/unavailable; `tests/scripts/test_publish_catalog.py` offline-green). Gitea publication path remains BLOCKED on a valid PAT. No live publish failure canary yet. - [ ] T046 [P0/P1/P2] Fresh external-client chain preserves authoritative context and complete baseline pin without raw ORM/REST repair; no frontend agent controls/routes/requests. Implement at the existing 050 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md. **Status (2026-09-11):** a fresh REST client chain (canary v2 `4eebfab3`) AND the fresh MCP-external chain (T029m CLOSED) both preserve the server-resolved binding and the verified authoritative context live (`context_authority=verified` at the register boundary; binding adopted server-side and exercised by live providers). The complete baseline pin is still blocked on the Gitea PAT. Frontend boundary for this package: manual CRUD/editor, human review/approval, monitoring and read-only evidence/evaluation only; all agent interaction is external MCP. No agent chat/prompt/assistant editing/proposal generation/workspace/start/handoff controls. Runtime removal is OPEN, not performed by this spec refresh. Optional approved performance baseline is outside scope.