fix(scenario): orthogonal review hardening — binding fingerprint guard, strict image digest, typed publisher errors, T029m human-loop closure

This commit is contained in:
2026-09-11 10:10:01 +03:00
parent 1d48625b10
commit 1411c03f9b
19 changed files with 401 additions and 31 deletions

View File

@@ -124,7 +124,14 @@ def upsert_live_binding(
query_model = DashboardQueryModel.model_validate(request.query_model_snapshot)
except ValidationError:
raise HTTPException(status_code=422, detail="LIVE_BINDING_QUERY_MODEL_INVALID") from None
if query_model.environment_id != binding.environment_id or query_model.dashboard_id != binding.dashboard_id:
if (
query_model.environment_id != binding.environment_id
or query_model.dashboard_id != binding.dashboard_id
or query_model.query_model_fingerprint != binding.query_model_fingerprint
):
# The runtime matcher requires the query-model fingerprint to equal the binding's
# (LiveBinding.Runtime._runtime_matches); accepting a mismatched pair would persist an
# enabled binding that can never dispatch a live Superset step.
raise HTTPException(status_code=422, detail="LIVE_BINDING_QUERY_MODEL_MISMATCH")
record = ScenarioLiveExecutionBindingConfig(
enabled=request.enabled,

View File

@@ -172,6 +172,10 @@ def main(argv: list[str] | None = None) -> int:
except PublishError as exc:
print(json.dumps({"result": "failed", "code": exc.code, "detail": exc.detail}))
return 1
except httpx.HTTPError as exc:
# Transport failures (ConnectError/ReadTimeout/...) are not OSError; keep the typed envelope.
print(json.dumps({"result": "failed", "code": "PUBLISH_UNAVAILABLE", "detail": f"{type(exc).__name__}: {exc}"}))
return 1
except OSError as exc:
print(json.dumps({"result": "failed", "code": "PUBLISH_CATALOG_UNREADABLE", "detail": str(exc)}))
return 1

View File

@@ -163,8 +163,10 @@ def _normalize_provider_response(raw: dict[str, Any], spec: AgentEvaluationSpec)
payload = dict(raw)
payload.pop("spec_id", None)
payload["evaluation_id"] = payload.get("evaluation_id") or str(_uuid.uuid4())
payload["operation_id"] = payload.get("operation_id") or str(_uuid.uuid4())
# Server-owned identity: a provider-supplied evaluation_id/operation_id must never survive
# (it could force an EVALUATION_ALREADY_EXISTS self-collision or forge operation linkage).
payload["evaluation_id"] = str(_uuid.uuid4())
payload["operation_id"] = str(_uuid.uuid4())
payload["provider_id"] = spec.provider_id
payload["provider_version"] = spec.provider_version
payload["model_id"] = spec.model_id
@@ -273,8 +275,8 @@ def _stamp_provenance(
"started_at": payload.get("started_at") or now,
"finished_at": payload.get("finished_at") or now,
})
payload.setdefault("evaluation_id", str(uuid.uuid4()))
payload.setdefault("operation_id", str(uuid.uuid4()))
payload["evaluation_id"] = str(uuid.uuid4())
payload["operation_id"] = str(uuid.uuid4())
return payload
# #endregion ScenarioExecution.EvaluationAdapter.Provenance

View File

@@ -14,6 +14,7 @@ from __future__ import annotations
from hashlib import sha256
from typing import Any
from .artifacts import is_valid_sha256
from .mime_sniff import sniff_mime
IMAGE_CONTENT_TYPES = frozenset({"image/jpeg", "image/png", "image/webp"})
@@ -53,9 +54,10 @@ def image_payloads_from_manifest(
continue
if not isinstance(data, (bytes, bytearray)) or not data:
continue
# Defense in depth: the loaded bytes must match the manifest digest before reaching the model.
# Defense in depth: the loaded bytes must match a valid manifest digest before reaching the
# model; a missing/malformed digest is rejected, never silently trusted.
expected_digest = item.get("sha256")
if isinstance(expected_digest, str) and sha256(bytes(data)).hexdigest() != expected_digest.lower():
if not is_valid_sha256(expected_digest) or sha256(bytes(data)).hexdigest() != str(expected_digest).lower():
continue
content_type = sniff_mime(bytes(data))
if content_type not in IMAGE_CONTENT_TYPES:

View File

@@ -4,8 +4,9 @@
# @RELATION CALLED_BY -> [ScenarioExecution.ArtifactContent]
# @RELATION CALLED_BY -> [ScenarioExecution.ScreenshotProvider]
# @RELATION CALLED_BY -> [ScenarioExecution.BrowserProvider]
# @INVARIANT Detection is byte-sniffing only (no path, name, or provider hint) so stored MIME can
# never drift from the actual bytes (e.g. a WebP archive mislabeled image/jpeg).
# @INVARIANT Detection is byte-sniffing only (no path, name, or provider hint). Callers that must
# store a non-null content type apply a nominal fallback ONLY for unrecognized bytes; the
# artifact-delivery path re-sniffs and rejects a bytes/label mismatch (production-chain).
# @RATIONALE Extracted from ArtifactContent so evidence providers can stamp per-ref MIME from the
# captured bytes without depending on the HTTP artifact-delivery service layer.
# @REJECTED Trusting the provider's nominal format (screenshot→jpeg, browser→png) was rejected: the

View File

@@ -151,11 +151,14 @@ def _resolve_configured_live_binding(
dashboard_ids = {
step.get("dashboard_id")
for step in (plan.get("steps") or [])
if isinstance(step, dict) and isinstance(step.get("dashboard_id"), int)
if isinstance(step, dict)
and isinstance(step.get("dashboard_id"), int)
and not isinstance(step.get("dashboard_id"), bool)
}
# Compiled (MCP/bootstrap) graphs carry no per-step dashboard identity; the registry entry's
# dashboard is server-owned, so it is a lawful binding-match key (T029m live replay defect).
if scenario_dashboard_id is not None:
# Compiled (MCP/bootstrap) graphs carry no per-step dashboard identity; only then is the
# registry entry's server-owned dashboard a lawful binding-match key (T029m live replay defect).
# When the plan DOES declare step identity, that pinned set stays authoritative (no widening).
if not dashboard_ids and scenario_dashboard_id is not None:
dashboard_ids.add(int(scenario_dashboard_id))
if not dashboard_ids:
return None

View File

@@ -124,10 +124,12 @@ def create_scenario(
"action_registry_hash": action_registry_fingerprint(),
}
# T029m live-replay defect: compiled steps carry no per-step target identity, so the
# runner's live-binding admission (target match) cannot fire. The verified
# dashboard_context is server-owned (context_authority evaluated at the register
# boundary), so stamping its identity onto every step is authoritative, and steps that
# already declare identity (legacy/authored graphs) are never overwritten.
# runner's live-binding admission (target match) cannot fire. Identity is taken from the
# scenario's declared dashboard_context and materialized server-side; for a VERIFIED pack
# that context is the server-owned live model (context_authority evaluated at the register
# boundary), while a fail-open `unverified`/legacy pack keeps its client-declared target —
# PROD is separately blocked for unverified contexts (CONTEXT_AUTHORITY_REQUIRED_FOR_PROD).
# Steps that already declare identity (authored graphs) are never overwritten.
_context = graph.get("dashboard_context") or {}
_identity = {
"environment_id": _context.get("environment_id"),

View File

@@ -149,6 +149,12 @@ def test_put_rejects_invalid_and_mismatched_query_model():
json=_put_body(query_model_snapshot={**_QUERY_MODEL, "environment_id": "other-env"}),
)
assert mismatched.status_code == 422
fingerprint_mismatch = env.client.put(
f"/api/scenario-live-bindings/{_BINDING_REF}",
json=_put_body(query_model_snapshot={**_QUERY_MODEL, "query_model_fingerprint": "z" * 64}),
)
assert fingerprint_mismatch.status_code == 422
assert env.config_manager.get_config().settings.scenario_live_execution_bindings == []

View File

@@ -161,5 +161,27 @@ def test_main_publishes_end_to_end_with_fake_client(monkeypatch, tmp_path):
assert code == 0
assert seen["kwargs"]["base_url"] == "http://gitea.test"
assert seen["kwargs"]["headers"] == {"Authorization": "token token-value"}
def test_main_types_transport_errors(monkeypatch, tmp_path, capsys):
import httpx
import src.scripts.publish_catalog as module
catalog_file = tmp_path / "catalog.json"
catalog_file.write_bytes(_RAW)
monkeypatch.setenv("PUBLISHED_CATALOG_GITEA_URL", "http://gitea.test")
monkeypatch.setenv("PUBLISHED_CATALOG_GITEA_TOKEN", "token-value")
monkeypatch.setenv("PUBLISHED_CATALOG_GITEA_REPO", "o/r")
class _BoomClient(_FakeClient):
def __init__(self, **kwargs) -> None:
super().__init__(_FakeResponse(404), _FakeResponse(201, {}))
def get(self, url, params=None):
raise httpx.ConnectError("boom")
monkeypatch.setattr(module.httpx, "Client", lambda **kwargs: _BoomClient(**kwargs))
code = main(["--catalog", str(catalog_file), "--baseline-set", "ss-prod-visual", "--version", "1"])
assert code == 1
assert "PUBLISH_UNAVAILABLE" in capsys.readouterr().out
# #endregion Test.Tooling.PublishCatalog.Main
# #endregion Test.Tooling.PublishCatalog

View File

@@ -331,6 +331,30 @@ def test_adapter_drops_provider_supplied_baseline_pin():
# #endregion Test.ScenarioExecution.EvaluationAdapter.Identity
# #region Test.ScenarioExecution.EvaluationAdapter.ServerIdentity [C:2] [TYPE Function]
# @BRIEF Provider-supplied evaluation_id/operation_id are replaced by server-generated ids.
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter.Normalize: provider text never authors
# evaluation/operation identity. -> VERIFIED_BY: test_adapter_replaces_provider_identity_ids
def test_adapter_replaces_provider_identity_ids():
record = _run_adapter_with_response({
"status": "succeeded",
"verdict": "inconclusive",
"confidence": 0.2,
"findings": [{
"criterion_id": "crit-visual",
"evidence_artifact_ids": ["draft:run-norm:" + "e" * 64],
"message": "identity claim",
}],
"reason_codes": [],
"evaluation_id": "provider-chosen-eval",
"operation_id": "provider-chosen-op",
"usage": {"input_tokens": 1, "output_tokens": 1},
})
assert record["evaluation_id"] != "provider-chosen-eval"
assert record["operation_id"] != "provider-chosen-op"
# #endregion Test.ScenarioExecution.EvaluationAdapter.ServerIdentity
# #region Test.ScenarioExecution.EvaluationAdapter.Empty [C:2] [TYPE Function]
# @BRIEF Missing prior-step evidence is fail-closed; the adapter never synthesizes PASS.
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter: empty completed fails closed. -> VERIFIED_BY: adapter_empty_completed_fail_closed
@@ -559,5 +583,9 @@ def test_image_payloads_skip_digest_mismatch():
store = _RetrievingStore({"ref-a": _PNG_BYTES})
tampered = {**_image_manifest_item("ref-a", _PNG_BYTES), "sha256": "0" * 64}
assert image_payloads_from_manifest([tampered], store, max_images=8) == []
missing_digest = {**_image_manifest_item("ref-a", _PNG_BYTES)}
missing_digest.pop("sha256")
assert image_payloads_from_manifest([missing_digest], store, max_images=8) == []
# #endregion Test.ScenarioExecution.EvaluationAdapter.Multimodal
# #endregion Test.ScenarioExecution.AgentEvaluation

View File

@@ -257,6 +257,22 @@ def test_resolution_uses_registry_dashboard_for_identity_less_steps():
config_manager=manager, scenario_dashboard_id=99,
)
assert foreign is None
# No widening: when the plan declares its own step identity, the registry dashboard is NOT unioned.
mismatched = _resolve_configured_live_binding(
environment_id=_BINDING_ENV, plan={"steps": [{"logical_step_id": "s", "dashboard_id": 5}]},
principal_fingerprint=hashlib.sha256(_BINDING_ACTOR.encode()).hexdigest(),
config_manager=manager, scenario_dashboard_id=_BINDING_DASHBOARD,
)
assert mismatched is None
# A bool is not a dashboard id (True is an int subclass).
bool_step = _resolve_configured_live_binding(
environment_id=_BINDING_ENV, plan={"steps": [{"logical_step_id": "s", "dashboard_id": True}]},
principal_fingerprint=hashlib.sha256(_BINDING_ACTOR.encode()).hexdigest(),
config_manager=manager, scenario_dashboard_id=_BINDING_DASHBOARD,
)
assert bool_step is not None # falls back to the registry dashboard, since the bool step is invalid
# #endregion Test.ScenarioExecution.StartRunServerAuthority.BindingMatrix

View File

@@ -168,6 +168,13 @@ async def test_mcp_bootstrap_run_automation_and_prod_gate(isolated_mcp_db, monke
assert revision and revision.activation_status == "current"
# T029h: the fail-open register marker materializes as a server-owned graph field.
assert revision.graph_snapshot.get("context_authority") == "unverified"
# T029m: compiled steps carry no per-step identity, so the registry materialization
# stamps the dashboard_context target identity onto every step (live-binding admission).
stamped_steps = revision.graph_snapshot.get("steps") or []
assert stamped_steps and all(
step.get("environment_id") == "env-prod-01" and step.get("dashboard_id") == 80
for step in stamped_steps
)
visible = await server.call_tool("list_scenario_schedules", {"scenario_id": scenario_id})
assert _unwrap(visible) == []

View File

@@ -75,3 +75,33 @@ remains blocked by the Gitea PAT (published catalog unavailable); `050 T045` and
`E2E-EXT-002` (T029m) → **CLOSED** by this trace. Remaining MCP-surface work is limited to the
PAT-blocked baseline-pinned variant (T045/T046-pin) and `Expected.threshold` (038 schema authority,
separate architect decision).
## Human-checkpoint loop (T029m formulation completeness)
`live_mcp_replay.py`'s B01 graph contains no HumanStep, so the task's
`list_checkpoints`/`decide_checkpoint` element was exercised by a companion replay
`specs/044-dashboard-scenario-execution/prototype/live_mcp_human_loop.py` (same scripted client):
| Stage | Result |
|---|---|
| `inspect_scenario` B05 with live derived capabilities | one step: `phase-5-B05-human_checkpoint-1` (`tool=human`) — the script refuses any graph with a non-human step, so no mutation case is started |
| validate → draft-pack → `register_draft_pack` | valid, save_eligible, `context_authority=verified` |
| bootstrap → `start_scenario_run` (ss-prod, manual) | `pending_approval` → approved → run `6dcb9b8b-4f96-4491-9480-a3c74a67ee42` |
| run state | `waiting_human` |
| MCP `list_checkpoints(run_id)` | pending checkpoint `35039c17-6469-4cdf-90a1-753a357c0e08`, `decision_version=1` |
| MCP `decide_checkpoint(confirm, expected_version=1)` | ok; checkpoint `decided`, `decision_version=2` (CAS consumed) |
| terminal | **passed** — the immutable 044 mapping `confirm → passed` names the human's conformance observation, not a synthesized PASS |
This closes the aligned-label human loop live; combined with the B01 chain above, every element of
the T029m task text is now backed by a live trace.
## Review residuals (recorded, not defects)
- The compile input declared `screenshot=true` as a caller capability; `screenshot` is not a
server-derived dashboard fact, so `merge_capabilities` passes it through verbatim. The
derived-wins axis that FR-028 protects (verifiable facts never downgraded to `human_checkpoint`)
is intact and `capability_authority.status=derived`; the caller-key nuance belongs to the 038/T029k
authority-boundary backlog, not to this replay.
- `live_mcp_replay.py` now asserts the honest typed non-pass terminal (a `passed` terminal on the
B01 chain raises `UNEXPECTED_SYNTHETIC_PASS`) and asserts the idempotent retry reuses the same
`run_id`.

View File

@@ -0,0 +1,208 @@
"""T029m human-checkpoint loop — live external MCP replay of the waiting_human decision cycle.
Completes the T029m acceptance (`... → list_checkpoints/decide_checkpoint human loop with aligned
labels`) that the B01 browser chain cannot exercise: a compiled HumanCheckpoint case drives the run
to `waiting_human`, then the external client lists the checkpoint and disposes it via the MCP CAS
tool, and the run continues to a terminal state.
Reuses the proven scripted client from live_mcp_replay.py (same OAuth/PKCE + SSE transport); no new
client machinery. Fail-closed: no mutation case is ever started (only a graph whose steps are all
tool=human is accepted).
"""
import json
import os
import sys
import time
import uuid
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from live_mcp_replay import ( # noqa: E402
CHECKLIST_CATALOG_VERSION,
DASHBOARD_ID,
DASHBOARD_NAME,
ENVIRONMENT_ID,
POLL_TIMEOUT_SECONDS,
_TERMINAL_STATUSES,
_step_report,
LiveMcpReplay,
ReplayError,
)
HUMAN_CASE_CANDIDATES = ("B05", "B06", "B07", "B08", "B09", "C01", "C03")
WAITING_HUMAN = "waiting_human"
# #region ScenarioExecution.LiveMcpHumanLoop.Chain [C:4] [TYPE Function] [SEMANTICS scenario,execution,mcp,human-checkpoint,replay]
# @ingroup ScenarioExecution
# @BRIEF Drive one HumanCheckpoint case to waiting_human, list_checkpoints, decide_checkpoint, finish.
# @PRE Live stand ready; the MCP admin principal has scenario RUN/RUN_PROD.
# @POST Returns the full evidence dict; every refusal raises ReplayError (never synthesized success).
# @INVARIANT Only a compiled graph whose every step is tool=human is started — a mutation-capable graph
# is rejected by the script before any run/gate exists.
# @REJECTED Auto-confirming without an explicit disposition was rejected — the loop proves the human
# decision surface, and `confirm` is the operator's aligned-label choice for conformance.
def run_human_loop() -> dict:
replay = LiveMcpReplay()
evidence: dict = {"base": replay.http.base_url, "environment_id": ENVIRONMENT_ID, "dashboard_id": DASHBOARD_ID}
user_jwt = replay.login()
evidence["login"] = "ok"
replay.oauth_pkce_token(user_jwt)
replay.mcp("initialize", {"protocolVersion": "2025-06-18", "capabilities": {},
"clientInfo": {"name": "t029m-human-loop", "version": "1.0"}})
replay.mcp("notifications/initialized", notify=True)
evidence["tools_listed"] = len(replay.mcp("tools/list", {}).get("tools", []))
context = replay.call("inspect_dashboard_context", {
"request": {"environment_id": ENVIRONMENT_ID, "dashboard_id": DASHBOARD_ID},
})
if context.get("status") != "ok":
raise ReplayError(f"inspect_dashboard_context blocked: {json.dumps(context)[:300]}")
evidence["query_model_fingerprint"] = context["query_model_fingerprint"]
agent_run = replay.call("create_agent_run", {"request": {
"dashboard_id": DASHBOARD_ID, "environment_id": ENVIRONMENT_ID,
"dashboard_name": DASHBOARD_NAME, "idempotency_key": f"t029m-human-agent-{uuid.uuid4().hex[:12]}",
}})
if agent_run.get("status") != "ok":
raise ReplayError(f"create_agent_run blocked: {json.dumps(agent_run)[:300]}")
agent_run_id = agent_run["run_id"]
evidence["agent_run_id"] = agent_run_id
capabilities = {**context["derived_capabilities"]["capabilities"], "screenshot": False, "baseline": False}
scenario, case_id = _compile_human_case(replay, context, agent_run_id, capabilities)
evidence["case_id"] = case_id
evidence["compiled_steps"] = [
{"id": step["id"], "tool": step["tool"], "action": step["action"], "risk": step["risk"]}
for step in scenario["steps"]
]
validation = replay.call("validate_scenario", {"scenario": scenario})
evidence["validation"] = {"valid": validation["valid"], "coverage": validation["coverage"]}
if not validation["valid"]:
raise ReplayError(f"human-loop scenario invalid: {json.dumps(validation['errors'])[:300]}")
draft = replay.call("generate_draft_pack", {"request": {"scenario": scenario}})
if draft["status"] != "save_eligible":
raise ReplayError(f"human-loop draft pack not save-eligible: {json.dumps(draft)[:300]}")
registered = replay.call("register_draft_pack", {"request": {"agent_run_id": agent_run_id, "scenario": scenario}})
if registered.get("status") != "save_eligible":
raise ReplayError(f"register_draft_pack blocked: {json.dumps(registered)[:300]}")
evidence["register"] = {"status": registered.get("status"), "context_authority": registered.get("context_authority")}
boot = replay.call("bootstrap_authoring_scenario", {"request": {
"idempotency_key": f"t029m-human-bootstrap-{uuid.uuid4().hex[:12]}",
"title": "T029m human loop replay", "dashboard_id": DASHBOARD_ID,
"allowed_environment_ids": [ENVIRONMENT_ID], "selected_case_ids": [case_id],
"objective": "T029m live MCP human-checkpoint loop replay",
"compiled_handle_id": registered["compiled_handle_id"],
"draft_pack_id": registered["draft_pack_handle_id"],
"draft_pack_digest": registered["draft_pack_digest"],
}})
evidence["bootstrap"] = {"scenario_id": boot["scenario_id"], "revision_id": boot["revision_id"]}
started = replay.call("start_scenario_run", {"request": {
"scenario_id": boot["scenario_id"], "revision_id": boot["revision_id"],
"environment_id": ENVIRONMENT_ID, "idempotency_key": f"t029m-human-run-{uuid.uuid4().hex[:12]}",
}})
evidence["start"] = {"status": started.get("status"), "run_id": started.get("run_id"), "error": started.get("error")}
if started.get("status") != "pending_approval":
raise ReplayError(f"human plan did not reach the PROD gate: {json.dumps(started)[:300]}")
run_id = started["run_id"]
approve = replay.http.post(f"/api/scenario-runs/{run_id}/approval/decision",
json={"decision": "approve", "comment": "T029m human-loop replay"})
if approve.status_code != 200:
raise ReplayError(f"PROD gate approval failed: {approve.status_code} {approve.text[:300]}")
evidence["approval"] = approve.json().get("status")
status = _poll_for(replay, run_id, {WAITING_HUMAN})
evidence["waiting_human"] = status
if status != WAITING_HUMAN:
raise ReplayError(f"run did not reach {WAITING_HUMAN}: {status}")
checkpoints = replay.call("list_checkpoints", {"run_id": run_id})
pending = [c for c in checkpoints.get("checkpoints", []) if c.get("status") == "pending"]
if checkpoints.get("status") != "ok" or not pending:
raise ReplayError(f"no pending checkpoint via MCP: {json.dumps(checkpoints)[:300]}")
target = pending[0]
evidence["checkpoint"] = {"id": target["id"], "logical_step_id": target["logical_step_id"],
"decision_version": target["decision_version"]}
decided = replay.call("decide_checkpoint", {"request": {
"run_id": run_id, "disposition": "confirm",
"expected_version": int(target["decision_version"]),
"comment": "T029m live human-loop replay — conformance confirmed",
}})
evidence["decision"] = {"status": decided.get("status"), "checkpoint_status": decided.get("checkpoint_status"),
"disposition": decided.get("disposition"), "decision_version": decided.get("decision_version")}
if decided.get("status") != "ok":
raise ReplayError(f"decide_checkpoint refused: {json.dumps(decided)[:300]}")
terminal = _poll_for(replay, run_id, _TERMINAL_STATUSES)
detail = replay.http.get(f"/api/scenario-runs/{run_id}").json()
evidence["terminal"] = {"status": terminal, "phase": detail.get("phase"), "error_code": detail.get("error_code")}
evidence["steps"] = _step_report(run_id)
return evidence
# #endregion ScenarioExecution.LiveMcpHumanLoop.Chain
# #region ScenarioExecution.LiveMcpHumanLoop.CompileHuman [C:3] [TYPE Function] [SEMANTICS mcp,compile,human-checkpoint,guard]
# @ingroup ScenarioExecution
# @BRIEF Compile the first candidate case whose graph is entirely tool=human; refuse mutation graphs.
def _compile_human_case(replay: LiveMcpReplay, context: dict, agent_run_id: str, capabilities: dict) -> tuple[dict, str]:
for case_id in HUMAN_CASE_CANDIDATES:
compiled = replay.call("inspect_scenario", {"request": {
"agent_run_id": agent_run_id,
"objective": {"goal": f"t029m human loop {case_id}", "selected_case_ids": [case_id],
"rationale": "external human-checkpoint loop replay"},
"query_model": context["query_model"],
"checklist_catalog_version": CHECKLIST_CATALOG_VERSION,
"baseline_version": "unpinned-live-replay",
"capabilities": capabilities,
"parameters": {
"row_key": {"type": "string", "required": False, "default": "replay-row"},
"comment": {"type": "string", "required": False, "default": "replay"},
"selector_hint": {"type": "selector_hint", "required": False, "default": "live-replay"},
},
"has_dataset_fields": bool(context["derived_capabilities"]["has_dataset_fields"]),
"environment_id": ENVIRONMENT_ID, "dashboard_id": DASHBOARD_ID, "dashboard_name": DASHBOARD_NAME,
}})
steps = compiled.get("scenario", {}).get("steps") or []
if steps and all(step.get("tool") == "human" for step in steps):
return compiled["scenario"], case_id
raise ReplayError("no candidate case compiled to a pure human-checkpoint graph")
def _poll_for(replay: LiveMcpReplay, run_id: str, targets: frozenset | set) -> str:
deadline = time.monotonic() + POLL_TIMEOUT_SECONDS
status = "queued"
while time.monotonic() < deadline:
status = replay.http.get(f"/api/scenario-runs/{run_id}").json().get("status", "unknown")
if status in targets:
return status
if status in _TERMINAL_STATUSES:
return status
time.sleep(5)
return status
# #endregion ScenarioExecution.LiveMcpHumanLoop.CompileHuman
# #region ScenarioExecution.LiveMcpHumanLoop.Main [C:2] [TYPE Function] [SEMANTICS replay,entrypoint,report]
# @BRIEF Entry point: run the human-checkpoint loop and dump the evidence JSON (fail-closed).
def main() -> None:
try:
evidence = run_human_loop()
evidence["result"] = "ok"
except ReplayError as exc:
evidence = {"result": "blocked", "error": str(exc)}
print("[t029m-human] RESULT")
print(json.dumps(evidence, indent=1, ensure_ascii=False, default=str))
result_path = os.environ.get("SS_REPLAY_RESULT", "/tmp/kilo/live_mcp_human_loop_result.json")
with open(result_path, "w") as handle:
json.dump(evidence, handle, indent=1, default=str)
if __name__ == "__main__":
main()
# #endregion ScenarioExecution.LiveMcpHumanLoop.Main

View File

@@ -23,6 +23,7 @@ tests/test_mcp_initial_scenario_e2e.py — only the transport is live (httpx + S
import base64
from hashlib import sha256
import json
import os
import secrets
import sys
import time
@@ -31,22 +32,24 @@ import uuid
from dotenv import load_dotenv
load_dotenv("/home/busya/dev/ss-tools/backend/.env")
sys.path.insert(0, "/home/busya/dev/ss-tools/backend")
_REPO_ROOT = os.path.dirname(os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))))
load_dotenv(os.path.join(_REPO_ROOT, "backend", ".env"))
sys.path.insert(0, os.path.join(_REPO_ROOT, "backend"))
import httpx # noqa: E402
from src.core.database import SessionLocal # noqa: E402
from src.models.scenario_run import ScenarioStepRun # noqa: E402
BASE = "http://127.0.0.1:8000"
BASE = os.environ.get("SS_REPLAY_BASE", "http://127.0.0.1:8000")
ENVIRONMENT_ID = "ss-prod"
DASHBOARD_ID = 11
DASHBOARD_NAME = "Sales Dashboard"
ACTOR = "admin"
ACTOR_PASSWORD = "admin"
ACTOR = os.environ.get("SS_REPLAY_USER", "admin")
ACTOR_PASSWORD = os.environ.get("SS_REPLAY_PASSWORD", "admin")
CHECKLIST_CATALOG_VERSION = 1
POLL_TIMEOUT_SECONDS = 300
_TERMINAL_STATUSES = frozenset({"passed", "failed", "blocked", "inconclusive", "cancelled"})
# #region ScenarioExecution.LiveMcpReplay.Client [C:5] [TYPE Class] [SEMANTICS mcp,oauth,pkce,sse,client]
@@ -283,9 +286,13 @@ def run_chain() -> dict:
"environment_id": ENVIRONMENT_ID, "idempotency_key": start_key,
}})
evidence["start"] = {"status": started.get("status"), "run_id": started.get("run_id"),
"retry_status": retry.get("status"), "error": started.get("error")}
"retry_status": retry.get("status"), "retry_run_id": retry.get("run_id"),
"error": started.get("error")}
if started.get("status") != "pending_approval" or retry.get("status") != "pending_approval":
raise ReplayError(f"PROD start did not gate: {json.dumps({'first': started, 'retry': retry})[:400]}")
# Idempotency proof: the retry must reuse the same durable run/gate, not create a second one.
if started.get("run_id") != retry.get("run_id"):
raise ReplayError(f"idempotent retry created a different run: {started.get('run_id')} != {retry.get('run_id')}")
run_id = started["run_id"]
approve = replay.http.post(f"/api/scenario-runs/{run_id}/approval/decision",
@@ -300,10 +307,18 @@ def run_chain() -> dict:
while time.monotonic() < deadline:
detail = replay.http.get(f"/api/scenario-runs/{run_id}").json()
status = detail.get("status", "unknown")
if status in {"passed", "failed", "blocked", "inconclusive", "cancelled"}:
if status in _TERMINAL_STATUSES:
break
time.sleep(5)
evidence["terminal"] = {"status": status, "phase": detail.get("phase"), "error_code": detail.get("error_code")}
if status not in _TERMINAL_STATUSES:
raise ReplayError(f"run did not reach a terminal status within {POLL_TIMEOUT_SECONDS}s (last={status})")
# Fail-closed evidence guard: the compiled B01 graph's browser action (apply_native_filter) is not
# supported by the isolated transport, so this chain's honest terminal is a typed non-pass. A
# `passed` terminal would mean the unsupported step was silently synthesized into success.
if status == "passed":
raise ReplayError("UNEXPECTED_SYNTHETIC_PASS: the typed non-pass terminal became passed")
evidence["terminal"] = {"status": status, "phase": detail.get("phase"), "error_code": detail.get("error_code"),
"synthetic_pass_guard": "non_pass_confirmed"}
evidence["steps"] = _step_report(run_id)
return evidence
@@ -333,7 +348,9 @@ def main() -> None:
evidence = {"result": "blocked", "error": str(exc)}
print("[t029m] RESULT")
print(json.dumps(evidence, indent=1, ensure_ascii=False, default=str))
json.dump(evidence, open("/tmp/kilo/live_mcp_replay_result.json", "w"), indent=1, default=str)
result_path = os.environ.get("SS_REPLAY_RESULT", "/tmp/kilo/live_mcp_replay_result.json")
with open(result_path, "w") as handle:
json.dump(evidence, handle, indent=1, default=str)
if __name__ == "__main__":

View File

@@ -26,7 +26,7 @@
| Visual chain | Compiler emits browser→capture→comparison→evaluation→policy (Slice E) | One-action templates; live Playwright canary PASSED 2026-09-10 (browser `open_dashboard` + screenshot capture vs ss-prod dashboard 11; comparison/evaluation steps not in the canary graph) |
| Live providers | Registered browser/screenshot + multimodal LLM | Binding `ss-prod-d11-live-001` resolved **server-side at REST start** (no binding in the request body); `/api/ready`: browser/screenshot ready, bindings 1; canary v2 live trace: `docs/reports/agentic-runtime-live-canary-v2-2026-09-10.md` |
**Open live gaps (2026-09-10, for the next agent):** multimodal image attachment (prompts are text-only → visual verdicts stay inconclusive); `baseline_pin` in the persisted `AgentEvaluation` record is `{}` for baseline-backed plans; screenshot/browser evidence MIME is hardcoded (`image/jpeg`/`image/png`) and would mismatch a WebP archive; live baseline pin is blocked on a valid Gitea PAT.
**Open live gaps (2026-09-10, for the next agent):** ~~multimodal image attachment (prompts are text-only → visual verdicts stay inconclusive); `baseline_pin` in the persisted `AgentEvaluation` record is `{}` for baseline-backed plans; screenshot/browser evidence MIME is hardcoded (`image/jpeg`/`image/png`) and would mismatch a WebP archive~~ → all three CLOSED 2026-09-10/11 (commit `792bb125`): multimodal evidence attachment (`evaluation_images.py` + `_evaluation_messages`), server-authoritative `baseline_pin` stamping (`baseline_resolver.stamp_baseline_pin`, live canary v3 `9e6f59c0`: `verdict=pass`, visual explanation), per-ref MIME sniffing (`mime_sniff.py`). Still open: live baseline pin is blocked on a valid Gitea PAT (publisher ready: `backend/src/scripts/publish_catalog.py`).
The former agent-driven product-UI flow (dashboard → `/agent` workspace → agent-generated scenario) is SUPERSEDED. 044 is a headless execution engine; agent interaction is external MCP (050). Product UI (045) is read-only evidence plus human approvals.

View File

@@ -84,3 +84,17 @@ Historical rows above identify prior tests/code only; removed agent UI paths are
| SCEX-FR-028; external-MCP-only UI | manual editor/review; read-only evidence | [production tasks](tasks.md) | No frontend agent prompt/chat/assistant editing/proposal generation/workspace/start/handoff routes or requests; human approval remains usable. | OPEN |
Sources: [production gap](../../docs/reports/ss-prod-agentic-e2e-production-gap-2026-09-08.md), [coverage gap](../../docs/reports/ss-prod-agentic-e2e-spec-coverage-2026-09-08.md), [baseline gap](../../docs/reports/ss-prod-agentic-e2e-baseline-gap-2026-09-08.md). Spec schema/static checks prove contract structure only; live canary/runtime closure and optional approved performance baseline are not claimed.
## Live-binding admin surface + catalog publisher — evidence 2026-09-11
Two deployment/operator seams added by commits `792bb125` + `d04927bd`; both are server-owned
surfaces outside the 050 MCP catalog (no spec row existed before this note):
| Seam | Contract | Surface | Evidence |
|---|---|---|---|
| `Api.ScenarioLiveBindings` | `ScenarioExecution.LiveBinding` + `Core.ConfigManager` | admin REST `GET/PUT/PATCH /api/scenario-live-bindings` (`admin:settings`); validates `LiveExecutionBinding.from_snapshot` + `DashboardQueryModel` (env/dashboard/fingerprint) and writes `settings.scenario_live_execution_bindings` atomically | `backend/tests/api/test_scenario_live_bindings_api.py` (7); first live exercise: re-registered `ss-prod-d11-live-001` with the correct principal, then the T029m run adopted it server-side |
| `Tooling.PublishCatalog` | T045/D7 published-catalog source (`PUBLISHED_CATALOG_*`) | CLI `backend/src/scripts/publish_catalog.py` — pre-validates catalog bytes, then Gitea contents PUT (typed auth/conflict/unavailable) | `backend/tests/scripts/test_publish_catalog.py` (7, offline); publication itself blocked on a valid Gitea PAT |
Cross-reference: the live external MCP replay that exercised the binding path end-to-end (and exposed
the identity-less-compiled-step defect) is `docs/2026-09-11-sales-prod-mcp-replay.md` (050 T029m /
`E2E-EXT-002`, CLOSED).

View File

@@ -302,7 +302,8 @@ raw-ORM insert, which masked the gap (test-honesty rule, ADR-0024 §3).
chain with zero non-MCP seeding and the reachability pin is hard-green
(`E2E-EXT-001` CLOSED); MCPX-FR-028 landed as `ScenarioGraph.CapabilityAuthority`
(T029k, `CAP-001` CLOSED); MCPX-FR-029 landed as outcome-based labels/descriptions
(T029l, `DISP-001` CLOSED). Remaining: live-stand replay `E2E-EXT-002` (T029m).
(T029l, `DISP-001` CLOSED). Remaining: none for Phase 2d — the live-stand replay `E2E-EXT-002`
(T029m) CLOSED 2026-09-11 (see the gate row below).
| Stage | Input | Authoritative owner | Output handle | Digest | Persistence | Failure / no-side-effect rule | Test id |
|---|---|---|---|---|---|---|---|
@@ -423,7 +424,7 @@ Required evidence rows produced by the live external run against ss-prod
| Evidence row | Required trace | Evidence required | Status |
|---|---|---|---|
| `E2E-EXT-001` | `create_agent_run` -> `register_draft_pack` -> `bootstrap_authoring_scenario` -> `start_scenario_run` | External MCP-only chain on a fresh DB with ZERO non-MCP prerequisite seeding; the strict-xfail pin `tests/test_mcp_agent_run_reachability.py` flipped to green and unmarked (MCPX-FR-027, T029i) | `[x] CLOSED 2026-09-07` — `tests/test_mcp_initial_scenario_e2e.py` converted (AgentRun minted via MCP `create_agent_run` tools/call; no service/ORM seeding remains); reachability pin unmarked+green; `tests/test_mcp_agent_run_tools.py` (5) covers create/replay/read/RBAC; slice `70 passed`, full suite `11357 passed` |
| `E2E-EXT-002` | live stand replay of the 2026-09-07 sales run | `inspect_dashboard_context` -> derived-capability compile -> register -> bootstrap -> queued/gated run on ss-prod with retained evidence (T029m) | `[ ] OPEN` |
| `E2E-EXT-002` | live stand replay of the 2026-09-07 sales run | `inspect_dashboard_context` -> derived-capability compile -> `create_agent_run` -> register -> bootstrap -> queued/gated run on ss-prod + `list_checkpoints`/`decide_checkpoint` human loop, with retained evidence (T029m) | `[x] CLOSED 2026-09-11` — committed replay client `specs/044-dashboard-scenario-execution/prototype/live_mcp_replay.py` replayed the full external chain on the live stand (run `110a6517…`: inspect derived capabilities → create_agent_run → compile B01 → register `context_authority=verified` → bootstrap → PROD gate `pending_approval` (identical retry = one durable gate) → approval → live `capture_screenshot` passed (8 refs) + typed `BROWSER_ACTION_NOT_SUPPORTED` → honest `inconclusive`); the human loop was exercised live by `live_mcp_human_loop.py` (compiled B05 HumanCheckpoint → `waiting_human` → MCP `list_checkpoints` (v1) → `decide_checkpoint confirm` (v2 CAS) → terminal `passed`). Two fail-closed defects found and fixed (binding resolution for identity-less compiled steps; per-step target-identity stamping at bootstrap). Trace: `docs/2026-09-11-sales-prod-mcp-replay.md` |
| `CAP-001` | `inspect_dashboard_context` -> compile capabilities | Server-derived capability map proven: verifiably-available dataset fields/native filters classify `automated`, not `human_checkpoint`; unsafe-mutation cases stay `human_checkpoint` (MCPX-FR-028, 038 amendment, T029k) | `[x] CLOSED 2026-09-07` — `ScenarioGraph.CapabilityAuthority` (derivation/derived-wins merge/boundary choke point) wired into MCP `inspect_scenario`+`inspect_dashboard_context` and REST `api_compile_scenario`; CAP-001 classification-fix test through `map_all` on the sales-shape fixture (B/T automated, C04–C06 unsupported, mutation cases legitimately human) + tool/REST parity pins; slice `67 passed` |
| `DISP-001` | disposition label/vocabulary audit | RU/EN labels and MCP tool descriptions name the persisted outcome (`confirm`→passed); no defect-confirming wording or destructive styling on a passing choice; mapping pinned by tests (MCPX-FR-029, 044/045 amendment, T029l) | `[x] CLOSED 2026-09-07` — ru/en labels renamed to outcome-based wording; confirm buttons `bg-destructive`→`bg-primary` (HumanCheckpointPanel + WaitingForMeView); `decide_checkpoint` docstring carries the immutable outcome table; vitest DISP-001 style/label/dispatch pin; frontend `3507 passed`, lint 0 errors, build OK |
| `TEST-001` | vertical-test honesty remediation | `test_mcp_initial_scenario_e2e.py::_pack()` creates the AgentRun through the production `create_agent_run` service boundary (no raw-ORM prerequisite seeding); reachability requirement pinned as `strict=True` xfail; typed zero-side-effect denial pinned (ADR-0024 §3, T029j) | `[x] CLOSED 2026-09-07` — `pytest tests/test_mcp_initial_scenario_e2e.py tests/test_mcp_agent_run_reachability.py` green (1 passed + 1 xfailed(strict) + denial pin); metadata records the open T029i gap. *(Superseded the same day by T029i closure: the strict-xfail flip ritual executed as designed — pin unmarked and hardened, vertical converted to the fully external chain; denial pin retained.)* |

View File

@@ -70,7 +70,7 @@
- [x] T029j Test-honesty remediation (ADR-0024 §3, binding rule): vertical/E2E tests obtain every prerequisite through a boundary the principal under test can reach; raw-ORM seeding of a chain prerequisite in a test claiming external reachability is forbidden. Executed: `test_mcp_initial_scenario_e2e.py::_pack()` replaced the raw `AgentRun(...)` insert with the production `create_agent_run` service boundary with honest test metadata; new `tests/test_mcp_agent_run_reachability.py` pins (a) the external-reachability REQUIREMENT as `strict=True` xfail (flips the suite red the moment `create_agent_run` lands without spec follow-through) and (b) the current typed zero-side-effect denial (`DRAFT_PACK_ACCESS_DENIED`, no handle rows). Evidence: see `TEST-001` row in spec.md release gates (targeted pytest green 2026-09-07). *(Update 2026-09-07, T029i closure: строгий xfail-пин сработал как спроектирован — конвертирован в обычный requirement-тест (unmarked, green) вместе с конвертацией вертикали в fully external chain; denial-pin сохранён.)*
- [x] T029k Context-authority-derived capability map (MCPX-FR-028; 038 amendment 2026-09-07): server-owned derivation of `capabilities`/`has_dataset_fields` from the authoritative `DashboardQueryModel` (the same live inspection `context_authority` binds), environment policy and provider readiness — exposed through `inspect_dashboard_context` (derived-capability section) and consumed by the persisted compile boundaries (MCP `scenario_compile` + REST `api_compile_scenario`); caller-declared capabilities remain accepted only as an explicit subset narrowing, never as an authority that downgrades verifiable facts into `human_checkpoint`; unresolved facts stay `needs_context`/`needs_selector`/`needs_baseline`; genuinely unsafe PROD mutation contexts keep `human_checkpoint`; HumanStep revisions remain automation-ineligible (SCEX-FR-004a stands). Evidence: `CAP-001` row + capability-derivation unit/E2E tests. Доказательство (2026-09-07): новый `src/services/dashboard_testing/scenario/capability_authority.py` (236 LOC, `ScenarioGraph.CapabilityAuthority`): `derive_capabilities` — только верифицируемые факты (native_filters; text_filter по STRING-фильтру; time_rollover по DATE/TIME/TIME_GRAIN; table_filter+pagination по executable table-viz; xlsx_export из capabilities-флага; dataset_field_read+has_dataset_fields по accessible dataset с колонками; browser — только при readiness-факте из T040-снимка composition-root), `NEVER_DERIVED` = row_edit/bulk_edit/persistence_refresh/safe_test_data/safe_clock_fixture/cross_dashboard (unsafe-mutation автоматизация метаданными невозможна — module @INVARIANT); `merge_capabilities` — derived-wins в ОБЕ стороны (declared-false не понижает проверяемую правду; declared-true не фабрикует) + overrides-аудит; legacy/unparseable payload → дословный caller-declared passthrough (register-time context_authority остаётся жёстким гейтом). Wiring: MCP `inspect_scenario` (+additive `capability_authority` секция), `inspect_dashboard_context` (+`derived_capabilities`), REST `api_compile_scenario` (parity через общий `build_capability_authority` choke point + additive секция). Тесты: фикстура `query_model_sales.json` (форма sales-стенда); unit truth-table `tests/services/dashboard_testing/scenario/test_capability_authority.py` (7, включая **CAP-001 classification-fix через `map_all`**: полевого shape декларации → B01–B04/T01–T03 `automated`, C04–C06 `unsupported`, B05–B09/C01–C03/C02/C07 легитимно `human_checkpoint`); tool-level `tests/test_mcp_capability_authority.py` (3: derived-wins overrides `["text_filter","xlsx_export"]`, legacy caller_declared, derived_capabilities в inspect-ответе); REST-parity pin `test_scenario_routes.py::test_capability_authority_derived_wins`. Браузер-readiness seam запинен monkeypatch (детерминизм против глобального composition-root состояния в full-suite). Gate: 5-файловый срез `67 passed`; полный suite `11357 passed`.
- [x] T029l Disposition vocabulary clarity (MCPX-FR-029; 044/045 amendment 2026-09-07): the 044 lifecycle mapping (`confirm`→`passed`, `false_positive`/`inconclusive`→`inconclusive`, aliases `pass`/`fail`) is immutable and stays; human-facing wording aligns to the persisted outcome — RU labels «Подтвердить соответствие» / «Проблема не подтверждена» / «Недостаточно данных» (+ EN parity), confirm-button restyled from `bg-destructive` to a positive token, `decide_checkpoint`/`decide_approval` MCP tool descriptions and 045 monitor docs carry the outcome table, vitest pins updated (`WaitingForMeView.test.ts`, `HumanCheckpointPanel` tests). Evidence: `DISP-001` row. Доказательство (2026-09-07): `waiting_disposition_confirm/false_positive` в ru/en `dashboard-testing.json` именуют персистентный исход (confirm→passed: «Подтвердить соответствие»/"Confirm conformance"; false_positive: «Проблема не подтверждена»/"Issue not confirmed"; inconclusive без изменений); confirm-кнопки `HumanCheckpointPanel.svelte` и `WaitingForMeView.svelte` переведены `bg-destructive` → `bg-primary` (прецедент approve-кнопки ApprovalDecisionPanel) + @RATIONALE в контрактах компонентов; API-вокабуляр не переименован (continuity аудита/CAS); MCP `decide_checkpoint` docstring (видим в tools/list) несёт immutable outcome-mapping и правило «confirm = проверка пройдена, никогда не подтверждение дефекта»; vitest-пины обновлены (`RunMonitorViews.test.ts` — новый DISP-001 pin лейбла+стиля+dispatch "confirm", `WaitingForMeView.test.ts`, `run.ux.test.ts`). Lifecycle-маппинг не менялся — backend-тесты decision-пути зелёные без правок. Gate: frontend `3507 passed` (206 файлов), lint `0 errors / 364 warnings` (baseline), `npm run build` OK.
- [x] T029m Live-stand replay of the field run: after T029i/T029k/T029l, re-run the 2026-09-07 sales scenario externally against the live stand — `inspect_dashboard_context` → derived-capability compile/validate → `create_agent_run` → `register_draft_pack` → `bootstrap_authoring_scenario` → `start_scenario_run` (PROD gate observed, no unexpected high-risk approvals) → `list_checkpoints`/`decide_checkpoint` human loop with aligned labels; retain the run report beside `docs/2026-09-07-sales-prod-mcp-run.md`. Evidence: `E2E-EXT-002` row. **Status (2026-09-11): CLOSED.** Committed replay client `specs/044-dashboard-scenario-execution/prototype/live_mcp_replay.py` (transport-only adaptation of the proven scripted OAuth/PKCE + tool-chain flows, no new client machinery) replayed the FULL external chain on the live stand with zero non-MCP seeding: inspect (derived capabilities) → `create_agent_run` `a6168c7d…` → compile B01 (capability_authority `derived`) → validate → draft-pack `save_eligible` → `register_draft_pack` (**context_authority verified**) → bootstrap (scenario `e2687f03…`, current revision) → PROD start `pending_approval` (identical retry = one durable gate) → approval → live execution to an honest typed terminal (`capture_screenshot` **passed** with 8 durable refs; `apply_native_filter` typed `BROWSER_ACTION_NOT_SUPPORTED` — no synthesized PASS). Full trace + two fail-closed defects the replay exposed and fixed (binding resolution for identity-less compiled steps; per-step target-identity stamping at bootstrap materialization) + D5 binding-admin live exercise: `docs/2026-09-11-sales-prod-mcp-replay.md`. The baseline-pinned variant remains blocked on the Gitea PAT (050 T045 / T046-pin).
- [x] T029m Live-stand replay of the field run: after T029i/T029k/T029l, re-run the 2026-09-07 sales scenario externally against the live stand — `inspect_dashboard_context` → derived-capability compile/validate → `create_agent_run` → `register_draft_pack` → `bootstrap_authoring_scenario` → `start_scenario_run` (PROD gate observed, no unexpected high-risk approvals) → `list_checkpoints`/`decide_checkpoint` human loop with aligned labels; retain the run report beside `docs/2026-09-07-sales-prod-mcp-run.md`. Evidence: `E2E-EXT-002` row. **Status (2026-09-11): CLOSED.** Committed replay client `specs/044-dashboard-scenario-execution/prototype/live_mcp_replay.py` (transport-only adaptation of the proven scripted OAuth/PKCE + tool-chain flows, no new client machinery) replayed the FULL external chain on the live stand with zero non-MCP seeding: inspect (derived capabilities) → `create_agent_run` `a6168c7d…` → compile B01 (capability_authority `derived`) → validate → draft-pack `save_eligible` → `register_draft_pack` (**context_authority verified**) → bootstrap (scenario `e2687f03…`, current revision) → PROD start `pending_approval` (identical retry = one durable gate) → approval → live execution to an honest typed terminal (`capture_screenshot` **passed** with 8 durable refs; `apply_native_filter` typed `BROWSER_ACTION_NOT_SUPPORTED` — no synthesized PASS). Full trace + two fail-closed defects the replay exposed and fixed (binding resolution for identity-less compiled steps; per-step target-identity stamping at bootstrap materialization) + D5 binding-admin live exercise: `docs/2026-09-11-sales-prod-mcp-replay.md`. The human-checkpoint loop was exercised live by the companion `live_mcp_human_loop.py` (compiled B05 HumanCheckpoint → PROD gate → `waiting_human` → MCP `list_checkpoints` decision_version 1 → `decide_checkpoint confirm` decision_version 2 CAS → terminal `passed`), so every element of this task's formulation is covered. The baseline-pinned variant remains blocked on the Gitea PAT (050 T045 / T046-pin).
## Phase 3 — Frontend decommission (flag-driven)
@@ -93,7 +93,7 @@ Historical [x] rows above retain only their dated local/transport evidence; they
Contract: [Contract-complete public parity](contracts/modules.md).
- [ ] T044 [P0/P1/P2] REST/MCP lifecycle/read/auth errors and disabled automation validation are identical; service principal cannot decide human gate. Implement at the existing 050 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md. **Status (2026-09-11):** offline parity tests green; live REST-only canary v2 exercised start→gate→approve→dispatch (`4eebfab3`); the live MCP-external replay (T029m CLOSED, `docs/2026-09-11-sales-prod-mcp-replay.md`) exercised the same lifecycle through `/mcp` with the identical typed start/gate/terminal outcomes — remaining: dedicated REST-vs-MCP error-shape parity fixtures for this matrix.
- [ ] T045 [P0/P1/P2] consume/publish failures return typed errors/pending state with no legacy fallback; every prerequisite is externally MCP-reachable. Implement at the existing 050 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md. **Status (2026-09-10):** the caller-side published-catalog source returns typed `None`/fail-closed (commit `bbbd4ccf`); Gitea publication path BLOCKED on a valid PAT. No live publish failure canary yet.
- [ ] T045 [P0/P1/P2] consume/publish failures return typed errors/pending state with no legacy fallback; every prerequisite is externally MCP-reachable. Implement at the existing 050 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md. **Status (2026-09-11):** the caller-side published-catalog source returns typed `None`/fail-closed (commit `bbbd4ccf`); the publisher helper landed 2026-09-11 (`backend/src/scripts/publish_catalog.py`, pre-validate + Gitea contents PUT, typed auth/conflict/unavailable; `tests/scripts/test_publish_catalog.py` offline-green). Gitea publication path remains BLOCKED on a valid PAT. No live publish failure canary yet.
- [ ] T046 [P0/P1/P2] Fresh external-client chain preserves authoritative context and complete baseline pin without raw ORM/REST repair; no frontend agent controls/routes/requests. Implement at the existing 050 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md. **Status (2026-09-11):** a fresh REST client chain (canary v2 `4eebfab3`) AND the fresh MCP-external chain (T029m CLOSED) both preserve the server-resolved binding and the verified authoritative context live (`context_authority=verified` at the register boundary; binding adopted server-side and exercised by live providers). The complete baseline pin is still blocked on the Gitea PAT.
Frontend boundary for this package: manual CRUD/editor, human review/approval, monitoring and read-only evidence/evaluation only; all agent interaction is external MCP. No agent chat/prompt/assistant editing/proposal generation/workspace/start/handoff controls. Runtime removal is OPEN, not performed by this spec refresh. Optional approved performance baseline is outside scope.