fix(scenario): orthogonal review hardening — binding fingerprint guard, strict image digest, typed publisher errors, T029m human-loop closure
This commit is contained in:
@@ -124,7 +124,14 @@ def upsert_live_binding(
|
||||
query_model = DashboardQueryModel.model_validate(request.query_model_snapshot)
|
||||
except ValidationError:
|
||||
raise HTTPException(status_code=422, detail="LIVE_BINDING_QUERY_MODEL_INVALID") from None
|
||||
if query_model.environment_id != binding.environment_id or query_model.dashboard_id != binding.dashboard_id:
|
||||
if (
|
||||
query_model.environment_id != binding.environment_id
|
||||
or query_model.dashboard_id != binding.dashboard_id
|
||||
or query_model.query_model_fingerprint != binding.query_model_fingerprint
|
||||
):
|
||||
# The runtime matcher requires the query-model fingerprint to equal the binding's
|
||||
# (LiveBinding.Runtime._runtime_matches); accepting a mismatched pair would persist an
|
||||
# enabled binding that can never dispatch a live Superset step.
|
||||
raise HTTPException(status_code=422, detail="LIVE_BINDING_QUERY_MODEL_MISMATCH")
|
||||
record = ScenarioLiveExecutionBindingConfig(
|
||||
enabled=request.enabled,
|
||||
|
||||
@@ -172,6 +172,10 @@ def main(argv: list[str] | None = None) -> int:
|
||||
except PublishError as exc:
|
||||
print(json.dumps({"result": "failed", "code": exc.code, "detail": exc.detail}))
|
||||
return 1
|
||||
except httpx.HTTPError as exc:
|
||||
# Transport failures (ConnectError/ReadTimeout/...) are not OSError; keep the typed envelope.
|
||||
print(json.dumps({"result": "failed", "code": "PUBLISH_UNAVAILABLE", "detail": f"{type(exc).__name__}: {exc}"}))
|
||||
return 1
|
||||
except OSError as exc:
|
||||
print(json.dumps({"result": "failed", "code": "PUBLISH_CATALOG_UNREADABLE", "detail": str(exc)}))
|
||||
return 1
|
||||
|
||||
@@ -163,8 +163,10 @@ def _normalize_provider_response(raw: dict[str, Any], spec: AgentEvaluationSpec)
|
||||
|
||||
payload = dict(raw)
|
||||
payload.pop("spec_id", None)
|
||||
payload["evaluation_id"] = payload.get("evaluation_id") or str(_uuid.uuid4())
|
||||
payload["operation_id"] = payload.get("operation_id") or str(_uuid.uuid4())
|
||||
# Server-owned identity: a provider-supplied evaluation_id/operation_id must never survive
|
||||
# (it could force an EVALUATION_ALREADY_EXISTS self-collision or forge operation linkage).
|
||||
payload["evaluation_id"] = str(_uuid.uuid4())
|
||||
payload["operation_id"] = str(_uuid.uuid4())
|
||||
payload["provider_id"] = spec.provider_id
|
||||
payload["provider_version"] = spec.provider_version
|
||||
payload["model_id"] = spec.model_id
|
||||
@@ -273,8 +275,8 @@ def _stamp_provenance(
|
||||
"started_at": payload.get("started_at") or now,
|
||||
"finished_at": payload.get("finished_at") or now,
|
||||
})
|
||||
payload.setdefault("evaluation_id", str(uuid.uuid4()))
|
||||
payload.setdefault("operation_id", str(uuid.uuid4()))
|
||||
payload["evaluation_id"] = str(uuid.uuid4())
|
||||
payload["operation_id"] = str(uuid.uuid4())
|
||||
return payload
|
||||
# #endregion ScenarioExecution.EvaluationAdapter.Provenance
|
||||
|
||||
|
||||
@@ -14,6 +14,7 @@ from __future__ import annotations
|
||||
from hashlib import sha256
|
||||
from typing import Any
|
||||
|
||||
from .artifacts import is_valid_sha256
|
||||
from .mime_sniff import sniff_mime
|
||||
|
||||
IMAGE_CONTENT_TYPES = frozenset({"image/jpeg", "image/png", "image/webp"})
|
||||
@@ -53,9 +54,10 @@ def image_payloads_from_manifest(
|
||||
continue
|
||||
if not isinstance(data, (bytes, bytearray)) or not data:
|
||||
continue
|
||||
# Defense in depth: the loaded bytes must match the manifest digest before reaching the model.
|
||||
# Defense in depth: the loaded bytes must match a valid manifest digest before reaching the
|
||||
# model; a missing/malformed digest is rejected, never silently trusted.
|
||||
expected_digest = item.get("sha256")
|
||||
if isinstance(expected_digest, str) and sha256(bytes(data)).hexdigest() != expected_digest.lower():
|
||||
if not is_valid_sha256(expected_digest) or sha256(bytes(data)).hexdigest() != str(expected_digest).lower():
|
||||
continue
|
||||
content_type = sniff_mime(bytes(data))
|
||||
if content_type not in IMAGE_CONTENT_TYPES:
|
||||
|
||||
@@ -4,8 +4,9 @@
|
||||
# @RELATION CALLED_BY -> [ScenarioExecution.ArtifactContent]
|
||||
# @RELATION CALLED_BY -> [ScenarioExecution.ScreenshotProvider]
|
||||
# @RELATION CALLED_BY -> [ScenarioExecution.BrowserProvider]
|
||||
# @INVARIANT Detection is byte-sniffing only (no path, name, or provider hint) so stored MIME can
|
||||
# never drift from the actual bytes (e.g. a WebP archive mislabeled image/jpeg).
|
||||
# @INVARIANT Detection is byte-sniffing only (no path, name, or provider hint). Callers that must
|
||||
# store a non-null content type apply a nominal fallback ONLY for unrecognized bytes; the
|
||||
# artifact-delivery path re-sniffs and rejects a bytes/label mismatch (production-chain).
|
||||
# @RATIONALE Extracted from ArtifactContent so evidence providers can stamp per-ref MIME from the
|
||||
# captured bytes without depending on the HTTP artifact-delivery service layer.
|
||||
# @REJECTED Trusting the provider's nominal format (screenshot→jpeg, browser→png) was rejected: the
|
||||
|
||||
@@ -151,11 +151,14 @@ def _resolve_configured_live_binding(
|
||||
dashboard_ids = {
|
||||
step.get("dashboard_id")
|
||||
for step in (plan.get("steps") or [])
|
||||
if isinstance(step, dict) and isinstance(step.get("dashboard_id"), int)
|
||||
if isinstance(step, dict)
|
||||
and isinstance(step.get("dashboard_id"), int)
|
||||
and not isinstance(step.get("dashboard_id"), bool)
|
||||
}
|
||||
# Compiled (MCP/bootstrap) graphs carry no per-step dashboard identity; the registry entry's
|
||||
# dashboard is server-owned, so it is a lawful binding-match key (T029m live replay defect).
|
||||
if scenario_dashboard_id is not None:
|
||||
# Compiled (MCP/bootstrap) graphs carry no per-step dashboard identity; only then is the
|
||||
# registry entry's server-owned dashboard a lawful binding-match key (T029m live replay defect).
|
||||
# When the plan DOES declare step identity, that pinned set stays authoritative (no widening).
|
||||
if not dashboard_ids and scenario_dashboard_id is not None:
|
||||
dashboard_ids.add(int(scenario_dashboard_id))
|
||||
if not dashboard_ids:
|
||||
return None
|
||||
|
||||
@@ -124,10 +124,12 @@ def create_scenario(
|
||||
"action_registry_hash": action_registry_fingerprint(),
|
||||
}
|
||||
# T029m live-replay defect: compiled steps carry no per-step target identity, so the
|
||||
# runner's live-binding admission (target match) cannot fire. The verified
|
||||
# dashboard_context is server-owned (context_authority evaluated at the register
|
||||
# boundary), so stamping its identity onto every step is authoritative, and steps that
|
||||
# already declare identity (legacy/authored graphs) are never overwritten.
|
||||
# runner's live-binding admission (target match) cannot fire. Identity is taken from the
|
||||
# scenario's declared dashboard_context and materialized server-side; for a VERIFIED pack
|
||||
# that context is the server-owned live model (context_authority evaluated at the register
|
||||
# boundary), while a fail-open `unverified`/legacy pack keeps its client-declared target —
|
||||
# PROD is separately blocked for unverified contexts (CONTEXT_AUTHORITY_REQUIRED_FOR_PROD).
|
||||
# Steps that already declare identity (authored graphs) are never overwritten.
|
||||
_context = graph.get("dashboard_context") or {}
|
||||
_identity = {
|
||||
"environment_id": _context.get("environment_id"),
|
||||
|
||||
@@ -149,6 +149,12 @@ def test_put_rejects_invalid_and_mismatched_query_model():
|
||||
json=_put_body(query_model_snapshot={**_QUERY_MODEL, "environment_id": "other-env"}),
|
||||
)
|
||||
assert mismatched.status_code == 422
|
||||
|
||||
fingerprint_mismatch = env.client.put(
|
||||
f"/api/scenario-live-bindings/{_BINDING_REF}",
|
||||
json=_put_body(query_model_snapshot={**_QUERY_MODEL, "query_model_fingerprint": "z" * 64}),
|
||||
)
|
||||
assert fingerprint_mismatch.status_code == 422
|
||||
assert env.config_manager.get_config().settings.scenario_live_execution_bindings == []
|
||||
|
||||
|
||||
|
||||
@@ -161,5 +161,27 @@ def test_main_publishes_end_to_end_with_fake_client(monkeypatch, tmp_path):
|
||||
assert code == 0
|
||||
assert seen["kwargs"]["base_url"] == "http://gitea.test"
|
||||
assert seen["kwargs"]["headers"] == {"Authorization": "token token-value"}
|
||||
def test_main_types_transport_errors(monkeypatch, tmp_path, capsys):
|
||||
import httpx
|
||||
|
||||
import src.scripts.publish_catalog as module
|
||||
|
||||
catalog_file = tmp_path / "catalog.json"
|
||||
catalog_file.write_bytes(_RAW)
|
||||
monkeypatch.setenv("PUBLISHED_CATALOG_GITEA_URL", "http://gitea.test")
|
||||
monkeypatch.setenv("PUBLISHED_CATALOG_GITEA_TOKEN", "token-value")
|
||||
monkeypatch.setenv("PUBLISHED_CATALOG_GITEA_REPO", "o/r")
|
||||
|
||||
class _BoomClient(_FakeClient):
|
||||
def __init__(self, **kwargs) -> None:
|
||||
super().__init__(_FakeResponse(404), _FakeResponse(201, {}))
|
||||
|
||||
def get(self, url, params=None):
|
||||
raise httpx.ConnectError("boom")
|
||||
|
||||
monkeypatch.setattr(module.httpx, "Client", lambda **kwargs: _BoomClient(**kwargs))
|
||||
code = main(["--catalog", str(catalog_file), "--baseline-set", "ss-prod-visual", "--version", "1"])
|
||||
assert code == 1
|
||||
assert "PUBLISH_UNAVAILABLE" in capsys.readouterr().out
|
||||
# #endregion Test.Tooling.PublishCatalog.Main
|
||||
# #endregion Test.Tooling.PublishCatalog
|
||||
|
||||
@@ -331,6 +331,30 @@ def test_adapter_drops_provider_supplied_baseline_pin():
|
||||
# #endregion Test.ScenarioExecution.EvaluationAdapter.Identity
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.EvaluationAdapter.ServerIdentity [C:2] [TYPE Function]
|
||||
# @BRIEF Provider-supplied evaluation_id/operation_id are replaced by server-generated ids.
|
||||
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter.Normalize: provider text never authors
|
||||
# evaluation/operation identity. -> VERIFIED_BY: test_adapter_replaces_provider_identity_ids
|
||||
def test_adapter_replaces_provider_identity_ids():
|
||||
record = _run_adapter_with_response({
|
||||
"status": "succeeded",
|
||||
"verdict": "inconclusive",
|
||||
"confidence": 0.2,
|
||||
"findings": [{
|
||||
"criterion_id": "crit-visual",
|
||||
"evidence_artifact_ids": ["draft:run-norm:" + "e" * 64],
|
||||
"message": "identity claim",
|
||||
}],
|
||||
"reason_codes": [],
|
||||
"evaluation_id": "provider-chosen-eval",
|
||||
"operation_id": "provider-chosen-op",
|
||||
"usage": {"input_tokens": 1, "output_tokens": 1},
|
||||
})
|
||||
assert record["evaluation_id"] != "provider-chosen-eval"
|
||||
assert record["operation_id"] != "provider-chosen-op"
|
||||
# #endregion Test.ScenarioExecution.EvaluationAdapter.ServerIdentity
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.EvaluationAdapter.Empty [C:2] [TYPE Function]
|
||||
# @BRIEF Missing prior-step evidence is fail-closed; the adapter never synthesizes PASS.
|
||||
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter: empty completed fails closed. -> VERIFIED_BY: adapter_empty_completed_fail_closed
|
||||
@@ -559,5 +583,9 @@ def test_image_payloads_skip_digest_mismatch():
|
||||
store = _RetrievingStore({"ref-a": _PNG_BYTES})
|
||||
tampered = {**_image_manifest_item("ref-a", _PNG_BYTES), "sha256": "0" * 64}
|
||||
assert image_payloads_from_manifest([tampered], store, max_images=8) == []
|
||||
|
||||
missing_digest = {**_image_manifest_item("ref-a", _PNG_BYTES)}
|
||||
missing_digest.pop("sha256")
|
||||
assert image_payloads_from_manifest([missing_digest], store, max_images=8) == []
|
||||
# #endregion Test.ScenarioExecution.EvaluationAdapter.Multimodal
|
||||
# #endregion Test.ScenarioExecution.AgentEvaluation
|
||||
@@ -257,6 +257,22 @@ def test_resolution_uses_registry_dashboard_for_identity_less_steps():
|
||||
config_manager=manager, scenario_dashboard_id=99,
|
||||
)
|
||||
assert foreign is None
|
||||
|
||||
# No widening: when the plan declares its own step identity, the registry dashboard is NOT unioned.
|
||||
mismatched = _resolve_configured_live_binding(
|
||||
environment_id=_BINDING_ENV, plan={"steps": [{"logical_step_id": "s", "dashboard_id": 5}]},
|
||||
principal_fingerprint=hashlib.sha256(_BINDING_ACTOR.encode()).hexdigest(),
|
||||
config_manager=manager, scenario_dashboard_id=_BINDING_DASHBOARD,
|
||||
)
|
||||
assert mismatched is None
|
||||
|
||||
# A bool is not a dashboard id (True is an int subclass).
|
||||
bool_step = _resolve_configured_live_binding(
|
||||
environment_id=_BINDING_ENV, plan={"steps": [{"logical_step_id": "s", "dashboard_id": True}]},
|
||||
principal_fingerprint=hashlib.sha256(_BINDING_ACTOR.encode()).hexdigest(),
|
||||
config_manager=manager, scenario_dashboard_id=_BINDING_DASHBOARD,
|
||||
)
|
||||
assert bool_step is not None # falls back to the registry dashboard, since the bool step is invalid
|
||||
# #endregion Test.ScenarioExecution.StartRunServerAuthority.BindingMatrix
|
||||
|
||||
|
||||
|
||||
@@ -168,6 +168,13 @@ async def test_mcp_bootstrap_run_automation_and_prod_gate(isolated_mcp_db, monke
|
||||
assert revision and revision.activation_status == "current"
|
||||
# T029h: the fail-open register marker materializes as a server-owned graph field.
|
||||
assert revision.graph_snapshot.get("context_authority") == "unverified"
|
||||
# T029m: compiled steps carry no per-step identity, so the registry materialization
|
||||
# stamps the dashboard_context target identity onto every step (live-binding admission).
|
||||
stamped_steps = revision.graph_snapshot.get("steps") or []
|
||||
assert stamped_steps and all(
|
||||
step.get("environment_id") == "env-prod-01" and step.get("dashboard_id") == 80
|
||||
for step in stamped_steps
|
||||
)
|
||||
visible = await server.call_tool("list_scenario_schedules", {"scenario_id": scenario_id})
|
||||
assert _unwrap(visible) == []
|
||||
|
||||
|
||||
@@ -75,3 +75,33 @@ remains blocked by the Gitea PAT (published catalog unavailable); `050 T045` and
|
||||
`E2E-EXT-002` (T029m) → **CLOSED** by this trace. Remaining MCP-surface work is limited to the
|
||||
PAT-blocked baseline-pinned variant (T045/T046-pin) and `Expected.threshold` (038 schema authority,
|
||||
separate architect decision).
|
||||
|
||||
## Human-checkpoint loop (T029m formulation completeness)
|
||||
|
||||
`live_mcp_replay.py`'s B01 graph contains no HumanStep, so the task's
|
||||
`list_checkpoints`/`decide_checkpoint` element was exercised by a companion replay
|
||||
`specs/044-dashboard-scenario-execution/prototype/live_mcp_human_loop.py` (same scripted client):
|
||||
|
||||
| Stage | Result |
|
||||
|---|---|
|
||||
| `inspect_scenario` B05 with live derived capabilities | one step: `phase-5-B05-human_checkpoint-1` (`tool=human`) — the script refuses any graph with a non-human step, so no mutation case is started |
|
||||
| validate → draft-pack → `register_draft_pack` | valid, save_eligible, `context_authority=verified` |
|
||||
| bootstrap → `start_scenario_run` (ss-prod, manual) | `pending_approval` → approved → run `6dcb9b8b-4f96-4491-9480-a3c74a67ee42` |
|
||||
| run state | `waiting_human` |
|
||||
| MCP `list_checkpoints(run_id)` | pending checkpoint `35039c17-6469-4cdf-90a1-753a357c0e08`, `decision_version=1` |
|
||||
| MCP `decide_checkpoint(confirm, expected_version=1)` | ok; checkpoint `decided`, `decision_version=2` (CAS consumed) |
|
||||
| terminal | **passed** — the immutable 044 mapping `confirm → passed` names the human's conformance observation, not a synthesized PASS |
|
||||
|
||||
This closes the aligned-label human loop live; combined with the B01 chain above, every element of
|
||||
the T029m task text is now backed by a live trace.
|
||||
|
||||
## Review residuals (recorded, not defects)
|
||||
|
||||
- The compile input declared `screenshot=true` as a caller capability; `screenshot` is not a
|
||||
server-derived dashboard fact, so `merge_capabilities` passes it through verbatim. The
|
||||
derived-wins axis that FR-028 protects (verifiable facts never downgraded to `human_checkpoint`)
|
||||
is intact and `capability_authority.status=derived`; the caller-key nuance belongs to the 038/T029k
|
||||
authority-boundary backlog, not to this replay.
|
||||
- `live_mcp_replay.py` now asserts the honest typed non-pass terminal (a `passed` terminal on the
|
||||
B01 chain raises `UNEXPECTED_SYNTHETIC_PASS`) and asserts the idempotent retry reuses the same
|
||||
`run_id`.
|
||||
|
||||
@@ -0,0 +1,208 @@
|
||||
"""T029m human-checkpoint loop — live external MCP replay of the waiting_human decision cycle.
|
||||
|
||||
Completes the T029m acceptance (`... → list_checkpoints/decide_checkpoint human loop with aligned
|
||||
labels`) that the B01 browser chain cannot exercise: a compiled HumanCheckpoint case drives the run
|
||||
to `waiting_human`, then the external client lists the checkpoint and disposes it via the MCP CAS
|
||||
tool, and the run continues to a terminal state.
|
||||
|
||||
Reuses the proven scripted client from live_mcp_replay.py (same OAuth/PKCE + SSE transport); no new
|
||||
client machinery. Fail-closed: no mutation case is ever started (only a graph whose steps are all
|
||||
tool=human is accepted).
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
import uuid
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
from live_mcp_replay import ( # noqa: E402
|
||||
CHECKLIST_CATALOG_VERSION,
|
||||
DASHBOARD_ID,
|
||||
DASHBOARD_NAME,
|
||||
ENVIRONMENT_ID,
|
||||
POLL_TIMEOUT_SECONDS,
|
||||
_TERMINAL_STATUSES,
|
||||
_step_report,
|
||||
LiveMcpReplay,
|
||||
ReplayError,
|
||||
)
|
||||
|
||||
HUMAN_CASE_CANDIDATES = ("B05", "B06", "B07", "B08", "B09", "C01", "C03")
|
||||
WAITING_HUMAN = "waiting_human"
|
||||
|
||||
|
||||
# #region ScenarioExecution.LiveMcpHumanLoop.Chain [C:4] [TYPE Function] [SEMANTICS scenario,execution,mcp,human-checkpoint,replay]
|
||||
# @ingroup ScenarioExecution
|
||||
# @BRIEF Drive one HumanCheckpoint case to waiting_human, list_checkpoints, decide_checkpoint, finish.
|
||||
# @PRE Live stand ready; the MCP admin principal has scenario RUN/RUN_PROD.
|
||||
# @POST Returns the full evidence dict; every refusal raises ReplayError (never synthesized success).
|
||||
# @INVARIANT Only a compiled graph whose every step is tool=human is started — a mutation-capable graph
|
||||
# is rejected by the script before any run/gate exists.
|
||||
# @REJECTED Auto-confirming without an explicit disposition was rejected — the loop proves the human
|
||||
# decision surface, and `confirm` is the operator's aligned-label choice for conformance.
|
||||
def run_human_loop() -> dict:
|
||||
replay = LiveMcpReplay()
|
||||
evidence: dict = {"base": replay.http.base_url, "environment_id": ENVIRONMENT_ID, "dashboard_id": DASHBOARD_ID}
|
||||
|
||||
user_jwt = replay.login()
|
||||
evidence["login"] = "ok"
|
||||
replay.oauth_pkce_token(user_jwt)
|
||||
replay.mcp("initialize", {"protocolVersion": "2025-06-18", "capabilities": {},
|
||||
"clientInfo": {"name": "t029m-human-loop", "version": "1.0"}})
|
||||
replay.mcp("notifications/initialized", notify=True)
|
||||
evidence["tools_listed"] = len(replay.mcp("tools/list", {}).get("tools", []))
|
||||
|
||||
context = replay.call("inspect_dashboard_context", {
|
||||
"request": {"environment_id": ENVIRONMENT_ID, "dashboard_id": DASHBOARD_ID},
|
||||
})
|
||||
if context.get("status") != "ok":
|
||||
raise ReplayError(f"inspect_dashboard_context blocked: {json.dumps(context)[:300]}")
|
||||
evidence["query_model_fingerprint"] = context["query_model_fingerprint"]
|
||||
|
||||
agent_run = replay.call("create_agent_run", {"request": {
|
||||
"dashboard_id": DASHBOARD_ID, "environment_id": ENVIRONMENT_ID,
|
||||
"dashboard_name": DASHBOARD_NAME, "idempotency_key": f"t029m-human-agent-{uuid.uuid4().hex[:12]}",
|
||||
}})
|
||||
if agent_run.get("status") != "ok":
|
||||
raise ReplayError(f"create_agent_run blocked: {json.dumps(agent_run)[:300]}")
|
||||
agent_run_id = agent_run["run_id"]
|
||||
evidence["agent_run_id"] = agent_run_id
|
||||
|
||||
capabilities = {**context["derived_capabilities"]["capabilities"], "screenshot": False, "baseline": False}
|
||||
scenario, case_id = _compile_human_case(replay, context, agent_run_id, capabilities)
|
||||
evidence["case_id"] = case_id
|
||||
evidence["compiled_steps"] = [
|
||||
{"id": step["id"], "tool": step["tool"], "action": step["action"], "risk": step["risk"]}
|
||||
for step in scenario["steps"]
|
||||
]
|
||||
|
||||
validation = replay.call("validate_scenario", {"scenario": scenario})
|
||||
evidence["validation"] = {"valid": validation["valid"], "coverage": validation["coverage"]}
|
||||
if not validation["valid"]:
|
||||
raise ReplayError(f"human-loop scenario invalid: {json.dumps(validation['errors'])[:300]}")
|
||||
draft = replay.call("generate_draft_pack", {"request": {"scenario": scenario}})
|
||||
if draft["status"] != "save_eligible":
|
||||
raise ReplayError(f"human-loop draft pack not save-eligible: {json.dumps(draft)[:300]}")
|
||||
registered = replay.call("register_draft_pack", {"request": {"agent_run_id": agent_run_id, "scenario": scenario}})
|
||||
if registered.get("status") != "save_eligible":
|
||||
raise ReplayError(f"register_draft_pack blocked: {json.dumps(registered)[:300]}")
|
||||
evidence["register"] = {"status": registered.get("status"), "context_authority": registered.get("context_authority")}
|
||||
|
||||
boot = replay.call("bootstrap_authoring_scenario", {"request": {
|
||||
"idempotency_key": f"t029m-human-bootstrap-{uuid.uuid4().hex[:12]}",
|
||||
"title": "T029m human loop replay", "dashboard_id": DASHBOARD_ID,
|
||||
"allowed_environment_ids": [ENVIRONMENT_ID], "selected_case_ids": [case_id],
|
||||
"objective": "T029m live MCP human-checkpoint loop replay",
|
||||
"compiled_handle_id": registered["compiled_handle_id"],
|
||||
"draft_pack_id": registered["draft_pack_handle_id"],
|
||||
"draft_pack_digest": registered["draft_pack_digest"],
|
||||
}})
|
||||
evidence["bootstrap"] = {"scenario_id": boot["scenario_id"], "revision_id": boot["revision_id"]}
|
||||
|
||||
started = replay.call("start_scenario_run", {"request": {
|
||||
"scenario_id": boot["scenario_id"], "revision_id": boot["revision_id"],
|
||||
"environment_id": ENVIRONMENT_ID, "idempotency_key": f"t029m-human-run-{uuid.uuid4().hex[:12]}",
|
||||
}})
|
||||
evidence["start"] = {"status": started.get("status"), "run_id": started.get("run_id"), "error": started.get("error")}
|
||||
if started.get("status") != "pending_approval":
|
||||
raise ReplayError(f"human plan did not reach the PROD gate: {json.dumps(started)[:300]}")
|
||||
run_id = started["run_id"]
|
||||
|
||||
approve = replay.http.post(f"/api/scenario-runs/{run_id}/approval/decision",
|
||||
json={"decision": "approve", "comment": "T029m human-loop replay"})
|
||||
if approve.status_code != 200:
|
||||
raise ReplayError(f"PROD gate approval failed: {approve.status_code} {approve.text[:300]}")
|
||||
evidence["approval"] = approve.json().get("status")
|
||||
|
||||
status = _poll_for(replay, run_id, {WAITING_HUMAN})
|
||||
evidence["waiting_human"] = status
|
||||
if status != WAITING_HUMAN:
|
||||
raise ReplayError(f"run did not reach {WAITING_HUMAN}: {status}")
|
||||
|
||||
checkpoints = replay.call("list_checkpoints", {"run_id": run_id})
|
||||
pending = [c for c in checkpoints.get("checkpoints", []) if c.get("status") == "pending"]
|
||||
if checkpoints.get("status") != "ok" or not pending:
|
||||
raise ReplayError(f"no pending checkpoint via MCP: {json.dumps(checkpoints)[:300]}")
|
||||
target = pending[0]
|
||||
evidence["checkpoint"] = {"id": target["id"], "logical_step_id": target["logical_step_id"],
|
||||
"decision_version": target["decision_version"]}
|
||||
|
||||
decided = replay.call("decide_checkpoint", {"request": {
|
||||
"run_id": run_id, "disposition": "confirm",
|
||||
"expected_version": int(target["decision_version"]),
|
||||
"comment": "T029m live human-loop replay — conformance confirmed",
|
||||
}})
|
||||
evidence["decision"] = {"status": decided.get("status"), "checkpoint_status": decided.get("checkpoint_status"),
|
||||
"disposition": decided.get("disposition"), "decision_version": decided.get("decision_version")}
|
||||
if decided.get("status") != "ok":
|
||||
raise ReplayError(f"decide_checkpoint refused: {json.dumps(decided)[:300]}")
|
||||
|
||||
terminal = _poll_for(replay, run_id, _TERMINAL_STATUSES)
|
||||
detail = replay.http.get(f"/api/scenario-runs/{run_id}").json()
|
||||
evidence["terminal"] = {"status": terminal, "phase": detail.get("phase"), "error_code": detail.get("error_code")}
|
||||
evidence["steps"] = _step_report(run_id)
|
||||
return evidence
|
||||
# #endregion ScenarioExecution.LiveMcpHumanLoop.Chain
|
||||
|
||||
|
||||
# #region ScenarioExecution.LiveMcpHumanLoop.CompileHuman [C:3] [TYPE Function] [SEMANTICS mcp,compile,human-checkpoint,guard]
|
||||
# @ingroup ScenarioExecution
|
||||
# @BRIEF Compile the first candidate case whose graph is entirely tool=human; refuse mutation graphs.
|
||||
def _compile_human_case(replay: LiveMcpReplay, context: dict, agent_run_id: str, capabilities: dict) -> tuple[dict, str]:
|
||||
for case_id in HUMAN_CASE_CANDIDATES:
|
||||
compiled = replay.call("inspect_scenario", {"request": {
|
||||
"agent_run_id": agent_run_id,
|
||||
"objective": {"goal": f"t029m human loop {case_id}", "selected_case_ids": [case_id],
|
||||
"rationale": "external human-checkpoint loop replay"},
|
||||
"query_model": context["query_model"],
|
||||
"checklist_catalog_version": CHECKLIST_CATALOG_VERSION,
|
||||
"baseline_version": "unpinned-live-replay",
|
||||
"capabilities": capabilities,
|
||||
"parameters": {
|
||||
"row_key": {"type": "string", "required": False, "default": "replay-row"},
|
||||
"comment": {"type": "string", "required": False, "default": "replay"},
|
||||
"selector_hint": {"type": "selector_hint", "required": False, "default": "live-replay"},
|
||||
},
|
||||
"has_dataset_fields": bool(context["derived_capabilities"]["has_dataset_fields"]),
|
||||
"environment_id": ENVIRONMENT_ID, "dashboard_id": DASHBOARD_ID, "dashboard_name": DASHBOARD_NAME,
|
||||
}})
|
||||
steps = compiled.get("scenario", {}).get("steps") or []
|
||||
if steps and all(step.get("tool") == "human" for step in steps):
|
||||
return compiled["scenario"], case_id
|
||||
raise ReplayError("no candidate case compiled to a pure human-checkpoint graph")
|
||||
|
||||
|
||||
def _poll_for(replay: LiveMcpReplay, run_id: str, targets: frozenset | set) -> str:
|
||||
deadline = time.monotonic() + POLL_TIMEOUT_SECONDS
|
||||
status = "queued"
|
||||
while time.monotonic() < deadline:
|
||||
status = replay.http.get(f"/api/scenario-runs/{run_id}").json().get("status", "unknown")
|
||||
if status in targets:
|
||||
return status
|
||||
if status in _TERMINAL_STATUSES:
|
||||
return status
|
||||
time.sleep(5)
|
||||
return status
|
||||
# #endregion ScenarioExecution.LiveMcpHumanLoop.CompileHuman
|
||||
|
||||
|
||||
# #region ScenarioExecution.LiveMcpHumanLoop.Main [C:2] [TYPE Function] [SEMANTICS replay,entrypoint,report]
|
||||
# @BRIEF Entry point: run the human-checkpoint loop and dump the evidence JSON (fail-closed).
|
||||
def main() -> None:
|
||||
try:
|
||||
evidence = run_human_loop()
|
||||
evidence["result"] = "ok"
|
||||
except ReplayError as exc:
|
||||
evidence = {"result": "blocked", "error": str(exc)}
|
||||
print("[t029m-human] RESULT")
|
||||
print(json.dumps(evidence, indent=1, ensure_ascii=False, default=str))
|
||||
result_path = os.environ.get("SS_REPLAY_RESULT", "/tmp/kilo/live_mcp_human_loop_result.json")
|
||||
with open(result_path, "w") as handle:
|
||||
json.dump(evidence, handle, indent=1, default=str)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
# #endregion ScenarioExecution.LiveMcpHumanLoop.Main
|
||||
@@ -23,6 +23,7 @@ tests/test_mcp_initial_scenario_e2e.py — only the transport is live (httpx + S
|
||||
import base64
|
||||
from hashlib import sha256
|
||||
import json
|
||||
import os
|
||||
import secrets
|
||||
import sys
|
||||
import time
|
||||
@@ -31,22 +32,24 @@ import uuid
|
||||
|
||||
from dotenv import load_dotenv
|
||||
|
||||
load_dotenv("/home/busya/dev/ss-tools/backend/.env")
|
||||
sys.path.insert(0, "/home/busya/dev/ss-tools/backend")
|
||||
_REPO_ROOT = os.path.dirname(os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))))
|
||||
load_dotenv(os.path.join(_REPO_ROOT, "backend", ".env"))
|
||||
sys.path.insert(0, os.path.join(_REPO_ROOT, "backend"))
|
||||
|
||||
import httpx # noqa: E402
|
||||
|
||||
from src.core.database import SessionLocal # noqa: E402
|
||||
from src.models.scenario_run import ScenarioStepRun # noqa: E402
|
||||
|
||||
BASE = "http://127.0.0.1:8000"
|
||||
BASE = os.environ.get("SS_REPLAY_BASE", "http://127.0.0.1:8000")
|
||||
ENVIRONMENT_ID = "ss-prod"
|
||||
DASHBOARD_ID = 11
|
||||
DASHBOARD_NAME = "Sales Dashboard"
|
||||
ACTOR = "admin"
|
||||
ACTOR_PASSWORD = "admin"
|
||||
ACTOR = os.environ.get("SS_REPLAY_USER", "admin")
|
||||
ACTOR_PASSWORD = os.environ.get("SS_REPLAY_PASSWORD", "admin")
|
||||
CHECKLIST_CATALOG_VERSION = 1
|
||||
POLL_TIMEOUT_SECONDS = 300
|
||||
_TERMINAL_STATUSES = frozenset({"passed", "failed", "blocked", "inconclusive", "cancelled"})
|
||||
|
||||
|
||||
# #region ScenarioExecution.LiveMcpReplay.Client [C:5] [TYPE Class] [SEMANTICS mcp,oauth,pkce,sse,client]
|
||||
@@ -283,9 +286,13 @@ def run_chain() -> dict:
|
||||
"environment_id": ENVIRONMENT_ID, "idempotency_key": start_key,
|
||||
}})
|
||||
evidence["start"] = {"status": started.get("status"), "run_id": started.get("run_id"),
|
||||
"retry_status": retry.get("status"), "error": started.get("error")}
|
||||
"retry_status": retry.get("status"), "retry_run_id": retry.get("run_id"),
|
||||
"error": started.get("error")}
|
||||
if started.get("status") != "pending_approval" or retry.get("status") != "pending_approval":
|
||||
raise ReplayError(f"PROD start did not gate: {json.dumps({'first': started, 'retry': retry})[:400]}")
|
||||
# Idempotency proof: the retry must reuse the same durable run/gate, not create a second one.
|
||||
if started.get("run_id") != retry.get("run_id"):
|
||||
raise ReplayError(f"idempotent retry created a different run: {started.get('run_id')} != {retry.get('run_id')}")
|
||||
run_id = started["run_id"]
|
||||
|
||||
approve = replay.http.post(f"/api/scenario-runs/{run_id}/approval/decision",
|
||||
@@ -300,10 +307,18 @@ def run_chain() -> dict:
|
||||
while time.monotonic() < deadline:
|
||||
detail = replay.http.get(f"/api/scenario-runs/{run_id}").json()
|
||||
status = detail.get("status", "unknown")
|
||||
if status in {"passed", "failed", "blocked", "inconclusive", "cancelled"}:
|
||||
if status in _TERMINAL_STATUSES:
|
||||
break
|
||||
time.sleep(5)
|
||||
evidence["terminal"] = {"status": status, "phase": detail.get("phase"), "error_code": detail.get("error_code")}
|
||||
if status not in _TERMINAL_STATUSES:
|
||||
raise ReplayError(f"run did not reach a terminal status within {POLL_TIMEOUT_SECONDS}s (last={status})")
|
||||
# Fail-closed evidence guard: the compiled B01 graph's browser action (apply_native_filter) is not
|
||||
# supported by the isolated transport, so this chain's honest terminal is a typed non-pass. A
|
||||
# `passed` terminal would mean the unsupported step was silently synthesized into success.
|
||||
if status == "passed":
|
||||
raise ReplayError("UNEXPECTED_SYNTHETIC_PASS: the typed non-pass terminal became passed")
|
||||
evidence["terminal"] = {"status": status, "phase": detail.get("phase"), "error_code": detail.get("error_code"),
|
||||
"synthetic_pass_guard": "non_pass_confirmed"}
|
||||
evidence["steps"] = _step_report(run_id)
|
||||
return evidence
|
||||
|
||||
@@ -333,7 +348,9 @@ def main() -> None:
|
||||
evidence = {"result": "blocked", "error": str(exc)}
|
||||
print("[t029m] RESULT")
|
||||
print(json.dumps(evidence, indent=1, ensure_ascii=False, default=str))
|
||||
json.dump(evidence, open("/tmp/kilo/live_mcp_replay_result.json", "w"), indent=1, default=str)
|
||||
result_path = os.environ.get("SS_REPLAY_RESULT", "/tmp/kilo/live_mcp_replay_result.json")
|
||||
with open(result_path, "w") as handle:
|
||||
json.dump(evidence, handle, indent=1, default=str)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -26,7 +26,7 @@
|
||||
| Visual chain | Compiler emits browser→capture→comparison→evaluation→policy (Slice E) | One-action templates; live Playwright canary PASSED 2026-09-10 (browser `open_dashboard` + screenshot capture vs ss-prod dashboard 11; comparison/evaluation steps not in the canary graph) |
|
||||
| Live providers | Registered browser/screenshot + multimodal LLM | Binding `ss-prod-d11-live-001` resolved **server-side at REST start** (no binding in the request body); `/api/ready`: browser/screenshot ready, bindings 1; canary v2 live trace: `docs/reports/agentic-runtime-live-canary-v2-2026-09-10.md` |
|
||||
|
||||
**Open live gaps (2026-09-10, for the next agent):** multimodal image attachment (prompts are text-only → visual verdicts stay inconclusive); `baseline_pin` in the persisted `AgentEvaluation` record is `{}` for baseline-backed plans; screenshot/browser evidence MIME is hardcoded (`image/jpeg`/`image/png`) and would mismatch a WebP archive; live baseline pin is blocked on a valid Gitea PAT.
|
||||
**Open live gaps (2026-09-10, for the next agent):** ~~multimodal image attachment (prompts are text-only → visual verdicts stay inconclusive); `baseline_pin` in the persisted `AgentEvaluation` record is `{}` for baseline-backed plans; screenshot/browser evidence MIME is hardcoded (`image/jpeg`/`image/png`) and would mismatch a WebP archive~~ → all three CLOSED 2026-09-10/11 (commit `792bb125`): multimodal evidence attachment (`evaluation_images.py` + `_evaluation_messages`), server-authoritative `baseline_pin` stamping (`baseline_resolver.stamp_baseline_pin`, live canary v3 `9e6f59c0`: `verdict=pass`, visual explanation), per-ref MIME sniffing (`mime_sniff.py`). Still open: live baseline pin is blocked on a valid Gitea PAT (publisher ready: `backend/src/scripts/publish_catalog.py`).
|
||||
|
||||
The former agent-driven product-UI flow (dashboard → `/agent` workspace → agent-generated scenario) is SUPERSEDED. 044 is a headless execution engine; agent interaction is external MCP (050). Product UI (045) is read-only evidence plus human approvals.
|
||||
|
||||
|
||||
@@ -84,3 +84,17 @@ Historical rows above identify prior tests/code only; removed agent UI paths are
|
||||
| SCEX-FR-028; external-MCP-only UI | manual editor/review; read-only evidence | [production tasks](tasks.md) | No frontend agent prompt/chat/assistant editing/proposal generation/workspace/start/handoff routes or requests; human approval remains usable. | OPEN |
|
||||
|
||||
Sources: [production gap](../../docs/reports/ss-prod-agentic-e2e-production-gap-2026-09-08.md), [coverage gap](../../docs/reports/ss-prod-agentic-e2e-spec-coverage-2026-09-08.md), [baseline gap](../../docs/reports/ss-prod-agentic-e2e-baseline-gap-2026-09-08.md). Spec schema/static checks prove contract structure only; live canary/runtime closure and optional approved performance baseline are not claimed.
|
||||
|
||||
## Live-binding admin surface + catalog publisher — evidence 2026-09-11
|
||||
|
||||
Two deployment/operator seams added by commits `792bb125` + `d04927bd`; both are server-owned
|
||||
surfaces outside the 050 MCP catalog (no spec row existed before this note):
|
||||
|
||||
| Seam | Contract | Surface | Evidence |
|
||||
|---|---|---|---|
|
||||
| `Api.ScenarioLiveBindings` | `ScenarioExecution.LiveBinding` + `Core.ConfigManager` | admin REST `GET/PUT/PATCH /api/scenario-live-bindings` (`admin:settings`); validates `LiveExecutionBinding.from_snapshot` + `DashboardQueryModel` (env/dashboard/fingerprint) and writes `settings.scenario_live_execution_bindings` atomically | `backend/tests/api/test_scenario_live_bindings_api.py` (7); first live exercise: re-registered `ss-prod-d11-live-001` with the correct principal, then the T029m run adopted it server-side |
|
||||
| `Tooling.PublishCatalog` | T045/D7 published-catalog source (`PUBLISHED_CATALOG_*`) | CLI `backend/src/scripts/publish_catalog.py` — pre-validates catalog bytes, then Gitea contents PUT (typed auth/conflict/unavailable) | `backend/tests/scripts/test_publish_catalog.py` (7, offline); publication itself blocked on a valid Gitea PAT |
|
||||
|
||||
Cross-reference: the live external MCP replay that exercised the binding path end-to-end (and exposed
|
||||
the identity-less-compiled-step defect) is `docs/2026-09-11-sales-prod-mcp-replay.md` (050 T029m /
|
||||
`E2E-EXT-002`, CLOSED).
|
||||
|
||||
@@ -302,7 +302,8 @@ raw-ORM insert, which masked the gap (test-honesty rule, ADR-0024 §3).
|
||||
chain with zero non-MCP seeding and the reachability pin is hard-green
|
||||
(`E2E-EXT-001` CLOSED); MCPX-FR-028 landed as `ScenarioGraph.CapabilityAuthority`
|
||||
(T029k, `CAP-001` CLOSED); MCPX-FR-029 landed as outcome-based labels/descriptions
|
||||
(T029l, `DISP-001` CLOSED). Remaining: live-stand replay `E2E-EXT-002` (T029m).
|
||||
(T029l, `DISP-001` CLOSED). Remaining: none for Phase 2d — the live-stand replay `E2E-EXT-002`
|
||||
(T029m) CLOSED 2026-09-11 (see the gate row below).
|
||||
|
||||
| Stage | Input | Authoritative owner | Output handle | Digest | Persistence | Failure / no-side-effect rule | Test id |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
@@ -423,7 +424,7 @@ Required evidence rows produced by the live external run against ss-prod
|
||||
| Evidence row | Required trace | Evidence required | Status |
|
||||
|---|---|---|---|
|
||||
| `E2E-EXT-001` | `create_agent_run` -> `register_draft_pack` -> `bootstrap_authoring_scenario` -> `start_scenario_run` | External MCP-only chain on a fresh DB with ZERO non-MCP prerequisite seeding; the strict-xfail pin `tests/test_mcp_agent_run_reachability.py` flipped to green and unmarked (MCPX-FR-027, T029i) | `[x] CLOSED 2026-09-07` — `tests/test_mcp_initial_scenario_e2e.py` converted (AgentRun minted via MCP `create_agent_run` tools/call; no service/ORM seeding remains); reachability pin unmarked+green; `tests/test_mcp_agent_run_tools.py` (5) covers create/replay/read/RBAC; slice `70 passed`, full suite `11357 passed` |
|
||||
| `E2E-EXT-002` | live stand replay of the 2026-09-07 sales run | `inspect_dashboard_context` -> derived-capability compile -> register -> bootstrap -> queued/gated run on ss-prod with retained evidence (T029m) | `[ ] OPEN` |
|
||||
| `E2E-EXT-002` | live stand replay of the 2026-09-07 sales run | `inspect_dashboard_context` -> derived-capability compile -> `create_agent_run` -> register -> bootstrap -> queued/gated run on ss-prod + `list_checkpoints`/`decide_checkpoint` human loop, with retained evidence (T029m) | `[x] CLOSED 2026-09-11` — committed replay client `specs/044-dashboard-scenario-execution/prototype/live_mcp_replay.py` replayed the full external chain on the live stand (run `110a6517…`: inspect derived capabilities → create_agent_run → compile B01 → register `context_authority=verified` → bootstrap → PROD gate `pending_approval` (identical retry = one durable gate) → approval → live `capture_screenshot` passed (8 refs) + typed `BROWSER_ACTION_NOT_SUPPORTED` → honest `inconclusive`); the human loop was exercised live by `live_mcp_human_loop.py` (compiled B05 HumanCheckpoint → `waiting_human` → MCP `list_checkpoints` (v1) → `decide_checkpoint confirm` (v2 CAS) → terminal `passed`). Two fail-closed defects found and fixed (binding resolution for identity-less compiled steps; per-step target-identity stamping at bootstrap). Trace: `docs/2026-09-11-sales-prod-mcp-replay.md` |
|
||||
| `CAP-001` | `inspect_dashboard_context` -> compile capabilities | Server-derived capability map proven: verifiably-available dataset fields/native filters classify `automated`, not `human_checkpoint`; unsafe-mutation cases stay `human_checkpoint` (MCPX-FR-028, 038 amendment, T029k) | `[x] CLOSED 2026-09-07` — `ScenarioGraph.CapabilityAuthority` (derivation/derived-wins merge/boundary choke point) wired into MCP `inspect_scenario`+`inspect_dashboard_context` and REST `api_compile_scenario`; CAP-001 classification-fix test through `map_all` on the sales-shape fixture (B/T automated, C04–C06 unsupported, mutation cases legitimately human) + tool/REST parity pins; slice `67 passed` |
|
||||
| `DISP-001` | disposition label/vocabulary audit | RU/EN labels and MCP tool descriptions name the persisted outcome (`confirm`→passed); no defect-confirming wording or destructive styling on a passing choice; mapping pinned by tests (MCPX-FR-029, 044/045 amendment, T029l) | `[x] CLOSED 2026-09-07` — ru/en labels renamed to outcome-based wording; confirm buttons `bg-destructive`→`bg-primary` (HumanCheckpointPanel + WaitingForMeView); `decide_checkpoint` docstring carries the immutable outcome table; vitest DISP-001 style/label/dispatch pin; frontend `3507 passed`, lint 0 errors, build OK |
|
||||
| `TEST-001` | vertical-test honesty remediation | `test_mcp_initial_scenario_e2e.py::_pack()` creates the AgentRun through the production `create_agent_run` service boundary (no raw-ORM prerequisite seeding); reachability requirement pinned as `strict=True` xfail; typed zero-side-effect denial pinned (ADR-0024 §3, T029j) | `[x] CLOSED 2026-09-07` — `pytest tests/test_mcp_initial_scenario_e2e.py tests/test_mcp_agent_run_reachability.py` green (1 passed + 1 xfailed(strict) + denial pin); metadata records the open T029i gap. *(Superseded the same day by T029i closure: the strict-xfail flip ritual executed as designed — pin unmarked and hardened, vertical converted to the fully external chain; denial pin retained.)* |
|
||||
|
||||
@@ -70,7 +70,7 @@
|
||||
- [x] T029j Test-honesty remediation (ADR-0024 §3, binding rule): vertical/E2E tests obtain every prerequisite through a boundary the principal under test can reach; raw-ORM seeding of a chain prerequisite in a test claiming external reachability is forbidden. Executed: `test_mcp_initial_scenario_e2e.py::_pack()` replaced the raw `AgentRun(...)` insert with the production `create_agent_run` service boundary with honest test metadata; new `tests/test_mcp_agent_run_reachability.py` pins (a) the external-reachability REQUIREMENT as `strict=True` xfail (flips the suite red the moment `create_agent_run` lands without spec follow-through) and (b) the current typed zero-side-effect denial (`DRAFT_PACK_ACCESS_DENIED`, no handle rows). Evidence: see `TEST-001` row in spec.md release gates (targeted pytest green 2026-09-07). *(Update 2026-09-07, T029i closure: строгий xfail-пин сработал как спроектирован — конвертирован в обычный requirement-тест (unmarked, green) вместе с конвертацией вертикали в fully external chain; denial-pin сохранён.)*
|
||||
- [x] T029k Context-authority-derived capability map (MCPX-FR-028; 038 amendment 2026-09-07): server-owned derivation of `capabilities`/`has_dataset_fields` from the authoritative `DashboardQueryModel` (the same live inspection `context_authority` binds), environment policy and provider readiness — exposed through `inspect_dashboard_context` (derived-capability section) and consumed by the persisted compile boundaries (MCP `scenario_compile` + REST `api_compile_scenario`); caller-declared capabilities remain accepted only as an explicit subset narrowing, never as an authority that downgrades verifiable facts into `human_checkpoint`; unresolved facts stay `needs_context`/`needs_selector`/`needs_baseline`; genuinely unsafe PROD mutation contexts keep `human_checkpoint`; HumanStep revisions remain automation-ineligible (SCEX-FR-004a stands). Evidence: `CAP-001` row + capability-derivation unit/E2E tests. Доказательство (2026-09-07): новый `src/services/dashboard_testing/scenario/capability_authority.py` (236 LOC, `ScenarioGraph.CapabilityAuthority`): `derive_capabilities` — только верифицируемые факты (native_filters; text_filter по STRING-фильтру; time_rollover по DATE/TIME/TIME_GRAIN; table_filter+pagination по executable table-viz; xlsx_export из capabilities-флага; dataset_field_read+has_dataset_fields по accessible dataset с колонками; browser — только при readiness-факте из T040-снимка composition-root), `NEVER_DERIVED` = row_edit/bulk_edit/persistence_refresh/safe_test_data/safe_clock_fixture/cross_dashboard (unsafe-mutation автоматизация метаданными невозможна — module @INVARIANT); `merge_capabilities` — derived-wins в ОБЕ стороны (declared-false не понижает проверяемую правду; declared-true не фабрикует) + overrides-аудит; legacy/unparseable payload → дословный caller-declared passthrough (register-time context_authority остаётся жёстким гейтом). Wiring: MCP `inspect_scenario` (+additive `capability_authority` секция), `inspect_dashboard_context` (+`derived_capabilities`), REST `api_compile_scenario` (parity через общий `build_capability_authority` choke point + additive секция). Тесты: фикстура `query_model_sales.json` (форма sales-стенда); unit truth-table `tests/services/dashboard_testing/scenario/test_capability_authority.py` (7, включая **CAP-001 classification-fix через `map_all`**: полевого shape декларации → B01–B04/T01–T03 `automated`, C04–C06 `unsupported`, B05–B09/C01–C03/C02/C07 легитимно `human_checkpoint`); tool-level `tests/test_mcp_capability_authority.py` (3: derived-wins overrides `["text_filter","xlsx_export"]`, legacy caller_declared, derived_capabilities в inspect-ответе); REST-parity pin `test_scenario_routes.py::test_capability_authority_derived_wins`. Браузер-readiness seam запинен monkeypatch (детерминизм против глобального composition-root состояния в full-suite). Gate: 5-файловый срез `67 passed`; полный suite `11357 passed`.
|
||||
- [x] T029l Disposition vocabulary clarity (MCPX-FR-029; 044/045 amendment 2026-09-07): the 044 lifecycle mapping (`confirm`→`passed`, `false_positive`/`inconclusive`→`inconclusive`, aliases `pass`/`fail`) is immutable and stays; human-facing wording aligns to the persisted outcome — RU labels «Подтвердить соответствие» / «Проблема не подтверждена» / «Недостаточно данных» (+ EN parity), confirm-button restyled from `bg-destructive` to a positive token, `decide_checkpoint`/`decide_approval` MCP tool descriptions and 045 monitor docs carry the outcome table, vitest pins updated (`WaitingForMeView.test.ts`, `HumanCheckpointPanel` tests). Evidence: `DISP-001` row. Доказательство (2026-09-07): `waiting_disposition_confirm/false_positive` в ru/en `dashboard-testing.json` именуют персистентный исход (confirm→passed: «Подтвердить соответствие»/"Confirm conformance"; false_positive: «Проблема не подтверждена»/"Issue not confirmed"; inconclusive без изменений); confirm-кнопки `HumanCheckpointPanel.svelte` и `WaitingForMeView.svelte` переведены `bg-destructive` → `bg-primary` (прецедент approve-кнопки ApprovalDecisionPanel) + @RATIONALE в контрактах компонентов; API-вокабуляр не переименован (continuity аудита/CAS); MCP `decide_checkpoint` docstring (видим в tools/list) несёт immutable outcome-mapping и правило «confirm = проверка пройдена, никогда не подтверждение дефекта»; vitest-пины обновлены (`RunMonitorViews.test.ts` — новый DISP-001 pin лейбла+стиля+dispatch "confirm", `WaitingForMeView.test.ts`, `run.ux.test.ts`). Lifecycle-маппинг не менялся — backend-тесты decision-пути зелёные без правок. Gate: frontend `3507 passed` (206 файлов), lint `0 errors / 364 warnings` (baseline), `npm run build` OK.
|
||||
- [x] T029m Live-stand replay of the field run: after T029i/T029k/T029l, re-run the 2026-09-07 sales scenario externally against the live stand — `inspect_dashboard_context` → derived-capability compile/validate → `create_agent_run` → `register_draft_pack` → `bootstrap_authoring_scenario` → `start_scenario_run` (PROD gate observed, no unexpected high-risk approvals) → `list_checkpoints`/`decide_checkpoint` human loop with aligned labels; retain the run report beside `docs/2026-09-07-sales-prod-mcp-run.md`. Evidence: `E2E-EXT-002` row. **Status (2026-09-11): CLOSED.** Committed replay client `specs/044-dashboard-scenario-execution/prototype/live_mcp_replay.py` (transport-only adaptation of the proven scripted OAuth/PKCE + tool-chain flows, no new client machinery) replayed the FULL external chain on the live stand with zero non-MCP seeding: inspect (derived capabilities) → `create_agent_run` `a6168c7d…` → compile B01 (capability_authority `derived`) → validate → draft-pack `save_eligible` → `register_draft_pack` (**context_authority verified**) → bootstrap (scenario `e2687f03…`, current revision) → PROD start `pending_approval` (identical retry = one durable gate) → approval → live execution to an honest typed terminal (`capture_screenshot` **passed** with 8 durable refs; `apply_native_filter` typed `BROWSER_ACTION_NOT_SUPPORTED` — no synthesized PASS). Full trace + two fail-closed defects the replay exposed and fixed (binding resolution for identity-less compiled steps; per-step target-identity stamping at bootstrap materialization) + D5 binding-admin live exercise: `docs/2026-09-11-sales-prod-mcp-replay.md`. The baseline-pinned variant remains blocked on the Gitea PAT (050 T045 / T046-pin).
|
||||
- [x] T029m Live-stand replay of the field run: after T029i/T029k/T029l, re-run the 2026-09-07 sales scenario externally against the live stand — `inspect_dashboard_context` → derived-capability compile/validate → `create_agent_run` → `register_draft_pack` → `bootstrap_authoring_scenario` → `start_scenario_run` (PROD gate observed, no unexpected high-risk approvals) → `list_checkpoints`/`decide_checkpoint` human loop with aligned labels; retain the run report beside `docs/2026-09-07-sales-prod-mcp-run.md`. Evidence: `E2E-EXT-002` row. **Status (2026-09-11): CLOSED.** Committed replay client `specs/044-dashboard-scenario-execution/prototype/live_mcp_replay.py` (transport-only adaptation of the proven scripted OAuth/PKCE + tool-chain flows, no new client machinery) replayed the FULL external chain on the live stand with zero non-MCP seeding: inspect (derived capabilities) → `create_agent_run` `a6168c7d…` → compile B01 (capability_authority `derived`) → validate → draft-pack `save_eligible` → `register_draft_pack` (**context_authority verified**) → bootstrap (scenario `e2687f03…`, current revision) → PROD start `pending_approval` (identical retry = one durable gate) → approval → live execution to an honest typed terminal (`capture_screenshot` **passed** with 8 durable refs; `apply_native_filter` typed `BROWSER_ACTION_NOT_SUPPORTED` — no synthesized PASS). Full trace + two fail-closed defects the replay exposed and fixed (binding resolution for identity-less compiled steps; per-step target-identity stamping at bootstrap materialization) + D5 binding-admin live exercise: `docs/2026-09-11-sales-prod-mcp-replay.md`. The human-checkpoint loop was exercised live by the companion `live_mcp_human_loop.py` (compiled B05 HumanCheckpoint → PROD gate → `waiting_human` → MCP `list_checkpoints` decision_version 1 → `decide_checkpoint confirm` decision_version 2 CAS → terminal `passed`), so every element of this task's formulation is covered. The baseline-pinned variant remains blocked on the Gitea PAT (050 T045 / T046-pin).
|
||||
|
||||
## Phase 3 — Frontend decommission (flag-driven)
|
||||
|
||||
@@ -93,7 +93,7 @@ Historical [x] rows above retain only their dated local/transport evidence; they
|
||||
Contract: [Contract-complete public parity](contracts/modules.md).
|
||||
|
||||
- [ ] T044 [P0/P1/P2] REST/MCP lifecycle/read/auth errors and disabled automation validation are identical; service principal cannot decide human gate. Implement at the existing 050 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md. **Status (2026-09-11):** offline parity tests green; live REST-only canary v2 exercised start→gate→approve→dispatch (`4eebfab3`); the live MCP-external replay (T029m CLOSED, `docs/2026-09-11-sales-prod-mcp-replay.md`) exercised the same lifecycle through `/mcp` with the identical typed start/gate/terminal outcomes — remaining: dedicated REST-vs-MCP error-shape parity fixtures for this matrix.
|
||||
- [ ] T045 [P0/P1/P2] consume/publish failures return typed errors/pending state with no legacy fallback; every prerequisite is externally MCP-reachable. Implement at the existing 050 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md. **Status (2026-09-10):** the caller-side published-catalog source returns typed `None`/fail-closed (commit `bbbd4ccf`); Gitea publication path BLOCKED on a valid PAT. No live publish failure canary yet.
|
||||
- [ ] T045 [P0/P1/P2] consume/publish failures return typed errors/pending state with no legacy fallback; every prerequisite is externally MCP-reachable. Implement at the existing 050 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md. **Status (2026-09-11):** the caller-side published-catalog source returns typed `None`/fail-closed (commit `bbbd4ccf`); the publisher helper landed 2026-09-11 (`backend/src/scripts/publish_catalog.py`, pre-validate + Gitea contents PUT, typed auth/conflict/unavailable; `tests/scripts/test_publish_catalog.py` offline-green). Gitea publication path remains BLOCKED on a valid PAT. No live publish failure canary yet.
|
||||
- [ ] T046 [P0/P1/P2] Fresh external-client chain preserves authoritative context and complete baseline pin without raw ORM/REST repair; no frontend agent controls/routes/requests. Implement at the existing 050 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md. **Status (2026-09-11):** a fresh REST client chain (canary v2 `4eebfab3`) AND the fresh MCP-external chain (T029m CLOSED) both preserve the server-resolved binding and the verified authoritative context live (`context_authority=verified` at the register boundary; binding adopted server-side and exercised by live providers). The complete baseline pin is still blocked on the Gitea PAT.
|
||||
|
||||
Frontend boundary for this package: manual CRUD/editor, human review/approval, monitoring and read-only evidence/evaluation only; all agent interaction is external MCP. No agent chat/prompt/assistant editing/proposal generation/workspace/start/handoff controls. Runtime removal is OPEN, not performed by this spec refresh. Optional approved performance baseline is outside scope.
|
||||
|
||||
Reference in New Issue
Block a user