feat(scenario): complete offline agentic runtime chain
This commit is contained in:
79
backend/alembic/versions/0022_agent_evaluation.py
Normal file
79
backend/alembic/versions/0022_agent_evaluation.py
Normal file
@@ -0,0 +1,79 @@
|
||||
# #region Migrations.AgentEvaluation [C:3] [TYPE Module] [SEMANTICS migration,scenario,evaluation]
|
||||
# @BRIEF Create the append-only AgentEvaluation projection after context authority.
|
||||
# @RATIONALE A dedicated table preserves immutable provider output and the identity CAS boundary.
|
||||
# @RATIONALE Idempotent inspector guards (as in 0019) are mandatory: 0001_baseline runs
|
||||
# Base.metadata.create_all, which already creates every model table including this one on
|
||||
# fresh databases; the guarded DDL only materializes on databases migrated before 0022.
|
||||
# @REJECTED Unguarded create_table/add_column was rejected — it collides with the 0001 create_all
|
||||
# bootstrap and breaks every fresh SQLite test schema.
|
||||
"""agent evaluation append-only projection"""
|
||||
from alembic import op
|
||||
import sqlalchemy as sa
|
||||
|
||||
revision = "0022_agent_evaluation"
|
||||
down_revision = "0021_context_authority"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
inspector = sa.inspect(op.get_bind())
|
||||
tables = set(inspector.get_table_names())
|
||||
if "agent_evaluations" not in tables:
|
||||
op.create_table(
|
||||
"agent_evaluations",
|
||||
sa.Column("evaluation_id", sa.String(36), primary_key=True),
|
||||
sa.Column("schema_version", sa.Integer(), nullable=False),
|
||||
sa.Column("scenario_run_id", sa.String(36), nullable=False),
|
||||
sa.Column("logical_step_id", sa.String(128), nullable=False),
|
||||
sa.Column("attempt", sa.Integer(), nullable=False),
|
||||
sa.Column("operation_id", sa.String(36), nullable=False),
|
||||
sa.Column("evaluation_spec_hash", sa.String(64), nullable=False),
|
||||
sa.Column("provider_id", sa.String(128), nullable=False),
|
||||
sa.Column("provider_version", sa.String(128), nullable=False),
|
||||
sa.Column("model_id", sa.String(128), nullable=False),
|
||||
sa.Column("model_version", sa.String(128), nullable=False),
|
||||
sa.Column("prompt_template_id", sa.String(128), nullable=False),
|
||||
sa.Column("prompt_template_version", sa.String(64), nullable=False),
|
||||
sa.Column("prompt_template_hash", sa.String(64), nullable=False),
|
||||
sa.Column("output_schema_hash", sa.String(64), nullable=False),
|
||||
sa.Column("input_manifest_hash", sa.String(64), nullable=False),
|
||||
sa.Column("input_manifest", sa.JSON(), nullable=False),
|
||||
sa.Column("baseline_pin", sa.JSON(), nullable=False),
|
||||
sa.Column("comparison_ids", sa.JSON(), nullable=False),
|
||||
sa.Column("status", sa.String(32), nullable=False),
|
||||
sa.Column("verdict", sa.String(32), nullable=False),
|
||||
sa.Column("confidence", sa.Float(), nullable=False),
|
||||
sa.Column("findings", sa.JSON(), nullable=False),
|
||||
sa.Column("reason_codes", sa.JSON(), nullable=False),
|
||||
sa.Column("raw_response_artifact_ref", sa.String(256)),
|
||||
sa.Column("raw_response_sha256", sa.String(64)),
|
||||
sa.Column("trust_policy_hash", sa.String(64), nullable=False),
|
||||
sa.Column("usage", sa.JSON(), nullable=False),
|
||||
sa.Column("started_at", sa.DateTime(), nullable=False),
|
||||
sa.Column("finished_at", sa.DateTime(), nullable=False),
|
||||
sa.UniqueConstraint(
|
||||
"scenario_run_id", "logical_step_id", "attempt", "operation_id",
|
||||
name="uq_agent_evaluations_identity",
|
||||
),
|
||||
)
|
||||
op.create_index("ix_agent_evaluations_run_step", "agent_evaluations", ["scenario_run_id", "logical_step_id"])
|
||||
artifact_columns = {col["name"] for col in inspector.get_columns("scenario_artifacts")}
|
||||
if "content_type" not in artifact_columns:
|
||||
op.add_column("scenario_artifacts", sa.Column("content_type", sa.String(128), nullable=True))
|
||||
if "byte_length" not in artifact_columns:
|
||||
op.add_column("scenario_artifacts", sa.Column("byte_length", sa.Integer(), nullable=True))
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
inspector = sa.inspect(op.get_bind())
|
||||
tables = set(inspector.get_table_names())
|
||||
if "agent_evaluations" in tables:
|
||||
op.drop_index("ix_agent_evaluations_run_step", table_name="agent_evaluations")
|
||||
op.drop_table("agent_evaluations")
|
||||
artifact_columns = {col["name"] for col in inspector.get_columns("scenario_artifacts")}
|
||||
if "byte_length" in artifact_columns:
|
||||
op.drop_column("scenario_artifacts", "byte_length")
|
||||
if "content_type" in artifact_columns:
|
||||
op.drop_column("scenario_artifacts", "content_type")
|
||||
# #endregion Migrations.AgentEvaluation
|
||||
@@ -7,6 +7,7 @@
|
||||
# @RELATION DEPENDS_ON -> [Api.DashboardTesting.StructureDiff]
|
||||
# @RELATION DEPENDS_ON -> [Api.DashboardTesting.StructureSnapshot]
|
||||
# @RELATION DEPENDS_ON -> [Api.DashboardTesting.VerificationRuns]
|
||||
# @RELATION DEPENDS_ON -> [Api.ScenarioArtifactContent]
|
||||
# @INVARIANT The exported `router` includes all submodule routes under /api/dashboard-testing.
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -20,6 +21,9 @@ from src.api.routes.dashboard_testing.scenario import router as scenario_router
|
||||
from src.api.routes.dashboard_testing.scenario_analytics import router as scenario_analytics_router
|
||||
from src.api.routes.dashboard_testing.scenario_automation import router as scenario_automation_router
|
||||
from src.api.routes.dashboard_testing.scenario_run_center import router as scenario_run_center_router
|
||||
from src.api.routes.dashboard_testing.scenario_artifact_content import (
|
||||
router as scenario_artifact_content_router,
|
||||
)
|
||||
from src.api.routes.dashboard_testing.scenario_runs import (
|
||||
history_router as scenario_runs_history_router,
|
||||
runs_router as scenario_runs_router,
|
||||
@@ -43,6 +47,7 @@ router.include_router(inheritance_router)
|
||||
router.include_router(scenario_router)
|
||||
router.include_router(scenarios_router)
|
||||
router.include_router(scenario_runs_router)
|
||||
router.include_router(scenario_artifact_content_router)
|
||||
router.include_router(scenario_runs_history_router)
|
||||
router.include_router(scenario_run_center_router)
|
||||
router.include_router(scenario_analytics_router)
|
||||
|
||||
@@ -12,6 +12,10 @@
|
||||
# (ConfigManager stage/is_production) — never derived from the environment name.
|
||||
# @INVARIANT 046 routes pass a server-owned automation origin to 044. A revision with a human
|
||||
# step is rejected as manual-run-only before a ScenarioRun, gate, or notification exists.
|
||||
# @INVARIANT All operational reads (schedules/trigger-rules/policies/notifications/metrics/retention)
|
||||
# require an authenticated principal (SEC-01); parity with MCP reads whose permission is
|
||||
# human-only. Schedule/trigger-rule persistence additionally runs the shared revision-bound
|
||||
# eligibility guard before any row is written (DEF-02) — REST and MCP never diverge.
|
||||
# @REJECTED A parallel scheduler was rejected — this surface persists configuration; APScheduler
|
||||
# job registration stays server-owned via the 037 framework.
|
||||
# @REJECTED str(environment_id).startswith("prod") PROD classification was rejected — a client-name
|
||||
@@ -26,6 +30,7 @@ from pydantic import BaseModel, Field
|
||||
from src.dependencies import get_config_manager, get_current_user, get_db, get_scheduler_service, has_permission
|
||||
from src.models.scenario_automation import AutomationPolicy, ScenarioNotificationEvent, ScenarioSchedule, ScenarioTriggerRule
|
||||
from src.services.dashboard_testing.automation.metrics import automation_metrics
|
||||
from src.services.dashboard_testing.automation.eligibility import assert_automation_eligible
|
||||
from src.services.dashboard_testing.automation.retention import tier_limits
|
||||
from src.services.dashboard_testing.automation.trigger import dispatch_trigger_event
|
||||
from src.services.dashboard_testing.execution.environment_policy import (
|
||||
@@ -125,13 +130,17 @@ def _validate_missed_policy(policy: str) -> None:
|
||||
# @INVARIANT Schedule CRUD may register work but never claims or executes a queued ScenarioRun;
|
||||
# the separate 044 queued dispatcher remains the sole initial execution authority.
|
||||
@router.get("/schedules")
|
||||
def list_schedules(db=_DB):
|
||||
def list_schedules(db=_DB, _user=_USER):
|
||||
return db.query(ScenarioSchedule).order_by(ScenarioSchedule.created_at.desc()).all()
|
||||
|
||||
|
||||
@router.post("/schedules", status_code=status.HTTP_201_CREATED)
|
||||
def create_schedule(body: ScheduleRequest, db=_DB, _perm=_MANAGE):
|
||||
def create_schedule(body: ScheduleRequest, db=_DB, _perm=_MANAGE, _user=_USER):
|
||||
_validate_missed_policy(body.missed_execution_policy)
|
||||
try:
|
||||
assert_automation_eligible(db, body.scenario_id, body.revision_id)
|
||||
except ValueError as exc:
|
||||
raise HTTPException(status_code=409, detail={"code": str(exc)}) from exc
|
||||
schedule = ScenarioSchedule(**body.model_dump())
|
||||
db.add(schedule)
|
||||
db.commit()
|
||||
@@ -154,11 +163,15 @@ def create_schedule(body: ScheduleRequest, db=_DB, _perm=_MANAGE):
|
||||
|
||||
|
||||
@router.patch("/schedules/{schedule_id}")
|
||||
def update_schedule(schedule_id: str, body: ScheduleRequest, db=_DB, _perm=_MANAGE):
|
||||
def update_schedule(schedule_id: str, body: ScheduleRequest, db=_DB, _perm=_MANAGE, _user=_USER):
|
||||
schedule = db.query(ScenarioSchedule).filter(ScenarioSchedule.id == schedule_id).first()
|
||||
if schedule is None:
|
||||
raise HTTPException(status_code=404, detail={"code": "SCHEDULE_NOT_FOUND"})
|
||||
_validate_missed_policy(body.missed_execution_policy)
|
||||
try:
|
||||
assert_automation_eligible(db, body.scenario_id, body.revision_id)
|
||||
except ValueError as exc:
|
||||
raise HTTPException(status_code=409, detail={"code": str(exc)}) from exc
|
||||
for key, value in body.model_dump(exclude_unset=True).items():
|
||||
setattr(schedule, key, value)
|
||||
db.commit()
|
||||
@@ -197,13 +210,17 @@ def delete_schedule(schedule_id: str, db=_DB, _perm=_MANAGE):
|
||||
# #region Api.ScenarioAutomation.TriggerRules [C:3] [TYPE Block] [SEMANTICS scenario,automation,api,trigger,crud]
|
||||
# @ingroup Api
|
||||
@router.get("/trigger-rules")
|
||||
def list_trigger_rules(db=_DB):
|
||||
def list_trigger_rules(db=_DB, _user=_USER):
|
||||
return db.query(ScenarioTriggerRule).order_by(ScenarioTriggerRule.created_at.desc()).all()
|
||||
|
||||
|
||||
@router.post("/trigger-rules", status_code=status.HTTP_201_CREATED)
|
||||
def create_trigger_rule(body: TriggerRuleRequest, db=_DB, _perm=_MANAGE):
|
||||
def create_trigger_rule(body: TriggerRuleRequest, db=_DB, _perm=_MANAGE, _user=_USER):
|
||||
_validate_rule_trigger(body.trigger)
|
||||
try:
|
||||
assert_automation_eligible(db, body.scenario_id, body.revision_id)
|
||||
except ValueError as exc:
|
||||
raise HTTPException(status_code=409, detail={"code": str(exc)}) from exc
|
||||
rule = ScenarioTriggerRule(**body.model_dump())
|
||||
db.add(rule)
|
||||
db.commit()
|
||||
@@ -212,11 +229,15 @@ def create_trigger_rule(body: TriggerRuleRequest, db=_DB, _perm=_MANAGE):
|
||||
|
||||
|
||||
@router.patch("/trigger-rules/{rule_id}")
|
||||
def update_trigger_rule(rule_id: str, body: TriggerRuleRequest, db=_DB, _perm=_MANAGE):
|
||||
def update_trigger_rule(rule_id: str, body: TriggerRuleRequest, db=_DB, _perm=_MANAGE, _user=_USER):
|
||||
rule = db.query(ScenarioTriggerRule).filter(ScenarioTriggerRule.id == rule_id).first()
|
||||
if rule is None:
|
||||
raise HTTPException(status_code=404, detail={"code": "TRIGGER_RULE_NOT_FOUND"})
|
||||
_validate_rule_trigger(body.trigger)
|
||||
try:
|
||||
assert_automation_eligible(db, body.scenario_id, body.revision_id)
|
||||
except ValueError as exc:
|
||||
raise HTTPException(status_code=409, detail={"code": str(exc)}) from exc
|
||||
for key, value in body.model_dump(exclude_unset=True).items():
|
||||
setattr(rule, key, value)
|
||||
db.commit()
|
||||
@@ -238,7 +259,7 @@ def delete_trigger_rule(rule_id: str, db=_DB, _perm=_MANAGE):
|
||||
# #region Api.ScenarioAutomation.Policies [C:3] [TYPE Block] [SEMANTICS scenario,automation,api,policy,crud]
|
||||
# @ingroup Api
|
||||
@router.get("/policies")
|
||||
def list_policies(db=_DB):
|
||||
def list_policies(db=_DB, _user=_USER):
|
||||
return db.query(AutomationPolicy).order_by(AutomationPolicy.created_at.desc()).all()
|
||||
|
||||
|
||||
@@ -277,7 +298,7 @@ def delete_policy(policy_id: str, db=_DB, _perm=_MANAGE):
|
||||
# #region Api.ScenarioAutomation.Notifications [C:2] [TYPE Function] [SEMANTICS scenario,automation,api,notifications]
|
||||
# @ingroup Api
|
||||
@router.get("/notifications")
|
||||
def list_notifications(limit: int = 100, db=_DB):
|
||||
def list_notifications(limit: int = 100, db=_DB, _user=_USER):
|
||||
return db.query(ScenarioNotificationEvent).order_by(ScenarioNotificationEvent.created_at.desc()).limit(limit).all()
|
||||
# #endregion Api.ScenarioAutomation.Notifications
|
||||
|
||||
@@ -324,7 +345,7 @@ def api_dispatch_event(
|
||||
# @ingroup Api
|
||||
# @BRIEF Operational metrics over persisted schedules, trigger rules, runs and notifications.
|
||||
@router.get("/metrics")
|
||||
def metrics(db=_DB):
|
||||
def metrics(db=_DB, _user=_USER):
|
||||
schedules = db.query(ScenarioSchedule).all()
|
||||
rules = db.query(ScenarioTriggerRule).all()
|
||||
notifications = db.query(ScenarioNotificationEvent).all()
|
||||
@@ -350,7 +371,7 @@ def metrics(db=_DB):
|
||||
# @ingroup Api
|
||||
# @BRIEF Expose the canonical layered retention tier horizons for the management UI.
|
||||
@router.get("/retention")
|
||||
def retention_defaults():
|
||||
def retention_defaults(_user=_USER):
|
||||
return {"tiers": tier_limits()}
|
||||
# #endregion Api.ScenarioAutomation.RetentionDefaults
|
||||
|
||||
|
||||
@@ -37,6 +37,7 @@ from src.services.dashboard_testing.execution.lifecycle import (
|
||||
retry_step,
|
||||
)
|
||||
from src.services.dashboard_testing.execution.result import build_result
|
||||
from src.services.dashboard_testing.execution.baseline_resolver import BASELINE_RESOLVE_ERRORS
|
||||
from src.services.dashboard_testing.execution.runner import (
|
||||
continue_after_human_decision,
|
||||
continue_after_infrastructure_resume,
|
||||
@@ -65,6 +66,7 @@ class StartRunRequest(BaseModel):
|
||||
is_prod: bool = False
|
||||
dashboard_release_id: str | None = None
|
||||
baseline_set: str | None = None
|
||||
baseline_set_version: str | None = None
|
||||
execution_toggles: dict[str, bool] = {}
|
||||
|
||||
class HumanDecisionRequest(BaseModel):
|
||||
@@ -125,7 +127,7 @@ def _require_run_prod(current_user) -> None:
|
||||
# @RELATION CALLS -> [ScenarioExecution.Runner.Start]
|
||||
# @RELATION CALLS -> [ScenarioExecution.EnvironmentPolicy.Resolve]
|
||||
# @POST 201 with queued/pending-approval run; 409 IDEMPOTENCY_KEY_REUSED / RUN_START_CONFLICT;
|
||||
# 403 on missing RBAC; 422 ENVIRONMENT_NOT_CONFIGURED before any durable side effect.
|
||||
# 403 on missing RBAC; 422 ENVIRONMENT_NOT_CONFIGURED / BASELINE_* before any durable side effect.
|
||||
# @INVARIANT This authenticated analyst route supplies the trusted manual origin itself; clients
|
||||
# cannot select trigger_source in StartRunRequest.
|
||||
# @INVARIANT HTTP start/replay is persistence-only: it creates or returns a queued/pending row but
|
||||
@@ -156,6 +158,7 @@ def api_start_run(
|
||||
auto_advance=False,
|
||||
dashboard_release_id=body.dashboard_release_id,
|
||||
baseline_set=body.baseline_set,
|
||||
baseline_set_version=body.baseline_set_version,
|
||||
execution_toggles=body.execution_toggles,
|
||||
trigger_source="manual",
|
||||
)
|
||||
@@ -166,7 +169,7 @@ def api_start_run(
|
||||
raise HTTPException(status_code=403, detail={"code": "PROD_APPROVAL_REQUIRED", "detail": str(exc)}) from exc
|
||||
except ValueError as exc:
|
||||
db.rollback()
|
||||
if str(exc) == "ENVIRONMENT_NOT_CONFIGURED":
|
||||
if str(exc) == "ENVIRONMENT_NOT_CONFIGURED" or str(exc) in BASELINE_RESOLVE_ERRORS:
|
||||
raise HTTPException(
|
||||
status_code=422,
|
||||
detail={"code": str(exc), "detail": str(exc)},
|
||||
|
||||
@@ -131,6 +131,9 @@ def initialize_live_execution_composition() -> int:
|
||||
# @INVARIANT The trusted live-composition bootstrap runs after AsyncJobRunner initialization and
|
||||
# before scheduler startup; queued dispatch therefore cannot observe an unbootstrapped
|
||||
# default root during normal application startup.
|
||||
# @INVARIANT Exactly one ProviderEventLoop thread is started before composition bootstrap (ordered
|
||||
# startup per 044 ProviderRuntime: loop ready -> bindings/providers -> scheduler admission),
|
||||
# stopped on bootstrap failure (fail-closed rollback) and joined on shutdown after the task drain.
|
||||
# @RATIONALE Alembic migrations run before the application process starts.
|
||||
# init_db() only records that schema initialization is owned by Alembic;
|
||||
# runtime create_all() and inline ALTER TABLE repairs are intentionally absent.
|
||||
@@ -223,8 +226,17 @@ async def lifespan(app: FastAPI):
|
||||
|
||||
logger.reason("Initializing AsyncJobRunner")
|
||||
get_async_job_runner() # Initialize singleton with running event loop BEFORE scheduler starts
|
||||
from .services.dashboard_testing.execution.provider_runtime import get_provider_event_loop
|
||||
logger.reason("Starting provider event loop")
|
||||
_provider_loop = get_provider_event_loop()
|
||||
_provider_loop.start()
|
||||
logger.reason("Bootstrapping fail-closed ScenarioRun live composition root")
|
||||
live_bindings = initialize_live_execution_composition()
|
||||
try:
|
||||
live_bindings = initialize_live_execution_composition()
|
||||
except Exception:
|
||||
from .services.dashboard_testing.execution.provider_runtime import stop_provider_event_loop
|
||||
stop_provider_event_loop()
|
||||
raise
|
||||
logger.reason("ScenarioRun live composition bootstrap complete", payload={"bindings": live_bindings})
|
||||
logger.reason("Starting scheduler")
|
||||
scheduler = get_scheduler_service()
|
||||
@@ -263,6 +275,8 @@ async def lifespan(app: FastAPI):
|
||||
except Exception as _e:
|
||||
logger.explore("Graceful task drain during shutdown encountered error", error=str(_e))
|
||||
await mcp_lifespan.__aexit__(None, None, None)
|
||||
from .services.dashboard_testing.execution.provider_runtime import stop_provider_event_loop
|
||||
stop_provider_event_loop()
|
||||
|
||||
|
||||
# #endregion App.AppModule.Lifespan
|
||||
@@ -492,11 +506,11 @@ async def network_error_handler(request: Request, exc: NetworkError):
|
||||
# spam the structured log (they remain visible in the uvicorn access log):
|
||||
# - /api/tasks* task progress polling (every 1.5s during operations)
|
||||
# - /api/health/summary health monitoring polling
|
||||
# - /api/auth/session/activity session activity heartbeats
|
||||
# - /api/settings/consolidated settings polling
|
||||
# - GET /api/scenario-runs?waiting_for_me=true sidebar approval-badge polling (60s)
|
||||
# Session activity heartbeats are POST-only — suppressed via _POLLING_POST_PATHS.
|
||||
_POLLING_EXACT_PATHS = frozenset({
|
||||
"/api/health/summary",
|
||||
"/api/auth/session/activity",
|
||||
"/api/settings/consolidated",
|
||||
"/api/environments",
|
||||
"/api/auth/me",
|
||||
@@ -513,9 +527,10 @@ _POLLING_EXACT_PATHS = frozenset({
|
||||
"/api/reports/task-log-gaps",
|
||||
})
|
||||
|
||||
# Read-only batch endpoints polled via POST — semantically polls, framing suppressed too.
|
||||
# Read-only batch/poll endpoints called via POST — semantically polls, framing suppressed too.
|
||||
_POLLING_POST_PATHS = frozenset({
|
||||
"/api/git/repositories/status/batch",
|
||||
"/api/auth/session/activity",
|
||||
})
|
||||
|
||||
|
||||
@@ -529,9 +544,14 @@ def _is_suppressed_request(request: Request) -> bool:
|
||||
path = request.url.path
|
||||
if request.method == "POST" and path in _POLLING_POST_PATHS:
|
||||
return True
|
||||
return request.method == "GET" and (
|
||||
path in _POLLING_EXACT_PATHS or path.startswith("/api/tasks")
|
||||
)
|
||||
if request.method != "GET":
|
||||
return False
|
||||
# Sidebar approval-badge poll (Stores.ScenarioRuns.WaitingStore, every 60s):
|
||||
# GET /api/scenario-runs?waiting_for_me=true&page_size=1 — only `total` is consumed.
|
||||
# Run-center list fetches (without waiting_for_me) keep full framing.
|
||||
if path == "/api/scenario-runs" and request.query_params.get("waiting_for_me") == "true":
|
||||
return True
|
||||
return path in _POLLING_EXACT_PATHS or path.startswith("/api/tasks")
|
||||
|
||||
|
||||
# #region App.AppModule.LogRequests [C:3] [TYPE Function]
|
||||
|
||||
@@ -67,6 +67,7 @@ class ScenarioStartInput(BaseModel):
|
||||
idempotency_key: str = Field(min_length=1, max_length=256)
|
||||
dashboard_release_id: str | None = Field(default=None, max_length=256)
|
||||
baseline_set: str | None = Field(default=None, max_length=256)
|
||||
baseline_set_version: str | None = Field(default=None, max_length=256)
|
||||
execution_toggles: dict[str, bool] = Field(default_factory=dict)
|
||||
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ from src.models.scenario_run import ScenarioRun
|
||||
from src.models.scenario_registry import ScenarioRegistryEntry, ScenarioRevision
|
||||
from src.services.dashboard_testing.automation.metrics import automation_metrics
|
||||
from src.services.dashboard_testing.automation.schedule import derive_aps_params, validate_timezone
|
||||
from src.services.dashboard_testing.execution.runner_plan import derive_runner_plan
|
||||
from src.services.dashboard_testing.automation.eligibility import assert_automation_eligible
|
||||
|
||||
_MISSED_POLICIES = {"skip", "run_latest", "queue_all"}
|
||||
_TRIGGER_TYPES = {"deploy_to_preprod", "release_created", "etl_completed", "api"}
|
||||
@@ -71,13 +71,7 @@ def register_automation_tools(server: Any) -> None:
|
||||
def _revision(db, request: AutomationBase) -> None:
|
||||
if request.revision_id is None:
|
||||
raise ValueError("revision_id is required")
|
||||
revision = db.get(ScenarioRevision, request.revision_id)
|
||||
entry = db.get(ScenarioRegistryEntry, request.scenario_id)
|
||||
if revision is None or entry is None or revision.scenario_id != request.scenario_id or entry.current_revision_id != revision.revision_id:
|
||||
raise ValueError("revision must be the active scenario revision")
|
||||
plan = derive_runner_plan(db, request.scenario_id, request.revision_id)
|
||||
if plan.get("manual_run_only"):
|
||||
raise ValueError("AUTOMATION_INELIGIBLE_HUMAN_STEP")
|
||||
assert_automation_eligible(db, request.scenario_id, request.revision_id)
|
||||
|
||||
def _policy(db, policy_id: str | None) -> None:
|
||||
if policy_id is not None and db.get(AutomationPolicy, policy_id) is None:
|
||||
|
||||
@@ -402,6 +402,7 @@ def register_scenario_tools(server) -> None:
|
||||
actor=access.subject, idempotency_key=request.idempotency_key,
|
||||
config_manager=get_config_manager(), auto_advance=False,
|
||||
dashboard_release_id=request.dashboard_release_id, baseline_set=request.baseline_set,
|
||||
baseline_set_version=request.baseline_set_version,
|
||||
execution_toggles=request.execution_toggles, trigger_source="manual",
|
||||
)
|
||||
db.commit()
|
||||
|
||||
@@ -21,5 +21,6 @@ from . import (
|
||||
scenario_registry as _scenario_registry, # noqa: F401
|
||||
scenario_run as _scenario_run, # noqa: F401
|
||||
scenario_worker as _scenario_worker, # noqa: F401
|
||||
scenario_evaluation as _scenario_evaluation, # noqa: F401
|
||||
verification_run as _verification_run, # noqa: F401
|
||||
)
|
||||
|
||||
@@ -48,6 +48,8 @@ class ScenarioArtifact(Base):
|
||||
name = Column(String(255), nullable=False)
|
||||
content_ref = Column(String(256), nullable=False)
|
||||
sha256 = Column(String(64), nullable=False)
|
||||
content_type = Column(String(128), nullable=True)
|
||||
byte_length = Column(Integer, nullable=True)
|
||||
retention_class = Column(String(32), nullable=False, default="standard")
|
||||
logical_step_id = Column(String(128), nullable=True, index=True)
|
||||
attempt = Column(Integer, nullable=True)
|
||||
|
||||
43
backend/src/models/scenario_evaluation.py
Normal file
43
backend/src/models/scenario_evaluation.py
Normal file
@@ -0,0 +1,43 @@
|
||||
# #region Models.ScenarioExecution.AgentEvaluation [C:4] [TYPE Class] [SEMANTICS scenario,evaluation,append-only]
|
||||
# @BRIEF Append-only ORM projection of the normative AgentEvaluation record.
|
||||
# @INVARIANT The identity tuple is unique and no update/delete service path exists.
|
||||
# @RATIONALE JSON columns preserve the immutable schema payload without lossy relational projection.
|
||||
# @REJECTED Mutable upsert semantics were rejected because evaluations are audit evidence.
|
||||
from __future__ import annotations
|
||||
from sqlalchemy import Column, DateTime, Float, Index, Integer, JSON, String, UniqueConstraint
|
||||
from .mapping import Base
|
||||
|
||||
class AgentEvaluation(Base):
|
||||
__tablename__ = "agent_evaluations"
|
||||
evaluation_id = Column(String(36), primary_key=True)
|
||||
schema_version = Column(Integer, nullable=False)
|
||||
scenario_run_id = Column(String(36), nullable=False, index=True)
|
||||
logical_step_id = Column(String(128), nullable=False)
|
||||
attempt = Column(Integer, nullable=False)
|
||||
operation_id = Column(String(36), nullable=False)
|
||||
evaluation_spec_hash = Column(String(64), nullable=False)
|
||||
provider_id = Column(String(128), nullable=False)
|
||||
provider_version = Column(String(128), nullable=False)
|
||||
model_id = Column(String(128), nullable=False)
|
||||
model_version = Column(String(128), nullable=False)
|
||||
prompt_template_id = Column(String(128), nullable=False)
|
||||
prompt_template_version = Column(String(64), nullable=False)
|
||||
prompt_template_hash = Column(String(64), nullable=False)
|
||||
output_schema_hash = Column(String(64), nullable=False)
|
||||
input_manifest_hash = Column(String(64), nullable=False)
|
||||
input_manifest = Column(JSON, nullable=False)
|
||||
baseline_pin = Column(JSON, nullable=False)
|
||||
comparison_ids = Column(JSON, nullable=False)
|
||||
status = Column(String(32), nullable=False)
|
||||
verdict = Column(String(32), nullable=False)
|
||||
confidence = Column(Float, nullable=False)
|
||||
findings = Column(JSON, nullable=False)
|
||||
reason_codes = Column(JSON, nullable=False)
|
||||
raw_response_artifact_ref = Column(String(256), nullable=True)
|
||||
raw_response_sha256 = Column(String(64), nullable=True)
|
||||
trust_policy_hash = Column(String(64), nullable=False)
|
||||
usage = Column(JSON, nullable=False)
|
||||
started_at = Column(DateTime, nullable=False)
|
||||
finished_at = Column(DateTime, nullable=False)
|
||||
__table_args__ = (UniqueConstraint("scenario_run_id", "logical_step_id", "attempt", "operation_id", name="uq_agent_evaluations_identity"), Index("ix_agent_evaluations_run_step", "scenario_run_id", "logical_step_id"))
|
||||
# #endregion Models.ScenarioExecution.AgentEvaluation
|
||||
@@ -14,6 +14,7 @@ import json
|
||||
import uuid
|
||||
from typing import Any
|
||||
|
||||
from pydantic import ValidationError
|
||||
from sqlalchemy import select, update
|
||||
from sqlalchemy.exc import IntegrityError
|
||||
from sqlalchemy.orm import Session
|
||||
@@ -29,7 +30,10 @@ from src.models.agent_authoring_workspace import (
|
||||
from src.models.scenario_registry import ScenarioEditProposal, ScenarioRevision
|
||||
from src.services.dashboard_testing.editor.agent import agent_propose, graph_diff, save_proposal
|
||||
from src.services.dashboard_testing.registry.revisions import activate_current_revision
|
||||
from src.services.dashboard_testing.scenario.models import DashboardTestScenario
|
||||
from src.services.dashboard_testing.scenario.sql_guard import contains_unsafe_free_text
|
||||
from src.services.dashboard_testing.scenario.templates import REGISTERED_ACTIONS
|
||||
from src.services.dashboard_testing.scenario.validator import validate_scenario
|
||||
|
||||
|
||||
class WorkspaceError(ValueError):
|
||||
@@ -927,26 +931,44 @@ def _graph_digest(graph: dict[str, Any]) -> str:
|
||||
return hashlib.sha256(payload.encode("utf-8")).hexdigest()
|
||||
|
||||
|
||||
def _has_unsafe_graph_text(value: Any) -> bool:
|
||||
if isinstance(value, str):
|
||||
lowered = value.lower()
|
||||
return any(token in lowered for token in ("select ", "insert ", "update ", "delete ", "drop ", "--", "../", "\\"))
|
||||
if isinstance(value, dict):
|
||||
return any(_has_unsafe_graph_text(item) for item in value.values())
|
||||
if isinstance(value, list):
|
||||
return any(_has_unsafe_graph_text(item) for item in value)
|
||||
# #region Services.AgentAuthoringWorkspace.PromotionGraphValidation [C:3] [TYPE Function] [SEMANTICS agent,authoring,promotion,safety,validation]
|
||||
# @ingroup Services
|
||||
# @BRIEF Structure-first promotion gate: canonical graphs are validated by the 038 validator; legacy snapshots fall back to a free-text-only scan.
|
||||
# @RATIONALE DEF-01 (ss-prod E2E 2026-09-08): naive whole-graph token scan false-rejected a server-derived proposal; closed-schema structure is the authority, and promotion gates safety/cycle codes only — NEEDS_SELECTOR/NEEDS_BASELINE resolution errors belong to the request_save boundary, not the review boundary.
|
||||
# @REJECTED Whole-graph substring scanning ("select ", "--", "\\") — false positives on legitimate server strings; rejecting every 038 validation error at promotion — blocks review of merely save-blocked graphs.
|
||||
_PROMOTION_GATE_ERROR_CODES = frozenset({"FORBIDDEN_SQL", "FORBIDDEN_QUERY_CONTEXT", "PATH_TRAVERSAL", "CYCLE"})
|
||||
_FREE_TEXT_KEYS = frozenset({"title", "description", "label", "goal", "rationale", "notes", "message", "request_text"})
|
||||
|
||||
|
||||
def _has_unsafe_free_text(node: Any, key: str | None = None) -> bool:
|
||||
if isinstance(node, str):
|
||||
return key in _FREE_TEXT_KEYS and contains_unsafe_free_text(node)
|
||||
if isinstance(node, dict):
|
||||
return any(_has_unsafe_free_text(value, str(k).lower()) for k, value in node.items())
|
||||
if isinstance(node, list):
|
||||
return any(_has_unsafe_free_text(item, key) for item in node)
|
||||
return False
|
||||
|
||||
|
||||
def _validate_proposal_graph(graph: Any) -> dict[str, Any]:
|
||||
findings: list[str] = []
|
||||
if not isinstance(graph, dict):
|
||||
findings.append("proposed graph is not an object")
|
||||
elif not graph:
|
||||
findings.append("proposed graph is empty")
|
||||
elif _has_unsafe_graph_text(graph):
|
||||
findings.append("proposed graph contains unsafe SQL, executable, or path content")
|
||||
return {"status": "invalid", "findings": ["proposed graph is not an object"]}
|
||||
if not graph:
|
||||
return {"status": "invalid", "findings": ["proposed graph is empty"]}
|
||||
try:
|
||||
scenario = DashboardTestScenario.model_validate(graph)
|
||||
except ValidationError:
|
||||
# Persisted legacy snapshot: typed model cannot adjudicate; scan free-text fields only.
|
||||
if _has_unsafe_free_text(graph):
|
||||
return {"status": "invalid", "findings": ["proposed graph contains unsafe SQL, executable, or path content"]}
|
||||
return {"status": "valid", "findings": []}
|
||||
findings = [
|
||||
f"{finding.code}: {finding.message}"
|
||||
for finding in validate_scenario(scenario).errors
|
||||
if finding.code in _PROMOTION_GATE_ERROR_CODES
|
||||
]
|
||||
return {"status": "invalid" if findings else "valid", "findings": findings}
|
||||
# #endregion Services.AgentAuthoringWorkspace.PromotionGraphValidation
|
||||
|
||||
|
||||
def _promotion_projection(db: Session, proposal: ScenarioEditProposal, new_status: str, cas_version: int) -> dict[str, Any]:
|
||||
|
||||
@@ -21,6 +21,9 @@ from src.models.scenario_run import ScenarioRun, ScenarioStepRun
|
||||
|
||||
|
||||
# #region ScenarioAnalytics.Investigation.Queue [C:3] [TYPE Function] [SEMANTICS scenario,analytics,queue,project]
|
||||
# @BRIEF Idempotently project immutable run signals into analyst-owned queue items with an atomic dequeue lifecycle.
|
||||
# @INVARIANT Queue status lifecycle is queued -> in_case (open_case locks the row FOR UPDATE) -> closed (resolved/accepted disposition); a closed fingerprint only requeues as a NEW row from a later recurrence signal.
|
||||
# @INVARIANT Occurrence merging targets one active row per fingerprint (queued preferred over in_case); closed rows are immutable history.
|
||||
def queue_signal(
|
||||
db: Session,
|
||||
*,
|
||||
@@ -30,8 +33,8 @@ def queue_signal(
|
||||
severity: str = "warning",
|
||||
evidence: dict[str, Any] | None = None,
|
||||
) -> InvestigationQueueItem:
|
||||
"""Idempotently project immutable evidence into an analyst-owned queue item."""
|
||||
item = db.query(InvestigationQueueItem).filter(InvestigationQueueItem.fingerprint == fingerprint, InvestigationQueueItem.status == "queued").first()
|
||||
"""Idempotently project immutable evidence into the active (queued|in_case) queue item for the fingerprint."""
|
||||
item = db.query(InvestigationQueueItem).filter(InvestigationQueueItem.fingerprint == fingerprint, InvestigationQueueItem.status.in_(("queued", "in_case"))).order_by(InvestigationQueueItem.status.desc()).first()
|
||||
if item is not None:
|
||||
item.occurrence_count += 1
|
||||
if run_id and run_id not in (item.evidence_snapshot or {}).get("run_ids", []):
|
||||
@@ -190,6 +193,9 @@ def ingest_investigation_signal(db: Session, signal: dict[str, Any]) -> Investig
|
||||
# #endregion ScenarioAnalytics.Investigation.IngestSignal
|
||||
|
||||
# #region ScenarioAnalytics.Investigation.Case [C:4] [TYPE Function] [SEMANTICS scenario,analytics,case,open,disposition]
|
||||
# @BRIEF Explicit analyst-owned case lifecycle: open (atomic dequeue), disposition CAS with queue closure, object ACL.
|
||||
# @INVARIANT open_case locks the queued row FOR UPDATE and transitions it to in_case before creating the case; the queued-row precondition is enforced by the HTTP route (QUEUE_ITEM_INACTIVE 409), while direct service calls without a queue signal keep legacy case-creation semantics (DEF-04).
|
||||
# @INVARIANT resolved/accepted disposition closes every active (queued|in_case) queue row of the fingerprint, so an accepted case never reappears in the analyst queue; recurrence after closure creates a NEW queued row.
|
||||
|
||||
def auto_queue_failed_run(
|
||||
db: Session,
|
||||
@@ -218,8 +224,10 @@ def open_case(db: Session, *, fingerprint: str, scenario_id: str, actor_id: str)
|
||||
queue = db.query(InvestigationQueueItem).filter(
|
||||
InvestigationQueueItem.fingerprint == fingerprint,
|
||||
InvestigationQueueItem.status == "queued",
|
||||
).first()
|
||||
).with_for_update().first()
|
||||
snapshot = dict(queue.evidence_snapshot or {}) if queue else {}
|
||||
if queue is not None:
|
||||
queue.status = "in_case"
|
||||
run_ids = list(snapshot.get("run_ids", []))
|
||||
case = InvestigationCase(
|
||||
id=str(uuid.uuid4()),
|
||||
@@ -261,6 +269,12 @@ def set_disposition(
|
||||
else:
|
||||
raise ValueError("disposition must be resolved or accepted")
|
||||
case.disposition = disposition
|
||||
# synchronize_session="evaluate" keeps in-session queue rows consistent with the closed
|
||||
# status; a stale identity map would leak "in_case" into same-request projections.
|
||||
db.query(InvestigationQueueItem).filter(
|
||||
InvestigationQueueItem.fingerprint == case.fingerprint,
|
||||
InvestigationQueueItem.status.in_(("queued", "in_case")),
|
||||
).update({InvestigationQueueItem.status: "closed"}, synchronize_session="evaluate")
|
||||
case.decision_version += 1
|
||||
db.add(AgentAction(case_id=case.id, action_type="disposition", payload={"actor_id": actor_id, "disposition": disposition, "verification_evidence": verification_evidence or {}, "rationale": rationale}))
|
||||
from src.models.scenario_investigation import RecurringFailureEpisode
|
||||
|
||||
@@ -10,6 +10,7 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
from pydantic import ValidationError
|
||||
|
||||
from .ops import (
|
||||
AddStepOp,
|
||||
@@ -19,17 +20,14 @@ from .ops import (
|
||||
SetDependencyOp,
|
||||
SetParameterDefinitionOp,
|
||||
)
|
||||
from src.services.dashboard_testing.scenario.models import DashboardTestScenario, Expected, ScenarioStep, sha256_hex
|
||||
from src.services.dashboard_testing.scenario.templates import STEP_TEMPLATES, REGISTERED_ACTIONS
|
||||
from src.services.dashboard_testing.scenario.validator import validate_scenario
|
||||
from src.services.dashboard_testing.scenario.sql_guard import contains_unsafe_free_text
|
||||
|
||||
|
||||
def _contains_unsafe_text(value: Any) -> bool:
|
||||
if isinstance(value, str):
|
||||
lowered = value.lower()
|
||||
return any(token in lowered for token in ("select ", "insert ", "update ", "delete ", "drop ", "--", "../", "\\"))
|
||||
if isinstance(value, dict):
|
||||
return any(_contains_unsafe_text(item) for item in value.values())
|
||||
if isinstance(value, list):
|
||||
return any(_contains_unsafe_text(item) for item in value)
|
||||
return False
|
||||
return contains_unsafe_free_text(value)
|
||||
|
||||
|
||||
# #region ScenarioEditor.Apply.ValidateAssertion [C:3] [TYPE Function] [SEMANTICS scenario,editor,assertion,constrain]
|
||||
@@ -38,7 +36,7 @@ def _contains_unsafe_text(value: Any) -> bool:
|
||||
# @PRE edit is a SetAssertionOp.
|
||||
# @POST Returns a normalized assertion; rejects SQL/raw baseline values and missing thresholds.
|
||||
def validate_assertion(edit: SetAssertionOp) -> dict[str, Any]:
|
||||
if _contains_unsafe_text(edit.model_dump()):
|
||||
if _contains_unsafe_text(edit.baseline_ref):
|
||||
raise ValueError("unsafe assertion payload")
|
||||
if not edit.baseline_ref.startswith("baseline:"):
|
||||
raise ValueError("baseline_ref must reference an approved baseline")
|
||||
@@ -54,38 +52,122 @@ def validate_assertion(edit: SetAssertionOp) -> dict[str, Any]:
|
||||
# @PRE base_graph is a server-loaded revision snapshot; ops are the closed EditOperation union.
|
||||
# @POST Returns a new graph plus normalized validation findings; input graph is not mutated.
|
||||
def apply_ops(base_graph: dict[str, Any], ops: list[EditOperation]) -> dict[str, Any]:
|
||||
graph = {**base_graph}
|
||||
graph["parameters"] = {**(base_graph.get("parameters") or {})}
|
||||
graph["assertions"] = {**(base_graph.get("assertions") or {})}
|
||||
graph["steps"] = [dict(step) for step in (base_graph.get("steps") or [])]
|
||||
graph["dependencies"] = [dict(edge) for edge in (base_graph.get("dependencies") or [])]
|
||||
for operation in ops:
|
||||
if _contains_unsafe_text(operation.model_dump()):
|
||||
if any(contains_unsafe_free_text(getattr(operation, field, None)) for field in ("baseline_ref", "value")):
|
||||
raise ValueError("unsafe edit operation")
|
||||
try:
|
||||
scenario = DashboardTestScenario.model_validate(base_graph)
|
||||
except ValidationError:
|
||||
return _apply_legacy_ops(base_graph, ops)
|
||||
graph = scenario.model_dump(mode="json")
|
||||
for operation in ops:
|
||||
_apply_operation(graph, operation)
|
||||
if _has_cycle(graph["dependencies"]):
|
||||
candidate = DashboardTestScenario.model_validate(graph)
|
||||
validation = validate_scenario(candidate)
|
||||
if any(f.code == "CYCLE" for f in validation.errors):
|
||||
raise ValueError("dependency cycle rejected")
|
||||
graph["revision_hash"] = sha256_hex(DashboardTestScenario.model_validate(graph).canonical_bytes())
|
||||
return graph
|
||||
# #endregion ScenarioEditor.Apply.Ops
|
||||
|
||||
|
||||
# #region ScenarioEditor.Apply.LegacyOps [C:3] [TYPE Function] [SEMANTICS scenario,editor,legacy,compatibility]
|
||||
# @ingroup ScenarioEditor
|
||||
# @BRIEF Apply typed edits to persisted pre-038 graph snapshots without fabricating new fields.
|
||||
# @RATIONALE Existing ScenarioRevision rows may contain the historical editor shape; preserving that shape avoids corrupting durable revisions while new canonical graphs use the typed path.
|
||||
# @REJECTED Coercing legacy `parameters` mappings to lists and materializing absent `assertions`/`dependencies` keys — DEF-01: fabricated fields corrupted proposal diffs.
|
||||
def _apply_legacy_ops(base_graph: dict[str, Any], ops: list[EditOperation]) -> dict[str, Any]:
|
||||
graph = {**base_graph}
|
||||
graph["parameters"] = dict(base_graph.get("parameters") or {}) if isinstance(base_graph.get("parameters"), dict) else list(base_graph.get("parameters") or [])
|
||||
if "assertions" in base_graph:
|
||||
graph["assertions"] = {**(base_graph.get("assertions") or {})}
|
||||
graph["steps"] = [dict(step) for step in (base_graph.get("steps") or [])]
|
||||
if "dependencies" in base_graph:
|
||||
graph["dependencies"] = [dict(edge) for edge in (base_graph.get("dependencies") or [])]
|
||||
for operation in ops:
|
||||
if isinstance(operation, SetParameterDefinitionOp):
|
||||
if not isinstance(graph["parameters"], dict):
|
||||
raise ValueError("legacy graph parameters are not a mapping")
|
||||
graph["parameters"][operation.param_name] = operation.value
|
||||
elif isinstance(operation, SetAssertionOp):
|
||||
if "assertions" not in graph:
|
||||
graph["assertions"] = {}
|
||||
graph["assertions"][operation.logical_step_id] = validate_assertion(operation)
|
||||
elif isinstance(operation, AddStepOp):
|
||||
if "dependencies" not in graph:
|
||||
raise ValueError("legacy graph has no dependencies collection")
|
||||
if any(step.get("template") == operation.template for step in graph["steps"]):
|
||||
raise ValueError("duplicate step template")
|
||||
step_id = f"new:{operation.template}"
|
||||
graph["steps"].append({"template": operation.template, "logical_step_id": step_id})
|
||||
if operation.after_logical_step_id:
|
||||
graph["dependencies"].append({"source": operation.after_logical_step_id, "target": step_id})
|
||||
elif isinstance(operation, RemoveStepOp):
|
||||
graph["steps"] = [
|
||||
step for step in graph["steps"]
|
||||
if step.get("logical_step_id", step.get("id")) != operation.logical_step_id
|
||||
]
|
||||
if "dependencies" in graph:
|
||||
graph["dependencies"] = [
|
||||
edge for edge in graph["dependencies"]
|
||||
if operation.logical_step_id not in {edge.get("source"), edge.get("target")}
|
||||
]
|
||||
elif isinstance(operation, SetDependencyOp):
|
||||
if "dependencies" not in graph:
|
||||
raise ValueError("legacy graph has no dependencies collection")
|
||||
edge = {"source": operation.logical_step_id, "target": operation.target_logical_step_id}
|
||||
if operation.action == "add" and edge not in graph["dependencies"]:
|
||||
graph["dependencies"].append(edge)
|
||||
elif operation.action == "remove":
|
||||
graph["dependencies"] = [item for item in graph["dependencies"] if item != edge]
|
||||
if "dependencies" in graph and _has_cycle(graph["dependencies"]):
|
||||
raise ValueError("dependency cycle rejected")
|
||||
return graph
|
||||
# #endregion ScenarioEditor.Apply.LegacyOps
|
||||
|
||||
|
||||
def _apply_operation(graph: dict[str, Any], operation: EditOperation) -> None:
|
||||
steps = graph["steps"]
|
||||
ids = {step["id"] for step in steps}
|
||||
if isinstance(operation, SetParameterDefinitionOp):
|
||||
graph["parameters"][operation.param_name] = operation.value
|
||||
parameter = next((p for p in graph["parameters"] if p["name"] == operation.param_name), None)
|
||||
if parameter is None:
|
||||
raise ValueError("parameter not found")
|
||||
parameter["value"] = operation.value
|
||||
parameter["status"] = "resolved"
|
||||
elif isinstance(operation, SetAssertionOp):
|
||||
graph["assertions"][operation.logical_step_id] = validate_assertion(operation)
|
||||
step = next((s for s in steps if s["id"] == operation.logical_step_id), None)
|
||||
if step is None:
|
||||
raise ValueError("step not found")
|
||||
step["expected"] = {"kind": "baseline_ref", "ref": operation.baseline_ref, "predicate": operation.comparison}
|
||||
if operation.threshold is not None:
|
||||
step["expected"]["description"] = f"threshold {operation.threshold}"
|
||||
elif isinstance(operation, AddStepOp):
|
||||
if any(step.get("template") == operation.template for step in graph["steps"]):
|
||||
raise ValueError("duplicate step template")
|
||||
graph["steps"].append({"template": operation.template, "logical_step_id": f"new:{operation.template}"})
|
||||
template = STEP_TEMPLATES.get(operation.template)
|
||||
if template is None:
|
||||
raise ValueError("unknown step template")
|
||||
action, tool, phase = template
|
||||
step_id = f"step-{operation.template.replace('_', '-')}-{len(steps) + 1}"
|
||||
if step_id in ids:
|
||||
raise ValueError("duplicate step id")
|
||||
depends = [operation.after_logical_step_id] if operation.after_logical_step_id else []
|
||||
steps.append(ScenarioStep(id=step_id, phase=phase, title=operation.template, tool=tool, action=action,
|
||||
expected=Expected(kind="structural", description=f"{action} completes"),
|
||||
depends_on=depends, automation_status="ready", risk=REGISTERED_ACTIONS[action]["risk"]).model_dump(mode="json"))
|
||||
elif isinstance(operation, RemoveStepOp):
|
||||
graph["steps"] = [step for step in graph["steps"] if step.get("logical_step_id") != operation.logical_step_id]
|
||||
if operation.logical_step_id not in ids:
|
||||
raise ValueError("step not found")
|
||||
graph["steps"] = [step for step in steps if step["id"] != operation.logical_step_id]
|
||||
for step in graph["steps"]:
|
||||
step["depends_on"] = [dep for dep in step.get("depends_on", []) if dep != operation.logical_step_id]
|
||||
elif isinstance(operation, SetDependencyOp):
|
||||
edge = {"source": operation.logical_step_id, "target": operation.target_logical_step_id}
|
||||
if operation.action == "add" and edge not in graph["dependencies"]:
|
||||
graph["dependencies"].append(edge)
|
||||
if operation.logical_step_id not in ids or operation.target_logical_step_id not in ids:
|
||||
raise ValueError("step not found")
|
||||
step = next(s for s in steps if s["id"] == operation.logical_step_id)
|
||||
if operation.action == "add" and operation.target_logical_step_id not in step["depends_on"]:
|
||||
step["depends_on"].append(operation.target_logical_step_id)
|
||||
elif operation.action == "remove":
|
||||
graph["dependencies"] = [item for item in graph["dependencies"] if item != edge]
|
||||
step["depends_on"] = [d for d in step["depends_on"] if d != operation.target_logical_step_id]
|
||||
|
||||
|
||||
def _has_cycle(edges: list[dict[str, Any]]) -> bool:
|
||||
|
||||
@@ -0,0 +1,245 @@
|
||||
# #region ScenarioExecution.AgentEvaluation [C:5] [TYPE Module] [SEMANTICS scenario,evaluation,parser,store]
|
||||
# @defgroup ScenarioExecution Strict AgentEvaluation boundary: schema, criterion binding, and append-only persistence.
|
||||
# @INVARIANT Raw response evidence is bound before a succeeded record can be published.
|
||||
# @RATIONALE This module keeps the normative 038 boundary independent from provider transport so tests can use typed mocks.
|
||||
# @REJECTED A permissive dict passthrough was rejected because it allows findings to claim undeclared deterministic authority.
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import UTC, datetime
|
||||
from typing import Any, Literal
|
||||
import uuid
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from src.models.scenario_evaluation import AgentEvaluation as AgentEvaluationRow
|
||||
from src.services.dashboard_testing.scenario.models import AgentEvaluationSpec
|
||||
from src.models.scenario_artifact import ScenarioArtifact
|
||||
from src.models.scenario_run import ScenarioRun, ScenarioStepRun
|
||||
from .artifacts import is_valid_sha256
|
||||
from .capacity import CapacityUnavailable, claim_capacity, release_capacity
|
||||
|
||||
|
||||
class EvaluationManifestItem(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
artifact_id: str
|
||||
sha256: str = Field(pattern=r"^[a-f0-9]{64}$")
|
||||
content_type: Literal["image/jpeg", "image/png", "image/webp", "application/json"]
|
||||
byte_length: int = Field(ge=1)
|
||||
role: Literal["actual", "baseline", "comparison", "context"]
|
||||
|
||||
|
||||
class EvaluationFinding(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
finding_id: str = Field(min_length=1)
|
||||
severity: Literal["info", "warning", "error", "critical"]
|
||||
message: str = Field(min_length=1, max_length=2000)
|
||||
evidence_artifact_ids: list[str] = Field(min_length=1, max_length=100)
|
||||
region: dict[str, float] | None = None
|
||||
criterion_id: str = Field(min_length=1, max_length=128)
|
||||
criterion_kind: Literal["semantic", "deterministic_comparison"]
|
||||
|
||||
|
||||
class EvaluationUsage(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
input_tokens: int | None = Field(default=None, ge=0)
|
||||
output_tokens: int | None = Field(default=None, ge=0)
|
||||
cost_amount: str | None = Field(default=None, pattern=r"^[0-9]+(\.[0-9]+)?$")
|
||||
currency: str | None = None
|
||||
pricing_version: str | None = None
|
||||
|
||||
|
||||
class AgentEvaluation(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
schema_version: Literal[1]
|
||||
evaluation_id: str
|
||||
scenario_run_id: str
|
||||
logical_step_id: str
|
||||
attempt: int = Field(ge=1)
|
||||
operation_id: str
|
||||
evaluation_spec_hash: str = Field(pattern=r"^[a-f0-9]{64}$")
|
||||
provider_id: str = Field(min_length=1)
|
||||
provider_version: str = Field(min_length=1)
|
||||
model_id: str = Field(min_length=1)
|
||||
model_version: str = Field(min_length=1)
|
||||
prompt_template_id: str = Field(min_length=1)
|
||||
prompt_template_version: str = Field(min_length=1)
|
||||
prompt_template_hash: str = Field(pattern=r"^[a-f0-9]{64}$")
|
||||
output_schema_hash: str = Field(pattern=r"^[a-f0-9]{64}$")
|
||||
input_manifest_hash: str = Field(pattern=r"^[a-f0-9]{64}$")
|
||||
input_manifest: list[EvaluationManifestItem] = Field(min_length=1, max_length=100)
|
||||
baseline_pin: dict[str, Any]
|
||||
comparison_ids: list[str] = Field(min_length=1, max_length=100)
|
||||
status: Literal["succeeded", "provider_error", "parser_error", "budget_exceeded", "cancelled", "timed_out"]
|
||||
verdict: Literal["pass", "fail", "inconclusive"]
|
||||
confidence: float = Field(ge=0, le=1)
|
||||
findings: list[EvaluationFinding] = Field(max_length=100)
|
||||
reason_codes: list[str] = Field(max_length=100)
|
||||
raw_response_artifact_ref: str | None = None
|
||||
raw_response_sha256: str | None = Field(default=None, pattern=r"^[a-f0-9]{64}$")
|
||||
trust_policy_hash: str = Field(pattern=r"^[a-f0-9]{64}$")
|
||||
usage: EvaluationUsage
|
||||
started_at: datetime
|
||||
finished_at: datetime
|
||||
|
||||
@model_validator(mode="after")
|
||||
def validate_uuid_fields(self) -> "AgentEvaluation":
|
||||
# logical_step_id is deliberately excluded: runtime step identity is a registry slug
|
||||
# (ScenarioStepRun.logical_step_id is String(128) and carries ids like "s4_assert_revenue"),
|
||||
# while the 038 schema's uuid claim targets the deferred five-program IR. Enforcing uuid here
|
||||
# would block evaluation persistence for every current slug-identified step.
|
||||
for field in ("evaluation_id", "scenario_run_id", "operation_id"):
|
||||
try:
|
||||
uuid.UUID(getattr(self, field))
|
||||
except (ValueError, AttributeError, TypeError) as exc:
|
||||
raise ValueError(f"{field} must be a UUID") from exc
|
||||
return self
|
||||
|
||||
@model_validator(mode="after")
|
||||
def validate_failure_shape(self) -> "AgentEvaluation":
|
||||
if self.status == "succeeded" and (not self.raw_response_artifact_ref or not self.raw_response_sha256):
|
||||
raise ValueError("EVALUATION_RAW_RESPONSE_REQUIRED")
|
||||
if self.status != "succeeded" and (self.verdict != "inconclusive" or self.confidence != 0 or self.findings or not self.reason_codes):
|
||||
raise ValueError("EVALUATION_FAILURE_SHAPE_INVALID")
|
||||
return self
|
||||
|
||||
|
||||
def parse_evaluation_response(raw: dict[str, Any], *, spec: AgentEvaluationSpec) -> AgentEvaluation:
|
||||
"""Parse and cross-check a provider response; malformed responses become typed parser errors."""
|
||||
try:
|
||||
candidate = {**raw, "schema_version": 1}
|
||||
result = AgentEvaluation.model_validate(candidate)
|
||||
criteria = {item.criterion_id: item.criterion_kind for item in spec.criteria}
|
||||
for finding in result.findings:
|
||||
if criteria.get(finding.criterion_id) != finding.criterion_kind:
|
||||
raise ValueError("EVALUATION_CRITERION_MISMATCH")
|
||||
return result
|
||||
except Exception:
|
||||
now = datetime.now(UTC)
|
||||
return AgentEvaluation.model_validate({
|
||||
"schema_version": 1, "evaluation_id": str(uuid.uuid4()), "scenario_run_id": raw.get("scenario_run_id", str(uuid.uuid4())),
|
||||
"logical_step_id": raw.get("logical_step_id", str(uuid.uuid4())), "attempt": raw.get("attempt", 1),
|
||||
"operation_id": raw.get("operation_id", str(uuid.uuid4())), "evaluation_spec_hash": "0" * 64,
|
||||
"provider_id": "unknown", "provider_version": "unknown", "model_id": "unknown", "model_version": "unknown",
|
||||
"prompt_template_id": "unknown", "prompt_template_version": "unknown", "prompt_template_hash": "0" * 64,
|
||||
"output_schema_hash": "0" * 64, "input_manifest_hash": "0" * 64, "input_manifest": [{"artifact_id": str(uuid.uuid4()), "sha256": "0" * 64, "content_type": "application/json", "byte_length": 1, "role": "context"}],
|
||||
"baseline_pin": {}, "comparison_ids": [str(uuid.uuid4())], "status": "parser_error", "verdict": "inconclusive", "confidence": 0,
|
||||
"findings": [], "reason_codes": ["EVALUATION_RESPONSE_INVALID"], "raw_response_artifact_ref": None, "raw_response_sha256": None,
|
||||
"trust_policy_hash": "0" * 64, "usage": {"input_tokens": None, "output_tokens": None, "cost_amount": None, "currency": None, "pricing_version": None},
|
||||
"started_at": now, "finished_at": now,
|
||||
})
|
||||
|
||||
|
||||
def persist_agent_evaluation(db: Session, record: AgentEvaluation) -> AgentEvaluationRow:
|
||||
"""Persist one immutable identity; a duplicate identity is rejected rather than updated."""
|
||||
if db.query(AgentEvaluationRow).filter_by(
|
||||
scenario_run_id=record.scenario_run_id, logical_step_id=record.logical_step_id,
|
||||
attempt=record.attempt, operation_id=record.operation_id,
|
||||
).first() is not None:
|
||||
raise ValueError("EVALUATION_ALREADY_EXISTS")
|
||||
row = AgentEvaluationRow(**record.model_dump())
|
||||
db.add(row)
|
||||
db.flush()
|
||||
return row
|
||||
|
||||
|
||||
# #region ScenarioExecution.AgentEvaluation.Evidence [C:4] [TYPE Function] [SEMANTICS evaluation,evidence,binding,p0-1]
|
||||
# @BRIEF Bind manifest/findings to same-run active artifacts; keep raw-response on this step+attempt.
|
||||
# @INVARIANT input_manifest and finding evidence_artifact_ids may come from prior steps of this run (owner_type=scenario_run, is_active). Raw-response must match this logical_step_id+attempt.
|
||||
# @INVARIANT Foreign-run artifacts stay rejected. Inactive prior-step rows are not evidence.
|
||||
# @RATIONALE The production adapter assembles input_manifest from completed prior-step artifacts; requiring those rows to share the evaluation step identity would force EVALUATION_EVIDENCE_NOT_FOUND.
|
||||
# @REJECTED Duplicating previous-step artifact rows onto the evaluation step was rejected — that forges a second provenance identity for the same digest (D1).
|
||||
def validate_evaluation_evidence(
|
||||
db: Session, *, run_id: str, logical_step_id: str, attempt: int,
|
||||
input_manifest: list[dict[str, Any]], raw_response_artifact_ref: str | None,
|
||||
raw_response_sha256: str | None, findings: list[dict[str, Any]],
|
||||
succeeded: bool = True,
|
||||
) -> None:
|
||||
"""Verify evaluation artifacts against run-owned evidence; raw-response stays step-bound."""
|
||||
run = db.query(ScenarioRun).filter(ScenarioRun.id == run_id).one_or_none()
|
||||
step = db.query(ScenarioStepRun).filter(
|
||||
ScenarioStepRun.run_id == run_id, ScenarioStepRun.logical_step_id == logical_step_id,
|
||||
ScenarioStepRun.attempt == attempt,
|
||||
).one_or_none()
|
||||
if run is None or step is None:
|
||||
raise ValueError("EVALUATION_EVIDENCE_OWNER_INVALID")
|
||||
refs = {str(item.get("artifact_id")): item for item in input_manifest}
|
||||
finding_refs = {str(ref) for finding in findings for ref in finding.get("evidence_artifact_ids", [])}
|
||||
owned_rows = db.query(ScenarioArtifact).filter(
|
||||
ScenarioArtifact.owner_type == "scenario_run", ScenarioArtifact.owner_id == run_id,
|
||||
ScenarioArtifact.is_active.is_(True),
|
||||
).all()
|
||||
owned_by_ref = {row.id: row for row in owned_rows} | {row.content_ref: row for row in owned_rows}
|
||||
step_rows = [
|
||||
row for row in owned_rows
|
||||
if row.logical_step_id == logical_step_id and row.attempt == attempt
|
||||
]
|
||||
step_by_ref = {row.id: row for row in step_rows} | {row.content_ref: row for row in step_rows}
|
||||
for ref, item in refs.items():
|
||||
row = owned_by_ref.get(ref)
|
||||
if row is None:
|
||||
raise ValueError("EVALUATION_EVIDENCE_NOT_FOUND")
|
||||
if row.owner_id != run_id or row.owner_type != "scenario_run":
|
||||
raise ValueError("EVALUATION_EVIDENCE_OWNER_INVALID")
|
||||
if item.get("sha256") != row.sha256 or item.get("content_type") != row.content_type or item.get("byte_length") != row.byte_length:
|
||||
raise ValueError("EVALUATION_EVIDENCE_METADATA_MISMATCH")
|
||||
for ref in finding_refs:
|
||||
row = owned_by_ref.get(ref)
|
||||
if row is None:
|
||||
raise ValueError("EVALUATION_EVIDENCE_NOT_FOUND")
|
||||
if row.owner_id != run_id or row.owner_type != "scenario_run":
|
||||
raise ValueError("EVALUATION_EVIDENCE_OWNER_INVALID")
|
||||
if succeeded:
|
||||
if not raw_response_artifact_ref or not is_valid_sha256(raw_response_sha256):
|
||||
raise ValueError("EVALUATION_EVIDENCE_RAW_REQUIRED")
|
||||
raw = step_by_ref.get(raw_response_artifact_ref)
|
||||
if raw is None or not raw.is_active:
|
||||
raise ValueError("EVALUATION_EVIDENCE_RAW_UNAVAILABLE")
|
||||
if raw.logical_step_id != logical_step_id or raw.attempt != attempt or raw.owner_id != run_id:
|
||||
raise ValueError("EVALUATION_EVIDENCE_OWNER_INVALID")
|
||||
if raw.sha256 != raw_response_sha256:
|
||||
raise ValueError("EVALUATION_EVIDENCE_RAW_DIGEST_MISMATCH")
|
||||
elif raw_response_artifact_ref:
|
||||
raw = step_by_ref.get(raw_response_artifact_ref)
|
||||
if raw is None:
|
||||
raise ValueError("EVALUATION_EVIDENCE_NOT_FOUND")
|
||||
# #endregion ScenarioExecution.AgentEvaluation.Evidence
|
||||
|
||||
|
||||
async def submit_evaluation(
|
||||
db: Session, *, spec: AgentEvaluationSpec, prompt: str, artifacts: list[bytes],
|
||||
environment_id: str, environment_class: str, run_id: str, logical_step_id: str,
|
||||
client: Any = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Call the existing JSON client under an agent_evaluation capacity lease."""
|
||||
from src.plugins.llm_analysis.models import LLMProviderType
|
||||
from src.plugins.llm_analysis.service import LLMClient
|
||||
from src.services.llm_provider import LLMProviderService
|
||||
|
||||
lease = None
|
||||
try:
|
||||
provider = LLMProviderService(db).get_provider(spec.provider_id)
|
||||
if provider is None:
|
||||
raise RuntimeError("EVALUATION_PROVIDER_MISSING")
|
||||
lease = claim_capacity(db, environment_id=environment_id, environment_class=environment_class,
|
||||
workload_class="agent_evaluation", provider_id=spec.provider_id,
|
||||
provider_version=spec.provider_version, run_id=run_id,
|
||||
logical_step_id=logical_step_id)
|
||||
api_key = LLMProviderService(db).get_decrypted_api_key(spec.provider_id)
|
||||
if not api_key:
|
||||
raise RuntimeError("EVALUATION_PROVIDER_MISSING")
|
||||
llm = client or LLMClient(LLMProviderType(provider.provider_type), api_key, provider.base_url, spec.model_id)
|
||||
result = await llm.get_json_completion([{"role": "user", "content": prompt}])
|
||||
if not isinstance(result, dict):
|
||||
raise RuntimeError("EVALUATION_RESPONSE_INVALID")
|
||||
return result
|
||||
except CapacityUnavailable:
|
||||
raise
|
||||
except TimeoutError as exc:
|
||||
raise RuntimeError("EVALUATION_TIMED_OUT") from exc
|
||||
except Exception as exc:
|
||||
raise RuntimeError("EVALUATION_PROVIDER_ERROR") from exc
|
||||
finally:
|
||||
if lease is not None:
|
||||
release_capacity(db, lease["lease_id"])
|
||||
# #endregion ScenarioExecution.AgentEvaluation
|
||||
@@ -0,0 +1,244 @@
|
||||
# #region ScenarioExecution.ArtifactContent [C:4] [TYPE Module] [SEMANTICS scenario,execution,artifact,content,digest,mime,acl]
|
||||
# @defgroup ScenarioExecution Authenticated ScenarioArtifact byte delivery (044 artifact-content.openapi.yaml).
|
||||
# @BRIEF Resolve a run-owned artifact receipt, verify digest/MIME/length, then return complete bytes.
|
||||
# @RELATION DEPENDS_ON -> [Models.ScenarioExecution.Artifact]
|
||||
# @RELATION DEPENDS_ON -> [Models.ScenarioExecution.Run]
|
||||
# @RELATION DEPENDS_ON -> [Services.AgentRuns.Artifacts]
|
||||
# @RELATION CALLED_BY -> [Api.ScenarioArtifactContent]
|
||||
# @INVARIANT Ownership (owner_type=scenario_run, owner_id=run_id), positive length, SHA-256 and
|
||||
# magic-byte MIME are verified before any success header or body byte is produced.
|
||||
# @INVARIANT Foreign/nonexistent children are indistinguishable 404; expired owner-visible rows are 410.
|
||||
# @INVARIANT Error messages never include storage paths, content_ref, secrets or provider credentials.
|
||||
# @RATIONALE Bytes leave storage only after a complete-object check so a corrupt spool cannot leak
|
||||
# through Content-Type or a partial body. ACL is parent-then-child: a known run without
|
||||
# scenario:result VIEW is 403, while a missing/foreign artifact stays 404.
|
||||
# @REJECTED Growing scenario_runs.py was rejected (INV_7 / D7). Reusing scenario:RUN for content was
|
||||
# rejected — result VIEW is the content ACL. Streaming then hashing, or honouring Range /
|
||||
# conditional cache headers, was rejected: v1 is full bounded objects after digest/ACL.
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
import base64
|
||||
import hashlib
|
||||
import uuid
|
||||
from typing import Any, NoReturn
|
||||
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from src.models.scenario_artifact import ScenarioArtifact
|
||||
from src.models.scenario_run import ScenarioRun
|
||||
from src.services.agent_runs.artifacts import get_draft_storage
|
||||
|
||||
IMAGE_MAX_BYTES = 10_485_760
|
||||
OTHER_MAX_BYTES = 26_214_400
|
||||
IMAGE_MIME = frozenset({"image/jpeg", "image/png", "image/webp"})
|
||||
KIND_ALLOWED_MIME: dict[str, frozenset[str]] = {
|
||||
"screenshot": IMAGE_MIME,
|
||||
"evidence": IMAGE_MIME | frozenset({"application/json"}),
|
||||
"report": frozenset({"application/json", "application/pdf"}),
|
||||
"xlsx": frozenset({"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"}),
|
||||
}
|
||||
_MIME_EXTENSION = {
|
||||
"image/jpeg": "jpg",
|
||||
"image/png": "png",
|
||||
"image/webp": "webp",
|
||||
"application/json": "json",
|
||||
"application/pdf": "pdf",
|
||||
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet": "xlsx",
|
||||
}
|
||||
_VIEW_RESOURCE = "scenario:result"
|
||||
_VIEW_ACTION = "VIEW"
|
||||
|
||||
|
||||
# #region ScenarioExecution.ArtifactContent.Error [C:1] [TYPE Class] [SEMANTICS scenario,artifact,error]
|
||||
# @BRIEF Typed content failure mapped 1:1 onto OpenAPI Error-Code enums.
|
||||
class ArtifactContentError(Exception):
|
||||
def __init__(self, status_code: int, code: str, message: str, *, retryable: bool = False) -> None:
|
||||
super().__init__(code)
|
||||
self.status_code = status_code
|
||||
self.code = code
|
||||
self.message = message
|
||||
self.retryable = retryable
|
||||
# #endregion ScenarioExecution.ArtifactContent.Error
|
||||
|
||||
|
||||
# #region ScenarioExecution.ArtifactContent.Payload [C:1] [TYPE Class] [SEMANTICS scenario,artifact,bytes]
|
||||
# @BRIEF Verified complete object plus the success headers OpenAPI requires.
|
||||
@dataclass(frozen=True)
|
||||
class ArtifactContent:
|
||||
body: bytes
|
||||
content_type: str
|
||||
content_length: int
|
||||
content_disposition: str
|
||||
etag: str
|
||||
content_digest: str
|
||||
# #endregion ScenarioExecution.ArtifactContent.Payload
|
||||
|
||||
|
||||
# #region ScenarioExecution.ArtifactContent.Fail [C:1] [TYPE Function]
|
||||
def _fail(status_code: int, code: str, message: str, *, retryable: bool = False) -> NoReturn:
|
||||
raise ArtifactContentError(status_code, code, message, retryable=retryable)
|
||||
# #endregion ScenarioExecution.ArtifactContent.Fail
|
||||
|
||||
|
||||
# #region ScenarioExecution.ArtifactContent.HasView [C:2] [TYPE Function] [SEMANTICS scenario,artifact,rbac]
|
||||
# @BRIEF Mirror has_permission("scenario:result", "VIEW") including the admin bypass.
|
||||
def _has_result_view(user: object) -> bool:
|
||||
for role in getattr(user, "roles", None) or []:
|
||||
if getattr(role, "is_admin", False):
|
||||
return True
|
||||
for perm in getattr(role, "permissions", None) or []:
|
||||
if getattr(perm, "resource", None) == _VIEW_RESOURCE and getattr(perm, "action", None) == _VIEW_ACTION:
|
||||
return True
|
||||
return False
|
||||
# #endregion ScenarioExecution.ArtifactContent.HasView
|
||||
|
||||
|
||||
# #region ScenarioExecution.ArtifactContent.SniffMime [C:2] [TYPE Function] [SEMANTICS scenario,artifact,mime,magic]
|
||||
# @BRIEF Detect allowlisted MIME from complete-object magic bytes; JSON is BOM/whitespace then { or [.
|
||||
def sniff_mime(data: bytes) -> str | None:
|
||||
if data.startswith(b"\xff\xd8\xff"):
|
||||
return "image/jpeg"
|
||||
if data.startswith(b"\x89PNG\r\n\x1a\n"):
|
||||
return "image/png"
|
||||
if len(data) >= 12 and data.startswith(b"RIFF") and data[8:12] == b"WEBP":
|
||||
return "image/webp"
|
||||
if data.startswith(b"%PDF-"):
|
||||
return "application/pdf"
|
||||
if data.startswith(b"PK\x03\x04"):
|
||||
return "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"
|
||||
stripped = data.lstrip(b"\xef\xbb\xbf \t\r\n")
|
||||
if stripped[:1] in (b"{", b"["):
|
||||
return "application/json"
|
||||
return None
|
||||
# #endregion ScenarioExecution.ArtifactContent.SniffMime
|
||||
|
||||
|
||||
# #region ScenarioExecution.ArtifactContent.Filename [C:2] [TYPE Function] [SEMANTICS scenario,artifact,filename]
|
||||
# @BRIEF Build Content-Disposition from a sanitized artifact UUID plus MIME extension, never a path.
|
||||
def attachment_disposition(artifact_id: str, content_type: str) -> str:
|
||||
try:
|
||||
stem = str(uuid.UUID(str(artifact_id)))
|
||||
except (ValueError, TypeError, AttributeError):
|
||||
stem = "artifact"
|
||||
return f"attachment; filename={stem}.{_MIME_EXTENSION[content_type]}"
|
||||
# #endregion ScenarioExecution.ArtifactContent.Filename
|
||||
|
||||
|
||||
# #region ScenarioExecution.ArtifactContent.Service [C:4] [TYPE Class] [SEMANTICS scenario,artifact,content,service]
|
||||
# @BRIEF Load one run-owned artifact through injected storage (opaque draft:{run_id}:{sha256} refs).
|
||||
class ScenarioArtifactContentService:
|
||||
def __init__(self, db: Session, storage: Any | None = None) -> None:
|
||||
self._db = db
|
||||
self._storage = storage
|
||||
|
||||
# #region ScenarioExecution.ArtifactContent.Service.Open [C:4] [TYPE Function] [SEMANTICS scenario,artifact,acl,integrity]
|
||||
# @BRIEF Parent ACL, then child ownership, then Range, tombstone, declared limits, then stored bytes.
|
||||
# @POST Success payload is a complete verified object; failures raise ArtifactContentError with no bytes.
|
||||
def open(
|
||||
self,
|
||||
*,
|
||||
run_id: str,
|
||||
artifact_id: str,
|
||||
user: object,
|
||||
range_header: str | None = None,
|
||||
) -> ArtifactContent:
|
||||
self._require_parent_acl(run_id, user)
|
||||
artifact = self._require_owned_artifact(run_id, artifact_id)
|
||||
if range_header:
|
||||
_fail(416, "RANGE_NOT_SUPPORTED", "Range requests are not supported")
|
||||
if not artifact.is_active:
|
||||
_fail(410, "ARTIFACT_EXPIRED", "Artifact is no longer available")
|
||||
content_type = self._declared_content_type(artifact)
|
||||
byte_length = artifact.byte_length
|
||||
if not isinstance(byte_length, int) or byte_length < 1:
|
||||
_fail(409, "ARTIFACT_INTEGRITY_FAILED", "Artifact integrity check failed")
|
||||
limit = IMAGE_MAX_BYTES if content_type in IMAGE_MIME else OTHER_MAX_BYTES
|
||||
if byte_length > limit:
|
||||
_fail(413, "ARTIFACT_TOO_LARGE", "Artifact exceeds the allowed size")
|
||||
data = self._retrieve(artifact)
|
||||
self._verify_complete_object(artifact, content_type, byte_length, data)
|
||||
digest = hashlib.sha256(data).digest()
|
||||
hex_digest = digest.hex()
|
||||
return ArtifactContent(
|
||||
body=data,
|
||||
content_type=content_type,
|
||||
content_length=len(data),
|
||||
content_disposition=attachment_disposition(str(artifact.id), content_type),
|
||||
etag=f'"{hex_digest}"',
|
||||
content_digest=f"sha-256=:{base64.b64encode(digest).decode('ascii')}:",
|
||||
)
|
||||
# #endregion ScenarioExecution.ArtifactContent.Service.Open
|
||||
|
||||
# #region ScenarioExecution.ArtifactContent.Service.ParentAcl [C:2] [TYPE Function] [SEMANTICS scenario,artifact,acl]
|
||||
# @BRIEF Known parent without VIEW is 403; missing parent is 404 (no permission leak on unknown runs).
|
||||
def _require_parent_acl(self, run_id: str, user: object) -> None:
|
||||
run = self._db.query(ScenarioRun).filter(ScenarioRun.id == run_id).first()
|
||||
if run is None:
|
||||
_fail(404, "NOT_FOUND", "Artifact not found")
|
||||
if not _has_result_view(user):
|
||||
_fail(403, "PERMISSION_DENIED", "Permission denied")
|
||||
# #endregion ScenarioExecution.ArtifactContent.Service.ParentAcl
|
||||
|
||||
# #region ScenarioExecution.ArtifactContent.Service.Owned [C:2] [TYPE Function] [SEMANTICS scenario,artifact,owner]
|
||||
# @BRIEF Child must be scenario_run-owned by this run_id; any other row is the same 404.
|
||||
def _require_owned_artifact(self, run_id: str, artifact_id: str) -> ScenarioArtifact:
|
||||
artifact = (
|
||||
self._db.query(ScenarioArtifact)
|
||||
.filter(
|
||||
ScenarioArtifact.id == artifact_id,
|
||||
ScenarioArtifact.owner_type == "scenario_run",
|
||||
ScenarioArtifact.owner_id == run_id,
|
||||
)
|
||||
.first()
|
||||
)
|
||||
if artifact is None:
|
||||
_fail(404, "NOT_FOUND", "Artifact not found")
|
||||
return artifact
|
||||
# #endregion ScenarioExecution.ArtifactContent.Service.Owned
|
||||
|
||||
# #region ScenarioExecution.ArtifactContent.Service.DeclaredType [C:2] [TYPE Function] [SEMANTICS scenario,artifact,kind,mime]
|
||||
# @BRIEF Declared MIME must be allowlisted for the artifact kind; unknown kinds cannot download.
|
||||
def _declared_content_type(self, artifact: ScenarioArtifact) -> str:
|
||||
declared = artifact.content_type
|
||||
allowed = KIND_ALLOWED_MIME.get(str(artifact.kind))
|
||||
if not isinstance(declared, str) or allowed is None or declared not in allowed:
|
||||
_fail(409, "ARTIFACT_INTEGRITY_FAILED", "Artifact integrity check failed")
|
||||
return declared
|
||||
# #endregion ScenarioExecution.ArtifactContent.Service.DeclaredType
|
||||
|
||||
# #region ScenarioExecution.ArtifactContent.Service.Retrieve [C:3] [TYPE Function] [SEMANTICS scenario,artifact,storage]
|
||||
# @BRIEF Read opaque draft bytes; missing storage is 409, provider/IO outage is 503.
|
||||
def _retrieve(self, artifact: ScenarioArtifact) -> bytes:
|
||||
storage = self._storage if self._storage is not None else get_draft_storage()
|
||||
try:
|
||||
data = storage.retrieve(artifact.content_ref)
|
||||
except ValueError:
|
||||
_fail(409, "ARTIFACT_INTEGRITY_FAILED", "Artifact integrity check failed")
|
||||
except ArtifactContentError:
|
||||
raise
|
||||
except Exception:
|
||||
_fail(503, "ARTIFACT_STORAGE_UNAVAILABLE", "Artifact storage is temporarily unavailable", retryable=True)
|
||||
if data is None:
|
||||
_fail(409, "ARTIFACT_MISSING", "Artifact bytes are not available")
|
||||
return data
|
||||
# #endregion ScenarioExecution.ArtifactContent.Service.Retrieve
|
||||
|
||||
# #region ScenarioExecution.ArtifactContent.Service.Verify [C:2] [TYPE Function] [SEMANTICS scenario,artifact,digest,mime]
|
||||
# @BRIEF Complete stored bytes must match declared length, SHA-256 and magic-byte MIME.
|
||||
def _verify_complete_object(
|
||||
self,
|
||||
artifact: ScenarioArtifact,
|
||||
content_type: str,
|
||||
byte_length: int,
|
||||
data: bytes,
|
||||
) -> None:
|
||||
digest = hashlib.sha256(data).hexdigest()
|
||||
expected = str(artifact.sha256 or "").lower()
|
||||
sniffed = sniff_mime(data)
|
||||
if len(data) != byte_length or digest != expected or sniffed != content_type:
|
||||
_fail(409, "ARTIFACT_INTEGRITY_FAILED", "Artifact integrity check failed")
|
||||
# #endregion ScenarioExecution.ArtifactContent.Service.Verify
|
||||
# #endregion ScenarioExecution.ArtifactContent.Service
|
||||
|
||||
# #endregion ScenarioExecution.ArtifactContent
|
||||
@@ -79,6 +79,8 @@ def register_artifact(
|
||||
retention_class: str = "standard",
|
||||
logical_step_id: str | None = None,
|
||||
attempt: int | None = None,
|
||||
content_type: str | None = None,
|
||||
byte_length: int | None = None,
|
||||
) -> ScenarioArtifact:
|
||||
if owner_type not in _OWNER_TYPES:
|
||||
raise ValueError(f"unsupported artifact owner_type: {owner_type}")
|
||||
@@ -97,6 +99,8 @@ def register_artifact(
|
||||
name=name,
|
||||
content_ref=content_ref,
|
||||
sha256=sha256.lower(),
|
||||
content_type=content_type,
|
||||
byte_length=byte_length,
|
||||
retention_class=retention_class,
|
||||
logical_step_id=logical_step_id,
|
||||
attempt=attempt,
|
||||
@@ -107,6 +111,22 @@ def register_artifact(
|
||||
# #endregion ScenarioExecution.Artifacts.Register
|
||||
|
||||
|
||||
# #region ScenarioExecution.Artifacts.EvidenceMetadata [C:2] [TYPE Function] [SEMANTICS scenario,execution,artifact,mime,length]
|
||||
# @BRIEF Read per-ref MIME/length from the executor outcome, falling back to a single-ref payload.
|
||||
def _evidence_metadata(step_outcome: dict[str, Any], ref: str) -> tuple[str | None, int | None]:
|
||||
types = step_outcome.get("artifact_content_types")
|
||||
lengths = step_outcome.get("artifact_byte_lengths")
|
||||
content_type = types.get(ref) if isinstance(types, dict) else None
|
||||
if not isinstance(content_type, str):
|
||||
content_type = step_outcome.get("content_type") if isinstance(step_outcome.get("content_type"), str) else None
|
||||
byte_length = lengths.get(ref) if isinstance(lengths, dict) else None
|
||||
if not isinstance(byte_length, int):
|
||||
raw_length = step_outcome.get("byte_length")
|
||||
byte_length = raw_length if isinstance(raw_length, int) else None
|
||||
return content_type, byte_length
|
||||
# #endregion ScenarioExecution.Artifacts.EvidenceMetadata
|
||||
|
||||
|
||||
# #region ScenarioExecution.Artifacts.RegisterStepEvidence [C:4] [TYPE Function] [SEMANTICS scenario,execution,artifact,evidence,digest,integrity]
|
||||
# @ingroup ScenarioExecution
|
||||
# @BRIEF Register verified executor evidence refs; retain a typed integrity condition for bad refs.
|
||||
@@ -114,6 +134,8 @@ def register_artifact(
|
||||
# @POST Valid evidence refs persist with their exact supplied/local digest; invalid refs return an
|
||||
# inconclusive integrity payload and remain unregistered.
|
||||
# @REJECTED Filling absent digest with zeros was rejected — it makes unverifiable evidence appear durable.
|
||||
# @RATIONALE Evaluation evidence binding compares MIME/length; omitting them from the row makes a
|
||||
# valid adapter payload fail EVALUATION_EVIDENCE_METADATA_MISMATCH (D2).
|
||||
def register_step_evidence(
|
||||
db: Session,
|
||||
*,
|
||||
@@ -138,6 +160,7 @@ def register_step_evidence(
|
||||
unregistered.append(ref)
|
||||
invalid_digest = invalid_digest or digest is not None
|
||||
continue
|
||||
content_type, byte_length = _evidence_metadata(step_outcome, ref)
|
||||
register_artifact(
|
||||
db,
|
||||
owner_type="scenario_run",
|
||||
@@ -148,6 +171,8 @@ def register_step_evidence(
|
||||
sha256=digest,
|
||||
logical_step_id=logical_step_id,
|
||||
attempt=attempt,
|
||||
content_type=content_type,
|
||||
byte_length=byte_length,
|
||||
)
|
||||
attach_artifact_refs(db, run_id, logical_step_id, [ref])
|
||||
if not unregistered:
|
||||
|
||||
@@ -0,0 +1,353 @@
|
||||
# #region ScenarioExecution.BaselineResolver [C:4] [TYPE Module] [SEMANTICS scenario,execution,baseline,pin,catalog,published]
|
||||
# @defgroup ScenarioExecution Server-resolve BaselineSelectionPin from published 037 catalog bytes.
|
||||
# @BRIEF Fail-closed pin of an explicit published catalog generation before start_run I/O, gate, or lease.
|
||||
# @LAYER Service
|
||||
# @RELATION CALLED_BY -> [ScenarioExecution.Runner.Start]
|
||||
# @RELATION DEPENDS_ON -> [DTO:BaselineSelectionPin]
|
||||
# @RELATION DEPENDS_ON -> [EXT:jsonschema]
|
||||
# @INVARIANT Client digests never establish pin authority; there is no silent latest/default selection.
|
||||
# @INVARIANT Absent baseline_set is legal only when the graph/plan has no baseline refs (null pin).
|
||||
# @INVARIANT A graph with compare_to_baseline or baseline refs and no set/version is BASELINE_MISSING.
|
||||
# @INVARIANT Working-tree YAML catalogs cannot populate pin identity and are never a published snapshot.
|
||||
# @RATIONALE CatalogRevision publication fields plus envelope identity (release_id, baseline_family,
|
||||
# baseline_set_id/version) are the only source that can satisfy baseline-pin.schema.json.
|
||||
# Existing YAML load_catalog validates metric/visual entries but has no catalog_revision_id,
|
||||
# release_id, publication receipt, or baseline_family — so resolution consumes an injected
|
||||
# published snapshot via load_published_catalog, never Git and never a catalog ORM.
|
||||
# @REJECTED Resolving "latest" approved YAML from the working tree — that is not a published generation
|
||||
# and cannot populate the pin. Copying a client-supplied pin/digest — client bytes are not
|
||||
# catalog authority. Inventing a CatalogRevision ORM/migration — D11 forbids a new Alembic
|
||||
# revision; the pin lives on RunnerPlan / target_snapshot JSON.
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from pathlib import Path
|
||||
import re
|
||||
from typing import Any
|
||||
|
||||
import jsonschema
|
||||
|
||||
from src.core.logger import logger
|
||||
|
||||
_SRC = "ScenarioExecution.BaselineResolver"
|
||||
_PIN_SCHEMA_CACHE: dict[str, Any] | None = None
|
||||
|
||||
BASELINE_MISSING = "BASELINE_MISSING"
|
||||
BASELINE_STALE = "BASELINE_STALE"
|
||||
BASELINE_AMBIGUOUS = "BASELINE_AMBIGUOUS"
|
||||
BASELINE_NOT_PUBLISHED = "BASELINE_NOT_PUBLISHED"
|
||||
BASELINE_EVIDENCE_UNAVAILABLE = "BASELINE_EVIDENCE_UNAVAILABLE"
|
||||
BASELINE_RESOLVE_ERRORS = frozenset({
|
||||
BASELINE_MISSING,
|
||||
BASELINE_STALE,
|
||||
BASELINE_AMBIGUOUS,
|
||||
BASELINE_NOT_PUBLISHED,
|
||||
BASELINE_EVIDENCE_UNAVAILABLE,
|
||||
})
|
||||
_COMPARE_ACTION = "compare_to_baseline"
|
||||
_SHA256 = r"^[a-f0-9]{64}$"
|
||||
_COMMIT = r"^[a-f0-9]{40}$"
|
||||
|
||||
|
||||
# #region ScenarioExecution.BaselineResolver.Reject [C:1] [TYPE Function]
|
||||
# @BRIEF Raise a typed start_run ValueError whose message is exactly the D11 code.
|
||||
def _reject(code: str) -> None:
|
||||
raise ValueError(code)
|
||||
# #endregion ScenarioExecution.BaselineResolver.Reject
|
||||
|
||||
|
||||
# #region ScenarioExecution.BaselineResolver.LoadPublished [C:3] [TYPE Function] [SEMANTICS baseline,catalog,published,seam]
|
||||
# @BRIEF [EXT] seam: accept an already-loaded published snapshot; never fetch Git or network.
|
||||
# @PRE snapshot is None, JSON bytes/text, or a dict envelope wrapping CatalogRevision.
|
||||
# @POST Returns a dict snapshot or None when no published generation was supplied.
|
||||
# @RATIONALE Tests inject a hardcoded published fixture that satisfies the pin schema. Production
|
||||
# callers pass the same snapshot they already loaded from durable published bytes.
|
||||
# YAML working-tree catalogs stay on load_catalog and cannot become this snapshot.
|
||||
def load_published_catalog(snapshot: dict[str, Any] | bytes | str | None = None) -> dict[str, Any] | None:
|
||||
if snapshot is None:
|
||||
return None
|
||||
payload: Any = snapshot
|
||||
if isinstance(snapshot, (bytes, bytearray)):
|
||||
try:
|
||||
payload = json.loads(bytes(snapshot).decode("utf-8"))
|
||||
except (UnicodeDecodeError, json.JSONDecodeError):
|
||||
logger.explore("Published catalog bytes are not JSON", src=_SRC, error_code=BASELINE_NOT_PUBLISHED)
|
||||
_reject(BASELINE_NOT_PUBLISHED)
|
||||
elif isinstance(snapshot, str):
|
||||
try:
|
||||
payload = json.loads(snapshot)
|
||||
except json.JSONDecodeError:
|
||||
logger.explore("Published catalog text is not JSON", src=_SRC, error_code=BASELINE_NOT_PUBLISHED)
|
||||
_reject(BASELINE_NOT_PUBLISHED)
|
||||
if not isinstance(payload, dict):
|
||||
_reject(BASELINE_NOT_PUBLISHED)
|
||||
return payload
|
||||
# #endregion ScenarioExecution.BaselineResolver.LoadPublished
|
||||
|
||||
|
||||
# #region ScenarioExecution.BaselineResolver.HasRefs [C:2] [TYPE Function] [SEMANTICS baseline,graph,refs]
|
||||
# @BRIEF True when the graph/plan declares compare_to_baseline or a non-empty baselines mapping.
|
||||
def graph_has_baseline_refs(graph_or_plan: dict[str, Any]) -> bool:
|
||||
baselines = graph_or_plan.get("baselines")
|
||||
if isinstance(baselines, dict) and baselines:
|
||||
return True
|
||||
pinned = graph_or_plan.get("pinned_baselines")
|
||||
if isinstance(pinned, dict) and pinned:
|
||||
return True
|
||||
for step in graph_or_plan.get("steps") or []:
|
||||
if not isinstance(step, dict):
|
||||
continue
|
||||
descriptor = step.get("action_descriptor") if isinstance(step.get("action_descriptor"), dict) else {}
|
||||
action = step.get("action") or descriptor.get("action")
|
||||
if action == _COMPARE_ACTION:
|
||||
return True
|
||||
return False
|
||||
# #endregion ScenarioExecution.BaselineResolver.HasRefs
|
||||
|
||||
|
||||
# #region ScenarioExecution.BaselineResolver.Canonical [C:1] [TYPE Function]
|
||||
# @BRIEF Canonical JSON for request-hash identity (sorted keys, no whitespace).
|
||||
def canonical_pin_json(pin: dict[str, Any] | None) -> str:
|
||||
return json.dumps(pin, sort_keys=True, separators=(",", ":"))
|
||||
# #endregion ScenarioExecution.BaselineResolver.Canonical
|
||||
|
||||
|
||||
# #region ScenarioExecution.BaselineResolver.Attach [C:2] [TYPE Function] [SEMANTICS baseline,pin,runnerplan]
|
||||
# @BRIEF Merge the resolved pin onto a derived plan and recompute plan_hash.
|
||||
def attach_baseline_pin(plan: dict[str, Any], pin: dict[str, Any] | None) -> dict[str, Any]:
|
||||
attached = dict(plan)
|
||||
attached["baseline_pin"] = pin
|
||||
attached.pop("plan_hash", None)
|
||||
attached["plan_hash"] = hashlib.sha256(
|
||||
json.dumps(attached, sort_keys=True, separators=(",", ":")).encode()
|
||||
).hexdigest()
|
||||
return attached
|
||||
# #endregion ScenarioExecution.BaselineResolver.Attach
|
||||
|
||||
|
||||
def _blank(value: Any) -> str | None:
|
||||
if value is None:
|
||||
return None
|
||||
text = str(value).strip()
|
||||
return text or None
|
||||
|
||||
|
||||
def _matches(pattern: str, value: Any) -> bool:
|
||||
return isinstance(value, str) and re.match(pattern, value) is not None
|
||||
|
||||
|
||||
# #region ScenarioExecution.BaselineResolver.PinSchema [C:2] [TYPE Function]
|
||||
# @BRIEF Load baseline-pin.schema.json once; structural fail-closed on a constructed pin.
|
||||
def _load_pin_schema() -> dict[str, Any]:
|
||||
global _PIN_SCHEMA_CACHE
|
||||
if _PIN_SCHEMA_CACHE is None:
|
||||
repo_root = Path(__file__).resolve().parents[5]
|
||||
schema_path = (
|
||||
repo_root / "specs" / "037-superset-baseline-engine" / "contracts" / "baseline-pin.schema.json"
|
||||
)
|
||||
_PIN_SCHEMA_CACHE = json.loads(schema_path.read_text(encoding="utf-8"))
|
||||
return _PIN_SCHEMA_CACHE
|
||||
|
||||
|
||||
def _validate_constructed_pin(pin: dict[str, Any]) -> None:
|
||||
try:
|
||||
jsonschema.validate(instance=pin, schema=_load_pin_schema())
|
||||
except jsonschema.ValidationError:
|
||||
logger.explore("Constructed pin failed schema", src=_SRC, error_code=BASELINE_EVIDENCE_UNAVAILABLE)
|
||||
_reject(BASELINE_EVIDENCE_UNAVAILABLE)
|
||||
# #endregion ScenarioExecution.BaselineResolver.PinSchema
|
||||
|
||||
|
||||
# #region ScenarioExecution.BaselineResolver.Snapshot [C:3] [TYPE Function] [SEMANTICS baseline,catalog,revision,envelope]
|
||||
# @BRIEF Split envelope identity from CatalogRevision; YAML-shaped catalogs are not published.
|
||||
def _split_published_snapshot(snapshot: dict[str, Any]) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
revision = snapshot.get("catalog_revision")
|
||||
if isinstance(revision, dict):
|
||||
return snapshot, revision
|
||||
if isinstance(snapshot.get("publication"), dict) and snapshot.get("catalog_revision_id"):
|
||||
logger.explore("CatalogRevision without envelope identity cannot populate the pin", src=_SRC, error_code=BASELINE_NOT_PUBLISHED)
|
||||
_reject(BASELINE_NOT_PUBLISHED)
|
||||
logger.explore("Snapshot is not a published CatalogRevision envelope", src=_SRC, error_code=BASELINE_NOT_PUBLISHED)
|
||||
_reject(BASELINE_NOT_PUBLISHED)
|
||||
return {}, {}
|
||||
# #endregion ScenarioExecution.BaselineResolver.Snapshot
|
||||
|
||||
|
||||
# #region ScenarioExecution.BaselineResolver.Publication [C:3] [TYPE Function] [SEMANTICS baseline,publication,receipt]
|
||||
# @BRIEF Only state=published with commit + receipt is a runnable generation.
|
||||
def _require_published(revision: dict[str, Any]) -> str:
|
||||
publication = revision.get("publication")
|
||||
if not isinstance(publication, dict):
|
||||
_reject(BASELINE_NOT_PUBLISHED)
|
||||
if publication.get("state") != "published":
|
||||
logger.explore("Catalog generation is not published", src=_SRC, error_code=BASELINE_NOT_PUBLISHED, payload={"state": publication.get("state")})
|
||||
_reject(BASELINE_NOT_PUBLISHED)
|
||||
commit_hash = publication.get("commit_hash")
|
||||
receipt = publication.get("published_receipt_id")
|
||||
if not _matches(_COMMIT, commit_hash) or _blank(receipt) is None:
|
||||
_reject(BASELINE_NOT_PUBLISHED)
|
||||
return str(commit_hash)
|
||||
# #endregion ScenarioExecution.BaselineResolver.Publication
|
||||
|
||||
|
||||
# #region ScenarioExecution.BaselineResolver.Selector [C:2] [TYPE Function]
|
||||
# @BRIEF Requested set/version must equal the published envelope; no alias/latest fallback.
|
||||
def _require_selector_match(envelope: dict[str, Any], baseline_set: str, baseline_set_version: str) -> None:
|
||||
set_id = _blank(envelope.get("baseline_set_id"))
|
||||
set_version = _blank(envelope.get("baseline_set_version"))
|
||||
if set_id is None or set_version is None:
|
||||
_reject(BASELINE_NOT_PUBLISHED)
|
||||
if set_id != baseline_set or set_version != baseline_set_version:
|
||||
logger.explore(
|
||||
"Requested baseline generation is not this published snapshot",
|
||||
src=_SRC, error_code=BASELINE_MISSING,
|
||||
payload={"requested": baseline_set, "requested_version": baseline_set_version},
|
||||
)
|
||||
_reject(BASELINE_MISSING)
|
||||
# #endregion ScenarioExecution.BaselineResolver.Selector
|
||||
|
||||
|
||||
# #region ScenarioExecution.BaselineResolver.Entries [C:3] [TYPE Function] [SEMANTICS baseline,approved,coordinate]
|
||||
# @BRIEF Collect approved revisions; duplicate coordinates are BASELINE_AMBIGUOUS, never first-match.
|
||||
def _approved_revisions(revision: dict[str, Any]) -> list[dict[str, Any]]:
|
||||
raw = revision.get("entry_revisions")
|
||||
if not isinstance(raw, list):
|
||||
_reject(BASELINE_NOT_PUBLISHED)
|
||||
approved = [item for item in raw if isinstance(item, dict) and item.get("status") == "approved"]
|
||||
if not approved:
|
||||
_reject(BASELINE_MISSING)
|
||||
for field in ("coordinate_hash", "baseline_id", "baseline_revision_id"):
|
||||
values = [item.get(field) for item in approved]
|
||||
if len(values) != len(set(values)):
|
||||
logger.explore("Ambiguous approved coordinate in published catalog", src=_SRC, error_code=BASELINE_AMBIGUOUS, payload={"field": field})
|
||||
_reject(BASELINE_AMBIGUOUS)
|
||||
return approved
|
||||
# #endregion ScenarioExecution.BaselineResolver.Entries
|
||||
|
||||
|
||||
# #region ScenarioExecution.BaselineResolver.Release [C:2] [TYPE Function]
|
||||
# @BRIEF Mixed release across the selected set is stale, not a runnable pin.
|
||||
def _unique_release(approved: list[dict[str, Any]]) -> tuple[str, str]:
|
||||
versions: set[str] = set()
|
||||
commits: set[str] = set()
|
||||
for item in approved:
|
||||
entry = item.get("entry") if isinstance(item.get("entry"), dict) else {}
|
||||
version = _blank(entry.get("release_version"))
|
||||
commit = _blank(entry.get("release_commit_hash"))
|
||||
if version is None or not _matches(_COMMIT, commit):
|
||||
_reject(BASELINE_STALE)
|
||||
versions.add(version)
|
||||
commits.add(str(commit))
|
||||
if len(versions) != 1 or len(commits) != 1:
|
||||
logger.explore("Mixed release in published catalog", src=_SRC, error_code=BASELINE_STALE)
|
||||
_reject(BASELINE_STALE)
|
||||
return next(iter(versions)), next(iter(commits))
|
||||
# #endregion ScenarioExecution.BaselineResolver.Release
|
||||
|
||||
|
||||
# #region ScenarioExecution.BaselineResolver.PinEntry [C:3] [TYPE Function] [SEMANTICS baseline,pin,entry,evidence]
|
||||
# @BRIEF Map one approved CatalogRevision wrapper to a pin entry; missing bytes fail closed.
|
||||
def _pin_entry(item: dict[str, Any]) -> dict[str, Any]:
|
||||
entry = item.get("entry") if isinstance(item.get("entry"), dict) else {}
|
||||
kind = "visual" if entry.get("kind") == "visual" else "metric"
|
||||
source_hash = entry.get("source_response_hash")
|
||||
capture_artifact_id = item.get("capture_artifact_id")
|
||||
capture_profile_hash = item.get("capture_profile_hash")
|
||||
image_hash = entry.get("expected_image_sha256") if kind == "visual" else None
|
||||
if not _matches(_SHA256, source_hash):
|
||||
_reject(BASELINE_EVIDENCE_UNAVAILABLE)
|
||||
if not _matches(_SHA256, capture_profile_hash) or _blank(capture_artifact_id) is None:
|
||||
_reject(BASELINE_EVIDENCE_UNAVAILABLE)
|
||||
if kind == "visual" and not _matches(_SHA256, image_hash):
|
||||
_reject(BASELINE_EVIDENCE_UNAVAILABLE)
|
||||
if not _matches(_SHA256, item.get("coordinate_hash")) or not _matches(_SHA256, item.get("entry_digest")):
|
||||
_reject(BASELINE_STALE)
|
||||
return {
|
||||
"baseline_id": item.get("baseline_id"),
|
||||
"baseline_revision_id": item.get("baseline_revision_id"),
|
||||
"kind": kind,
|
||||
"coordinate_hash": item.get("coordinate_hash"),
|
||||
"entry_digest": item.get("entry_digest"),
|
||||
"source_response_hash": source_hash,
|
||||
"expected_image_sha256": image_hash if kind == "visual" else None,
|
||||
"capture_artifact_id": capture_artifact_id,
|
||||
"capture_profile_hash": capture_profile_hash,
|
||||
}
|
||||
# #endregion ScenarioExecution.BaselineResolver.PinEntry
|
||||
|
||||
|
||||
# #region ScenarioExecution.BaselineResolver.Identity [C:3] [TYPE Function] [SEMANTICS baseline,digest,release,family]
|
||||
# @BRIEF Envelope identity plus revision digest; mismatched fingerprints are stale, missing identity is unpublished.
|
||||
def _require_pin_identity(envelope: dict[str, Any], revision: dict[str, Any]) -> tuple[str, str, str]:
|
||||
envelope_digest = _blank(envelope.get("catalog_digest"))
|
||||
revision_digest = revision.get("catalog_digest")
|
||||
if not _matches(_SHA256, revision_digest):
|
||||
_reject(BASELINE_STALE)
|
||||
if envelope_digest is not None and envelope_digest != revision_digest:
|
||||
logger.explore("Envelope catalog digest does not match revision", src=_SRC, error_code=BASELINE_STALE)
|
||||
_reject(BASELINE_STALE)
|
||||
release_id = _blank(envelope.get("release_id"))
|
||||
baseline_family = envelope.get("baseline_family")
|
||||
if release_id is None or not _matches(_SHA256, baseline_family):
|
||||
_reject(BASELINE_NOT_PUBLISHED)
|
||||
return str(revision_digest), release_id, str(baseline_family)
|
||||
# #endregion ScenarioExecution.BaselineResolver.Identity
|
||||
|
||||
|
||||
# #region ScenarioExecution.BaselineResolver.Resolve [C:4] [TYPE Function] [SEMANTICS baseline,pin,resolve,fail-closed]
|
||||
# @ingroup ScenarioExecution
|
||||
# @BRIEF Resolve BaselineSelectionPin from a published snapshot before a ScenarioRun row exists.
|
||||
# @PRE graph_or_plan is the derived plan or revision graph; selector is the explicit start request.
|
||||
# @POST Returns the pin dict or None; raises ValueError with a D11 code otherwise.
|
||||
# @INVARIANT No candidate/raw expected value, unpublished YAML, mixed release, or ambiguous
|
||||
# coordinate becomes a runnable pin.
|
||||
def resolve_baseline_pin(
|
||||
*,
|
||||
graph_or_plan: dict[str, Any],
|
||||
baseline_set: str | None,
|
||||
baseline_set_version: str | None = None,
|
||||
published_catalog: dict[str, Any] | None = None,
|
||||
) -> dict[str, Any] | None:
|
||||
set_id = _blank(baseline_set)
|
||||
set_version = _blank(baseline_set_version)
|
||||
needs_pin = graph_has_baseline_refs(graph_or_plan)
|
||||
if set_id is None and set_version is None and not needs_pin:
|
||||
return None
|
||||
if set_id is None or set_version is None:
|
||||
logger.explore("Baseline refs present without explicit set/version", src=_SRC, error_code=BASELINE_MISSING)
|
||||
_reject(BASELINE_MISSING)
|
||||
snapshot = load_published_catalog(published_catalog)
|
||||
if snapshot is None:
|
||||
logger.explore("No published catalog snapshot for baseline resolve", src=_SRC, error_code=BASELINE_NOT_PUBLISHED)
|
||||
_reject(BASELINE_NOT_PUBLISHED)
|
||||
envelope, revision = _split_published_snapshot(snapshot)
|
||||
_require_selector_match(envelope, set_id, set_version)
|
||||
publication_commit_hash = _require_published(revision)
|
||||
catalog_digest, release_id, baseline_family = _require_pin_identity(envelope, revision)
|
||||
approved = _approved_revisions(revision)
|
||||
release_version, release_commit_hash = _unique_release(approved)
|
||||
entries = [_pin_entry(item) for item in sorted(approved, key=lambda item: str(item.get("baseline_id") or ""))]
|
||||
pin = {
|
||||
"schema_version": 1,
|
||||
"baseline_set_id": set_id,
|
||||
"baseline_set_version": set_version,
|
||||
"catalog_revision_id": revision.get("catalog_revision_id"),
|
||||
"catalog_digest": catalog_digest,
|
||||
"release_id": release_id,
|
||||
"release_version": release_version,
|
||||
"release_commit_hash": release_commit_hash,
|
||||
"publication_commit_hash": publication_commit_hash,
|
||||
"baseline_family": baseline_family,
|
||||
"entries": entries,
|
||||
}
|
||||
_validate_constructed_pin(pin)
|
||||
logger.reflect(
|
||||
"Resolved BaselineSelectionPin from published catalog",
|
||||
src=_SRC,
|
||||
payload={"baseline_set_id": set_id, "baseline_set_version": set_version, "catalog_revision_id": pin["catalog_revision_id"]},
|
||||
)
|
||||
return pin
|
||||
# #endregion ScenarioExecution.BaselineResolver.Resolve
|
||||
|
||||
# #endregion ScenarioExecution.BaselineResolver
|
||||
@@ -0,0 +1,372 @@
|
||||
# #region ScenarioExecution.DecisionPolicy [C:5] [TYPE Module] [SEMANTICS scenario,execution,decision,policy,truth-table,stepoutcome]
|
||||
# @defgroup ScenarioExecution Sole StepOutcome mapper (SCEX-FR-028): first-match 038 truth table.
|
||||
# @BRIEF Total first-match mapping from deterministic comparison + immutable evaluation to StepOutcome.
|
||||
# @RELATION DEPENDS_ON -> [ScenarioGraph.DecisionPolicyV1]
|
||||
# @RELATION IMPLEMENTS -> [ScenarioGraph.DecisionPolicyV1]
|
||||
# @RELATION CALLED_BY -> [ScenarioExecution.Runner.Walker]
|
||||
# @RELATION CALLED_BY -> [ScenarioExecution.Lifecycle.Decide]
|
||||
# @INVARIANT decide_step_outcome is a pure function: no DB access, no logging, no mutation — repeated
|
||||
# canonical inputs yield identical status and reason codes.
|
||||
# @INVARIANT Every policy decision carries at least one reason_code (038: "Each outcome carries
|
||||
# at least one reason_code"); an empty code list is a contract defect, never an outcome.
|
||||
# @INVARIANT The 15 rows of decision-policy.md partition the entire input domain: unrecognized
|
||||
# envelopes, missing verdicts and invalid numbers resolve to row 15, never to a crash.
|
||||
# @RATIONALE The module is the runtime of the normative 038 table the executions pipeline has lacked:
|
||||
# the walker previously took step status directly from executor payloads, so no code path
|
||||
# could distinguish "deterministic comparison truth" from "advisory model claims". First-match
|
||||
# precedence (rows 4-6 before 8-14) keeps a passing model from erasing deterministic failures,
|
||||
# and disabled/advisory evaluation modes (v1) keep model claims as annotations only.
|
||||
# @REJECTED Treating executor payload statuses as the StepOutcome authority was rejected — SCEX-FR-028
|
||||
# names the DecisionPolicy truth table the sole mapper, so ad-hoc executor statuses would
|
||||
# bypass immutable policy. Fabricating an empty reason_codes list was rejected by the same
|
||||
# contract. Dynamic waiting_human from disagreement was rejected (038 row-sequence invariant):
|
||||
# an automated revision never acquires an undeclared HumanCheckpoint.
|
||||
# @DATA_CONTRACT (EvidenceManifest, ComparisonInput, EvaluationInput|None) ->
|
||||
# PolicyDecision{status, reason_codes}; reason codes are the exact strings of
|
||||
# decision-policy.md rows (BASELINE_PASS, BASELINE_MISMATCH, EVIDENCE_UNAVAILABLE,
|
||||
# EVIDENCE_BYTES_CORRUPT, POLICY_INVALID, EVALUATION_DISAGREEMENT, SEMANTIC_FAILURE, ...).
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from typing import Any, Literal
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field, ValidationError
|
||||
|
||||
# #region ScenarioExecution.DecisionPolicy.Constants [C:1] [TYPE Model] [SEMANTICS policy,model,038,constant]
|
||||
# @BRIEF Exact 038 decision-policy.schema.json shape, closed to unknown keys.
|
||||
# @INVARIANT policy_id/version are Literal consts — only baseline-semantic/1.0.0 may pin a plan.
|
||||
# @INVARIANT Every enum field mirrors the schema const/enum; extra fields are forbidden so a
|
||||
# server-owned pin can never drift from the 038 contract.
|
||||
class DecisionPolicy(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
|
||||
policy_id: Literal["baseline-semantic"]
|
||||
version: Literal["1.0.0"]
|
||||
confidence_threshold: float = Field(default=0.7, ge=0, le=1)
|
||||
evaluation_mode: Literal["disabled", "advisory", "required"] = "disabled"
|
||||
deterministic_hard_failure: Literal["failed"] = "failed"
|
||||
high_confidence_failure: Literal["failed"] = "failed"
|
||||
low_confidence: Literal["inconclusive"] = "inconclusive"
|
||||
disagreement: Literal["inconclusive"] = "inconclusive"
|
||||
missing_evidence: Literal["blocked"] = "blocked"
|
||||
provider_error: Literal["inconclusive"] = "inconclusive"
|
||||
human_checkpoint: Literal["declared_only"] = "declared_only"
|
||||
# #endregion ScenarioExecution.DecisionPolicy.Constants
|
||||
|
||||
|
||||
# #region ScenarioExecution.DecisionPolicy.DefaultV1 [C:1] [TYPE Constant] [SEMANTICS decision,policy,default,v1]
|
||||
# @BRIEF The single pinned baseline-semantic/1.0.0 instance for 044 v1 plans and legacy fallback.
|
||||
# @INVARIANT evaluation_mode stays disabled while no registry tool can emit an AgentEvaluation
|
||||
# (Slice E/F); derive_runner_plan overrides this field before pinning, never mutates it.
|
||||
BASELINE_SEMANTIC_V1 = DecisionPolicy(policy_id="baseline-semantic", version="1.0.0")
|
||||
# #endregion ScenarioExecution.DecisionPolicy.DefaultV1
|
||||
|
||||
|
||||
# #region ScenarioExecution.DecisionPolicy.EvidenceManifest [C:1] [TYPE Model] [SEMANTICS decision,evidence,manifest]
|
||||
# @BRIEF Verified win-attempt evidence facts consumed by truth-table row 3.
|
||||
# @INVARIANT digests_valid/owner_verified are only set by a verified manifest boundary (walk artifact
|
||||
# registration), never by caller prose or artifact metadata.
|
||||
class EvidenceManifest(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
|
||||
required: bool = False
|
||||
refs: list[str] = Field(default_factory=list)
|
||||
digests_valid: bool = True
|
||||
owner_verified: bool = True
|
||||
# #endregion ScenarioExecution.DecisionPolicy.EvidenceManifest
|
||||
|
||||
|
||||
# #region ScenarioExecution.DecisionPolicy.ComparisonInput [C:1] [TYPE Model] [SEMANTICS decision,comparison,input]
|
||||
# @BRIEF One 037 comparison status fact with the comparison's stable identity.
|
||||
# @INVARIANT status enumerates the result-evidence comparison schema enum — any other value is
|
||||
# unrecognized input for row 15, never a pass/fail vote.
|
||||
class ComparisonInput(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
|
||||
comparison_id: str
|
||||
status: Literal[
|
||||
"pass", "fail", "immutability_violation", "missing_baseline", "stale_baseline",
|
||||
"stale_visual_baseline", "ambiguous_baseline", "invalidated_baseline",
|
||||
"permission_denied", "source_error", "inconclusive",
|
||||
]
|
||||
reason_codes: list[str] = Field(default_factory=list)
|
||||
# #endregion ScenarioExecution.DecisionPolicy.ComparisonInput
|
||||
|
||||
|
||||
# #region ScenarioExecution.DecisionPolicy.EvaluationInput [C:1] [TYPE Model] [SEMANTICS decision,evaluation,input]
|
||||
# @BRIEF One immutable AgentEvaluation verdict summary consumed by rows 9-14.
|
||||
# @INVARIANT confidence defaults to unknown; verdict/confidence are independently absent so a missing
|
||||
# verdict or invalid number stays representable and resolves to row 15, never row 14.
|
||||
class EvaluationInput(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
|
||||
evaluation_id: str
|
||||
status: Literal[
|
||||
"succeeded", "provider_error", "parser_error", "budget_exceeded", "cancelled", "timed_out",
|
||||
]
|
||||
verdict: Literal["pass", "fail", "inconclusive"] | None = None
|
||||
confidence: float | None = None
|
||||
has_error_critical_findings: bool = False
|
||||
conflicts_with_deterministic_criterion: bool = False
|
||||
fail_supported_by_findings: bool = False
|
||||
# #endregion ScenarioExecution.DecisionPolicy.EvaluationInput
|
||||
|
||||
|
||||
# #region ScenarioExecution.DecisionPolicy.StepInputs [C:2] [TYPE Model] [SEMANTICS decision,step,inputs,winning-attempt]
|
||||
# @BRIEF The complete mapper input domain for one winning step attempt.
|
||||
# @INVARIANT winning_attempt_exists=False and cancelled_or_late=True both suppress the winning
|
||||
# outcome (row 1) — a historical record only, never a StepOutcome publication.
|
||||
class StepPolicyInputs(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
|
||||
winning_attempt_exists: bool = True
|
||||
cancelled_or_late: bool = False
|
||||
identity_valid: bool = True
|
||||
evidence: EvidenceManifest = Field(default_factory=EvidenceManifest)
|
||||
comparisons: list[ComparisonInput] = Field(default_factory=list)
|
||||
evaluation: EvaluationInput | None = None
|
||||
# #endregion ScenarioExecution.DecisionPolicy.StepInputs
|
||||
|
||||
|
||||
# #region ScenarioExecution.DecisionPolicy.Decision [C:1] [TYPE Model] [SEMANTICS decision,stepoutcome,status]
|
||||
# @BRIEF One mapped StepOutcome: status and the non-empty reason_codes contract.
|
||||
# @INVARIANT status None means row 1 (cancelled/lost attempt) — the caller publishes no winning outcome.
|
||||
# @INVARIANT reason_codes holds at least one exact decision-policy.md reason string.
|
||||
class PolicyDecision(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
|
||||
status: Literal["passed", "failed", "inconclusive", "blocked"] | None
|
||||
reason_codes: list[str] = Field(min_length=1)
|
||||
# #endregion ScenarioExecution.DecisionPolicy.Decision
|
||||
|
||||
|
||||
_BASELINE_BLOCKED_REASONS: dict[str, str] = {
|
||||
"missing_baseline": "BASELINE_MISSING",
|
||||
"stale_baseline": "BASELINE_STALE",
|
||||
"stale_visual_baseline": "BASELINE_STALE",
|
||||
"ambiguous_baseline": "BASELINE_AMBIGUOUS",
|
||||
"invalidated_baseline": "BASELINE_INVALIDATED",
|
||||
}
|
||||
_INCONCLUSIVE_STATUSES: frozenset[str] = frozenset(
|
||||
{"inconclusive", "source_error", "permission_denied"}
|
||||
)
|
||||
_HUMAN_DISPOSITIONS: dict[str, tuple[str, list[str]]] = {
|
||||
"confirm": ("passed", ["HUMAN_CONFIRMED"]),
|
||||
"pass": ("passed", ["HUMAN_CONFIRMED"]),
|
||||
"false_positive": ("inconclusive", ["HUMAN_FALSE_POSITIVE"]),
|
||||
"inconclusive": ("inconclusive", ["HUMAN_INCONCLUSIVE"]),
|
||||
}
|
||||
|
||||
|
||||
# #region ScenarioExecution.DecisionPolicy.Decide [C:5] [TYPE Function] [SEMANTICS decision,truth-table,mapper,pure]
|
||||
# @ingroup ScenarioExecution
|
||||
# @BRIEF First-match all 15 rows of decision-policy.md to one PolicyDecision.
|
||||
# @PRE inputs and policy describe one winning attempt; comparisons carry 037 statuses; evaluation is
|
||||
# None until an immutable AgentEvaluation exists (Slice F) — absent here is a legal v1 fact, not
|
||||
# a malformed envelope.
|
||||
# @POST Repeated canonical inputs return identical status and reason_codes (determinism); every
|
||||
# returned decision has a non-empty reason_codes list; unrecognized envelopes never raise.
|
||||
# @POST Rows execute strictly in table order: cancelled/lost (row 1) -> identity (2) -> evidence (3)
|
||||
# -> immutability (4) -> fail (5) -> baseline (6) -> inconclusive family (7) -> evaluation
|
||||
# rows (8-14) -> garbage (15). Deterministic failures precede and shadow model claims.
|
||||
# @INVARIANT Success requires exactly high-confidence clean evidence hierarchy: only an explicit
|
||||
# schema-valid pass with no error/critical finding reaches row 14; equality of confidence
|
||||
# with threshold is high (row 10 never fires at exactly the threshold).
|
||||
# @INVARIANT This function is pure: no DB, logger, clock, or mutation of its inputs.
|
||||
# @RATIONALE Comparisons reduce to a set (an outcome may reference several): the mapper scans the
|
||||
# whole comparison set in table-row priority (immutability > fail > baseline-block >
|
||||
# inconclusive family > all-pass) so first-match precedence holds deterministically even
|
||||
# when a step carries multiple 037 comparisons.
|
||||
# @REJECTED Accepting any unrecognized comparison/evaluation state silently as pass was rejected —
|
||||
# the total-domain row 15 must remain blocked so a client cannot vote an invalid envelope
|
||||
# into success.
|
||||
def decide_step_outcome(
|
||||
inputs: StepPolicyInputs | dict[str, Any],
|
||||
policy: DecisionPolicy | dict[str, Any],
|
||||
) -> PolicyDecision:
|
||||
try:
|
||||
normalized_inputs = (
|
||||
inputs if isinstance(inputs, StepPolicyInputs) else StepPolicyInputs.model_validate(inputs)
|
||||
)
|
||||
normalized_policy = (
|
||||
policy if isinstance(policy, DecisionPolicy) else DecisionPolicy.model_validate(policy)
|
||||
)
|
||||
except ValidationError:
|
||||
return PolicyDecision(status="blocked", reason_codes=["POLICY_INPUT_INVALID"])
|
||||
|
||||
if normalized_inputs.cancelled_or_late:
|
||||
return PolicyDecision(status=None, reason_codes=["CANCELLED"])
|
||||
if not normalized_inputs.winning_attempt_exists:
|
||||
return PolicyDecision(status=None, reason_codes=["LATE_RESPONSE"])
|
||||
if not normalized_inputs.identity_valid:
|
||||
return PolicyDecision(status="blocked", reason_codes=["POLICY_INVALID"])
|
||||
|
||||
evidence = normalized_inputs.evidence
|
||||
if evidence.required:
|
||||
if not evidence.refs:
|
||||
return PolicyDecision(status="blocked", reason_codes=["EVIDENCE_UNAVAILABLE"])
|
||||
if not evidence.digests_valid or not evidence.owner_verified:
|
||||
return PolicyDecision(status="blocked", reason_codes=["EVIDENCE_CORRUPT"])
|
||||
|
||||
statuses = [comparison.status for comparison in normalized_inputs.comparisons]
|
||||
if "immutability_violation" in statuses:
|
||||
return PolicyDecision(status="failed", reason_codes=["IMMUTABILITY_VIOLATION"])
|
||||
if "fail" in statuses:
|
||||
return PolicyDecision(status="failed", reason_codes=["BASELINE_MISMATCH"])
|
||||
for baseline_status, code in _BASELINE_BLOCKED_REASONS.items():
|
||||
if baseline_status in statuses:
|
||||
return PolicyDecision(status="blocked", reason_codes=[code])
|
||||
for inconclusive_status in _INCONCLUSIVE_STATUSES:
|
||||
if inconclusive_status in statuses:
|
||||
return PolicyDecision(status="inconclusive", reason_codes=["COMPARISON_INCONCLUSIVE"])
|
||||
if not statuses or any(status != "pass" for status in statuses):
|
||||
return PolicyDecision(status="blocked", reason_codes=["POLICY_INPUT_INVALID"])
|
||||
|
||||
if normalized_policy.evaluation_mode in {"disabled", "advisory"}:
|
||||
return PolicyDecision(status="passed", reason_codes=["BASELINE_PASS"])
|
||||
|
||||
evaluation = normalized_inputs.evaluation
|
||||
if evaluation is None:
|
||||
return PolicyDecision(status="inconclusive", reason_codes=["EVALUATION_UNAVAILABLE"])
|
||||
if evaluation.status != "succeeded":
|
||||
return PolicyDecision(status="inconclusive", reason_codes=["EVALUATION_UNAVAILABLE"])
|
||||
if (
|
||||
evaluation.verdict is None
|
||||
or evaluation.confidence is None
|
||||
or not 0.0 <= evaluation.confidence <= 1.0
|
||||
):
|
||||
return PolicyDecision(status="blocked", reason_codes=["POLICY_INPUT_INVALID"])
|
||||
|
||||
high_confidence = evaluation.confidence >= normalized_policy.confidence_threshold
|
||||
if not high_confidence:
|
||||
return PolicyDecision(status="inconclusive", reason_codes=["LOW_CONFIDENCE"])
|
||||
if evaluation.verdict == "inconclusive":
|
||||
return PolicyDecision(status="inconclusive", reason_codes=["MODEL_INCONCLUSIVE"])
|
||||
if evaluation.conflicts_with_deterministic_criterion:
|
||||
return PolicyDecision(status="inconclusive", reason_codes=["EVALUATION_DISAGREEMENT"])
|
||||
if evaluation.verdict == "pass" and evaluation.has_error_critical_findings:
|
||||
return PolicyDecision(status="inconclusive", reason_codes=["EVALUATION_CONTRADICTORY"])
|
||||
if evaluation.verdict == "fail" and not evaluation.fail_supported_by_findings:
|
||||
return PolicyDecision(status="inconclusive", reason_codes=["EVALUATION_CONTRADICTORY"])
|
||||
if evaluation.verdict == "fail":
|
||||
return PolicyDecision(status="failed", reason_codes=["SEMANTIC_FAILURE"])
|
||||
if evaluation.verdict == "pass":
|
||||
return PolicyDecision(status="passed", reason_codes=["BASELINE_AND_SEMANTIC_PASS"])
|
||||
return PolicyDecision(status="blocked", reason_codes=["POLICY_INPUT_INVALID"])
|
||||
# #endregion ScenarioExecution.DecisionPolicy.Decide
|
||||
|
||||
|
||||
# #region ScenarioExecution.DecisionPolicy.HumanDisposition [C:3] [TYPE Function] [SEMANTICS decision,human,checkpoint,disposition]
|
||||
# @ingroup ScenarioExecution
|
||||
# @BRIEF Single source for the HumanCheckpoint disposition -> (status, reason_codes) table (038).
|
||||
# @PRE disposition is one of the five decide_checkpoint dispositions.
|
||||
# @POST confirm/pass -> (passed, HUMAN_CONFIRMED); false_positive/inconclusive -> (inconclusive,
|
||||
# HUMAN_FALSE_POSITIVE / HUMAN_INCONCLUSIVE) — never a fabricated PASS.
|
||||
# @INVARIANT fail is deliberately unmapped here: its failed status (which blocks dependents) stays
|
||||
# owned by ScenarioExecution.Lifecycle.Decide; the mapper cannot express a run-blocking
|
||||
# observation without the caller's failed-closure semantics.
|
||||
# @REJECTED Mapping false_positive to passed was rejected — the observation remains an explicit
|
||||
# non-pass checkpoint outcome even when dependents can collect more evidence.
|
||||
def map_human_disposition(disposition: str) -> tuple[str, list[str]]:
|
||||
if disposition == "fail":
|
||||
raise ValueError("HUMAN_DISPOSITION_FAIL_CALLER_OWNED")
|
||||
mapped = _HUMAN_DISPOSITIONS.get(disposition)
|
||||
if mapped is None:
|
||||
raise ValueError("invalid disposition")
|
||||
return mapped
|
||||
# #endregion ScenarioExecution.DecisionPolicy.HumanDisposition
|
||||
|
||||
|
||||
# #region ScenarioExecution.DecisionPolicy.PolicyDigest [C:2] [TYPE Function] [SEMANTICS decision,policy,digest,idempotency]
|
||||
# @ingroup ScenarioExecution
|
||||
# @BRIEF Canonical SHA-256 of a resolved policy for the run idempotency identity.
|
||||
# @POST Stable across processes for the same policy shape; a changed policy changes the request hash.
|
||||
# @INVARIANT Uses the canonical JSON dump (same separators as runner plan_hash) so derive-time and
|
||||
# start-time digests never diverge by formatting.
|
||||
def policy_digest(policy: DecisionPolicy) -> str:
|
||||
canonical = json.dumps(policy.model_dump(), sort_keys=True, separators=(",", ":")).encode()
|
||||
return hashlib.sha256(canonical).hexdigest()
|
||||
# #endregion ScenarioExecution.DecisionPolicy.PolicyDigest
|
||||
|
||||
|
||||
# #region ScenarioExecution.DecisionPolicy.OutcomeAdapter [C:4] [TYPE Function] [SEMANTICS decision,walker,outcome,adapter,manifest]
|
||||
# @ingroup ScenarioExecution
|
||||
# @BRIEF Collect StepPolicyInputs when the step is policy-normative; return None for non-verification steps.
|
||||
# @PRE outcome/step come from the walker finalization path only; artifacts_integrity is the
|
||||
# register_step_evidence integrity payload (None when every ref registered).
|
||||
# @POST Returns one StepPolicyInputs with exactly one ComparisonInput derived from the executor
|
||||
# comparison_status or the outcome status fallback; evaluation stays None in v1.
|
||||
# @INVARIANT The mapper applies only to steps that carry comparison authority: verification tools
|
||||
# (assertion/screenshot/sql_evidence) or an explicit 037 comparison payload. Side-effect
|
||||
# steps (browser navigation, xlsx, transform, report, artifact) keep their executor outcome
|
||||
# without a policy stamp — labeling a pure interaction BASELINE_PASS would be misleading.
|
||||
# @INVARIANT Evidence is "required" only for tools whose executor PASS contract mandates registered
|
||||
# evidence (screenshot, sql_evidence); register_step_evidence also audits opportunistic
|
||||
# refs (browser/superset), but policy row 3 applies only to declared required evidence —
|
||||
# otherwise the walker's inconclusive integrity branch would be wrongly escalated to blocked.
|
||||
# @INVARIANT digests_valid mirrors the artifact registration result: absent/invalid digests make a
|
||||
# required-evidence step resolve to row 3 EVIDENCE_* codes instead of row 7.
|
||||
# @RATIONALE The fallback from outcome.status is required because v1 executor payloads do not always
|
||||
# carry a 037 comparison_status; passing a synthesized pass/fail vote keeps the mapper
|
||||
# total while the 037 comparison records materialize in a later slice.
|
||||
# @REJECTED Deriving required-evidence from a raw stringly tool name alone was rejected — only the
|
||||
# executor's pass contract (refs mandatory for PASS) is the declaration boundary.
|
||||
_EVIDENCE_REQUIRED_TOOLS = frozenset({"screenshot", "sql_evidence"})
|
||||
_POLICY_NORMATIVE_TOOLS = frozenset({"assertion", "screenshot", "sql_evidence", "agent_evaluation"})
|
||||
_COMPARISON_PAYLOAD_KEYS = ("comparison_status", "comparison_id", "comparison")
|
||||
|
||||
|
||||
def policy_inputs_from_outcome(
|
||||
outcome: dict[str, Any],
|
||||
step: dict[str, Any],
|
||||
artifacts_integrity: dict[str, Any] | None,
|
||||
) -> StepPolicyInputs | None:
|
||||
step_outcome = outcome.get("step_outcome") if isinstance(outcome.get("step_outcome"), dict) else outcome
|
||||
tool = str(step.get("tool") or step_outcome.get("tool") or "")
|
||||
if tool not in _POLICY_NORMATIVE_TOOLS and not any(key in step_outcome for key in _COMPARISON_PAYLOAD_KEYS):
|
||||
return None
|
||||
status = str(outcome.get("status") or "inconclusive")
|
||||
comparison_status = step_outcome.get("comparison_status")
|
||||
if comparison_status is None:
|
||||
comparison_status = {"passed": "pass", "failed": "fail", "inconclusive": "inconclusive"}.get(
|
||||
status, "inconclusive"
|
||||
)
|
||||
refs = list(outcome.get("artifact_refs") or [])
|
||||
return StepPolicyInputs(
|
||||
evidence=EvidenceManifest(
|
||||
required=tool in _EVIDENCE_REQUIRED_TOOLS,
|
||||
refs=refs,
|
||||
digests_valid=artifacts_integrity is None,
|
||||
owner_verified=True,
|
||||
),
|
||||
comparisons=[
|
||||
ComparisonInput(
|
||||
comparison_id=str(
|
||||
step_outcome.get("comparison_id") or step.get("logical_step_id") or "step"
|
||||
),
|
||||
status=comparison_status,
|
||||
)
|
||||
],
|
||||
evaluation=(EvaluationInput.model_validate(step_outcome["evaluation_input"])
|
||||
if isinstance(step_outcome.get("evaluation_input"), dict) else None),
|
||||
)
|
||||
# #endregion ScenarioExecution.DecisionPolicy.OutcomeAdapter
|
||||
|
||||
|
||||
# #region ScenarioExecution.DecisionPolicy.VerifiedRefs [C:2] [TYPE Function] [SEMANTICS decision,evidence,refs,verified]
|
||||
# @BRIEF Return only artifact refs that passed walker evidence registration (integrity clean).
|
||||
# @POST The returned list excludes every ref named in artifacts_integrity.unregistered_refs; a clean
|
||||
# registration returns all refs unchanged.
|
||||
def verified_evidence_refs(
|
||||
outcome: dict[str, Any],
|
||||
artifacts_integrity: dict[str, Any] | None,
|
||||
) -> list[str]:
|
||||
refs = list(outcome.get("artifact_refs") or [])
|
||||
if not isinstance(artifacts_integrity, dict):
|
||||
return refs
|
||||
unregistered = set(artifacts_integrity.get("unregistered_refs") or [])
|
||||
return [ref for ref in refs if ref not in unregistered]
|
||||
# #endregion ScenarioExecution.DecisionPolicy.VerifiedRefs
|
||||
|
||||
# #endregion ScenarioExecution.DecisionPolicy
|
||||
@@ -0,0 +1,247 @@
|
||||
# #region ScenarioExecution.EvaluationAdapter [C:5] [TYPE Module] [SEMANTICS scenario,execution,evaluation,adapter,provider]
|
||||
# @defgroup ScenarioExecution Sync AgentEvaluation adapter: prior-step manifest, provider seam, raw bytes.
|
||||
# @BRIEF Close over storage/run_async like live Superset adapters; never synthesize a PASS.
|
||||
# @RELATION CALLS -> [ScenarioExecution.AgentEvaluation]
|
||||
# @RELATION DEPENDS_ON -> [Services.AgentRuns.Artifacts]
|
||||
# @INVARIANT Executor remains sync (step, completed) -> dict. Missing provider/timeout/invalid stay non-PASS; succeeded still requires bound raw-response bytes.
|
||||
# @RATIONALE input_manifest is assembled from completed prior-step artifacts so P0-1 evidence can bind without duplicating those rows onto the evaluation step.
|
||||
# @REJECTED Copying previous-step artifact rows onto the evaluation step was rejected — that forges a second provenance identity for the same digest (D1).
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
from collections.abc import Callable
|
||||
from datetime import UTC, datetime
|
||||
from hashlib import sha256
|
||||
from typing import Any
|
||||
import uuid
|
||||
|
||||
from src.core.database import SessionLocal
|
||||
from src.services.dashboard_testing.scenario.models import AgentEvaluationSpec
|
||||
|
||||
from .agent_evaluation import AgentEvaluation, parse_evaluation_response, submit_evaluation
|
||||
from .artifacts import is_valid_sha256
|
||||
from .live_binding import EvidenceStorage
|
||||
|
||||
_ALLOWED_CONTENT_TYPES = frozenset({"image/jpeg", "image/png", "image/webp", "application/json"})
|
||||
|
||||
|
||||
# #region ScenarioExecution.EvaluationAdapter.Hash [C:1] [TYPE Function] [SEMANTICS evaluation,hash]
|
||||
# @BRIEF Canonical SHA-256 of a JSON-serializable payload.
|
||||
def _canonical_hash(payload: Any) -> str:
|
||||
return sha256(json.dumps(payload, sort_keys=True, separators=(",", ":"), default=str).encode()).hexdigest()
|
||||
# #endregion ScenarioExecution.EvaluationAdapter.Hash
|
||||
|
||||
|
||||
# #region ScenarioExecution.EvaluationAdapter.Spec [C:2] [TYPE Function] [SEMANTICS evaluation,spec]
|
||||
# @BRIEF Load the pinned AgentEvaluationSpec from the dispatched step; missing/malformed is fail-closed.
|
||||
def _spec_from_step(step: dict[str, Any]) -> AgentEvaluationSpec:
|
||||
payload = step.get("step_meta") if isinstance(step.get("step_meta"), dict) else {}
|
||||
raw_spec = payload.get("agent_evaluation_spec")
|
||||
if raw_spec is None:
|
||||
raw_spec = step.get("agent_evaluation_spec")
|
||||
if not isinstance(raw_spec, dict):
|
||||
raise RuntimeError("EVALUATION_RESPONSE_INVALID")
|
||||
return AgentEvaluationSpec.model_validate(raw_spec)
|
||||
# #endregion ScenarioExecution.EvaluationAdapter.Spec
|
||||
|
||||
|
||||
# #region ScenarioExecution.EvaluationAdapter.Manifest [C:3] [TYPE Function] [SEMANTICS evaluation,manifest,completed]
|
||||
# @BRIEF Build input_manifest from completed prior-step artifact refs, never from this evaluation step.
|
||||
# @INVARIANT Items without a valid digest, allowed MIME, and byte_length >= 1 are omitted rather than invented.
|
||||
# @RATIONALE Walker has flushed prior evidence into completed outcomes; a second SessionLocal would not see uncommitted rows, so completed is the only lawful source at adapter time.
|
||||
# @REJECTED Querying ScenarioArtifact in a fresh session was rejected — walker has only flushed.
|
||||
def _manifest_from_completed(completed: dict[str, dict[str, Any]]) -> list[dict[str, Any]]:
|
||||
items: list[dict[str, Any]] = []
|
||||
for outcome in completed.values():
|
||||
if not isinstance(outcome, dict):
|
||||
continue
|
||||
refs = list(outcome.get("artifact_refs") or [])
|
||||
nested = outcome.get("step_outcome") if isinstance(outcome.get("step_outcome"), dict) else outcome
|
||||
if not isinstance(nested, dict):
|
||||
nested = outcome
|
||||
digests = nested.get("artifact_digests") if isinstance(nested.get("artifact_digests"), dict) else {}
|
||||
types = nested.get("artifact_content_types") if isinstance(nested.get("artifact_content_types"), dict) else {}
|
||||
lengths = nested.get("artifact_byte_lengths") if isinstance(nested.get("artifact_byte_lengths"), dict) else {}
|
||||
default_type = nested.get("content_type")
|
||||
if default_type not in _ALLOWED_CONTENT_TYPES:
|
||||
default_type = "image/jpeg" if nested.get("tool") == "screenshot" else "application/json"
|
||||
default_length = nested.get("byte_length")
|
||||
for index, ref in enumerate(refs):
|
||||
if not isinstance(ref, str) or not ref:
|
||||
continue
|
||||
digest = digests.get(ref) or (nested.get("sha256") if len(refs) == 1 else None)
|
||||
if not is_valid_sha256(digest):
|
||||
continue
|
||||
content_type = types.get(ref) if types.get(ref) in _ALLOWED_CONTENT_TYPES else default_type
|
||||
if content_type not in _ALLOWED_CONTENT_TYPES:
|
||||
continue
|
||||
byte_length = lengths.get(ref) if isinstance(lengths.get(ref), int) else default_length
|
||||
if not isinstance(byte_length, int) or byte_length < 1:
|
||||
continue
|
||||
items.append({
|
||||
"artifact_id": ref,
|
||||
"sha256": str(digest).lower(),
|
||||
"content_type": content_type,
|
||||
"byte_length": byte_length,
|
||||
"role": "actual" if index == 0 and not items else "context",
|
||||
})
|
||||
return items
|
||||
# #endregion ScenarioExecution.EvaluationAdapter.Manifest
|
||||
|
||||
|
||||
# #region ScenarioExecution.EvaluationAdapter.Input [C:2] [TYPE Function] [SEMANTICS evaluation,decision,input]
|
||||
# @BRIEF Project an immutable evaluation record into the DecisionPolicy EvaluationInput envelope.
|
||||
def _evaluation_input(record: AgentEvaluation) -> dict[str, Any]:
|
||||
findings = record.findings
|
||||
return {
|
||||
"evaluation_id": record.evaluation_id,
|
||||
"status": record.status,
|
||||
"verdict": record.verdict,
|
||||
"confidence": record.confidence,
|
||||
"has_error_critical_findings": any(item.severity in {"error", "critical"} for item in findings),
|
||||
"conflicts_with_deterministic_criterion": any(
|
||||
item.criterion_kind == "deterministic_comparison" for item in findings
|
||||
),
|
||||
"fail_supported_by_findings": record.verdict == "fail" and bool(findings),
|
||||
}
|
||||
# #endregion ScenarioExecution.EvaluationAdapter.Input
|
||||
|
||||
|
||||
# #region ScenarioExecution.EvaluationAdapter.Storage [C:2] [TYPE Function] [SEMANTICS evaluation,storage]
|
||||
# @BRIEF Resolve draft storage only at invoke time so default registry composition needs no config.
|
||||
# @RATIONALE Default registry is built without Admin storage settings; eager get_draft_storage() crashed queued dispatch.
|
||||
# @REJECTED Closing over get_draft_storage() at composition-root construction — it made every default-registry walk require a configured storage root.
|
||||
def _default_storage() -> EvidenceStorage:
|
||||
try:
|
||||
from src.services.agent_runs.artifacts import get_draft_storage
|
||||
|
||||
return get_draft_storage()
|
||||
except Exception as exc:
|
||||
raise RuntimeError("EVALUATION_PROVIDER_MISSING") from exc
|
||||
# #endregion ScenarioExecution.EvaluationAdapter.Storage
|
||||
|
||||
|
||||
# #region ScenarioExecution.EvaluationAdapter.Raw [C:3] [TYPE Function] [SEMANTICS evaluation,artifact,raw,storage]
|
||||
# @BRIEF Persist raw JSON bytes through owned storage and return the opaque ref plus digest.
|
||||
# @POST content_ref equals draft:{run_id}:{digest}; MIME is application/json.
|
||||
def _persist_raw_response(storage: EvidenceStorage, run_id: str, raw: Any) -> tuple[bytes, str, str]:
|
||||
raw_bytes = json.dumps(raw, sort_keys=True, separators=(",", ":"), default=str).encode()
|
||||
if not raw_bytes:
|
||||
raise RuntimeError("EVALUATION_RESPONSE_INVALID")
|
||||
digest = sha256(raw_bytes).hexdigest()
|
||||
content_ref = storage.store(run_id, digest, raw_bytes)
|
||||
expected = f"draft:{run_id}:{digest}"
|
||||
if content_ref != expected:
|
||||
raise RuntimeError("EVALUATION_RESPONSE_INVALID")
|
||||
return raw_bytes, digest, content_ref
|
||||
# #endregion ScenarioExecution.EvaluationAdapter.Raw
|
||||
|
||||
|
||||
# #region ScenarioExecution.EvaluationAdapter.Provenance [C:2] [TYPE Function] [SEMANTICS evaluation,provenance]
|
||||
# @BRIEF Overlay run/step identity, manifest, and raw-response refs; provider text is not identity authority.
|
||||
def _stamp_provenance(
|
||||
raw: Any,
|
||||
*,
|
||||
run_id: str,
|
||||
logical_step_id: str,
|
||||
attempt: int,
|
||||
manifest: list[dict[str, Any]],
|
||||
content_ref: str,
|
||||
digest: str,
|
||||
spec: AgentEvaluationSpec,
|
||||
) -> dict[str, Any]:
|
||||
payload = dict(raw) if isinstance(raw, dict) else {}
|
||||
now = datetime.now(UTC)
|
||||
payload.update({
|
||||
"schema_version": 1,
|
||||
"scenario_run_id": run_id,
|
||||
"logical_step_id": logical_step_id,
|
||||
"attempt": attempt,
|
||||
"input_manifest": manifest,
|
||||
"input_manifest_hash": _canonical_hash(manifest),
|
||||
"raw_response_artifact_ref": content_ref,
|
||||
"raw_response_sha256": digest,
|
||||
"evaluation_spec_hash": _canonical_hash(spec.model_dump(mode="json")),
|
||||
"started_at": payload.get("started_at") or now,
|
||||
"finished_at": payload.get("finished_at") or now,
|
||||
})
|
||||
payload.setdefault("evaluation_id", str(uuid.uuid4()))
|
||||
payload.setdefault("operation_id", str(uuid.uuid4()))
|
||||
return payload
|
||||
# #endregion ScenarioExecution.EvaluationAdapter.Provenance
|
||||
|
||||
|
||||
# #region ScenarioExecution.EvaluationAdapter.Factory [C:5] [TYPE Function] [SEMANTICS evaluation,adapter,factory,provider]
|
||||
# @BRIEF Build the sync adapter closed over storage, run_async, and the submit_evaluation seam.
|
||||
# @PRE storage is the deployment-owned evidence boundary; tests inject submit/client, never a live LLM.
|
||||
# @POST Returns evaluation_input + evaluation_record + artifact_refs/digests/MIME/length for the walker.
|
||||
# @INVARIANT Empty prior-step evidence, missing spec, and provider failures raise rather than PASS.
|
||||
# @SIDE_EFFECT Stores raw JSON bytes; submit_evaluation claims agent_evaluation capacity on the live path.
|
||||
def evaluation_adapter_from(
|
||||
*,
|
||||
run_async: Callable[[Any], Any] | None = None,
|
||||
storage: EvidenceStorage | None = None,
|
||||
client: Any = None,
|
||||
submit: Callable[..., dict[str, Any]] | None = None,
|
||||
db_factory: Callable[[], Any] | None = None,
|
||||
) -> Callable[[dict[str, Any], dict[str, dict[str, Any]]], dict[str, Any]]:
|
||||
def invoke(step: dict[str, Any], completed: dict[str, dict[str, Any]]) -> dict[str, Any]:
|
||||
spec = _spec_from_step(step)
|
||||
manifest = _manifest_from_completed(completed)
|
||||
if not manifest:
|
||||
raise RuntimeError("EVALUATION_EVIDENCE_NOT_FOUND")
|
||||
run_id = step.get("scenario_run_id")
|
||||
logical_step_id = step.get("logical_step_id")
|
||||
if not isinstance(run_id, str) or not run_id or not isinstance(logical_step_id, str) or not logical_step_id:
|
||||
raise RuntimeError("EVALUATION_RESPONSE_INVALID")
|
||||
attempt = int(step.get("attempt") or 1)
|
||||
evidence = storage if storage is not None else _default_storage()
|
||||
prompt = json.dumps(
|
||||
{"evaluation_spec": spec.model_dump(mode="json"), "input_manifest": manifest},
|
||||
sort_keys=True, separators=(",", ":"), default=str,
|
||||
)
|
||||
if submit is not None:
|
||||
raw = submit(spec=spec, prompt=prompt, step=step, completed=completed)
|
||||
else:
|
||||
target = step.get("target_snapshot") if isinstance(step.get("target_snapshot"), dict) else {}
|
||||
meta = step.get("step_meta") if isinstance(step.get("step_meta"), dict) else {}
|
||||
environment_id = str(target.get("environment_id") or meta.get("environment_id") or "")
|
||||
environment_class = "PROD" if target.get("environment_class") == "PROD" or meta.get("environment_class") == "PROD" else "DEV"
|
||||
if not environment_id:
|
||||
raise RuntimeError("EVALUATION_PROVIDER_MISSING")
|
||||
db = (db_factory or SessionLocal)()
|
||||
try:
|
||||
coro = submit_evaluation(
|
||||
db, spec=spec, prompt=prompt, artifacts=[], environment_id=environment_id,
|
||||
environment_class=environment_class, run_id=run_id, logical_step_id=logical_step_id,
|
||||
client=client,
|
||||
)
|
||||
raw = run_async(coro) if run_async is not None else asyncio.run(coro)
|
||||
finally:
|
||||
close = getattr(db, "close", None)
|
||||
if callable(close):
|
||||
close()
|
||||
if not isinstance(raw, dict):
|
||||
raise RuntimeError("EVALUATION_RESPONSE_INVALID")
|
||||
raw_bytes, digest, content_ref = _persist_raw_response(evidence, run_id, raw)
|
||||
record = parse_evaluation_response(
|
||||
_stamp_provenance(
|
||||
raw, run_id=run_id, logical_step_id=logical_step_id, attempt=attempt,
|
||||
manifest=manifest, content_ref=content_ref, digest=digest, spec=spec,
|
||||
),
|
||||
spec=spec,
|
||||
)
|
||||
return {
|
||||
"evaluation_input": _evaluation_input(record),
|
||||
"evaluation_record": record.model_dump(mode="json"),
|
||||
"artifact_refs": [content_ref],
|
||||
"artifact_digests": {content_ref: digest},
|
||||
"content_type": "application/json",
|
||||
"byte_length": len(raw_bytes),
|
||||
}
|
||||
|
||||
return invoke
|
||||
# #endregion ScenarioExecution.EvaluationAdapter.Factory
|
||||
|
||||
# #endregion ScenarioExecution.EvaluationAdapter
|
||||
@@ -37,6 +37,7 @@ from .live_adapter import (
|
||||
dispatch_live_adapter,
|
||||
)
|
||||
|
||||
|
||||
_STATUS = {
|
||||
ComparisonStatus.PASS: "passed",
|
||||
ComparisonStatus.FAIL: "failed",
|
||||
@@ -444,6 +445,45 @@ def artifact(step: dict[str, Any], _completed: dict[str, dict[str, Any]]) -> dic
|
||||
# #endregion ScenarioExecution.Executors.Artifact
|
||||
|
||||
|
||||
# #region ScenarioExecution.Executors.AgentEvaluation [C:4] [TYPE Function] [SEMANTICS scenario,execution,evaluation,executor]
|
||||
# @BRIEF Fail-closed evaluation adapter boundary; only an injected mock/provider may produce a result.
|
||||
# @PRE The step contains a validated AgentEvaluationSpec and the adapter is explicitly composed.
|
||||
# @POST Missing provider never becomes PASS and returns a typed unavailable outcome.
|
||||
# @RATIONALE The executor accepts an adapter rather than reaching into provider globals, keeping tests offline and capacity ownership explicit. Fail-closed outcomes still emit comparison_status=pass so DecisionPolicy row 9 (EVALUATION_UNAVAILABLE) is reachable instead of row 7.
|
||||
# @REJECTED Synthesizing a successful evaluation from declared criteria was rejected because it would bypass raw-response provenance. Mapping missing adapter to COMPARISON_INCONCLUSIVE was rejected — that hides EVALUATION_UNAVAILABLE.
|
||||
def agent_evaluation(step: dict[str, Any], completed: dict[str, dict[str, Any]], *, adapter: Any = None) -> dict[str, Any]:
|
||||
# comparison_status=pass is not a baseline claim: this tool has no 037 comparison, so the
|
||||
# policy mapper can reach rows 9–14 instead of stalling on COMPARISON_INCONCLUSIVE.
|
||||
closed = {"comparison_status": "pass"}
|
||||
if adapter is None:
|
||||
return _outcome("agent_evaluation", "inconclusive", reason="EVALUATION_PROVIDER_MISSING", extra=closed)
|
||||
try:
|
||||
result = adapter(step, completed)
|
||||
except TimeoutError:
|
||||
return _outcome("agent_evaluation", "inconclusive", reason="EVALUATION_TIMED_OUT", extra=closed)
|
||||
except Exception:
|
||||
return _outcome("agent_evaluation", "inconclusive", reason="EVALUATION_PROVIDER_ERROR", extra=closed)
|
||||
if not isinstance(result, dict) or not isinstance(result.get("evaluation_input"), dict):
|
||||
return _outcome("agent_evaluation", "inconclusive", reason="EVALUATION_RESPONSE_INVALID", extra=closed)
|
||||
extra = {
|
||||
**closed,
|
||||
"evaluation_input": result["evaluation_input"],
|
||||
"evaluation_record": result.get("evaluation_record"),
|
||||
}
|
||||
for key in ("artifact_digests", "artifact_content_types", "artifact_byte_lengths"):
|
||||
if isinstance(result.get(key), dict):
|
||||
extra[key] = result[key]
|
||||
if isinstance(result.get("content_type"), str):
|
||||
extra["content_type"] = result["content_type"]
|
||||
if isinstance(result.get("byte_length"), int):
|
||||
extra["byte_length"] = result["byte_length"]
|
||||
return _outcome(
|
||||
"agent_evaluation", "passed", reason="EVALUATION_COMPLETE", extra=extra,
|
||||
refs=list(result.get("artifact_refs") or []),
|
||||
)
|
||||
# #endregion ScenarioExecution.Executors.AgentEvaluation
|
||||
|
||||
|
||||
# #region ScenarioExecution.Executors.RegisterDefaults [C:2] [TYPE Function] [SEMANTICS scenario,execution,executor,registry]
|
||||
# @BRIEF Register bounded built-in executors and optional injected live-I/O adapters.
|
||||
def _register_default_executors(
|
||||
@@ -452,6 +492,7 @@ def _register_default_executors(
|
||||
browser_adapter: BrowserExecutionAdapter | None = None,
|
||||
superset_adapter: SupersetExecutionAdapter | None = None,
|
||||
screenshot_adapter: ScreenshotExecutionAdapter | None = None,
|
||||
agent_evaluation_adapter: Any = None,
|
||||
) -> None:
|
||||
registry.register("assertion", assertion)
|
||||
registry.register("browser", lambda step, completed: browser(step, completed, adapter=browser_adapter))
|
||||
@@ -462,5 +503,6 @@ def _register_default_executors(
|
||||
registry.register("screenshot", lambda step, completed: screenshot(step, completed, adapter=screenshot_adapter))
|
||||
registry.register("report", report)
|
||||
registry.register("artifact", artifact)
|
||||
registry.register("agent_evaluation", lambda step, completed: agent_evaluation(step, completed, adapter=agent_evaluation_adapter))
|
||||
# #endregion ScenarioExecution.Executors.RegisterDefaults
|
||||
# #endregion ScenarioExecution.Executors
|
||||
|
||||
@@ -15,6 +15,7 @@ from src.models.scenario_run import ScenarioRun, ScenarioStepRun
|
||||
from src.models.scenario_worker import ScenarioStepLease
|
||||
|
||||
from .artifacts import invalidate_step_evidence
|
||||
from .decision_policy import map_human_disposition
|
||||
from .runner_plan import validate_pinned_runner_plan
|
||||
|
||||
_DEFAULT_CANCEL_DRAIN_SECONDS = 30
|
||||
@@ -177,13 +178,11 @@ def decide_checkpoint(db: Session, checkpoint_id: str, *, disposition: str, expe
|
||||
raise ValueError("checkpoint not found")
|
||||
run = db.query(ScenarioRun).filter(ScenarioRun.id == checkpoint.run_id).first()
|
||||
if run is not None:
|
||||
outcome_status = {
|
||||
"confirm": "passed",
|
||||
"pass": "passed",
|
||||
"false_positive": "inconclusive",
|
||||
"inconclusive": "inconclusive",
|
||||
"fail": "failed",
|
||||
}[disposition]
|
||||
if disposition == "fail":
|
||||
outcome_status = "failed"
|
||||
reason_codes = ["HUMAN_FAIL"]
|
||||
else:
|
||||
outcome_status, reason_codes = map_human_disposition(disposition)
|
||||
step = db.query(ScenarioStepRun).filter(
|
||||
ScenarioStepRun.run_id == run.id,
|
||||
ScenarioStepRun.logical_step_id == checkpoint.logical_step_id,
|
||||
@@ -198,6 +197,10 @@ def decide_checkpoint(db: Session, checkpoint_id: str, *, disposition: str, expe
|
||||
"status": outcome_status,
|
||||
"disposition": disposition,
|
||||
"comment": comment,
|
||||
"reason_codes": reason_codes,
|
||||
"decision_policy_id": "baseline-semantic",
|
||||
"decision_policy_version": "1.0.0",
|
||||
"decided_at": decided_at.isoformat(),
|
||||
}
|
||||
step.finished_at = checkpoint.decided_at
|
||||
run.status = "queued"
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
# #region ScenarioExecution.LiveCompositionRoot [C:5] [TYPE Module] [SEMANTICS scenario,execution,composition,live,browser,superset,screenshot]
|
||||
# @BRIEF Own the application-only registration of authorized live providers for persisted 044 bindings.
|
||||
# @RELATION CALLS -> [ScenarioExecution.LiveBinding.SupersetAdapter]
|
||||
# @RELATION CALLS -> [ScenarioExecution.EvaluationAdapter.Factory]
|
||||
# @RELATION DEPENDS_ON -> [BaselineEngine.QueryExecutor.ExecuteQueryEnvelope]
|
||||
# @RELATION DEPENDS_ON -> [Plugin.Service.ScreenshotService]
|
||||
# @INVARIANT No runtime client, secret, cookie, callable principal, or browser context is persisted;
|
||||
@@ -134,6 +135,17 @@ class LiveExecutionCompositionRoot:
|
||||
def screenshot_adapter(self):
|
||||
return self._adapter_for("SCREENSHOT", self._screenshot, require_browser_checkpoint=False)
|
||||
|
||||
# #region ScenarioExecution.LiveCompositionRoot.EvaluationAdapter [C:2] [TYPE Function] [SEMANTICS scenario,execution,composition,evaluation,adapter]
|
||||
# @BRIEF Expose the production AgentEvaluation adapter so the default registry is not always None.
|
||||
# @RELATION CALLS -> [ScenarioExecution.EvaluationAdapter.Factory]
|
||||
# @POST Missing LLM provider stays typed unavailable through submit_evaluation; never synthesizes PASS.
|
||||
def agent_evaluation_adapter(self):
|
||||
from .evaluation_adapter import evaluation_adapter_from
|
||||
|
||||
run_async = next((resolved.run_async for resolved in self._superset.values()), None)
|
||||
return evaluation_adapter_from(run_async=run_async)
|
||||
# #endregion ScenarioExecution.LiveCompositionRoot.EvaluationAdapter
|
||||
|
||||
def _adapter_for(
|
||||
self,
|
||||
tool_code: str,
|
||||
|
||||
@@ -2,11 +2,14 @@
|
||||
# @defgroup ScenarioExecution Start durable runs pinned to immutable revisions.
|
||||
# @BRIEF Create queued/pending-approval ScenarioRun records with deterministic idempotency.
|
||||
# @RELATION CALLS -> [ScenarioExecution.RunnerPlan.Derive]
|
||||
# @RELATION CALLS -> [ScenarioExecution.BaselineResolver.Resolve]
|
||||
# @RELATION CALLS -> [ScenarioExecution.Approval.CreateGate]
|
||||
# @RELATION CALLS -> [ScenarioExecution.LiveBinding.SupersetAdapter]
|
||||
# @RELATION CALLS -> [ScenarioAnalytics.Investigation.TerminalSignal]
|
||||
# @RELATION DEPENDS_ON -> [Models.ScenarioExecution.Run]
|
||||
# @INVARIANT Same idempotency key and request hash returns the same run; changed request rejects.
|
||||
# @INVARIANT Canonical request identity includes the resolved BaselineSelectionPin; a different
|
||||
# selector/version/pin cannot replay under the same idempotency key.
|
||||
# @INVARIANT PROD starts are pinned to a durable ActionApprovalGate; pending_approval -> queued
|
||||
# only through gate approval (036), never by silent dispatch.
|
||||
# @INVARIANT scenario_content_hash mirrors the pinned revision's content hash; request_hash
|
||||
@@ -35,6 +38,13 @@ from src.services.dashboard_testing.automation.notify import persist_notificatio
|
||||
|
||||
from .approval import create_prod_gate
|
||||
from .artifacts import invalidate_step_evidence, register_step_evidence
|
||||
from .baseline_resolver import attach_baseline_pin, load_published_catalog, resolve_baseline_pin
|
||||
from .decision_policy import (
|
||||
decide_step_outcome,
|
||||
policy_digest,
|
||||
policy_inputs_from_outcome,
|
||||
verified_evidence_refs,
|
||||
)
|
||||
from .dispatch import dispatch_step
|
||||
from .environment_policy import resolve_environment_execution_policy
|
||||
from .executor_registry import ScenarioExecutorRegistry
|
||||
@@ -42,8 +52,22 @@ from .executors import _register_default_executors
|
||||
from .lifecycle import suspend_for_human
|
||||
from .live_binding import LiveExecutionBinding, LiveExecutionBindingResolver, superset_adapter_from
|
||||
from .result import build_result
|
||||
from .runner_plan import derive_runner_plan, validate_pinned_runner_plan
|
||||
from .runner_plan import derive_runner_plan, resolve_pinned_policy, validate_pinned_runner_plan
|
||||
from .worker import claim_step
|
||||
from .agent_evaluation import AgentEvaluation, persist_agent_evaluation, validate_evaluation_evidence
|
||||
|
||||
|
||||
# #region ScenarioExecution.Runner.EvaluationRecord [C:2] [TYPE Function] [SEMANTICS scenario,execution,evaluation,outcome]
|
||||
# @BRIEF Unwrap evaluation_record from the executor envelope or a flattened step_outcome payload.
|
||||
def _evaluation_record_from_outcome(outcome: dict[str, Any] | None) -> dict[str, Any] | None:
|
||||
if not isinstance(outcome, dict):
|
||||
return None
|
||||
nested = outcome.get("step_outcome") if isinstance(outcome.get("step_outcome"), dict) else None
|
||||
if isinstance(nested, dict) and isinstance(nested.get("evaluation_record"), dict):
|
||||
return nested["evaluation_record"]
|
||||
record = outcome.get("evaluation_record")
|
||||
return record if isinstance(record, dict) else None
|
||||
# #endregion ScenarioExecution.Runner.EvaluationRecord
|
||||
|
||||
# #region ScenarioExecution.Runner.TriggerSource [C:3] [TYPE Block] [SEMANTICS scenario,execution,start,trigger,automation,human]
|
||||
# @BRIEF Enumerate server-owned run origins accepted by the execution boundary.
|
||||
@@ -96,7 +120,11 @@ def _request_hash(
|
||||
params: dict[str, Any],
|
||||
environment_id: str,
|
||||
environment_class: str,
|
||||
policy: str | None = None,
|
||||
live_binding_snapshot: dict[str, Any] | None = None,
|
||||
baseline_pin: dict[str, Any] | None = None,
|
||||
baseline_set: str | None = None,
|
||||
baseline_set_version: str | None = None,
|
||||
) -> str:
|
||||
payload = {
|
||||
"scenario_id": scenario_id,
|
||||
@@ -104,7 +132,12 @@ def _request_hash(
|
||||
"params": params,
|
||||
"environment_id": environment_id,
|
||||
"environment_class": environment_class,
|
||||
"baseline_pin": baseline_pin,
|
||||
"baseline_set": baseline_set,
|
||||
"baseline_set_version": baseline_set_version,
|
||||
}
|
||||
if policy is not None:
|
||||
payload["policy"] = policy
|
||||
if live_binding_snapshot is not None:
|
||||
payload["live_binding_snapshot"] = live_binding_snapshot
|
||||
return hashlib.sha256(
|
||||
@@ -129,6 +162,8 @@ def _request_hash(
|
||||
# ScenarioRun creation, PROD gate creation, dispatch, notification, or queue projection.
|
||||
# @INVARIANT Environment class is resolved only from the server-owned policy before idempotency and
|
||||
# gate creation. is_prod/approval_granted compatibility arguments are never authority.
|
||||
# @INVARIANT BaselineSelectionPin is resolved from published catalog bytes before the ScenarioRun
|
||||
# row, PROD gate, or request-hash lookup; a missing catalog cannot leave an orphan queued run.
|
||||
# @REJECTED Request-body is_prod was rejected as an execution authority because it could bypass or
|
||||
# fabricate PROD gating; only the configured Environment policy may select a gate.
|
||||
def start_run(
|
||||
@@ -146,9 +181,11 @@ def start_run(
|
||||
auto_advance: bool = False,
|
||||
dashboard_release_id: str | None = None,
|
||||
baseline_set: str | None = None,
|
||||
baseline_set_version: str | None = None,
|
||||
execution_toggles: dict[str, bool] | None = None,
|
||||
live_execution_binding: LiveExecutionBinding | dict[str, Any] | None = None,
|
||||
trigger_source: str = TRIGGER_SOURCE_MANUAL,
|
||||
published_catalog: dict[str, Any] | None = None,
|
||||
) -> ScenarioRun:
|
||||
del auto_advance, is_prod, approval_granted
|
||||
trigger_source = _require_trusted_trigger_source(trigger_source)
|
||||
@@ -164,6 +201,13 @@ def start_run(
|
||||
raise ValueError("scenario not found")
|
||||
plan = derive_runner_plan(db, scenario_id, revision_id)
|
||||
_reject_automated_human_plan(plan, trigger_source)
|
||||
baseline_pin = resolve_baseline_pin(
|
||||
graph_or_plan=plan,
|
||||
baseline_set=baseline_set,
|
||||
baseline_set_version=baseline_set_version,
|
||||
published_catalog=load_published_catalog(published_catalog),
|
||||
)
|
||||
plan = attach_baseline_pin(plan, baseline_pin)
|
||||
if environment_policy.is_prod:
|
||||
# T029h (option C): an explicitly evaluated but unverified dashboard-context binding
|
||||
# never reaches a PROD run. A missing marker is legacy/unregistered (REST packs and
|
||||
@@ -211,7 +255,11 @@ def start_run(
|
||||
binding_snapshot = binding.snapshot() if binding is not None else None
|
||||
request_hash = _request_hash(
|
||||
scenario_id, revision_id, params, environment_id, environment_policy.environment_class,
|
||||
policy_digest(resolve_pinned_policy(plan)),
|
||||
binding_snapshot,
|
||||
baseline_pin,
|
||||
baseline_set,
|
||||
baseline_set_version,
|
||||
)
|
||||
logger.reason(
|
||||
"Start scenario run request", src="ScenarioExecution.Runner.start_run",
|
||||
@@ -242,6 +290,8 @@ def start_run(
|
||||
"environment_class": environment_policy.environment_class,
|
||||
"dashboard_release_id": binding.dashboard_release_id if binding is not None else dashboard_release_id,
|
||||
"baseline_set": baseline_set,
|
||||
"baseline_set_version": baseline_set_version,
|
||||
"baseline_pin": baseline_pin,
|
||||
"execution_toggles": execution_toggles or {},
|
||||
},
|
||||
trigger_source=trigger_source, idempotency_key=idempotency_key, runner_plan=plan,
|
||||
@@ -310,6 +360,11 @@ def _build_default_registry(
|
||||
if composition_root is not None and composition_root.has_screenshot_provider
|
||||
else None
|
||||
),
|
||||
agent_evaluation_adapter=(
|
||||
composition_root.agent_evaluation_adapter()
|
||||
if composition_root is not None and hasattr(composition_root, "agent_evaluation_adapter")
|
||||
else None
|
||||
),
|
||||
)
|
||||
return registry
|
||||
# #endregion ScenarioExecution.Runner.DefaultRegistry
|
||||
@@ -1065,6 +1120,7 @@ def _advance_run(db: Session, run: ScenarioRun, registry: ScenarioExecutorRegist
|
||||
step.finished_at = datetime.now(UTC)
|
||||
if outcome.get("output_refs"):
|
||||
step.outputs = {"refs": outcome["output_refs"]}
|
||||
integrity = None
|
||||
if outcome.get("artifact_refs"):
|
||||
step.artifact_refs = list(outcome["artifact_refs"])
|
||||
integrity = register_step_evidence(
|
||||
@@ -1085,6 +1141,45 @@ def _advance_run(db: Session, run: ScenarioRun, registry: ScenarioExecutorRegist
|
||||
step.step_outcome = outcome
|
||||
step.status = "inconclusive"
|
||||
step.error_code = integrity["reason_code"]
|
||||
evaluation_record = _evaluation_record_from_outcome(step.step_outcome)
|
||||
evaluation_publish_failed = False
|
||||
if step_meta.get("tool") == "agent_evaluation" and isinstance(evaluation_record, dict):
|
||||
try:
|
||||
record = AgentEvaluation.model_validate(evaluation_record)
|
||||
validate_evaluation_evidence(
|
||||
db, run_id=run.id, logical_step_id=step_id, attempt=step.attempt,
|
||||
input_manifest=record.model_dump().get("input_manifest", []),
|
||||
raw_response_artifact_ref=record.raw_response_artifact_ref,
|
||||
raw_response_sha256=record.raw_response_sha256,
|
||||
findings=record.model_dump().get("findings", []),
|
||||
succeeded=record.status == "succeeded",
|
||||
)
|
||||
row = persist_agent_evaluation(db, record)
|
||||
step.step_outcome = {**step.step_outcome, "agent_evaluation_ids": [row.evaluation_id]}
|
||||
except ValueError as exc:
|
||||
evaluation_publish_failed = True
|
||||
step.status = "inconclusive"
|
||||
step.error_code = str(exc)
|
||||
step.step_outcome = {**step.step_outcome, "agent_evaluation_ids": []}
|
||||
decision_inputs = None if evaluation_publish_failed else policy_inputs_from_outcome(outcome, step_meta, integrity)
|
||||
if decision_inputs is not None:
|
||||
policy = resolve_pinned_policy(run.runner_plan or {})
|
||||
decision = decide_step_outcome(decision_inputs, policy)
|
||||
if decision.status is not None:
|
||||
step.step_outcome = {
|
||||
**step.step_outcome,
|
||||
"status": decision.status,
|
||||
"decision_policy_id": policy.policy_id,
|
||||
"decision_policy_version": policy.version,
|
||||
"reason_codes": decision.reason_codes,
|
||||
"comparison_ids": [comparison.comparison_id for comparison in decision_inputs.comparisons],
|
||||
"agent_evaluation_ids": step.step_outcome.get("agent_evaluation_ids", []),
|
||||
"deterministic_evidence_refs": verified_evidence_refs(outcome, integrity),
|
||||
"decided_at": datetime.now(UTC).isoformat(),
|
||||
}
|
||||
step.status = decision.status
|
||||
if decision.status in {"failed", "inconclusive", "blocked"} and not step.error_code:
|
||||
step.error_code = decision.reason_codes[0]
|
||||
db.flush()
|
||||
db.refresh(run)
|
||||
if run.status == "cancel_requested":
|
||||
@@ -1093,7 +1188,7 @@ def _advance_run(db: Session, run: ScenarioRun, registry: ScenarioExecutorRegist
|
||||
cancel_run(db, run.id, drain_in_flight=False)
|
||||
return build_result(run, db.query(ScenarioStepRun).filter(ScenarioStepRun.run_id == run.id).all())
|
||||
if step.status in {"passed", "failed", "inconclusive", "blocked"}:
|
||||
completed[step_id] = outcome
|
||||
completed[step_id] = step.step_outcome
|
||||
if run.status in {"pending_approval", "queued", "running", "waiting_human"}:
|
||||
# A step-level inconclusive result is aggregated after all reachable steps; it must not
|
||||
# make the run terminal before a later human checkpoint can safely suspend it.
|
||||
|
||||
@@ -23,10 +23,15 @@ import hashlib
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
from pydantic import ValidationError
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from src.core.logger import logger
|
||||
from src.models.scenario_registry import ScenarioRegistryEntry, ScenarioRevision
|
||||
from src.services.dashboard_testing.execution.decision_policy import (
|
||||
BASELINE_SEMANTIC_V1,
|
||||
DecisionPolicy,
|
||||
)
|
||||
from src.services.dashboard_testing.scenario.templates import validate_action_step
|
||||
|
||||
|
||||
@@ -54,6 +59,26 @@ def _topological_order(steps: list[dict[str, Any]], edges: list[dict[str, Any]])
|
||||
return result
|
||||
|
||||
|
||||
# #region ScenarioExecution.RunnerPlan.EvaluationMode [C:3] [TYPE Function] [SEMANTICS scenario,execution,runnerplan,evaluation,mode]
|
||||
# @ingroup ScenarioExecution
|
||||
# @BRIEF Derive the pinned evaluation_mode from the immutable graph's declared AgentEvaluation needs.
|
||||
# @INVARIANT evaluation_mode is "required" when at least one pinned step declares tool=agent_evaluation
|
||||
# or an agent_evaluation_spec; otherwise "disabled". "advisory" is never derived — it is
|
||||
# an explicit operator pin only.
|
||||
# @RATIONALE The registry has agent_evaluation/evaluate_declared_spec. Plans that declare that tool
|
||||
# pin required so DecisionPolicy rows 9–14 can fire; graphs without evaluation stay
|
||||
# disabled so v1 assertion walks do not manufacture EVALUATION_UNAVAILABLE.
|
||||
def _derive_evaluation_mode(steps: list[dict[str, Any]]) -> str:
|
||||
if any(
|
||||
isinstance(step, dict)
|
||||
and (step.get("agent_evaluation_spec") is not None or step.get("tool") == "agent_evaluation")
|
||||
for step in steps
|
||||
):
|
||||
return "required"
|
||||
return "disabled"
|
||||
# #endregion ScenarioExecution.RunnerPlan.EvaluationMode
|
||||
|
||||
|
||||
# #region ScenarioExecution.RunnerPlan.Derive [C:4] [TYPE Function] [SEMANTICS scenario,execution,runnerplan,derive]
|
||||
# @ingroup ScenarioExecution
|
||||
# @BRIEF Derive env targets, topological order and executor mapping from a revision snapshot.
|
||||
@@ -122,12 +147,16 @@ def derive_runner_plan(db: Session, scenario_id: str, revision_id: str) -> dict[
|
||||
for step_id in order
|
||||
}
|
||||
human_checkpoints = [step_id for step_id in order if by_id[step_id].get("tool") == "human"]
|
||||
pinned_policy = BASELINE_SEMANTIC_V1.model_copy(
|
||||
update={"evaluation_mode": _derive_evaluation_mode(pinned_steps)}
|
||||
)
|
||||
plan = {
|
||||
"scenario_revision_id": revision.revision_id,
|
||||
"scenario_content_hash": revision.content_hash,
|
||||
"verification_program_hash": revision.content_hash,
|
||||
"action_registry_version": registry_version,
|
||||
"action_registry_hash": registry_hash,
|
||||
"decision_policy": pinned_policy.model_dump(),
|
||||
"env_targets": graph.get("environment_ids", []),
|
||||
"resolved_params": graph.get("parameters", {}),
|
||||
"pinned_baselines": graph.get("baselines", {}),
|
||||
@@ -143,6 +172,26 @@ def derive_runner_plan(db: Session, scenario_id: str, revision_id: str) -> dict[
|
||||
# #endregion ScenarioExecution.RunnerPlan.Derive
|
||||
|
||||
|
||||
# #region ScenarioExecution.RunnerPlan.ResolvePolicy [C:3] [TYPE Function] [SEMANTICS scenario,execution,runnerplan,decision,policy]
|
||||
# @ingroup ScenarioExecution
|
||||
# @BRIEF Resolve the pinned DecisionPolicy for a plan, defaulting persisted legacy plans to v1.
|
||||
# @INVARIANT A persisted plan without decision_policy (pre-slice-C derivation) is a legal legacy plan and
|
||||
# resolves to BASELINE_SEMANTIC_V1; a present but malformed/unknown policy is a hard error.
|
||||
# @REJECTED Rejecting legacy queued plans at claim time was rejected — policies are server-owned pins and
|
||||
# legacy runs must be able to finish with the single v1 default rather than a synthetic 409.
|
||||
def resolve_pinned_policy(plan: dict[str, Any]) -> DecisionPolicy:
|
||||
pinned = plan.get("decision_policy")
|
||||
if pinned is None:
|
||||
return BASELINE_SEMANTIC_V1
|
||||
if not isinstance(pinned, dict):
|
||||
raise ValueError("DECISION_POLICY_INVALID")
|
||||
try:
|
||||
return DecisionPolicy.model_validate(pinned)
|
||||
except ValidationError as exc:
|
||||
raise ValueError("DECISION_POLICY_INVALID") from exc
|
||||
# #endregion ScenarioExecution.RunnerPlan.ResolvePolicy
|
||||
|
||||
|
||||
# #region ScenarioExecution.RunnerPlan.ValidatePinned [C:4] [TYPE Function] [SEMANTICS scenario,execution,runnerplan,action,preflight]
|
||||
# @BRIEF Verify a persisted plan still contains exact immutable 038 action descriptors before a worker claim.
|
||||
# @INVARIANT A legacy/malformed queued plan is blocked before a step lease or adapter call; descriptor
|
||||
@@ -150,6 +199,14 @@ def derive_runner_plan(db: Session, scenario_id: str, revision_id: str) -> dict[
|
||||
# @REJECTED Reconstructing a missing descriptor from tool metadata was rejected because mutable or
|
||||
# unknown effects would otherwise inherit a synthetic retry-safe contract.
|
||||
def validate_pinned_runner_plan(plan: dict[str, Any]) -> None:
|
||||
pinned_policy = plan.get("decision_policy")
|
||||
if pinned_policy is not None:
|
||||
if not isinstance(pinned_policy, dict):
|
||||
raise ValueError("DECISION_POLICY_INVALID")
|
||||
try:
|
||||
DecisionPolicy.model_validate(pinned_policy)
|
||||
except ValidationError as exc:
|
||||
raise ValueError("DECISION_POLICY_INVALID") from exc
|
||||
registry_version = plan.get("action_registry_version")
|
||||
registry_hash = plan.get("action_registry_hash")
|
||||
for step in plan.get("steps") or []:
|
||||
|
||||
@@ -0,0 +1,141 @@
|
||||
# #region ScenarioGraph.Compiler.ChainEmission [C:4] [TYPE Module] [SEMANTICS scenario,compiler,chain,browser,canonical]
|
||||
# @defgroup ScenarioGraph Visual/metric depends_on chain emission for compiled cases.
|
||||
# @LAYER Service
|
||||
# @RELATION CALLED_BY -> [ScenarioGraph.Compiler.CompileGraph]
|
||||
# @RELATION DEPENDS_ON -> [ScenarioGraph.Templates]
|
||||
# @RELATION DEPENDS_ON -> [ScenarioGraph.Compiler.BuildStep]
|
||||
# @INVARIANT Newly compiled steps never use legacy alias names (apply_filters, text_filter,
|
||||
# table_filter, row_edit, download_xlsx).
|
||||
# @INVARIANT Artifact registration is provider output; this module never emits artifact/register.
|
||||
# @INVARIANT DecisionPolicy is not a compiled step.
|
||||
# @RATIONALE Visual cases need a capture→compare chain so baseline comparison consumes screenshot
|
||||
# evidence rather than a dishonest compare-without-capture edge.
|
||||
# @REJECTED Emitting compare_to_baseline without capture when screenshot capability is absent —
|
||||
# that would claim visual evidence the graph cannot produce.
|
||||
# @REJECTED Compiling artifact/register as a DAG step — artifact bytes are provider output (D10).
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Callable
|
||||
|
||||
from src.services.dashboard_testing.scenario.capability_mapper import CapabilityMapping
|
||||
from src.services.dashboard_testing.scenario.models import (
|
||||
AgentEvaluationSpec,
|
||||
Finding,
|
||||
ScenarioParameter,
|
||||
ScenarioStep,
|
||||
)
|
||||
from src.services.dashboard_testing.scenario.templates import STEP_TEMPLATES
|
||||
|
||||
# Compile-time migrations only. These names are never written onto new ScenarioStep.action values.
|
||||
ACTION_ALIASES: dict[str, str] = {
|
||||
"apply_filters": "apply_native_filter",
|
||||
"text_filter": "apply_native_filter",
|
||||
"table_filter": "apply_table_filter",
|
||||
"row_edit": "edit_row",
|
||||
"download_xlsx": "download",
|
||||
}
|
||||
|
||||
_METRIC_TOOLS = frozenset({"superset_api", "xlsx", "assertion"})
|
||||
|
||||
BuildStep = Callable[..., ScenarioStep]
|
||||
HumanCheckpoint = Callable[[str, dict[str, int]], ScenarioStep]
|
||||
|
||||
|
||||
# #region ScenarioGraph.Compiler.CanonicalizeAction [C:1] [TYPE Function] [SEMANTICS scenario,compiler,alias]
|
||||
# @BRIEF Map a legacy browser alias to its canonical browser-actions.md name.
|
||||
def canonicalize_action(action: str) -> str:
|
||||
return ACTION_ALIASES.get(action, action)
|
||||
# #endregion ScenarioGraph.Compiler.CanonicalizeAction
|
||||
|
||||
|
||||
# #region ScenarioGraph.Compiler.EmitSelectedCase [C:4] [TYPE Function] [SEMANTICS scenario,compiler,chain,dag]
|
||||
# @BRIEF Emit the compiled step chain (or blocker) for one selected catalog case.
|
||||
# @PRE mapping is the capability classification for case_id; selected_case_ids are already sorted.
|
||||
# @POST Unsupported selected cases return no steps and an UNSUPPORTED_ACTION blocker.
|
||||
# @POST Human-checkpoint cases return the unchanged manual step.
|
||||
# @POST Visual/browser cases chain canonical action → capture (if screenshot) → compare (if baseline).
|
||||
# @POST Metric/observe cases keep the template action and append compare_to_baseline when baseline is present.
|
||||
def emit_selected_case(
|
||||
*,
|
||||
case_id: str,
|
||||
mapping: CapabilityMapping,
|
||||
capabilities: dict[str, bool],
|
||||
evaluation_spec: AgentEvaluationSpec | None,
|
||||
ordinal_counter: dict[str, int],
|
||||
parameters: list[ScenarioParameter],
|
||||
produced_refs: set[str],
|
||||
build_step: BuildStep,
|
||||
human_checkpoint: HumanCheckpoint,
|
||||
) -> tuple[list[ScenarioStep], list[Finding]]:
|
||||
if mapping.classification == "unsupported":
|
||||
return [], [Finding(
|
||||
code="UNSUPPORTED_ACTION",
|
||||
severity="blocker",
|
||||
message=f"selected case {case_id} is unsupported: {mapping.rationale}",
|
||||
case_id=case_id,
|
||||
recovery_options=["remove the case from the selection", "enable the required capability"],
|
||||
)]
|
||||
if mapping.classification == "human_checkpoint":
|
||||
return [human_checkpoint(case_id, ordinal_counter)], []
|
||||
|
||||
template = mapping.selected_template or "browser_apply_observe_assert"
|
||||
action, tool, _ = STEP_TEMPLATES[template]
|
||||
action = canonicalize_action(action)
|
||||
|
||||
def _step(
|
||||
step_action: str,
|
||||
step_tool: str,
|
||||
depends_on: list[str],
|
||||
spec: AgentEvaluationSpec | None = None,
|
||||
) -> ScenarioStep:
|
||||
return build_step(
|
||||
case_id,
|
||||
step_action,
|
||||
step_tool,
|
||||
mapping,
|
||||
ordinal_counter,
|
||||
parameters,
|
||||
produced_refs,
|
||||
depends_on=depends_on,
|
||||
evaluation_spec=spec,
|
||||
)
|
||||
|
||||
if tool == "human":
|
||||
return [_step(action, tool, [])], []
|
||||
|
||||
primary = _step(action, tool, [])
|
||||
steps = [primary]
|
||||
blockers: list[Finding] = []
|
||||
last_id = primary.id
|
||||
screenshot = bool(capabilities.get("screenshot"))
|
||||
baseline = bool(capabilities.get("baseline"))
|
||||
|
||||
if tool == "browser":
|
||||
if screenshot:
|
||||
capture = _step("capture_screenshot", "screenshot", [last_id])
|
||||
steps.append(capture)
|
||||
last_id = capture.id
|
||||
if baseline:
|
||||
compare = _step("compare_to_baseline", "assertion", [last_id])
|
||||
steps.append(compare)
|
||||
last_id = compare.id
|
||||
elif baseline:
|
||||
blockers.append(Finding(
|
||||
code="MISSING_SCREENSHOT",
|
||||
severity="blocker",
|
||||
message=f"case {case_id} cannot compare to baseline without screenshot capture",
|
||||
case_id=case_id,
|
||||
recovery_options=["enable screenshot capability", "remove baseline comparison"],
|
||||
))
|
||||
elif tool in _METRIC_TOOLS and baseline:
|
||||
compare = _step("compare_to_baseline", "assertion", [last_id])
|
||||
steps.append(compare)
|
||||
last_id = compare.id
|
||||
|
||||
if evaluation_spec is not None:
|
||||
steps.append(_step("evaluate_declared_spec", "agent_evaluation", [last_id], evaluation_spec))
|
||||
|
||||
return steps, blockers
|
||||
# #endregion ScenarioGraph.Compiler.EmitSelectedCase
|
||||
# #endregion ScenarioGraph.Compiler.ChainEmission
|
||||
@@ -3,10 +3,13 @@
|
||||
# @BRIEF Compile canonical inputs and mappings into a stable dashboard-specific DAG.
|
||||
# @PRE Intent, query model, catalog, baseline summary, and parameters have valid fingerprints.
|
||||
# @POST Same canonical inputs/compiler version yield byte-identical graph and stable ids/order.
|
||||
# @POST dashboard_context embeds the server-issued query model under the `query` key (038 schema),
|
||||
# so registration's context authority can revalidate it without a client echo (DEF-03 fix).
|
||||
# @SIDE_EFFECT None.
|
||||
# @SIDE_EFFECT Logging (REASON before compile; REFLECT with step count and hash after).
|
||||
# @DATA_CONTRACT CompileScenarioRequest -> DashboardTestScenario
|
||||
# @INVARIANT Steps consume only context/parameter/baseline/earlier-step refs.
|
||||
# @RELATION DEPENDS_ON -> [ScenarioGraph.Compiler.ChainEmission]
|
||||
# @RATIONALE Rule/template compilation makes the agent a planner/explainer, not an executable-code generator.
|
||||
# @REJECTED LLM-generated ids/dependencies/code — non-deterministic and unsafe.
|
||||
|
||||
@@ -21,7 +24,9 @@ from src.core.logger import logger
|
||||
from src.core.logger import belief_scope
|
||||
from src.services.dashboard_testing.scenario.capability_mapper import CapabilityMapping, map_all
|
||||
from src.services.dashboard_testing.scenario.checklist_catalog import catalog_fingerprint
|
||||
from src.services.dashboard_testing.scenario.chain_emission import emit_selected_case
|
||||
from src.services.dashboard_testing.scenario.models import (
|
||||
AgentEvaluationSpec,
|
||||
DashboardTestScenario,
|
||||
Expected,
|
||||
Finding,
|
||||
@@ -31,7 +36,7 @@ from src.services.dashboard_testing.scenario.models import (
|
||||
canonical_dump,
|
||||
sha256_hex,
|
||||
)
|
||||
from src.services.dashboard_testing.scenario.templates import REGISTERED_ACTIONS, STEP_TEMPLATES, assert_registered
|
||||
from src.services.dashboard_testing.scenario.templates import REGISTERED_ACTIONS, assert_registered
|
||||
|
||||
COMPILER_VERSION = "038.1.0"
|
||||
TEMPLATE_VERSION = "v1"
|
||||
@@ -54,6 +59,7 @@ class CompileScenarioRequest:
|
||||
environment_id: str = "env-default"
|
||||
dashboard_id: int = 0
|
||||
dashboard_name: str = "unknown"
|
||||
agent_evaluation_spec: AgentEvaluationSpec | None = None
|
||||
# #endregion ScenarioGraph.Compiler.CompileRequest
|
||||
|
||||
|
||||
@@ -175,31 +181,25 @@ def _compile_graph(req: CompileScenarioRequest) -> CompiledResult:
|
||||
warnings: list[Finding] = []
|
||||
blockers: list[Finding] = []
|
||||
produced_refs: set[str] = set()
|
||||
case_steps: dict[str, list[str]] = {}
|
||||
ordinal_counter: dict[str, int] = {}
|
||||
|
||||
# Selected cases drive the graph; deterministic order by case id.
|
||||
selected = objective["selected_case_ids"]
|
||||
for case_id in selected:
|
||||
mapping: CapabilityMapping = mappings_by_case[case_id]
|
||||
if mapping.classification == "unsupported":
|
||||
warnings.append(Finding(
|
||||
code="UNSUPPORTED_CASE",
|
||||
severity="warning",
|
||||
message=f"case {case_id} unsupported: {mapping.rationale}",
|
||||
case_id=case_id,
|
||||
))
|
||||
continue
|
||||
if mapping.classification == "human_checkpoint":
|
||||
steps.append(_human_checkpoint_step(case_id, ordinal_counter))
|
||||
case_steps.setdefault(case_id, []).append(_human_checkpoint_step(case_id, ordinal_counter).id)
|
||||
continue
|
||||
template = mapping.selected_template
|
||||
if template is None:
|
||||
template = "browser_apply_observe_assert"
|
||||
action, tool, _ = STEP_TEMPLATES[template]
|
||||
steps.append(_build_step(case_id, action, tool, mapping, ordinal_counter, parameters, produced_refs))
|
||||
case_steps.setdefault(case_id, []).append(_last_step_id(case_id, action, ordinal_counter))
|
||||
emitted, case_blockers = emit_selected_case(
|
||||
case_id=case_id,
|
||||
mapping=mapping,
|
||||
capabilities=req.capabilities,
|
||||
evaluation_spec=req.agent_evaluation_spec,
|
||||
ordinal_counter=ordinal_counter,
|
||||
parameters=parameters,
|
||||
produced_refs=produced_refs,
|
||||
build_step=_build_step,
|
||||
human_checkpoint=_human_checkpoint_step,
|
||||
)
|
||||
steps.extend(emitted)
|
||||
blockers.extend(case_blockers)
|
||||
|
||||
# Missing selector/context detection on browser-interaction steps without hints
|
||||
for step in steps:
|
||||
@@ -213,6 +213,7 @@ def _compile_graph(req: CompileScenarioRequest) -> CompiledResult:
|
||||
"environment_id": req.environment_id,
|
||||
"dashboard_id": req.dashboard_id,
|
||||
"dashboard_name": req.dashboard_name,
|
||||
"query": req.query_model,
|
||||
},
|
||||
objective=objective,
|
||||
input_fingerprints={
|
||||
@@ -266,6 +267,9 @@ def _build_step(
|
||||
ordinal_counter: dict[str, int],
|
||||
parameters: list[ScenarioParameter],
|
||||
produced_refs: set[str],
|
||||
*,
|
||||
depends_on: list[str] | None = None,
|
||||
evaluation_spec: AgentEvaluationSpec | None = None,
|
||||
) -> ScenarioStep:
|
||||
assert_registered(action, tool)
|
||||
entry = REGISTERED_ACTIONS[action]
|
||||
@@ -289,7 +293,7 @@ def _build_step(
|
||||
description="compare to approved baseline")
|
||||
|
||||
automation = "ready"
|
||||
if action == "download_xlsx" and not mapping.matched_capabilities:
|
||||
if action in {"download", "download_xlsx"} and not mapping.matched_capabilities:
|
||||
automation = "needs_context"
|
||||
if entry["risk"] == "browser_interaction" and not _has_selector(parameters):
|
||||
automation = "needs_selector"
|
||||
@@ -303,10 +307,11 @@ def _build_step(
|
||||
inputs=inputs,
|
||||
outputs=outputs,
|
||||
expected=expected,
|
||||
depends_on=[],
|
||||
depends_on=list(depends_on or []),
|
||||
automation_status=automation,
|
||||
checklist_case_ids=[case_id],
|
||||
risk=entry["risk"],
|
||||
agent_evaluation_spec=evaluation_spec if action == "evaluate_declared_spec" else None,
|
||||
)
|
||||
# #endregion ScenarioGraph.Compiler.BuildStep
|
||||
|
||||
@@ -360,14 +365,4 @@ def _apply_selector_check(step: ScenarioStep, parameters: list[ScenarioParameter
|
||||
recovery_options=["provide selector hint", "convert to human checkpoint", "remove step"],
|
||||
))
|
||||
# #endregion ScenarioGraph.Compiler.ApplySelectorCheck
|
||||
|
||||
|
||||
# #region ScenarioGraph.Compiler.LastStepId [C:1] [TYPE Function] [SEMANTICS scenario,id,step]
|
||||
# @ingroup ScenarioGraph
|
||||
# @BRIEF Return the stable id of the last step emitted for a case/action pair.
|
||||
# @POST Returns deterministic step id for coverage linkage.
|
||||
def _last_step_id(case_id: str, action: str, ordinal_counter: dict[str, int]) -> str:
|
||||
ordinal = ordinal_counter.get(f"{case_id}:{action}", 1)
|
||||
return _stable_step_id(PHASE_ORDER[1], case_id, action, ordinal)
|
||||
# #endregion ScenarioGraph.Compiler.LastStepId
|
||||
# #endregion ScenarioGraph.Compiler.Compile
|
||||
|
||||
@@ -15,10 +15,11 @@ import json
|
||||
import re
|
||||
from typing import Any, Literal
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field, field_validator
|
||||
from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator
|
||||
from src.core.logger import logger
|
||||
|
||||
from src.core.logger import belief_scope
|
||||
from src.services.dashboard_testing.execution.decision_policy import DecisionPolicy
|
||||
|
||||
SCHEMA_VERSION = 1
|
||||
_COMPILER_VERSION = "038.1.0"
|
||||
@@ -134,6 +135,66 @@ class VlmFinding(BaseModel):
|
||||
# #endregion ScenarioGraph.Models.VlmFinding
|
||||
|
||||
|
||||
# #region ScenarioGraph.Models.AgentEvaluationSpec [C:4] [TYPE Class] [SEMANTICS scenario,evaluation,spec,038]
|
||||
# @ingroup ScenarioGraph
|
||||
# @BRIEF Strict 038 declaration of an immutable agent evaluation operation.
|
||||
# @RATIONALE The spec is owned by the scenario graph so registry validation and runner planning use
|
||||
# one schema boundary rather than accepting an untyped evaluation payload.
|
||||
# @REJECTED Allowing tool access or duplicate criterion ids was rejected by the 038 bounded-access
|
||||
# invariant; evaluations may inspect declared evidence only.
|
||||
class EvaluationLimits(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
timeout_ms: int = Field(ge=1, le=60000)
|
||||
max_images: int = Field(ge=1, le=50)
|
||||
max_input_tokens: int = Field(ge=1)
|
||||
max_output_tokens: int = Field(ge=1, le=8192)
|
||||
max_cost: str = Field(pattern=r"^[0-9]+(\.[0-9]+)?$")
|
||||
currency: str = Field(min_length=1)
|
||||
|
||||
|
||||
class EvaluationCriterion(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
criterion_id: str = Field(min_length=1, max_length=128)
|
||||
criterion_kind: Literal["semantic", "deterministic_comparison"]
|
||||
description: str = Field(min_length=1, max_length=2000)
|
||||
comparison_id: str | None = Field(default=None, max_length=128)
|
||||
|
||||
|
||||
class AgentEvaluationSpec(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
schema_version: Literal[1]
|
||||
spec_id: str
|
||||
provider_id: str = Field(min_length=1)
|
||||
provider_version: str = Field(min_length=1)
|
||||
model_id: str = Field(min_length=1)
|
||||
model_version: str = Field(min_length=1)
|
||||
prompt_template_id: str = Field(min_length=1)
|
||||
prompt_template_version: str = Field(min_length=1)
|
||||
prompt_template_hash: str = Field(pattern=_SHA256_RE)
|
||||
evidence_refs: list[str] = Field(min_length=1, max_length=100)
|
||||
comparison_refs: list[str] = Field(min_length=1, max_length=100)
|
||||
tool_allowlist: list[str] = Field(default_factory=list, max_length=0)
|
||||
output_schema: Literal["agent-evaluation.schema.json"]
|
||||
decision_policy: DecisionPolicy
|
||||
limits: EvaluationLimits
|
||||
trust_policy_hash: str = Field(pattern=_SHA256_RE)
|
||||
criteria: list[EvaluationCriterion] = Field(min_length=1, max_length=100)
|
||||
|
||||
@model_validator(mode="after")
|
||||
def validate_criteria(self) -> "AgentEvaluationSpec":
|
||||
ids = [criterion.criterion_id for criterion in self.criteria]
|
||||
if len(ids) != len(set(ids)):
|
||||
raise ValueError("EVALUATION_CRITERION_ID_DUPLICATE")
|
||||
comparisons = set(self.comparison_refs)
|
||||
for criterion in self.criteria:
|
||||
if criterion.criterion_kind == "deterministic_comparison" and criterion.comparison_id not in comparisons:
|
||||
raise ValueError("EVALUATION_COMPARISON_REF_REQUIRED")
|
||||
if criterion.criterion_kind == "semantic" and criterion.comparison_id is not None:
|
||||
raise ValueError("EVALUATION_SEMANTIC_COMPARISON_FORBIDDEN")
|
||||
return self
|
||||
# #endregion ScenarioGraph.Models.AgentEvaluationSpec
|
||||
|
||||
|
||||
# #region ScenarioGraph.Models.ScenarioStep [C:2] [TYPE Class] [SEMANTICS scenario,step,dag]
|
||||
# @ingroup ScenarioGraph
|
||||
# @BRIEF One node in the scenario DAG: tool, action, typed refs, expectation, status, risk.
|
||||
@@ -144,7 +205,7 @@ class ScenarioStep(BaseModel):
|
||||
phase: Literal["setup", "interact", "observe", "assert", "evidence", "report"]
|
||||
title: str = Field(max_length=300)
|
||||
description: str | None = Field(default=None, max_length=2000)
|
||||
tool: Literal["browser", "superset_api", "xlsx", "assertion", "screenshot", "report", "artifact", "human"]
|
||||
tool: Literal["browser", "superset_api", "sql_evidence", "transform", "xlsx", "assertion", "screenshot", "report", "artifact", "human", "agent_evaluation"]
|
||||
action: str = Field(pattern=r"^[a-z][a-z0-9_]{1,63}$")
|
||||
inputs: list[Ref] = Field(default_factory=list)
|
||||
outputs: list[Ref] = Field(default_factory=list)
|
||||
@@ -157,6 +218,8 @@ class ScenarioStep(BaseModel):
|
||||
risk: Literal["read", "browser_interaction", "draft_write", "human"]
|
||||
capture_spec: CaptureSpec | None = None
|
||||
vlm_analysis: VlmAnalysis | None = None
|
||||
agent_evaluation_spec: AgentEvaluationSpec | None = None
|
||||
decision_policy: DecisionPolicy | None = None
|
||||
# #endregion ScenarioGraph.Models.ScenarioStep
|
||||
|
||||
|
||||
|
||||
42
backend/src/services/dashboard_testing/scenario/sql_guard.py
Normal file
42
backend/src/services/dashboard_testing/scenario/sql_guard.py
Normal file
@@ -0,0 +1,42 @@
|
||||
# #region ScenarioGraph.SqlGuard [C:2] [TYPE Module] [SEMANTICS scenario,safety,sql]
|
||||
# @defgroup ScenarioGraph Structure-aware SQL and free-text safety checks.
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
import sqlparse
|
||||
from sqlparse import tokens as T
|
||||
|
||||
_SQL_TYPES = {"SELECT", "INSERT", "UPDATE", "DELETE", "DROP", "ALTER", "CREATE", "TRUNCATE", "GRANT", "EXEC"}
|
||||
|
||||
|
||||
# #region ScenarioGraph.SqlGuard.ContainsSql [C:2] [TYPE Function] [SEMANTICS scenario,safety,sql]
|
||||
# @BRIEF Detect executable SQL statements via sqlparse statement typing instead of substring tokens.
|
||||
# @POST True only when a statement types as DML/DDL/DCL or carries such a leading keyword token; stacked statements are checked recursively.
|
||||
# @REJECTED Bare SQL-comment flagging ("--" anywhere) — prose like "2026-09-08--2026-09-09" or "well--done" produced false positives (same defect class as DEF-01).
|
||||
# A residual limitation is accepted: prose starting with an exact SQL verb ("Select rows to export") is still flagged; authority
|
||||
# for typed references remains the closed 038 schema, this detector is defense on free-text fields only.
|
||||
def contains_sql_statement(text: str | None) -> bool:
|
||||
"""Detect executable SQL statements without matching ordinary prose."""
|
||||
if not isinstance(text, str) or not text.strip():
|
||||
return False
|
||||
statements = [statement for statement in sqlparse.parse(text) if str(statement).strip()]
|
||||
for statement in statements:
|
||||
if statement.get_type().upper() in _SQL_TYPES:
|
||||
return True
|
||||
for token in statement.flatten():
|
||||
if token.ttype in T.Keyword.DML or token.ttype in T.Keyword.DDL or token.ttype in T.Keyword.DCL:
|
||||
if str(token).upper() in _SQL_TYPES:
|
||||
return True
|
||||
return len(statements) > 1 and any(contains_sql_statement(str(item)) for item in statements[1:])
|
||||
# #endregion ScenarioGraph.SqlGuard.ContainsSql
|
||||
|
||||
|
||||
# #region ScenarioGraph.SqlGuard.ContainsUnsafeFreeText [C:2] [TYPE Function] [SEMANTICS scenario,safety,path]
|
||||
# @BRIEF Apply SQL and path-traversal checks only to fields authored as free text.
|
||||
def contains_unsafe_free_text(value: Any) -> bool:
|
||||
"""Apply SQL/path checks only to fields authored as free text."""
|
||||
return isinstance(value, str) and (contains_sql_statement(value) or "../" in value or "\\" in value)
|
||||
# #endregion ScenarioGraph.SqlGuard.ContainsUnsafeFreeText
|
||||
|
||||
# #endregion ScenarioGraph.SqlGuard
|
||||
@@ -14,21 +14,36 @@ import json
|
||||
from typing import Any
|
||||
|
||||
PHASES = ("setup", "interact", "observe", "assert", "evidence", "report")
|
||||
TOOLS = ("browser", "superset_api", "sql_evidence", "transform", "xlsx", "assertion", "screenshot", "report", "artifact", "human")
|
||||
TOOLS = ("browser", "superset_api", "sql_evidence", "transform", "xlsx", "assertion", "screenshot", "report", "artifact", "human", "agent_evaluation")
|
||||
RISKS = ("read", "browser_interaction", "draft_write", "human")
|
||||
ACTION_REGISTRY_VERSION = "038.2.0"
|
||||
ACTION_REGISTRY_VERSION = "038.4.0" # additive: canonical browser-actions.md names
|
||||
|
||||
# Registered tool/action pairs (action -> phase, risk, required capability)
|
||||
# Registered tool/action pairs (action -> phase, risk, required capability).
|
||||
# Canonical browser names are emitted on new graphs. Legacy aliases stay registered
|
||||
# so historical descriptors still resolve; the compiler never emits those names.
|
||||
REGISTERED_ACTIONS: dict[str, dict[str, Any]] = {
|
||||
"open_dashboard": {"tool": "browser", "phase": "setup", "risk": "read", "capability": "browser"},
|
||||
"navigate_tab": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "browser"},
|
||||
"apply_native_filter": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "browser"},
|
||||
"inspect_filter_state": {"tool": "browser", "phase": "observe", "risk": "read", "capability": "browser"},
|
||||
"apply_table_filter": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "table_filter"},
|
||||
"pagination": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "pagination"},
|
||||
"navigate_dashboard": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "cross_dashboard"},
|
||||
"extract_table": {"tool": "browser", "phase": "observe", "risk": "read", "capability": "browser"},
|
||||
"scroll_to": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "browser"},
|
||||
"inspect_columns": {"tool": "browser", "phase": "observe", "risk": "read", "capability": "browser"},
|
||||
"click": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "browser"},
|
||||
"select_rows": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "browser"},
|
||||
"edit_row": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "row_edit"},
|
||||
"bulk_edit": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "bulk_edit"},
|
||||
"download": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "xlsx_export"},
|
||||
"refresh": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "browser"},
|
||||
"wait_for_state": {"tool": "browser", "phase": "observe", "risk": "read", "capability": "browser"},
|
||||
"apply_filters": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "browser"},
|
||||
"text_filter": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "text_filter"},
|
||||
"table_filter": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "table_filter"},
|
||||
"pagination": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "pagination"},
|
||||
"row_edit": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "row_edit"},
|
||||
"bulk_edit": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "bulk_edit"},
|
||||
"download_xlsx": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "xlsx_export"},
|
||||
"navigate_dashboard": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "cross_dashboard"},
|
||||
"execute_metric": {"tool": "superset_api", "phase": "observe", "risk": "read", "capability": "superset_metric"},
|
||||
"capture_sql_evidence": {"tool": "sql_evidence", "phase": "observe", "risk": "read", "capability": "superset_query_envelope"},
|
||||
"transform_result": {"tool": "transform", "phase": "observe", "risk": "read", "capability": "bounded_projection"},
|
||||
@@ -43,6 +58,7 @@ REGISTERED_ACTIONS: dict[str, dict[str, Any]] = {
|
||||
"generate_report": {"tool": "report", "phase": "report", "risk": "draft_write", "capability": None},
|
||||
"register_artifact": {"tool": "artifact", "phase": "report", "risk": "draft_write", "capability": "repository_write"},
|
||||
"human_checkpoint": {"tool": "human", "phase": "assert", "risk": "human", "capability": None},
|
||||
"evaluate_declared_spec": {"tool": "agent_evaluation", "phase": "assert", "risk": "read", "capability": None},
|
||||
}
|
||||
|
||||
|
||||
@@ -73,14 +89,17 @@ class ActionExecutionDescriptor:
|
||||
|
||||
# External mutations only. generate_report is a local deterministic report write (report executor,
|
||||
# no external side effect) and is NOT a mutation. register_artifact writes to a Git repository
|
||||
# (capability=repository_write) and remains mutating; row_edit/bulk_edit are Superset browser
|
||||
# mutations. Reclassification (2026-09-06) bumps ACTION_REGISTRY_VERSION to 038.2.0.
|
||||
_MUTATING_ACTIONS = frozenset({"row_edit", "bulk_edit", "register_artifact"})
|
||||
# (capability=repository_write) and remains mutating; edit_row/row_edit/bulk_edit are Superset
|
||||
# browser mutations. Canonical edit_row is additive in 038.4.0; row_edit remains an alias.
|
||||
_MUTATING_ACTIONS = frozenset({"row_edit", "edit_row", "bulk_edit", "register_artifact"})
|
||||
_BROWSER_ACTIONS = frozenset({
|
||||
"open_dashboard", "apply_filters", "text_filter", "table_filter", "pagination",
|
||||
"row_edit", "bulk_edit", "download_xlsx", "navigate_dashboard",
|
||||
"open_dashboard", "navigate_tab", "apply_native_filter", "inspect_filter_state",
|
||||
"apply_table_filter", "pagination", "navigate_dashboard", "extract_table", "scroll_to",
|
||||
"inspect_columns", "click", "select_rows", "edit_row", "bulk_edit", "download", "refresh",
|
||||
"wait_for_state", "apply_filters", "text_filter", "table_filter", "row_edit", "download_xlsx",
|
||||
})
|
||||
_TIMEOUTS = {
|
||||
"download": 60000,
|
||||
"download_xlsx": 60000,
|
||||
"capture_screenshot": 30000,
|
||||
"execute_metric": 30000,
|
||||
@@ -150,20 +169,20 @@ def validate_action_step(
|
||||
return descriptor
|
||||
# #endregion ScenarioGraph.Templates.ActionRegistry
|
||||
|
||||
# Step templates: mapping name -> (action, tool, phase)
|
||||
# Step templates: mapping name -> (canonical action, tool, phase). Aliases never appear here.
|
||||
STEP_TEMPLATES: dict[str, tuple[str, str, str]] = {
|
||||
"browser_apply_observe_assert": ("open_dashboard", "browser", "setup"),
|
||||
"browser_text_filter_assert": ("text_filter", "browser", "interact"),
|
||||
"browser_table_filter_assert": ("table_filter", "browser", "interact"),
|
||||
"browser_apply_observe_assert": ("apply_native_filter", "browser", "interact"),
|
||||
"browser_text_filter_assert": ("apply_native_filter", "browser", "interact"),
|
||||
"browser_table_filter_assert": ("apply_table_filter", "browser", "interact"),
|
||||
"browser_pagination_assert": ("pagination", "browser", "interact"),
|
||||
"browser_edit_refresh_evidence": ("row_edit", "browser", "interact"),
|
||||
"browser_edit_refresh_evidence": ("edit_row", "browser", "interact"),
|
||||
"browser_bulk_edit_evidence": ("bulk_edit", "browser", "interact"),
|
||||
"browser_bulk_blank_before_after": ("bulk_edit", "browser", "interact"),
|
||||
"browser_bulk_large_evidence": ("bulk_edit", "browser", "interact"),
|
||||
"browser_refresh_assert_screenshot": ("row_edit", "browser", "interact"),
|
||||
"browser_chain_evidence": ("apply_filters", "browser", "interact"),
|
||||
"browser_refresh_assert_screenshot": ("edit_row", "browser", "interact"),
|
||||
"browser_chain_evidence": ("apply_native_filter", "browser", "interact"),
|
||||
"browser_api_distinct_comments": ("execute_metric", "superset_api", "observe"),
|
||||
"browser_download_xlsx_parse": ("download_xlsx", "browser", "interact"),
|
||||
"browser_download_xlsx_parse": ("download", "browser", "interact"),
|
||||
"browser_api_xlsx_assert": ("execute_metric", "superset_api", "observe"),
|
||||
"browser_xlsx_rowset_assert": ("xlsx_rowset_assert", "xlsx", "assert"),
|
||||
"browser_navigation_evidence": ("navigate_dashboard", "browser", "interact"),
|
||||
|
||||
@@ -20,8 +20,7 @@ from src.core.logger import logger
|
||||
from src.core.logger import belief_scope
|
||||
from src.services.dashboard_testing.scenario.models import DashboardTestScenario, Finding
|
||||
from src.services.dashboard_testing.scenario.templates import REGISTERED_ACTIONS, TOOLS
|
||||
|
||||
_SQL_TOKENS = ("select ", "insert ", "update ", "delete from", "drop table", "create table", "select *", "join ")
|
||||
from src.services.dashboard_testing.scenario.sql_guard import contains_sql_statement
|
||||
|
||||
|
||||
@dataclass
|
||||
@@ -73,10 +72,7 @@ def _find_cycles(steps: list[dict[str, Any]]) -> list[str]:
|
||||
|
||||
|
||||
def _detect_sql(text: str | None) -> bool:
|
||||
if not text:
|
||||
return False
|
||||
lowered = text.lower()
|
||||
return any(tok in lowered for tok in _SQL_TOKENS)
|
||||
return contains_sql_statement(text)
|
||||
|
||||
|
||||
# #region ScenarioGraph.Validator.ValidateCore [C:4] [TYPE Function] [SEMANTICS scenario,validator,checks]
|
||||
@@ -201,10 +197,14 @@ def _check_safety(steps: list[dict[str, Any]], result: ScenarioValidationResult)
|
||||
# @RATIONALE dashboard_context.query is embedded verbatim into canonical bytes and graph_snapshot;
|
||||
# without this scan a client could persist raw query_context/SQL payloads that the
|
||||
# step-level safety checks never see.
|
||||
# @RATIONALE The dashboard_context.query subtree is exempt from the free-text SQL scan: it carries the
|
||||
# server-issued DashboardQueryModel whose authority is the context_authority fingerprint
|
||||
# recomputation at registration, not token heuristics (query_context keys remain forbidden
|
||||
# recursively; DEF-03 fix would otherwise false-positive on legitimate model strings).
|
||||
def _check_dashboard_context(scenario: DashboardTestScenario, result: ScenarioValidationResult) -> None:
|
||||
max_findings = 20
|
||||
|
||||
def walk(node: Any, path: str, depth: int) -> None:
|
||||
def walk(node: Any, path: str, depth: int, server_query: bool = False) -> None:
|
||||
if len(result.errors) >= max_findings or depth > 24:
|
||||
return
|
||||
if isinstance(node, dict):
|
||||
@@ -214,11 +214,11 @@ def _check_dashboard_context(scenario: DashboardTestScenario, result: ScenarioVa
|
||||
result.errors.append(_err("FORBIDDEN_QUERY_CONTEXT", f"dashboard_context embeds raw query_context at {key_path}"))
|
||||
if len(result.errors) >= max_findings:
|
||||
return
|
||||
walk(node[key], key_path, depth + 1)
|
||||
walk(node[key], key_path, depth + 1, server_query or str(key).lower() == "query")
|
||||
elif isinstance(node, list):
|
||||
for index, item in enumerate(node):
|
||||
walk(item, f"{path}[{index}]", depth + 1)
|
||||
elif isinstance(node, str) and _detect_sql(node):
|
||||
walk(item, f"{path}[{index}]", depth + 1, server_query)
|
||||
elif isinstance(node, str) and not server_query and _detect_sql(node):
|
||||
result.errors.append(_err("FORBIDDEN_SQL", f"dashboard_context contains SQL text at {path}"))
|
||||
|
||||
walk(scenario.dashboard_context, "dashboard_context", 0)
|
||||
|
||||
358
backend/tests/api/test_scenario_artifact_content_api.py
Normal file
358
backend/tests/api/test_scenario_artifact_content_api.py
Normal file
@@ -0,0 +1,358 @@
|
||||
# #region Test.Api.ScenarioArtifactContent [C:3] [TYPE Module] [SEMANTICS test,api,scenario,artifact,content,head]
|
||||
# @BRIEF HTTP GET/HEAD coverage for artifact-content.openapi.yaml via an isolated FastAPI app.
|
||||
# @RELATION BINDS_TO -> [Api.ScenarioArtifactContent]
|
||||
# @TEST_CONTRACT: GET/HEAD /api/scenario-runs/{run_id}/artifacts/{artifact_id}/content -> bytes|envelope
|
||||
# @TEST_FIXTURE: JPEG_BYTES / PNG_BYTES / JSON_BYTES -> INLINE magic-byte fixtures with hardcoded SHA-256
|
||||
# @TEST_EDGE: missing_artifact -> 404 NOT_FOUND identical to foreign child
|
||||
# @TEST_EDGE: invalid_digest -> 409 ARTIFACT_INTEGRITY_FAILED with no artifact bytes
|
||||
# @TEST_EDGE: storage_fail -> 503 ARTIFACT_STORAGE_UNAVAILABLE retryable
|
||||
# @TEST_INVARIANT Api.ScenarioArtifactContent: GET 200 returns verified bytes and required headers;
|
||||
# HEAD 200 is empty with the same headers. -> VERIFIED_BY: test_get_jpeg_png_json_200,
|
||||
# test_head_200_empty_body_same_headers
|
||||
# @TEST_INVARIANT Api.ScenarioArtifactContent: 401 unauthenticated; 403 known run without VIEW; 404
|
||||
# foreign/wrong-run/missing share NOT_FOUND. -> VERIFIED_BY: test_unauthenticated_401,
|
||||
# test_known_run_missing_view_403, test_foreign_wrong_run_missing_404
|
||||
# @TEST_INVARIANT Api.ScenarioArtifactContent: 409/410/413/416/503 typed Error-Code; HEAD errors have
|
||||
# empty bodies; Range is rejected after ACL. -> VERIFIED_BY: test_digest_and_mime_409_no_bytes,
|
||||
# test_inactive_410, test_range_416_after_acl, test_oversized_413, test_storage_503,
|
||||
# test_head_error_empty_body
|
||||
from __future__ import annotations
|
||||
|
||||
from fastapi import FastAPI
|
||||
from fastapi.testclient import TestClient
|
||||
from sqlalchemy import create_engine, event
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.pool import StaticPool
|
||||
|
||||
from src.api.routes.dashboard_testing.scenario_artifact_content import (
|
||||
get_artifact_content_storage,
|
||||
router,
|
||||
)
|
||||
from src.dependencies import get_current_user, get_db
|
||||
from src.models.auth import Permission, Role, User
|
||||
from src.models.scenario_artifact import ScenarioArtifact
|
||||
from src.models.scenario_registry import ScenarioRegistryEntry
|
||||
from src.models.scenario_run import ScenarioRun
|
||||
|
||||
JPEG_BYTES = b"\xff\xd8\xff\xe0\x00\x10JFIF\x00\x01\x01\x00\x00\x01\x00\x01\x00\x00\xff\xd9"
|
||||
JPEG_SHA256 = "d20f6ffd523b78a86cd2f916fa34af5d1918d75f7b142237c752ad6b254213ab"
|
||||
JPEG_DIGEST = "sha-256=:0g9v/VI7eKhs0vkW+jSvXRkY1197FCI3x1KtayVCE6s=:"
|
||||
PNG_BYTES = b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR\x00\x00\x00\x01\x00\x00\x00\x01\x08\x02\x00\x00\x00\x90wS\xde\x00\x00\x00\x00IEND\xaeB`\x82"
|
||||
PNG_SHA256 = "cb9ee84a55dfe3cb7c73189dfce37bc0ccb2dd549f8d83093eff135d12614191"
|
||||
PNG_DIGEST = "sha-256=:y57oSlXf48t8cxid/ON7wMyy3VSfjYMJPv8TXRJhQZE=:"
|
||||
JSON_BYTES = b'{"ok":true}'
|
||||
JSON_SHA256 = "4062edaf750fb8074e7e83e0c9028c94e32468a8b6f1614774328ef045150f93"
|
||||
JSON_DIGEST = "sha-256=:QGLtr3UPuAdOfoPgyQKMlOMkaKi28WFHdDKO8EUVD5M=:"
|
||||
|
||||
RUN_ID = "11111111-1111-4111-8111-111111111111"
|
||||
OTHER_RUN = "55555555-5555-4555-8555-555555555555"
|
||||
JPEG_ID = "22222222-2222-4222-8222-222222222222"
|
||||
PNG_ID = "33333333-3333-4333-8333-333333333333"
|
||||
JSON_ID = "44444444-4444-4444-8444-444444444444"
|
||||
FOREIGN_ID = "66666666-6666-4666-8666-666666666666"
|
||||
EXPIRED_ID = "77777777-7777-4777-8777-777777777777"
|
||||
BAD_DIGEST_ID = "88888888-8888-4888-8888-888888888888"
|
||||
BAD_MIME_ID = "99999999-9999-4999-8999-999999999999"
|
||||
OVERSIZE_ID = "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaa1"
|
||||
MISSING_STORE_ID = "bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbb1"
|
||||
UNAVAIL_ID = "cccccccc-cccc-4ccc-8ccc-ccccccccccc1"
|
||||
SCENARIO_ID = "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa"
|
||||
|
||||
|
||||
# #region Test.Api.ScenarioArtifactContent.Helpers [C:1] [TYPE Function]
|
||||
class MemoryStorage:
|
||||
def __init__(self, blobs: dict[str, bytes] | None = None, error: Exception | None = None) -> None:
|
||||
self.blobs = blobs or {}
|
||||
self.error = error
|
||||
|
||||
def retrieve(self, content_ref: str) -> bytes | None:
|
||||
if self.error is not None:
|
||||
raise self.error
|
||||
return self.blobs.get(content_ref)
|
||||
|
||||
|
||||
def _user(*perms: tuple[str, str]) -> User:
|
||||
role = Role(id=f"role-{perms}", name="content", is_admin=False)
|
||||
role.permissions = [Permission(resource=resource, action=action) for resource, action in perms]
|
||||
user = User(id="user-content", username="viewer", email="viewer@test.com")
|
||||
user.roles = [role]
|
||||
return user
|
||||
|
||||
|
||||
VIEWER = _user(("scenario:result", "VIEW"))
|
||||
OUTSIDER = _user()
|
||||
|
||||
|
||||
class ContentRouteEnv:
|
||||
def __init__(self) -> None:
|
||||
self.engine = create_engine("sqlite:///:memory:", poolclass=StaticPool, connect_args={"check_same_thread": False})
|
||||
event.listen(self.engine, "connect", lambda conn, _: conn.execute("PRAGMA foreign_keys=ON"))
|
||||
from src.models.mapping import Base
|
||||
|
||||
Base.metadata.create_all(self.engine)
|
||||
self.session_factory = sessionmaker(bind=self.engine)
|
||||
self.storage = MemoryStorage()
|
||||
|
||||
def seed(self) -> None:
|
||||
session = self.session_factory()
|
||||
try:
|
||||
session.add(ScenarioRegistryEntry(
|
||||
scenario_id=SCENARIO_ID, scenario_key="content-fixture", name="Content fixture",
|
||||
dashboard_id=80, owner_id="user-content", owner_username="viewer",
|
||||
environment_ids=["env-preprod-02"],
|
||||
))
|
||||
for run_id in (RUN_ID, OTHER_RUN):
|
||||
session.add(ScenarioRun(
|
||||
id=run_id, scenario_id=SCENARIO_ID, scenario_revision_id="bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbb1",
|
||||
scenario_content_hash="d" * 64, environment_id="env-preprod-02",
|
||||
idempotency_key=f"idem-{run_id}",
|
||||
))
|
||||
rows = [
|
||||
dict(id=JPEG_ID, kind="screenshot", name="jpeg", content_ref=f"draft:{RUN_ID}:{JPEG_SHA256}",
|
||||
sha256=JPEG_SHA256, content_type="image/jpeg", byte_length=22),
|
||||
dict(id=PNG_ID, kind="screenshot", name="png", content_ref=f"draft:{RUN_ID}:{PNG_SHA256}",
|
||||
sha256=PNG_SHA256, content_type="image/png", byte_length=45),
|
||||
dict(id=JSON_ID, kind="evidence", name="json", content_ref=f"draft:{RUN_ID}:{JSON_SHA256}",
|
||||
sha256=JSON_SHA256, content_type="application/json", byte_length=11),
|
||||
dict(id=FOREIGN_ID, owner_id=OTHER_RUN, kind="screenshot", name="foreign",
|
||||
content_ref=f"draft:{OTHER_RUN}:{JPEG_SHA256}", sha256=JPEG_SHA256,
|
||||
content_type="image/jpeg", byte_length=22),
|
||||
dict(id=EXPIRED_ID, kind="screenshot", name="expired", content_ref=f"draft:{RUN_ID}:{JPEG_SHA256}",
|
||||
sha256=JPEG_SHA256, content_type="image/jpeg", byte_length=22, is_active=False),
|
||||
dict(id=BAD_DIGEST_ID, kind="screenshot", name="bad-digest", content_ref=f"draft:{RUN_ID}:{JPEG_SHA256}",
|
||||
sha256=JSON_SHA256, content_type="image/jpeg", byte_length=22),
|
||||
dict(id=BAD_MIME_ID, kind="screenshot", name="bad-mime", content_ref=f"draft:{RUN_ID}:{JPEG_SHA256}",
|
||||
sha256=JPEG_SHA256, content_type="image/png", byte_length=22),
|
||||
dict(id=OVERSIZE_ID, kind="screenshot", name="huge", content_ref=f"draft:{RUN_ID}:{JPEG_SHA256}",
|
||||
sha256=JPEG_SHA256, content_type="image/jpeg", byte_length=10_485_761),
|
||||
dict(id=MISSING_STORE_ID, kind="screenshot", name="missing", content_ref=f"draft:{RUN_ID}:{'a' * 64}",
|
||||
sha256="a" * 64, content_type="image/jpeg", byte_length=22),
|
||||
dict(id=UNAVAIL_ID, kind="screenshot", name="down", content_ref=f"draft:{RUN_ID}:{JPEG_SHA256}",
|
||||
sha256=JPEG_SHA256, content_type="image/jpeg", byte_length=22),
|
||||
]
|
||||
for fields in rows:
|
||||
session.add(ScenarioArtifact(owner_type="scenario_run", owner_id=fields.pop("owner_id", RUN_ID), **fields))
|
||||
session.commit()
|
||||
finally:
|
||||
session.close()
|
||||
self.storage.blobs = {
|
||||
f"draft:{RUN_ID}:{JPEG_SHA256}": JPEG_BYTES,
|
||||
f"draft:{RUN_ID}:{PNG_SHA256}": PNG_BYTES,
|
||||
f"draft:{RUN_ID}:{JSON_SHA256}": JSON_BYTES,
|
||||
f"draft:{OTHER_RUN}:{JPEG_SHA256}": JPEG_BYTES,
|
||||
}
|
||||
|
||||
def client(self, *, user: User | None = VIEWER, storage: MemoryStorage | None = None) -> TestClient:
|
||||
app = FastAPI()
|
||||
app.include_router(router)
|
||||
|
||||
def _override_db():
|
||||
session = self.session_factory()
|
||||
try:
|
||||
yield session
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
app.dependency_overrides[get_db] = _override_db
|
||||
app.dependency_overrides[get_artifact_content_storage] = lambda: storage or self.storage
|
||||
if user is not None:
|
||||
app.dependency_overrides[get_current_user] = lambda: user
|
||||
return TestClient(app)
|
||||
|
||||
def close(self) -> None:
|
||||
self.engine.dispose()
|
||||
# #endregion Test.Api.ScenarioArtifactContent.Helpers
|
||||
|
||||
|
||||
def _url(artifact_id: str, run_id: str = RUN_ID) -> str:
|
||||
return f"/api/scenario-runs/{run_id}/artifacts/{artifact_id}/content"
|
||||
|
||||
|
||||
def _assert_verified_headers(resp, *, content_type: str, length: int, etag: str, digest: str, filename: str) -> None:
|
||||
assert resp.headers["Content-Type"].split(";")[0] == content_type
|
||||
assert resp.headers["Content-Length"] == str(length)
|
||||
assert resp.headers["Content-Disposition"] == f"attachment; filename={filename}"
|
||||
assert resp.headers["ETag"] == f'"{etag}"'
|
||||
assert resp.headers["Content-Digest"] == digest
|
||||
assert resp.headers["Cache-Control"] == "private, no-store"
|
||||
assert resp.headers["X-Content-Type-Options"] == "nosniff"
|
||||
assert resp.headers["Accept-Ranges"] == "none"
|
||||
assert "/" not in resp.headers["Content-Disposition"].split("filename=", 1)[1]
|
||||
|
||||
|
||||
# #region Test.Api.ScenarioArtifactContent.Success [C:2] [TYPE Function]
|
||||
# @BRIEF GET returns hardcoded JPEG/PNG/JSON bytes; HEAD is empty with the same verified headers.
|
||||
def test_get_jpeg_png_json_200():
|
||||
env = ContentRouteEnv()
|
||||
env.seed()
|
||||
try:
|
||||
client = env.client()
|
||||
jpeg = client.get(_url(JPEG_ID))
|
||||
assert jpeg.status_code == 200 and jpeg.content == JPEG_BYTES
|
||||
_assert_verified_headers(jpeg, content_type="image/jpeg", length=22, etag=JPEG_SHA256, digest=JPEG_DIGEST, filename=f"{JPEG_ID}.jpg")
|
||||
png = client.get(_url(PNG_ID))
|
||||
assert png.status_code == 200 and png.content == PNG_BYTES
|
||||
_assert_verified_headers(png, content_type="image/png", length=45, etag=PNG_SHA256, digest=PNG_DIGEST, filename=f"{PNG_ID}.png")
|
||||
payload = client.get(_url(JSON_ID))
|
||||
assert payload.status_code == 200 and payload.content == JSON_BYTES
|
||||
_assert_verified_headers(payload, content_type="application/json", length=11, etag=JSON_SHA256, digest=JSON_DIGEST, filename=f"{JSON_ID}.json")
|
||||
finally:
|
||||
env.close()
|
||||
|
||||
|
||||
def test_head_200_empty_body_same_headers():
|
||||
env = ContentRouteEnv()
|
||||
env.seed()
|
||||
try:
|
||||
resp = env.client().head(_url(JPEG_ID))
|
||||
assert resp.status_code == 200
|
||||
assert resp.content == b""
|
||||
_assert_verified_headers(resp, content_type="image/jpeg", length=22, etag=JPEG_SHA256, digest=JPEG_DIGEST, filename=f"{JPEG_ID}.jpg")
|
||||
finally:
|
||||
env.close()
|
||||
# #endregion Test.Api.ScenarioArtifactContent.Success
|
||||
|
||||
|
||||
# #region Test.Api.ScenarioArtifactContent.AclHttp [C:2] [TYPE Function]
|
||||
# @BRIEF Unauthenticated 401, known-run 403, and foreign/wrong-run/missing 404 share codes not bytes.
|
||||
def test_unauthenticated_401():
|
||||
env = ContentRouteEnv()
|
||||
env.seed()
|
||||
try:
|
||||
resp = env.client(user=None).get(_url(JPEG_ID))
|
||||
assert resp.status_code == 401
|
||||
assert resp.headers["Error-Code"] == "AUTHENTICATION_REQUIRED"
|
||||
body = resp.json()
|
||||
assert body["code"] == "AUTHENTICATION_REQUIRED"
|
||||
assert body["retryable"] is False
|
||||
assert "correlation_id" in body and JPEG_BYTES not in resp.content
|
||||
finally:
|
||||
env.close()
|
||||
|
||||
|
||||
def test_known_run_missing_view_403():
|
||||
env = ContentRouteEnv()
|
||||
env.seed()
|
||||
try:
|
||||
resp = env.client(user=OUTSIDER).get(_url(JPEG_ID))
|
||||
assert resp.status_code == 403
|
||||
assert resp.headers["Error-Code"] == "PERMISSION_DENIED"
|
||||
assert resp.json()["code"] == "PERMISSION_DENIED"
|
||||
assert JPEG_BYTES not in resp.content
|
||||
finally:
|
||||
env.close()
|
||||
|
||||
|
||||
def test_foreign_wrong_run_missing_404():
|
||||
env = ContentRouteEnv()
|
||||
env.seed()
|
||||
try:
|
||||
client = env.client()
|
||||
codes = []
|
||||
for resp in (
|
||||
client.get(_url(FOREIGN_ID)),
|
||||
client.get(_url(JPEG_ID, run_id=OTHER_RUN)),
|
||||
client.get(_url("00000000-0000-4000-8000-000000000000")),
|
||||
):
|
||||
assert resp.status_code == 404
|
||||
assert resp.headers["Error-Code"] == "NOT_FOUND"
|
||||
assert resp.json()["code"] == "NOT_FOUND"
|
||||
assert JPEG_BYTES not in resp.content
|
||||
codes.append(resp.json()["code"])
|
||||
assert codes == ["NOT_FOUND", "NOT_FOUND", "NOT_FOUND"]
|
||||
finally:
|
||||
env.close()
|
||||
# #endregion Test.Api.ScenarioArtifactContent.AclHttp
|
||||
|
||||
|
||||
# #region Test.Api.ScenarioArtifactContent.ErrorHttp [C:2] [TYPE Function]
|
||||
# @BRIEF Integrity/tombstone/range/size/storage errors use typed Error-Code and never return artifact bytes.
|
||||
def test_digest_and_mime_409_no_bytes():
|
||||
env = ContentRouteEnv()
|
||||
env.seed()
|
||||
try:
|
||||
client = env.client()
|
||||
digest = client.get(_url(BAD_DIGEST_ID))
|
||||
mime = client.get(_url(BAD_MIME_ID))
|
||||
for resp in (digest, mime):
|
||||
assert resp.status_code == 409
|
||||
assert resp.headers["Error-Code"] == "ARTIFACT_INTEGRITY_FAILED"
|
||||
assert resp.json()["code"] == "ARTIFACT_INTEGRITY_FAILED"
|
||||
assert resp.json()["retryable"] is False
|
||||
assert JPEG_BYTES not in resp.content
|
||||
finally:
|
||||
env.close()
|
||||
|
||||
|
||||
def test_inactive_410():
|
||||
env = ContentRouteEnv()
|
||||
env.seed()
|
||||
try:
|
||||
resp = env.client().get(_url(EXPIRED_ID))
|
||||
assert resp.status_code == 410
|
||||
assert resp.headers["Error-Code"] == "ARTIFACT_EXPIRED"
|
||||
assert JPEG_BYTES not in resp.content
|
||||
finally:
|
||||
env.close()
|
||||
|
||||
|
||||
def test_range_416_after_acl():
|
||||
env = ContentRouteEnv()
|
||||
env.seed()
|
||||
try:
|
||||
denied = env.client(user=OUTSIDER).get(_url(JPEG_ID), headers={"Range": "bytes=0-1"})
|
||||
assert denied.status_code == 403
|
||||
resp = env.client().get(_url(JPEG_ID), headers={"Range": "bytes=0-1"})
|
||||
assert resp.status_code == 416
|
||||
assert resp.headers["Error-Code"] == "RANGE_NOT_SUPPORTED"
|
||||
assert resp.headers["Accept-Ranges"] == "none"
|
||||
assert JPEG_BYTES not in resp.content
|
||||
finally:
|
||||
env.close()
|
||||
|
||||
|
||||
def test_oversized_413():
|
||||
env = ContentRouteEnv()
|
||||
env.seed()
|
||||
try:
|
||||
resp = env.client().get(_url(OVERSIZE_ID))
|
||||
assert resp.status_code == 413
|
||||
assert resp.headers["Error-Code"] == "ARTIFACT_TOO_LARGE"
|
||||
assert JPEG_BYTES not in resp.content
|
||||
finally:
|
||||
env.close()
|
||||
|
||||
|
||||
def test_storage_503():
|
||||
env = ContentRouteEnv()
|
||||
env.seed()
|
||||
try:
|
||||
resp = env.client(storage=MemoryStorage(error=OSError("disk"))).get(_url(UNAVAIL_ID))
|
||||
assert resp.status_code == 503
|
||||
assert resp.headers["Error-Code"] == "ARTIFACT_STORAGE_UNAVAILABLE"
|
||||
assert resp.json()["retryable"] is True
|
||||
assert JPEG_BYTES not in resp.content
|
||||
missing = env.client().get(_url(MISSING_STORE_ID))
|
||||
assert missing.status_code == 409
|
||||
assert missing.headers["Error-Code"] == "ARTIFACT_MISSING"
|
||||
finally:
|
||||
env.close()
|
||||
|
||||
|
||||
def test_head_error_empty_body():
|
||||
env = ContentRouteEnv()
|
||||
env.seed()
|
||||
try:
|
||||
unauth = env.client(user=None).head(_url(JPEG_ID))
|
||||
assert unauth.status_code == 401
|
||||
assert unauth.headers["Error-Code"] == "AUTHENTICATION_REQUIRED"
|
||||
assert unauth.content == b""
|
||||
missing = env.client().head(_url("00000000-0000-4000-8000-000000000000"))
|
||||
assert missing.status_code == 404
|
||||
assert missing.headers["Error-Code"] == "NOT_FOUND"
|
||||
assert missing.content == b""
|
||||
finally:
|
||||
env.close()
|
||||
# #endregion Test.Api.ScenarioArtifactContent.ErrorHttp
|
||||
|
||||
# #endregion Test.Api.ScenarioArtifactContent
|
||||
@@ -391,4 +391,29 @@ class TestScenarioAutomationRbac:
|
||||
verify.close()
|
||||
# #endregion Test.Api.ScenarioAutomation.Rbac.ManualOnly
|
||||
# #endregion Test.Api.ScenarioAutomation.Rbac
|
||||
|
||||
|
||||
# #region Test.Api.ScenarioAutomation.Sec01 [C:2] [TYPE Class] [SEMANTICS test,api,scenario,automation,rbac,anonymous]
|
||||
# @BRIEF SEC-01: operational reads require an authenticated principal; anonymous bearers are 401.
|
||||
# @TEST_INVARIANT Api.ScenarioAutomation: every operational read carries _USER; API-key/service
|
||||
# principals cannot bypass the OAuth2 bearer dependency.
|
||||
class TestScenarioAutomationAnonymousReads:
|
||||
_READ_PATHS = (
|
||||
"/api/scenario-automation/schedules",
|
||||
"/api/scenario-automation/trigger-rules",
|
||||
"/api/scenario-automation/policies",
|
||||
"/api/scenario-automation/notifications",
|
||||
"/api/scenario-automation/metrics",
|
||||
"/api/scenario-automation/retention",
|
||||
)
|
||||
|
||||
def test_anonymous_reads_require_authenticated_principal(self, dashboard_testing_client):
|
||||
# Remove the auth bypass so the real OAuth2PasswordBearer dependency rejects an
|
||||
# anonymous/API-key principal (no Authorization Bearer header) on every operational read.
|
||||
app.dependency_overrides.pop(get_current_user, None)
|
||||
anonymous = TestClient(app)
|
||||
for path in self._READ_PATHS:
|
||||
resp = anonymous.get(path)
|
||||
assert resp.status_code == 401, (path, resp.text)
|
||||
# #endregion Test.Api.ScenarioAutomation.Sec01
|
||||
# #endregion Test.Api.ScenarioAutomation
|
||||
|
||||
@@ -104,6 +104,46 @@ _INFRA_RESUME_GRAPH = {
|
||||
],
|
||||
"dependencies": [{"source": "before_resume", "target": "after_resume"}],
|
||||
}
|
||||
_REFRESH = json.loads(
|
||||
(
|
||||
Path(__file__).resolve().parents[3]
|
||||
/ "specs" / "044-dashboard-scenario-execution" / "fixtures" / "production-contract-refresh.json"
|
||||
).read_text(encoding="utf-8")
|
||||
)
|
||||
|
||||
|
||||
def _published_snapshot() -> dict:
|
||||
pin = _REFRESH["baseline_pin"]
|
||||
return {
|
||||
"baseline_set_id": pin["baseline_set_id"],
|
||||
"baseline_set_version": pin["baseline_set_version"],
|
||||
"release_id": pin["release_id"],
|
||||
"baseline_family": pin["baseline_family"],
|
||||
"catalog_digest": pin["catalog_digest"],
|
||||
"catalog_revision": _REFRESH["catalog_revision"],
|
||||
}
|
||||
|
||||
|
||||
def _start_body(**overrides) -> dict:
|
||||
body = {
|
||||
"scenario_id": _SCENARIO,
|
||||
"revision_id": _REV_CURRENT,
|
||||
"environment_id": "env-preprod-02",
|
||||
"params": {},
|
||||
"baseline_set": "ss-prod-visual",
|
||||
"baseline_set_version": "1",
|
||||
}
|
||||
body.update(overrides)
|
||||
return body
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _inject_published_catalog(monkeypatch):
|
||||
snapshot = _published_snapshot()
|
||||
monkeypatch.setattr(
|
||||
"src.services.dashboard_testing.execution.runner.load_published_catalog",
|
||||
lambda injected=None: injected if injected is not None else snapshot,
|
||||
)
|
||||
|
||||
|
||||
def _seed(session) -> None:
|
||||
@@ -251,7 +291,7 @@ class TestStart:
|
||||
resp = route_env.client().post(
|
||||
"/api/scenario-runs",
|
||||
headers={"Idempotency-Key": "start-1"},
|
||||
json={"scenario_id": _SCENARIO, "revision_id": _REV_CURRENT, "environment_id": "env-preprod-02", "params": {"region": "emea"}},
|
||||
json=_start_body(params={"region": "emea"}),
|
||||
)
|
||||
assert resp.status_code == 201, resp.text
|
||||
body = resp.json()
|
||||
@@ -259,6 +299,22 @@ class TestStart:
|
||||
assert body["scenario_revision_id"] == _REV_CURRENT
|
||||
assert body["scenario_content_hash"] == "d" * 64
|
||||
assert body["runner_plan"]["topological_order"][0] == "s1_setup_filters"
|
||||
assert body["runner_plan"]["baseline_pin"]["baseline_set_id"] == "ss-prod-visual"
|
||||
assert body["target_snapshot"]["baseline_pin"]["catalog_digest"] == "a" * 64
|
||||
|
||||
def test_start_without_baseline_set_is_422(self, route_env):
|
||||
resp = route_env.client().post(
|
||||
"/api/scenario-runs",
|
||||
headers={"Idempotency-Key": "start-missing-set"},
|
||||
json={
|
||||
"scenario_id": _SCENARIO,
|
||||
"revision_id": _REV_CURRENT,
|
||||
"environment_id": "env-preprod-02",
|
||||
"params": {},
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 422
|
||||
assert resp.json()["detail"]["code"] == "BASELINE_MISSING"
|
||||
|
||||
def test_idempotent_replay_returns_same_run(self, route_env):
|
||||
from src.models.scenario_automation import ScenarioNotificationEvent
|
||||
@@ -266,7 +322,7 @@ class TestStart:
|
||||
from src.models.scenario_run import ScenarioStepRun
|
||||
|
||||
client = route_env.client()
|
||||
payload = {"scenario_id": _SCENARIO, "revision_id": _REV_CURRENT, "environment_id": "env-preprod-02", "params": {}}
|
||||
payload = _start_body()
|
||||
first = client.post("/api/scenario-runs", headers={"Idempotency-Key": "start-2"}, json=payload)
|
||||
second = client.post("/api/scenario-runs", headers={"Idempotency-Key": "start-2"}, json=payload)
|
||||
assert first.status_code == 201 and second.status_code == 201
|
||||
@@ -282,7 +338,7 @@ class TestStart:
|
||||
|
||||
def test_changed_request_same_key_409(self, route_env):
|
||||
client = route_env.client()
|
||||
payload = {"scenario_id": _SCENARIO, "revision_id": _REV_CURRENT, "environment_id": "env-preprod-02", "params": {"a": 1}}
|
||||
payload = _start_body(params={"a": 1})
|
||||
client.post("/api/scenario-runs", headers={"Idempotency-Key": "start-3"}, json=payload)
|
||||
resp = client.post(
|
||||
"/api/scenario-runs", headers={"Idempotency-Key": "start-3"},
|
||||
@@ -295,7 +351,7 @@ class TestStart:
|
||||
resp = route_env.client().post(
|
||||
"/api/scenario-runs",
|
||||
headers={"Idempotency-Key": "start-4"},
|
||||
json={"scenario_id": _SCENARIO, "revision_id": _REV_STALE, "environment_id": "env-preprod-02", "params": {}},
|
||||
json=_start_body(revision_id=_REV_STALE),
|
||||
)
|
||||
assert resp.status_code == 409
|
||||
assert "revision mismatch" in resp.json()["detail"]["detail"]
|
||||
@@ -307,7 +363,7 @@ class TestStart:
|
||||
resp = env.client().post(
|
||||
"/api/scenario-runs",
|
||||
headers={"Idempotency-Key": "start-5"},
|
||||
json={"scenario_id": _SCENARIO, "revision_id": _REV_CURRENT, "environment_id": "env-preprod-02", "params": {}},
|
||||
json=_start_body(),
|
||||
)
|
||||
assert resp.status_code == 403
|
||||
finally:
|
||||
@@ -320,7 +376,7 @@ class TestStart:
|
||||
resp = env.client().post(
|
||||
"/api/scenario-runs",
|
||||
headers={"Idempotency-Key": "start-prod-1"},
|
||||
json={"scenario_id": _SCENARIO, "revision_id": _REV_CURRENT, "environment_id": "env-prod-01", "params": {}, "is_prod": False},
|
||||
json=_start_body(environment_id="env-prod-01", is_prod=False),
|
||||
)
|
||||
assert resp.status_code == 403
|
||||
finally:
|
||||
@@ -333,7 +389,7 @@ class TestStart:
|
||||
resp = env.client().post(
|
||||
"/api/scenario-runs",
|
||||
headers={"Idempotency-Key": "start-prod-2"},
|
||||
json={"scenario_id": _SCENARIO, "revision_id": _REV_CURRENT, "environment_id": "env-prod-01", "params": {}, "is_prod": False},
|
||||
json=_start_body(environment_id="env-prod-01", is_prod=False),
|
||||
)
|
||||
assert resp.status_code == 201, resp.text
|
||||
assert resp.json()["status"] == "pending_approval"
|
||||
@@ -344,13 +400,7 @@ class TestStart:
|
||||
response = route_env.client().post(
|
||||
"/api/scenario-runs",
|
||||
headers={"Idempotency-Key": "start-preprod-true-044"},
|
||||
json={
|
||||
"scenario_id": _SCENARIO,
|
||||
"revision_id": _REV_CURRENT,
|
||||
"environment_id": "env-preprod-02",
|
||||
"params": {},
|
||||
"is_prod": True,
|
||||
},
|
||||
json=_start_body(is_prod=True),
|
||||
)
|
||||
assert response.status_code == 201, response.text
|
||||
assert response.json()["status"] == "queued"
|
||||
@@ -366,12 +416,10 @@ class TestStart:
|
||||
env.seed()
|
||||
try:
|
||||
client = env.client()
|
||||
payload = {
|
||||
"scenario_id": _SCENARIO,
|
||||
"revision_id": _REV_CURRENT,
|
||||
"environment_id": "env-prod-01",
|
||||
"params": {"hardcoded": "prod-replay-044"},
|
||||
}
|
||||
payload = _start_body(
|
||||
environment_id="env-prod-01",
|
||||
params={"hardcoded": "prod-replay-044"},
|
||||
)
|
||||
first = client.post(
|
||||
"/api/scenario-runs",
|
||||
headers={"Idempotency-Key": "start-prod-replay-class-044"},
|
||||
@@ -405,12 +453,7 @@ class TestStart:
|
||||
response = route_env.client().post(
|
||||
"/api/scenario-runs",
|
||||
headers={"Idempotency-Key": "start-unknown-environment-044"},
|
||||
json={
|
||||
"scenario_id": _SCENARIO,
|
||||
"revision_id": _REV_CURRENT,
|
||||
"environment_id": "env-unknown-044",
|
||||
"params": {},
|
||||
},
|
||||
json=_start_body(environment_id="env-unknown-044"),
|
||||
)
|
||||
assert response.status_code == 422
|
||||
assert response.json()["detail"]["code"] == "ENVIRONMENT_NOT_CONFIGURED"
|
||||
@@ -433,13 +476,7 @@ class TestApprovalDecision:
|
||||
created = client.post(
|
||||
"/api/scenario-runs",
|
||||
headers={"Idempotency-Key": "approval-prod-1"},
|
||||
json={
|
||||
"scenario_id": _SCENARIO,
|
||||
"revision_id": _REV_CURRENT,
|
||||
"environment_id": "env-prod-01",
|
||||
"params": {},
|
||||
"is_prod": True,
|
||||
},
|
||||
json=_start_body(environment_id="env-prod-01", is_prod=True),
|
||||
)
|
||||
assert created.status_code == 201, created.text
|
||||
run_id = created.json()["id"]
|
||||
@@ -469,13 +506,7 @@ class TestApprovalDecision:
|
||||
created = creator.client().post(
|
||||
"/api/scenario-runs",
|
||||
headers={"Idempotency-Key": "approval-prod-2"},
|
||||
json={
|
||||
"scenario_id": _SCENARIO,
|
||||
"revision_id": _REV_CURRENT,
|
||||
"environment_id": "env-prod-01",
|
||||
"params": {},
|
||||
"is_prod": True,
|
||||
},
|
||||
json=_start_body(environment_id="env-prod-01", is_prod=True),
|
||||
)
|
||||
run_id = created.json()["id"]
|
||||
creator.user = _user_with(("scenario", "RUN"))
|
||||
@@ -732,7 +763,7 @@ class TestLifecycleRoutes:
|
||||
created = client.post(
|
||||
"/api/scenario-runs",
|
||||
headers={"Idempotency-Key": "life-1"},
|
||||
json={"scenario_id": _SCENARIO, "revision_id": _REV_CURRENT, "environment_id": "env-preprod-02", "params": {}},
|
||||
json=_start_body(),
|
||||
).json()
|
||||
resp = client.post(f"/api/scenario-runs/{created['id']}/cancel")
|
||||
assert resp.status_code == 200
|
||||
@@ -745,7 +776,7 @@ class TestLifecycleRoutes:
|
||||
created = client.post(
|
||||
"/api/scenario-runs",
|
||||
headers={"Idempotency-Key": "life-2"},
|
||||
json={"scenario_id": _SCENARIO, "revision_id": _REV_CURRENT, "environment_id": "env-preprod-02", "params": {}},
|
||||
json=_start_body(),
|
||||
).json()
|
||||
# Mint the pause token through the service (the pause action is server-internal).
|
||||
session = route_env.session_factory()
|
||||
@@ -769,7 +800,7 @@ class TestLifecycleRoutes:
|
||||
created = client.post(
|
||||
"/api/scenario-runs",
|
||||
headers={"Idempotency-Key": "life-3"},
|
||||
json={"scenario_id": _SCENARIO, "revision_id": _REV_CURRENT, "environment_id": "env-preprod-02", "params": {}},
|
||||
json=_start_body(),
|
||||
).json()
|
||||
session = route_env.session_factory()
|
||||
try:
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"compiler_version": "038.1.0",
|
||||
"action_registry_version": "038.2.0",
|
||||
"action_registry_hash": "666a2b6a2502b980de26bcd82b76962b54b9c96f023caf27f09c97a346203de1",
|
||||
"action_registry_version": "038.4.0",
|
||||
"action_registry_hash": "9839099380356b2ebf482eb3b3c34c64320dc88fcc4694288e8d4278291c7310",
|
||||
"scenario_id": "fi-0080_verify-filters-metric-xlsx",
|
||||
"revision_hash": "dddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd",
|
||||
"objective": {
|
||||
|
||||
@@ -0,0 +1,207 @@
|
||||
# #region Test.ScenarioExecution.AgentEvaluation [C:4] [TYPE Module] [SEMANTICS test,scenario,evaluation,parser,boundary]
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.AgentEvaluation]
|
||||
# @TEST_INVARIANT ScenarioExecution.AgentEvaluation: succeeded requires a bound raw response; failure
|
||||
# shapes force inconclusive; the strict parser maps malformed to parser_error.
|
||||
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter: production adapter builds manifest from completed,
|
||||
# stores raw bytes, and never synthesizes PASS. -> VERIFIED_BY: adapter_success_from_completed, adapter_empty_completed_fail_closed, composition_root_adapter_fail_closed
|
||||
from __future__ import annotations
|
||||
|
||||
import uuid
|
||||
from datetime import UTC, datetime
|
||||
|
||||
import pytest
|
||||
|
||||
from src.services.dashboard_testing.execution.agent_evaluation import (
|
||||
AgentEvaluation,
|
||||
parse_evaluation_response,
|
||||
)
|
||||
from src.services.dashboard_testing.scenario.models import AgentEvaluationSpec
|
||||
|
||||
_NOW = datetime(2026, 9, 10, 0, 0, 0, tzinfo=UTC)
|
||||
|
||||
|
||||
def _eval_dict(**overrides) -> dict:
|
||||
data: dict = {
|
||||
"schema_version": 1,
|
||||
"evaluation_id": str(uuid.uuid4()),
|
||||
"scenario_run_id": str(uuid.uuid4()),
|
||||
"logical_step_id": "step-eval-1",
|
||||
"attempt": 1,
|
||||
"operation_id": str(uuid.uuid4()),
|
||||
"evaluation_spec_hash": "a" * 64,
|
||||
"provider_id": "llm-provider-a",
|
||||
"provider_version": "1",
|
||||
"model_id": "vision-model",
|
||||
"model_version": "2026-08",
|
||||
"prompt_template_id": "agent-evaluation-prompt",
|
||||
"prompt_template_version": "1.0.0",
|
||||
"prompt_template_hash": "b" * 64,
|
||||
"output_schema_hash": "c" * 64,
|
||||
"input_manifest_hash": "d" * 64,
|
||||
"input_manifest": [{"artifact_id": "artifact-1", "sha256": "e" * 64, "content_type": "image/jpeg", "byte_length": 100, "role": "actual"}],
|
||||
"baseline_pin": {"catalog_revision_id": str(uuid.uuid4())},
|
||||
"comparison_ids": [str(uuid.uuid4())],
|
||||
"status": "succeeded",
|
||||
"verdict": "pass",
|
||||
"confidence": 0.93,
|
||||
"findings": [{"finding_id": "f-1", "severity": "info", "message": "matches", "evidence_artifact_ids": ["artifact-1"], "criterion_id": "crit-visual", "criterion_kind": "semantic"}],
|
||||
"reason_codes": ["EVALUATION_PASS"],
|
||||
"raw_response_artifact_ref": "artifact-1",
|
||||
"raw_response_sha256": "e" * 64,
|
||||
"trust_policy_hash": "f" * 64,
|
||||
"usage": {"input_tokens": 100, "output_tokens": 10, "cost_amount": "0.01", "currency": "USD", "pricing_version": "1"},
|
||||
"started_at": _NOW,
|
||||
"finished_at": _NOW,
|
||||
}
|
||||
data.update(overrides)
|
||||
return data
|
||||
|
||||
|
||||
def _spec() -> AgentEvaluationSpec:
|
||||
return AgentEvaluationSpec.model_validate({
|
||||
"schema_version": 1,
|
||||
"spec_id": str(uuid.uuid4()),
|
||||
"provider_id": "llm-provider-a", "provider_version": "1", "model_id": "vision-model", "model_version": "2026-08",
|
||||
"prompt_template_id": "agent-evaluation-prompt", "prompt_template_version": "1.0.0", "prompt_template_hash": "a" * 64,
|
||||
"evidence_refs": ["artifact-1"], "comparison_refs": ["44444444-4444-4444-8444-444444444444"],
|
||||
"output_schema": "agent-evaluation.schema.json",
|
||||
"decision_policy": {"policy_id": "baseline-semantic", "version": "1.0.0"},
|
||||
"limits": {"timeout_ms": 60000, "max_images": 4, "max_input_tokens": 32000, "max_output_tokens": 2000, "max_cost": "1.00", "currency": "USD"},
|
||||
"trust_policy_hash": "b" * 64,
|
||||
"criteria": [
|
||||
{"criterion_id": "crit-visual", "criterion_kind": "semantic", "description": "visual", "comparison_id": None},
|
||||
],
|
||||
})
|
||||
|
||||
|
||||
def test_succeeded_record_roundtrip():
|
||||
record = AgentEvaluation.model_validate(_eval_dict())
|
||||
assert record.status == "succeeded"
|
||||
assert record.findings[0].criterion_id == "crit-visual"
|
||||
|
||||
|
||||
def test_succeeded_requires_raw_response():
|
||||
with pytest.raises(ValueError, match="EVALUATION_RAW_RESPONSE_REQUIRED"):
|
||||
AgentEvaluation.model_validate(_eval_dict(raw_response_artifact_ref=None, raw_response_sha256=None))
|
||||
|
||||
|
||||
def test_failure_shape_forces_inconclusive():
|
||||
with pytest.raises(ValueError):
|
||||
AgentEvaluation.model_validate(_eval_dict(status="provider_error"))
|
||||
|
||||
|
||||
def test_logical_step_id_slug_is_allowed():
|
||||
record = AgentEvaluation.model_validate(_eval_dict(logical_step_id="s4_assert_revenue"))
|
||||
assert record.logical_step_id == "s4_assert_revenue"
|
||||
|
||||
|
||||
def test_parser_accepts_valid_response():
|
||||
result = parse_evaluation_response(_eval_dict(), spec=_spec())
|
||||
assert result.status == "succeeded"
|
||||
|
||||
|
||||
def test_parser_maps_malformed_to_parser_error():
|
||||
result = parse_evaluation_response({"garbage": True}, spec=_spec())
|
||||
assert result.status == "parser_error"
|
||||
assert result.verdict == "inconclusive"
|
||||
assert result.confidence == 0
|
||||
assert result.findings == []
|
||||
|
||||
|
||||
def test_parser_maps_criterion_kind_mismatch_to_parser_error():
|
||||
result = parse_evaluation_response(
|
||||
_eval_dict(findings=[{"finding_id": "f-1", "severity": "info", "message": "m", "evidence_artifact_ids": ["artifact-1"], "criterion_id": "crit-visual", "criterion_kind": "deterministic_comparison"}]),
|
||||
spec=_spec(),
|
||||
)
|
||||
assert result.status == "parser_error"
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.EvaluationAdapter.Store [C:1] [TYPE Class]
|
||||
class _EvidenceStore:
|
||||
def __init__(self) -> None:
|
||||
self.stored: list[tuple[str, str, bytes]] = []
|
||||
|
||||
def store(self, run_id: str, digest: str, data: bytes) -> str:
|
||||
self.stored.append((run_id, digest, data))
|
||||
return f"draft:{run_id}:{digest}"
|
||||
# #endregion Test.ScenarioExecution.EvaluationAdapter.Store
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.EvaluationAdapter.Success [C:2] [TYPE Function]
|
||||
# @BRIEF Injected submit + prior-step completed artifacts produce a bound evaluation_record.
|
||||
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter: completed prior-step artifacts become input_manifest. -> VERIFIED_BY: adapter_success_from_completed
|
||||
def test_evaluation_adapter_builds_manifest_and_returns_record():
|
||||
from src.services.dashboard_testing.execution.evaluation_adapter import evaluation_adapter_from
|
||||
|
||||
prior_digest = "e" * 64
|
||||
prior_ref = f"draft:run-eval:{prior_digest}"
|
||||
run_id = str(uuid.uuid4())
|
||||
storage = _EvidenceStore()
|
||||
adapter = evaluation_adapter_from(storage=storage, submit=lambda **_kwargs: _eval_dict())
|
||||
result = adapter(
|
||||
{
|
||||
"logical_step_id": "step-eval-1",
|
||||
"scenario_run_id": run_id,
|
||||
"attempt": 1,
|
||||
"agent_evaluation_spec": _spec().model_dump(mode="json"),
|
||||
},
|
||||
{
|
||||
"s1-shot": {
|
||||
"status": "passed",
|
||||
"artifact_refs": [prior_ref],
|
||||
"step_outcome": {
|
||||
"tool": "screenshot",
|
||||
"artifact_digests": {prior_ref: prior_digest},
|
||||
"content_type": "image/jpeg",
|
||||
"byte_length": 100,
|
||||
},
|
||||
}
|
||||
},
|
||||
)
|
||||
assert result["evaluation_input"]["status"] == "succeeded"
|
||||
assert result["evaluation_input"]["verdict"] == "pass"
|
||||
assert result["evaluation_input"]["confidence"] == 0.93
|
||||
assert result["evaluation_record"]["input_manifest"][0]["artifact_id"] == prior_ref
|
||||
assert result["content_type"] == "application/json"
|
||||
assert result["byte_length"] == len(storage.stored[0][2])
|
||||
assert result["artifact_refs"] == [f"draft:{run_id}:{storage.stored[0][1]}"]
|
||||
# #endregion Test.ScenarioExecution.EvaluationAdapter.Success
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.EvaluationAdapter.Empty [C:2] [TYPE Function]
|
||||
# @BRIEF Missing prior-step evidence is fail-closed; the adapter never synthesizes PASS.
|
||||
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter: empty completed fails closed. -> VERIFIED_BY: adapter_empty_completed_fail_closed
|
||||
def test_evaluation_adapter_empty_completed_fail_closed():
|
||||
from src.services.dashboard_testing.execution.evaluation_adapter import evaluation_adapter_from
|
||||
|
||||
adapter = evaluation_adapter_from(storage=_EvidenceStore(), submit=lambda **_kwargs: _eval_dict())
|
||||
with pytest.raises(RuntimeError, match="EVALUATION_EVIDENCE_NOT_FOUND"):
|
||||
adapter(
|
||||
{
|
||||
"logical_step_id": "step-eval-1",
|
||||
"scenario_run_id": str(uuid.uuid4()),
|
||||
"agent_evaluation_spec": _spec().model_dump(mode="json"),
|
||||
},
|
||||
{},
|
||||
)
|
||||
# #endregion Test.ScenarioExecution.EvaluationAdapter.Empty
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.EvaluationAdapter.Composition [C:2] [TYPE Function]
|
||||
# @BRIEF Composition root always exposes an adapter; missing spec/evidence stays non-PASS.
|
||||
# @TEST_INVARIANT ScenarioExecution.LiveCompositionRoot: agent_evaluation_adapter is composed and fail-closed. -> VERIFIED_BY: composition_root_adapter_fail_closed
|
||||
def test_composition_root_evaluation_adapter_is_composed_and_fail_closed():
|
||||
from src.services.dashboard_testing.execution.executors import agent_evaluation
|
||||
from src.services.dashboard_testing.execution.live_composition import LiveExecutionCompositionRoot
|
||||
|
||||
adapter = LiveExecutionCompositionRoot().agent_evaluation_adapter()
|
||||
outcome = agent_evaluation(
|
||||
{"logical_step_id": "step-eval-1", "scenario_run_id": str(uuid.uuid4())},
|
||||
{},
|
||||
adapter=adapter,
|
||||
)
|
||||
assert adapter is not None
|
||||
assert outcome["status"] == "inconclusive"
|
||||
assert outcome["error_code"] is not None
|
||||
# #endregion Test.ScenarioExecution.EvaluationAdapter.Composition
|
||||
# #endregion Test.ScenarioExecution.AgentEvaluation
|
||||
@@ -0,0 +1,251 @@
|
||||
# #region Test.ScenarioExecution.ArtifactContent [C:3] [TYPE Module] [SEMANTICS test,scenario,artifact,content,digest,mime]
|
||||
# @BRIEF Unit coverage for ScenarioArtifactContentService ACL and complete-object checks.
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.ArtifactContent]
|
||||
# @TEST_CONTRACT: (run_id, artifact_id, user, Range?, storage) -> ArtifactContent | ArtifactContentError
|
||||
# @TEST_FIXTURE: JPEG_BYTES / PNG_BYTES / JSON_BYTES -> INLINE magic-byte fixtures with hardcoded SHA-256
|
||||
# @TEST_EDGE: missing_artifact -> 404 NOT_FOUND
|
||||
# @TEST_EDGE: invalid_digest -> 409 ARTIFACT_INTEGRITY_FAILED
|
||||
# @TEST_EDGE: storage_fail -> 503 ARTIFACT_STORAGE_UNAVAILABLE
|
||||
# @TEST_INVARIANT ScenarioExecution.ArtifactContent: ownership, positive length, SHA-256 and magic-byte
|
||||
# MIME are verified before success bytes. -> VERIFIED_BY: test_open_jpeg_payload,
|
||||
# test_digest_mismatch_409, test_mime_mismatch_409
|
||||
# @TEST_INVARIANT ScenarioExecution.ArtifactContent: known parent without VIEW is 403; foreign/missing
|
||||
# child is indistinguishable 404. -> VERIFIED_BY: test_known_parent_without_view_403,
|
||||
# test_foreign_and_missing_are_404
|
||||
# @TEST_INVARIANT ScenarioExecution.ArtifactContent: Range after ACL is 416; inactive is 410; declared
|
||||
# oversize is 413; missing storage is 409; outage is 503. -> VERIFIED_BY:
|
||||
# test_range_416, test_inactive_410, test_oversized_413, test_missing_storage_409,
|
||||
# test_storage_outage_503
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
from sqlalchemy import create_engine, event
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.pool import StaticPool
|
||||
|
||||
from src.models.auth import Permission, Role, User
|
||||
from src.models.scenario_artifact import ScenarioArtifact
|
||||
from src.models.scenario_registry import ScenarioRegistryEntry
|
||||
from src.models.scenario_run import ScenarioRun
|
||||
from src.services.dashboard_testing.execution.artifact_content import (
|
||||
ArtifactContentError,
|
||||
ScenarioArtifactContentService,
|
||||
sniff_mime,
|
||||
)
|
||||
|
||||
JPEG_BYTES = b"\xff\xd8\xff\xe0\x00\x10JFIF\x00\x01\x01\x00\x00\x01\x00\x01\x00\x00\xff\xd9"
|
||||
JPEG_SHA256 = "d20f6ffd523b78a86cd2f916fa34af5d1918d75f7b142237c752ad6b254213ab"
|
||||
PNG_BYTES = b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR\x00\x00\x00\x01\x00\x00\x00\x01\x08\x02\x00\x00\x00\x90wS\xde\x00\x00\x00\x00IEND\xaeB`\x82"
|
||||
PNG_SHA256 = "cb9ee84a55dfe3cb7c73189dfce37bc0ccb2dd549f8d83093eff135d12614191"
|
||||
JSON_BYTES = b'{"ok":true}'
|
||||
JSON_SHA256 = "4062edaf750fb8074e7e83e0c9028c94e32468a8b6f1614774328ef045150f93"
|
||||
|
||||
RUN_ID = "11111111-1111-4111-8111-111111111111"
|
||||
OTHER_RUN = "55555555-5555-4555-8555-555555555555"
|
||||
ARTIFACT_ID = "22222222-2222-4222-8222-222222222222"
|
||||
FOREIGN_ID = "66666666-6666-4666-8666-666666666666"
|
||||
SCENARIO_ID = "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa"
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.ArtifactContent.Helpers [C:1] [TYPE Function]
|
||||
class MemoryStorage:
|
||||
def __init__(self, blobs: dict[str, bytes] | None = None, error: Exception | None = None) -> None:
|
||||
self.blobs = blobs or {}
|
||||
self.error = error
|
||||
|
||||
def retrieve(self, content_ref: str) -> bytes | None:
|
||||
if self.error is not None:
|
||||
raise self.error
|
||||
return self.blobs.get(content_ref)
|
||||
|
||||
|
||||
def _user(*perms: tuple[str, str], admin: bool = False) -> User:
|
||||
role = Role(id="role-content", name="content", is_admin=admin)
|
||||
role.permissions = [Permission(resource=resource, action=action) for resource, action in perms]
|
||||
user = User(id="user-content", username="viewer", email="viewer@test.com")
|
||||
user.roles = [role]
|
||||
return user
|
||||
|
||||
|
||||
def _session():
|
||||
engine = create_engine("sqlite:///:memory:", poolclass=StaticPool, connect_args={"check_same_thread": False})
|
||||
event.listen(engine, "connect", lambda conn, _: conn.execute("PRAGMA foreign_keys=ON"))
|
||||
from src.models.mapping import Base
|
||||
|
||||
Base.metadata.create_all(engine)
|
||||
factory = sessionmaker(bind=engine)
|
||||
session = factory()
|
||||
session.add(ScenarioRegistryEntry(
|
||||
scenario_id=SCENARIO_ID, scenario_key="content-fixture", name="Content fixture",
|
||||
dashboard_id=80, owner_id="user-content", owner_username="viewer",
|
||||
environment_ids=["env-preprod-02"],
|
||||
))
|
||||
for run_id in (RUN_ID, OTHER_RUN):
|
||||
session.add(ScenarioRun(
|
||||
id=run_id, scenario_id=SCENARIO_ID, scenario_revision_id="bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbb1",
|
||||
scenario_content_hash="d" * 64, environment_id="env-preprod-02",
|
||||
idempotency_key=f"idem-{run_id}",
|
||||
))
|
||||
session.commit()
|
||||
return session
|
||||
# #endregion Test.ScenarioExecution.ArtifactContent.Helpers
|
||||
|
||||
|
||||
def _add_artifact(session, **fields) -> ScenarioArtifact:
|
||||
values = {
|
||||
"id": ARTIFACT_ID,
|
||||
"owner_type": "scenario_run",
|
||||
"owner_id": RUN_ID,
|
||||
"kind": "screenshot",
|
||||
"name": "shot",
|
||||
"content_ref": f"draft:{RUN_ID}:{JPEG_SHA256}",
|
||||
"sha256": JPEG_SHA256,
|
||||
"content_type": "image/jpeg",
|
||||
"byte_length": len(JPEG_BYTES),
|
||||
"is_active": True,
|
||||
}
|
||||
values.update(fields)
|
||||
row = ScenarioArtifact(**values)
|
||||
session.add(row)
|
||||
session.commit()
|
||||
return row
|
||||
|
||||
|
||||
def _open(session, storage, user=None, **kwargs):
|
||||
service = ScenarioArtifactContentService(session, storage=storage)
|
||||
return service.open(
|
||||
run_id=kwargs.pop("run_id", RUN_ID),
|
||||
artifact_id=kwargs.pop("artifact_id", ARTIFACT_ID),
|
||||
user=user or _user(("scenario:result", "VIEW")),
|
||||
range_header=kwargs.pop("range_header", None),
|
||||
)
|
||||
|
||||
|
||||
def _expect(session, storage, status, code, **kwargs):
|
||||
with pytest.raises(ArtifactContentError) as caught:
|
||||
_open(session, storage, **kwargs)
|
||||
err = caught.value
|
||||
assert (err.status_code, err.code) == (status, code)
|
||||
assert "draft:" not in err.message
|
||||
assert "/" not in err.message
|
||||
return err
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.ArtifactContent.Sniff [C:2] [TYPE Function]
|
||||
# @BRIEF Magic-byte fixtures map to the OpenAPI allowlist; random bytes sniff as None.
|
||||
def test_sniff_mime_hardcoded_magic_bytes():
|
||||
assert sniff_mime(JPEG_BYTES) == "image/jpeg"
|
||||
assert sniff_mime(PNG_BYTES) == "image/png"
|
||||
assert sniff_mime(JSON_BYTES) == "application/json"
|
||||
assert sniff_mime(b"RIFF\x00\x00\x00\x00NOPE") is None
|
||||
assert sniff_mime(b"RIFF\x00\x00\x00\x00WEBP") == "image/webp"
|
||||
assert sniff_mime(b"%PDF-1.4") == "application/pdf"
|
||||
assert sniff_mime(b"PK\x03\x04rest") == "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"
|
||||
assert sniff_mime(b"not-a-real-file") is None
|
||||
# #endregion Test.ScenarioExecution.ArtifactContent.Sniff
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.ArtifactContent.OpenJpeg [C:2] [TYPE Function]
|
||||
# @BRIEF Verified JPEG payload carries quoted lowercase ETag and RFC 9530 Content-Digest.
|
||||
def test_open_jpeg_payload():
|
||||
session = _session()
|
||||
_add_artifact(session)
|
||||
storage = MemoryStorage({f"draft:{RUN_ID}:{JPEG_SHA256}": JPEG_BYTES})
|
||||
payload = _open(session, storage)
|
||||
assert payload.body == JPEG_BYTES
|
||||
assert payload.content_type == "image/jpeg"
|
||||
assert payload.content_length == 22
|
||||
assert payload.etag == f'"{JPEG_SHA256}"'
|
||||
assert payload.content_digest == "sha-256=:0g9v/VI7eKhs0vkW+jSvXRkY1197FCI3x1KtayVCE6s=:"
|
||||
assert payload.content_disposition == f"attachment; filename={ARTIFACT_ID}.jpg"
|
||||
# #endregion Test.ScenarioExecution.ArtifactContent.OpenJpeg
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.ArtifactContent.Acl [C:2] [TYPE Function]
|
||||
# @BRIEF Known parent without VIEW is 403; foreign/missing children share NOT_FOUND.
|
||||
def test_known_parent_without_view_403():
|
||||
session = _session()
|
||||
_add_artifact(session)
|
||||
_expect(session, MemoryStorage(), 403, "PERMISSION_DENIED", user=_user())
|
||||
|
||||
|
||||
def test_foreign_and_missing_are_404():
|
||||
session = _session()
|
||||
_add_artifact(session)
|
||||
_add_artifact(session, id=FOREIGN_ID, owner_id=OTHER_RUN, content_ref=f"draft:{OTHER_RUN}:{JPEG_SHA256}")
|
||||
missing = _expect(session, MemoryStorage(), 404, "NOT_FOUND", artifact_id="00000000-0000-4000-8000-000000000000")
|
||||
foreign = _expect(session, MemoryStorage(), 404, "NOT_FOUND", artifact_id=FOREIGN_ID)
|
||||
wrong_run = _expect(session, MemoryStorage(), 404, "NOT_FOUND", run_id=OTHER_RUN, artifact_id=ARTIFACT_ID)
|
||||
assert missing.code == foreign.code == wrong_run.code == "NOT_FOUND"
|
||||
# #endregion Test.ScenarioExecution.ArtifactContent.Acl
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.ArtifactContent.Integrity [C:2] [TYPE Function]
|
||||
# @BRIEF Digest/MIME/length failures are 409 with no payload object.
|
||||
def test_digest_mismatch_409():
|
||||
session = _session()
|
||||
_add_artifact(session, sha256=JSON_SHA256)
|
||||
_expect(session, MemoryStorage({f"draft:{RUN_ID}:{JPEG_SHA256}": JPEG_BYTES}), 409, "ARTIFACT_INTEGRITY_FAILED")
|
||||
|
||||
|
||||
def test_mime_mismatch_409():
|
||||
session = _session()
|
||||
_add_artifact(session, content_type="image/png")
|
||||
_expect(session, MemoryStorage({f"draft:{RUN_ID}:{JPEG_SHA256}": JPEG_BYTES}), 409, "ARTIFACT_INTEGRITY_FAILED")
|
||||
# #endregion Test.ScenarioExecution.ArtifactContent.Integrity
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.ArtifactContent.Limits [C:2] [TYPE Function]
|
||||
# @BRIEF Range, tombstone, declared oversize, missing bytes and storage outage map to typed codes.
|
||||
def test_range_416():
|
||||
session = _session()
|
||||
_add_artifact(session)
|
||||
_expect(session, MemoryStorage({f"draft:{RUN_ID}:{JPEG_SHA256}": JPEG_BYTES}), 416, "RANGE_NOT_SUPPORTED", range_header="bytes=0-1")
|
||||
|
||||
|
||||
def test_inactive_410():
|
||||
session = _session()
|
||||
_add_artifact(session, is_active=False)
|
||||
_expect(session, MemoryStorage({f"draft:{RUN_ID}:{JPEG_SHA256}": JPEG_BYTES}), 410, "ARTIFACT_EXPIRED")
|
||||
|
||||
|
||||
def test_oversized_413():
|
||||
session = _session()
|
||||
_add_artifact(session, byte_length=10_485_761)
|
||||
_expect(session, MemoryStorage(), 413, "ARTIFACT_TOO_LARGE")
|
||||
|
||||
|
||||
def test_missing_storage_409():
|
||||
session = _session()
|
||||
_add_artifact(session)
|
||||
_expect(session, MemoryStorage(), 409, "ARTIFACT_MISSING")
|
||||
|
||||
|
||||
def test_storage_outage_503():
|
||||
session = _session()
|
||||
_add_artifact(session)
|
||||
err = _expect(session, MemoryStorage(error=OSError("disk")), 503, "ARTIFACT_STORAGE_UNAVAILABLE")
|
||||
assert err.retryable is True
|
||||
# #endregion Test.ScenarioExecution.ArtifactContent.Limits
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.ArtifactContent.JsonPng [C:2] [TYPE Function]
|
||||
# @BRIEF PNG and JSON receipts succeed with kind-allowlisted MIME and hardcoded digests.
|
||||
def test_open_png_and_json():
|
||||
session = _session()
|
||||
_add_artifact(session, id="33333333-3333-4333-8333-333333333333", kind="screenshot", name="png",
|
||||
content_ref=f"draft:{RUN_ID}:{PNG_SHA256}", sha256=PNG_SHA256, content_type="image/png",
|
||||
byte_length=45)
|
||||
png = _open(session, MemoryStorage({f"draft:{RUN_ID}:{PNG_SHA256}": PNG_BYTES}),
|
||||
artifact_id="33333333-3333-4333-8333-333333333333")
|
||||
assert png.body == PNG_BYTES and png.content_length == 45
|
||||
_add_artifact(session, id="44444444-4444-4444-8444-444444444444", kind="evidence", name="json",
|
||||
content_ref=f"draft:{RUN_ID}:{JSON_SHA256}", sha256=JSON_SHA256, content_type="application/json",
|
||||
byte_length=11)
|
||||
payload = _open(session, MemoryStorage({f"draft:{RUN_ID}:{JSON_SHA256}": JSON_BYTES}),
|
||||
artifact_id="44444444-4444-4444-8444-444444444444")
|
||||
assert payload.body == JSON_BYTES
|
||||
assert payload.content_disposition.endswith(".json")
|
||||
# #endregion Test.ScenarioExecution.ArtifactContent.JsonPng
|
||||
|
||||
# #endregion Test.ScenarioExecution.ArtifactContent
|
||||
@@ -0,0 +1,298 @@
|
||||
# #region Test.ScenarioExecution.BaselineResolver [C:3] [TYPE Module] [SEMANTICS test,scenario,execution,baseline,pin,catalog]
|
||||
# @BRIEF Verify fail-closed BaselineSelectionPin resolve from published catalog bytes, not client digests.
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.BaselineResolver]
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.Runner.Start]
|
||||
# @TEST_CONTRACT: published CatalogRevision envelope + explicit set/version -> BaselineSelectionPin | D11 code
|
||||
# @TEST_FIXTURE published_catalog -> specs/044-dashboard-scenario-execution/fixtures/production-contract-refresh.json
|
||||
# @TEST_EDGE missing_field -> compare_to_baseline without baseline_set raises BASELINE_MISSING and creates no ScenarioRun
|
||||
# @TEST_EDGE invalid_type -> unpublished / mixed-release / ambiguous coordinate fail closed
|
||||
# @TEST_EDGE external_fail -> missing visual evidence bytes is BASELINE_EVIDENCE_UNAVAILABLE
|
||||
# @TEST_INVARIANT ScenarioExecution.BaselineResolver: pin fields are the published fixture identities, never resolve() output.
|
||||
# -> VERIFIED_BY: test_published_catalog_resolves_hardcoded_pin
|
||||
# @TEST_INVARIANT ScenarioExecution.Runner.Start: request hash includes the pin so a different generation cannot replay.
|
||||
# -> VERIFIED_BY: test_idempotency_rejects_different_pin
|
||||
from __future__ import annotations
|
||||
|
||||
from copy import deepcopy
|
||||
import json
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
from sqlalchemy import create_engine, event
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.pool import StaticPool
|
||||
|
||||
from src.services.dashboard_testing.execution.baseline_resolver import (
|
||||
BASELINE_AMBIGUOUS,
|
||||
BASELINE_EVIDENCE_UNAVAILABLE,
|
||||
BASELINE_MISSING,
|
||||
BASELINE_NOT_PUBLISHED,
|
||||
BASELINE_STALE,
|
||||
graph_has_baseline_refs,
|
||||
load_published_catalog,
|
||||
resolve_baseline_pin,
|
||||
)
|
||||
from src.services.dashboard_testing.execution.runner import start_run
|
||||
from src.services.dashboard_testing.scenario.templates import (
|
||||
ACTION_REGISTRY_VERSION,
|
||||
action_registry_fingerprint,
|
||||
)
|
||||
|
||||
_REFRESH_PATH = (
|
||||
Path(__file__).resolve().parents[5]
|
||||
/ "specs" / "044-dashboard-scenario-execution" / "fixtures" / "production-contract-refresh.json"
|
||||
)
|
||||
_REFRESH = json.loads(_REFRESH_PATH.read_text(encoding="utf-8"))
|
||||
EXPECTED_PIN = _REFRESH["baseline_pin"]
|
||||
_SCENARIO = "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaa1"
|
||||
_REVISION = "bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbb1"
|
||||
_GRAPH_WITH_REF = {
|
||||
"action_registry_version": ACTION_REGISTRY_VERSION,
|
||||
"action_registry_hash": action_registry_fingerprint(),
|
||||
"steps": [{"logical_step_id": "cmp", "tool": "assertion", "action": "compare_to_baseline"}],
|
||||
"dependencies": [],
|
||||
"baselines": {"revenue_sum": {"reference": "ss-prod-visual"}},
|
||||
}
|
||||
_GRAPH_NO_REF = {
|
||||
"action_registry_version": ACTION_REGISTRY_VERSION,
|
||||
"action_registry_hash": action_registry_fingerprint(),
|
||||
"steps": [{"logical_step_id": "assert", "tool": "assertion", "action": "structural_assert"}],
|
||||
"dependencies": [],
|
||||
}
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.BaselineResolver.Fixtures [C:1] [TYPE Function]
|
||||
# @BRIEF Published envelope identity plus CatalogRevision; YAML cannot populate these pin fields.
|
||||
def _published_snapshot() -> dict:
|
||||
pin = EXPECTED_PIN
|
||||
return {
|
||||
"baseline_set_id": pin["baseline_set_id"],
|
||||
"baseline_set_version": pin["baseline_set_version"],
|
||||
"release_id": pin["release_id"],
|
||||
"baseline_family": pin["baseline_family"],
|
||||
"catalog_digest": pin["catalog_digest"],
|
||||
"catalog_revision": deepcopy(_REFRESH["catalog_revision"]),
|
||||
}
|
||||
# #endregion Test.ScenarioExecution.BaselineResolver.Fixtures
|
||||
|
||||
|
||||
def _config_manager():
|
||||
return SimpleNamespace(
|
||||
get_environment=lambda environment_id: {
|
||||
"preprod": SimpleNamespace(stage="PREPROD", is_production=False),
|
||||
}.get(environment_id)
|
||||
)
|
||||
|
||||
|
||||
def _session_with_graph(graph: dict):
|
||||
engine = create_engine(
|
||||
"sqlite:///:memory:", poolclass=StaticPool, connect_args={"check_same_thread": False},
|
||||
)
|
||||
event.listen(engine, "connect", lambda c, _: c.execute("PRAGMA foreign_keys=ON"))
|
||||
from src.models.mapping import Base
|
||||
from src.models.scenario_registry import ScenarioRegistryEntry, ScenarioRevision
|
||||
|
||||
Base.metadata.create_all(engine)
|
||||
session = sessionmaker(bind=engine)()
|
||||
session.add(ScenarioRegistryEntry(
|
||||
scenario_id=_SCENARIO, scenario_key="baseline-resolver", name="Baseline resolver",
|
||||
dashboard_id=12, environment_ids=["preprod"], owner_id="user-1", owner_username="qa",
|
||||
current_revision_id=_REVISION, lifecycle_status="READY", validation_status="valid",
|
||||
))
|
||||
session.add(ScenarioRevision(
|
||||
revision_id=_REVISION, scenario_id=_SCENARIO, content_hash="d" * 64,
|
||||
graph_snapshot=graph, created_by="fixture", activation_status="current",
|
||||
))
|
||||
session.commit()
|
||||
return session
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.BaselineResolver.PublishedPin [C:2] [TYPE Function]
|
||||
# @BRIEF Hardcoded published catalog + set/version yields the spec fixture pin fields.
|
||||
def test_published_catalog_resolves_hardcoded_pin():
|
||||
pin = resolve_baseline_pin(
|
||||
graph_or_plan=_GRAPH_WITH_REF,
|
||||
baseline_set="ss-prod-visual",
|
||||
baseline_set_version="1",
|
||||
published_catalog=load_published_catalog(_published_snapshot()),
|
||||
)
|
||||
assert pin == EXPECTED_PIN
|
||||
assert pin["schema_version"] == 1
|
||||
assert pin["baseline_set_id"] == "ss-prod-visual"
|
||||
assert pin["baseline_set_version"] == "1"
|
||||
assert pin["catalog_revision_id"] == "11111111-1111-4111-8111-111111111111"
|
||||
assert pin["catalog_digest"] == "a" * 64
|
||||
assert pin["release_id"] == "11111111-1111-4111-8111-111111111111"
|
||||
assert pin["release_version"] == "v1.0.0"
|
||||
assert pin["release_commit_hash"] == "b" * 40
|
||||
assert pin["publication_commit_hash"] == "b" * 40
|
||||
assert pin["baseline_family"] == "a" * 64
|
||||
assert pin["entries"][0]["baseline_id"] == "44444444-4444-4444-8444-444444444444"
|
||||
assert pin["entries"][0]["kind"] == "visual"
|
||||
assert pin["entries"][0]["expected_image_sha256"] == "a" * 64
|
||||
assert pin["entries"][1]["baseline_id"] == "77777777-7777-4777-8777-777777777777"
|
||||
assert pin["entries"][1]["kind"] == "metric"
|
||||
assert pin["entries"][1]["expected_image_sha256"] is None
|
||||
# #endregion Test.ScenarioExecution.BaselineResolver.PublishedPin
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.BaselineResolver.MissingSelector [C:2] [TYPE Function]
|
||||
# @BRIEF compare_to_baseline without set/version is BASELINE_MISSING and leaves no ScenarioRun row.
|
||||
def test_compare_without_baseline_set_is_missing_and_creates_no_run():
|
||||
assert graph_has_baseline_refs(_GRAPH_WITH_REF) is True
|
||||
with pytest.raises(ValueError, match=BASELINE_MISSING):
|
||||
resolve_baseline_pin(graph_or_plan=_GRAPH_WITH_REF, baseline_set=None, baseline_set_version=None)
|
||||
|
||||
session = _session_with_graph(_GRAPH_WITH_REF)
|
||||
from src.models.scenario_run import ScenarioRun
|
||||
before = session.query(ScenarioRun).count()
|
||||
with pytest.raises(ValueError, match=BASELINE_MISSING):
|
||||
start_run(
|
||||
session, _SCENARIO, _REVISION, {}, "preprod",
|
||||
actor="qa", idempotency_key="missing-set",
|
||||
config_manager=_config_manager(),
|
||||
published_catalog=_published_snapshot(),
|
||||
)
|
||||
assert session.query(ScenarioRun).count() == before
|
||||
session.close()
|
||||
# #endregion Test.ScenarioExecution.BaselineResolver.MissingSelector
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.BaselineResolver.Unpublished [C:2] [TYPE Function]
|
||||
# @BRIEF Working-tree YAML and non-published CatalogRevision cannot resolve.
|
||||
def test_unpublished_and_yaml_catalog_are_not_published():
|
||||
yaml_shaped = {"schema_version": 1, "dashboard": {"id": 12}, "entries": []}
|
||||
with pytest.raises(ValueError, match=BASELINE_NOT_PUBLISHED):
|
||||
resolve_baseline_pin(
|
||||
graph_or_plan=_GRAPH_WITH_REF,
|
||||
baseline_set="ss-prod-visual", baseline_set_version="1",
|
||||
published_catalog=load_published_catalog(yaml_shaped),
|
||||
)
|
||||
unpublished = _published_snapshot()
|
||||
unpublished["catalog_revision"]["publication"] = {
|
||||
"state": "materialized",
|
||||
"publication_id": "11111111-1111-4111-8111-111111111111",
|
||||
"expected_branch_head": "b" * 40,
|
||||
"commit_hash": None,
|
||||
"error_code": None,
|
||||
"published_receipt_id": None,
|
||||
}
|
||||
with pytest.raises(ValueError, match=BASELINE_NOT_PUBLISHED):
|
||||
resolve_baseline_pin(
|
||||
graph_or_plan=_GRAPH_WITH_REF,
|
||||
baseline_set="ss-prod-visual", baseline_set_version="1",
|
||||
published_catalog=unpublished,
|
||||
)
|
||||
# #endregion Test.ScenarioExecution.BaselineResolver.Unpublished
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.BaselineResolver.Stale [C:2] [TYPE Function]
|
||||
# @BRIEF Digest mismatch and mixed release are BASELINE_STALE, never a silent pin.
|
||||
def test_stale_fingerprint_and_mixed_release():
|
||||
digest_mismatch = _published_snapshot()
|
||||
digest_mismatch["catalog_digest"] = "b" * 64
|
||||
with pytest.raises(ValueError, match=BASELINE_STALE):
|
||||
resolve_baseline_pin(
|
||||
graph_or_plan=_GRAPH_WITH_REF,
|
||||
baseline_set="ss-prod-visual", baseline_set_version="1",
|
||||
published_catalog=digest_mismatch,
|
||||
)
|
||||
mixed = _published_snapshot()
|
||||
mixed["catalog_revision"]["entry_revisions"][1]["entry"]["release_commit_hash"] = "c" * 40
|
||||
with pytest.raises(ValueError, match=BASELINE_STALE):
|
||||
resolve_baseline_pin(
|
||||
graph_or_plan=_GRAPH_WITH_REF,
|
||||
baseline_set="ss-prod-visual", baseline_set_version="1",
|
||||
published_catalog=mixed,
|
||||
)
|
||||
# #endregion Test.ScenarioExecution.BaselineResolver.Stale
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.BaselineResolver.Ambiguous [C:2] [TYPE Function]
|
||||
# @BRIEF Two approved entries at the same coordinate fail BASELINE_AMBIGUOUS, never first-match.
|
||||
def test_ambiguous_coordinate_is_rejected():
|
||||
snapshot = _published_snapshot()
|
||||
clone = deepcopy(snapshot["catalog_revision"]["entry_revisions"][0])
|
||||
clone["baseline_id"] = "55555555-5555-4555-8555-555555555555"
|
||||
clone["baseline_revision_id"] = "55555555-5555-4555-8555-555555555556"
|
||||
clone["entry"]["baseline_id"] = clone["baseline_id"]
|
||||
snapshot["catalog_revision"]["entry_revisions"].append(clone)
|
||||
with pytest.raises(ValueError, match=BASELINE_AMBIGUOUS):
|
||||
resolve_baseline_pin(
|
||||
graph_or_plan=_GRAPH_WITH_REF,
|
||||
baseline_set="ss-prod-visual", baseline_set_version="1",
|
||||
published_catalog=snapshot,
|
||||
)
|
||||
# #endregion Test.ScenarioExecution.BaselineResolver.Ambiguous
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.BaselineResolver.Evidence [C:2] [TYPE Function]
|
||||
# @BRIEF Missing visual image digest is BASELINE_EVIDENCE_UNAVAILABLE.
|
||||
def test_missing_visual_evidence_is_unavailable():
|
||||
snapshot = _published_snapshot()
|
||||
snapshot["catalog_revision"]["entry_revisions"][0]["entry"]["expected_image_sha256"] = None
|
||||
with pytest.raises(ValueError, match=BASELINE_EVIDENCE_UNAVAILABLE):
|
||||
resolve_baseline_pin(
|
||||
graph_or_plan=_GRAPH_WITH_REF,
|
||||
baseline_set="ss-prod-visual", baseline_set_version="1",
|
||||
published_catalog=snapshot,
|
||||
)
|
||||
# #endregion Test.ScenarioExecution.BaselineResolver.Evidence
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.BaselineResolver.NullPin [C:2] [TYPE Function]
|
||||
# @BRIEF Graph without baseline refs and no selector stores a null pin and still starts.
|
||||
def test_graph_without_baseline_refs_allows_null_pin_start():
|
||||
pin = resolve_baseline_pin(graph_or_plan=_GRAPH_NO_REF, baseline_set=None, baseline_set_version=None)
|
||||
assert pin is None
|
||||
session = _session_with_graph(_GRAPH_NO_REF)
|
||||
run = start_run(
|
||||
session, _SCENARIO, _REVISION, {}, "preprod",
|
||||
actor="qa", idempotency_key="null-pin",
|
||||
config_manager=_config_manager(),
|
||||
)
|
||||
assert run.status == "queued"
|
||||
assert run.runner_plan["baseline_pin"] is None
|
||||
assert run.target_snapshot["baseline_pin"] is None
|
||||
session.close()
|
||||
# #endregion Test.ScenarioExecution.BaselineResolver.NullPin
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.BaselineResolver.Idempotency [C:2] [TYPE Function]
|
||||
# @BRIEF Same idempotency key with a different resolved pin is IDEMPOTENCY_KEY_REUSED.
|
||||
def test_idempotency_rejects_different_pin():
|
||||
session = _session_with_graph(_GRAPH_WITH_REF)
|
||||
first = start_run(
|
||||
session, _SCENARIO, _REVISION, {}, "preprod",
|
||||
actor="qa", idempotency_key="pin-key",
|
||||
config_manager=_config_manager(),
|
||||
baseline_set="ss-prod-visual", baseline_set_version="1",
|
||||
published_catalog=_published_snapshot(),
|
||||
)
|
||||
assert first.runner_plan["baseline_pin"] == EXPECTED_PIN
|
||||
assert first.target_snapshot["baseline_pin"] == EXPECTED_PIN
|
||||
replay = start_run(
|
||||
session, _SCENARIO, _REVISION, {}, "preprod",
|
||||
actor="qa", idempotency_key="pin-key",
|
||||
config_manager=_config_manager(),
|
||||
baseline_set="ss-prod-visual", baseline_set_version="1",
|
||||
published_catalog=_published_snapshot(),
|
||||
)
|
||||
assert replay.id == first.id
|
||||
other = _published_snapshot()
|
||||
other["baseline_set_version"] = "2"
|
||||
other["catalog_digest"] = "c" * 64
|
||||
other["catalog_revision"]["catalog_digest"] = "c" * 64
|
||||
with pytest.raises(ValueError, match="IDEMPOTENCY_KEY_REUSED"):
|
||||
start_run(
|
||||
session, _SCENARIO, _REVISION, {}, "preprod",
|
||||
actor="qa", idempotency_key="pin-key",
|
||||
config_manager=_config_manager(),
|
||||
baseline_set="ss-prod-visual", baseline_set_version="2",
|
||||
published_catalog=other,
|
||||
)
|
||||
session.close()
|
||||
# #endregion Test.ScenarioExecution.BaselineResolver.Idempotency
|
||||
|
||||
# #endregion Test.ScenarioExecution.BaselineResolver
|
||||
@@ -0,0 +1,335 @@
|
||||
# #region Test.ScenarioExecution.DecisionPolicy [C:4] [TYPE Module] [SEMANTICS test,scenario,execution,decision,policy,truth-table]
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.DecisionPolicy]
|
||||
# @TEST_EDGE html_label_scope -> every row of decision-policy.md has a hardcoded case
|
||||
# @TEST_INVARIANT ScenarioExecution.DecisionPolicy: first-match 15-row table partitions the domain;
|
||||
# every decision has >=1 reason_code; repeated inputs are deterministic.
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest # noqa: F401
|
||||
|
||||
from src.services.dashboard_testing.execution.decision_policy import (
|
||||
BASELINE_SEMANTIC_V1,
|
||||
DecisionPolicy,
|
||||
EvaluationInput,
|
||||
EvidenceManifest,
|
||||
PolicyDecision,
|
||||
StepPolicyInputs,
|
||||
ComparisonInput,
|
||||
decide_step_outcome,
|
||||
map_human_disposition,
|
||||
policy_digest,
|
||||
policy_inputs_from_outcome,
|
||||
verified_evidence_refs,
|
||||
)
|
||||
|
||||
|
||||
def _inputs(
|
||||
*,
|
||||
comparisons: list[ComparisonInput] | None = None,
|
||||
evidence: EvidenceManifest | None = None,
|
||||
evaluation: EvaluationInput | None = None,
|
||||
**overrides,
|
||||
) -> StepPolicyInputs:
|
||||
defaults: dict = {
|
||||
"winning_attempt_exists": True,
|
||||
"cancelled_or_late": False,
|
||||
"identity_valid": True,
|
||||
"evidence": evidence or EvidenceManifest(),
|
||||
"comparisons": comparisons or [ComparisonInput(comparison_id="cmp-1", status="pass")],
|
||||
"evaluation": evaluation,
|
||||
}
|
||||
defaults.update(overrides)
|
||||
return StepPolicyInputs(**defaults)
|
||||
|
||||
|
||||
def _required_policy(**overrides) -> DecisionPolicy:
|
||||
data = {"evaluation_mode": "required"}
|
||||
data.update(overrides)
|
||||
return BASELINE_SEMANTIC_V1.model_copy(update=data)
|
||||
|
||||
|
||||
def _evaluation(**overrides) -> EvaluationInput:
|
||||
data: dict = {
|
||||
"evaluation_id": "eval-1",
|
||||
"status": "succeeded",
|
||||
"verdict": "pass",
|
||||
"confidence": 0.8,
|
||||
}
|
||||
data.update(overrides)
|
||||
return EvaluationInput(**data)
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.DecisionPolicy.Row1 [C:2] [TYPE Function]
|
||||
# @BRIEF Cancelled/lost attempts publish no winning outcome.
|
||||
def test_row1_cancelled_and_late_publish_none():
|
||||
cancelled = decide_step_outcome(_inputs(cancelled_or_late=True), BASELINE_SEMANTIC_V1)
|
||||
assert cancelled.status is None and cancelled.reason_codes == ["CANCELLED"]
|
||||
lost = decide_step_outcome(_inputs(winning_attempt_exists=False), BASELINE_SEMANTIC_V1)
|
||||
assert lost.status is None and lost.reason_codes == ["LATE_RESPONSE"]
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy.Row1
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.DecisionPolicy.Row2 [C:2] [TYPE Function]
|
||||
# @BRIEF Invalid policy/spec identity blocks before any comparison authority.
|
||||
def test_row2_invalid_identity_blocks():
|
||||
decision = decide_step_outcome(_inputs(identity_valid=False), BASELINE_SEMANTIC_V1)
|
||||
assert (decision.status, decision.reason_codes) == ("blocked", ["POLICY_INVALID"])
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy.Row2
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.DecisionPolicy.Row3 [C:2] [TYPE Function]
|
||||
# @BRIEF Required evidence absence/corruption/foreign owner blocks with EVIDENCE codes.
|
||||
def test_row3_required_evidence_blocks():
|
||||
missing = decide_step_outcome(
|
||||
_inputs(evidence=EvidenceManifest(required=True, refs=[], digests_valid=True)), BASELINE_SEMANTIC_V1
|
||||
)
|
||||
assert (missing.status, missing.reason_codes) == ("blocked", ["EVIDENCE_UNAVAILABLE"])
|
||||
corrupt = decide_step_outcome(
|
||||
_inputs(evidence=EvidenceManifest(required=True, refs=["r"], digests_valid=False)), BASELINE_SEMANTIC_V1
|
||||
)
|
||||
assert (corrupt.status, corrupt.reason_codes) == ("blocked", ["EVIDENCE_CORRUPT"])
|
||||
foreign = decide_step_outcome(
|
||||
_inputs(evidence=EvidenceManifest(required=True, refs=["r"], digests_valid=True, owner_verified=False)),
|
||||
BASELINE_SEMANTIC_V1,
|
||||
)
|
||||
assert (foreign.status, foreign.reason_codes) == ("blocked", ["EVIDENCE_CORRUPT"])
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy.Row3
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.DecisionPolicy.Rows46 [C:2] [TYPE Function]
|
||||
# @BRIEF Immutability and comparison fail dominate model claims; baseline block family is row 6.
|
||||
def test_row4_immutability_violation_fails():
|
||||
decision = decide_step_outcome(
|
||||
_inputs(comparisons=[ComparisonInput(comparison_id="c", status="immutability_violation")]),
|
||||
BASELINE_SEMANTIC_V1,
|
||||
)
|
||||
assert (decision.status, decision.reason_codes) == ("failed", ["IMMUTABILITY_VIOLATION"])
|
||||
|
||||
|
||||
def test_row5_comparison_fail_ignores_model_pass_claims():
|
||||
decision = decide_step_outcome(
|
||||
_inputs(
|
||||
comparisons=[ComparisonInput(comparison_id="c", status="fail")],
|
||||
evidence=EvidenceManifest(required=True, refs=["r"], digests_valid=True),
|
||||
),
|
||||
BASELINE_SEMANTIC_V1,
|
||||
)
|
||||
# A passing model (advisory/required) must not erase a deterministic comparison fail.
|
||||
assert (decision.status, decision.reason_codes) == ("failed", ["BASELINE_MISMATCH"])
|
||||
|
||||
|
||||
def test_row6_baseline_block_family():
|
||||
expected = {
|
||||
"missing_baseline": "BASELINE_MISSING",
|
||||
"stale_baseline": "BASELINE_STALE",
|
||||
"stale_visual_baseline": "BASELINE_STALE",
|
||||
"ambiguous_baseline": "BASELINE_AMBIGUOUS",
|
||||
"invalidated_baseline": "BASELINE_INVALIDATED",
|
||||
}
|
||||
for status, code in expected.items():
|
||||
decision = decide_step_outcome(
|
||||
_inputs(comparisons=[ComparisonInput(comparison_id="c", status=status)]), BASELINE_SEMANTIC_V1
|
||||
)
|
||||
assert (decision.status, decision.reason_codes) == ("blocked", [code])
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy.Rows46
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.DecisionPolicy.Row7 [C:2] [TYPE Function]
|
||||
# @BRIEF Comparison inconclusive family stays inconclusive.
|
||||
def test_row7_inconclusive_family():
|
||||
for status in ("inconclusive", "source_error", "permission_denied"):
|
||||
decision = decide_step_outcome(
|
||||
_inputs(comparisons=[ComparisonInput(comparison_id="c", status=status)]), BASELINE_SEMANTIC_V1
|
||||
)
|
||||
assert (decision.status, decision.reason_codes) == ("inconclusive", ["COMPARISON_INCONCLUSIVE"])
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy.Row7
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.DecisionPolicy.Row8 [C:2] [TYPE Function]
|
||||
# @BRIEF Pass + disabled/advisory evaluation mode: deterministic pass is the outcome; model is annotation.
|
||||
def test_row8_disabled_and_advisory_pass():
|
||||
for mode in ("disabled", "advisory"):
|
||||
policy = BASELINE_SEMANTIC_V1.model_copy(update={"evaluation_mode": mode})
|
||||
decision = decide_step_outcome(_inputs(), policy)
|
||||
assert (decision.status, decision.reason_codes) == ("passed", ["BASELINE_PASS"])
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy.Row8
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.DecisionPolicy.Row9 [C:2] [TYPE Function]
|
||||
# @BRIEF Pass + required but evaluation missing/failed -> inconclusive EVALUATION_UNAVAILABLE.
|
||||
def test_row9_required_evaluation_missing_or_failed():
|
||||
missing = decide_step_outcome(_inputs(), _required_policy())
|
||||
assert (missing.status, missing.reason_codes) == ("inconclusive", ["EVALUATION_UNAVAILABLE"])
|
||||
for status in ("provider_error", "parser_error", "budget_exceeded", "cancelled", "timed_out"):
|
||||
decision = decide_step_outcome(
|
||||
_inputs(evaluation=_evaluation(status=status)), _required_policy()
|
||||
)
|
||||
assert decision.status == "inconclusive"
|
||||
assert decision.reason_codes == ["EVALUATION_UNAVAILABLE"]
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy.Row9
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.DecisionPolicy.Row10 [C:2] [TYPE Function]
|
||||
# @BRIEF Below-threshold confidence is low; equality with threshold is high.
|
||||
def test_row10_low_confidence_and_threshold_equality():
|
||||
low = decide_step_outcome(
|
||||
_inputs(evaluation=_evaluation(verdict="fail", confidence=0.69)), _required_policy()
|
||||
)
|
||||
assert (low.status, low.reason_codes) == ("inconclusive", ["LOW_CONFIDENCE"])
|
||||
# confidence == threshold is high: a clean high-confidence pass reaches row 14, not row 10.
|
||||
at_threshold = decide_step_outcome(
|
||||
_inputs(evaluation=_evaluation(verdict="pass", confidence=0.7)), _required_policy()
|
||||
)
|
||||
assert at_threshold.status == "passed"
|
||||
assert at_threshold.reason_codes == ["BASELINE_AND_SEMANTIC_PASS"]
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy.Row10
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.DecisionPolicy.Row11 [C:2] [TYPE Function]
|
||||
# @BRIEF High-confidence inconclusive verdict -> MODEL_INCONCLUSIVE.
|
||||
def test_row11_high_confidence_inconclusive():
|
||||
decision = decide_step_outcome(
|
||||
_inputs(evaluation=_evaluation(verdict="inconclusive", confidence=0.9)), _required_policy()
|
||||
)
|
||||
assert (decision.status, decision.reason_codes) == ("inconclusive", ["MODEL_INCONCLUSIVE"])
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy.Row11
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.DecisionPolicy.Row12 [C:2] [TYPE Function]
|
||||
# @BRIEF Agreement family: disagreement with the deterministic criterion and contradictory claims stay inconclusive.
|
||||
def test_row12_disagreement_and_contradictions():
|
||||
disagreement = decide_step_outcome(
|
||||
_inputs(evaluation=_evaluation(verdict="pass", conflicts_with_deterministic_criterion=True)),
|
||||
_required_policy(),
|
||||
)
|
||||
assert (disagreement.status, disagreement.reason_codes) == ("inconclusive", ["EVALUATION_DISAGREEMENT"])
|
||||
|
||||
contradictory_pass = decide_step_outcome(
|
||||
_inputs(evaluation=_evaluation(verdict="pass", has_error_critical_findings=True)), _required_policy()
|
||||
)
|
||||
assert (contradictory_pass.status, contradictory_pass.reason_codes) == (
|
||||
"inconclusive", ["EVALUATION_CONTRADICTORY"]
|
||||
)
|
||||
|
||||
unsupported_fail = decide_step_outcome(
|
||||
_inputs(evaluation=_evaluation(verdict="fail", fail_supported_by_findings=False)), _required_policy()
|
||||
)
|
||||
assert (unsupported_fail.status, unsupported_fail.reason_codes) == ("inconclusive", ["EVALUATION_CONTRADICTORY"])
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy.Row12
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.DecisionPolicy.Row13 [C:2] [TYPE Function]
|
||||
# @BRIEF High-confidence fail supported by declared semantic findings -> SEMANTIC_FAILURE.
|
||||
def test_row13_semantic_failure():
|
||||
decision = decide_step_outcome(
|
||||
_inputs(evaluation=_evaluation(verdict="fail", confidence=0.95, fail_supported_by_findings=True)),
|
||||
_required_policy(),
|
||||
)
|
||||
assert (decision.status, decision.reason_codes) == ("failed", ["SEMANTIC_FAILURE"])
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy.Row13
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.DecisionPolicy.Row14 [C:2] [TYPE Function]
|
||||
# @BRIEF High-confidence clean pass -> BASELINE_AND_SEMANTIC_PASS.
|
||||
def test_row14_baseline_and_semantic_pass():
|
||||
decision = decide_step_outcome(
|
||||
_inputs(evaluation=_evaluation(verdict="pass", confidence=0.95)), _required_policy()
|
||||
)
|
||||
assert (decision.status, decision.reason_codes) == ("passed", ["BASELINE_AND_SEMANTIC_PASS"])
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy.Row14
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.DecisionPolicy.Row15 [C:2] [TYPE Function]
|
||||
# @BRIEF Garbage envelopes and malformed evaluation numbers block with POLICY_INPUT_INVALID, never crash.
|
||||
def test_row15_garbage_and_malformed_inputs():
|
||||
malformed_envelope = decide_step_outcome({"winning_attempt_exists": True}, {"policy_id": "nope", "version": "9"})
|
||||
assert (malformed_envelope.status, malformed_envelope.reason_codes) == ("blocked", ["POLICY_INPUT_INVALID"])
|
||||
too_low = decide_step_outcome(
|
||||
_inputs(evaluation=_evaluation(verdict="pass", confidence=-0.1)), _required_policy()
|
||||
)
|
||||
assert (too_low.status, too_low.reason_codes) == ("blocked", ["POLICY_INPUT_INVALID"])
|
||||
missing_verdict = decide_step_outcome(
|
||||
_inputs(evaluation=_evaluation(verdict=None, confidence=0.9)), _required_policy()
|
||||
)
|
||||
assert (missing_verdict.status, missing_verdict.reason_codes) == ("blocked", ["POLICY_INPUT_INVALID"])
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy.Row15
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.DecisionPolicy.Determinism [C:2] [TYPE Function]
|
||||
# @BRIEF Repeated canonical inputs yield identical decisions and never an empty reason_codes list.
|
||||
def test_determinism_and_nonempty_reason_codes():
|
||||
candidates = [
|
||||
_inputs(cancelled_or_late=True),
|
||||
_inputs(identity_valid=False),
|
||||
_inputs(evaluation=_evaluation(verdict="pass", confidence=0.95)),
|
||||
_inputs(comparisons=[ComparisonInput(comparison_id="c", status="immutability_violation")]),
|
||||
_inputs(evaluation=_evaluation(verdict="fail", confidence=0.9, fail_supported_by_findings=True)),
|
||||
]
|
||||
for candidate in candidates:
|
||||
first = decide_step_outcome(candidate, BASELINE_SEMANTIC_V1)
|
||||
second = decide_step_outcome(candidate, BASELINE_SEMANTIC_V1)
|
||||
assert first.model_dump() == second.model_dump()
|
||||
assert first.reason_codes, "every decision must carry at least one reason_code"
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy.Determinism
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.DecisionPolicy.Human [C:2] [TYPE Function]
|
||||
# @BRIEF HumanCheckpoint v1 mapping: confirm/pass pass, false_positive/inconclusive stay non-pass.
|
||||
def test_human_disposition_mapping():
|
||||
assert map_human_disposition("confirm") == ("passed", ["HUMAN_CONFIRMED"])
|
||||
assert map_human_disposition("pass") == ("passed", ["HUMAN_CONFIRMED"])
|
||||
assert map_human_disposition("false_positive") == ("inconclusive", ["HUMAN_FALSE_POSITIVE"])
|
||||
assert map_human_disposition("inconclusive") == ("inconclusive", ["HUMAN_INCONCLUSIVE"])
|
||||
for invalid in ("fail", "unknown"):
|
||||
with pytest.raises(ValueError):
|
||||
map_human_disposition(invalid)
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy.Human
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.DecisionPolicy.Adapter [C:2] [TYPE Function]
|
||||
# @BRIEF Outcome adapter gates the mapper to normative tools/comparison payloads and mirrors integrity.
|
||||
def test_adapter_skips_non_normative_tools():
|
||||
inputs = policy_inputs_from_outcome(
|
||||
{"status": "passed", "artifact_refs": []},
|
||||
{"tool": "browser", "logical_step_id": "navigate"},
|
||||
None,
|
||||
)
|
||||
assert inputs is None
|
||||
|
||||
|
||||
def test_adapter_synthesizes_comparison_for_normative_tool():
|
||||
inputs = policy_inputs_from_outcome(
|
||||
{"status": "passed", "artifact_refs": ["shot-1"]},
|
||||
{"tool": "screenshot", "logical_step_id": "capture"},
|
||||
None,
|
||||
)
|
||||
assert inputs is not None
|
||||
assert inputs.evidence.required is True
|
||||
assert inputs.evidence.digests_valid is True
|
||||
assert inputs.comparisons[0].status == "pass"
|
||||
|
||||
|
||||
def test_adapter_marks_integrity_corruption():
|
||||
inputs = policy_inputs_from_outcome(
|
||||
{"status": "passed", "artifact_refs": ["shot-1"]},
|
||||
{"tool": "screenshot", "logical_step_id": "capture"},
|
||||
{"status": "inconclusive", "reason_code": "ARTIFACT_DIGEST_MISSING", "unregistered_refs": ["shot-1"]},
|
||||
)
|
||||
assert inputs is not None
|
||||
assert inputs.evidence.digests_valid is False
|
||||
assert verified_evidence_refs(
|
||||
{"artifact_refs": ["shot-1"]}, {"unregistered_refs": ["shot-1"]}
|
||||
) == []
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy.Adapter
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.DecisionPolicy.PolicyDigest [C:2] [TYPE Function]
|
||||
# @BRIEF Policy digest is stable for identical shapes and changes when the policy changes.
|
||||
def test_policy_digest_stability():
|
||||
assert policy_digest(BASELINE_SEMANTIC_V1) == policy_digest(BASELINE_SEMANTIC_V1)
|
||||
assert policy_digest(BASELINE_SEMANTIC_V1) != policy_digest(
|
||||
BASELINE_SEMANTIC_V1.model_copy(update={"evaluation_mode": "required"})
|
||||
)
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy.PolicyDigest
|
||||
|
||||
# #endregion Test.ScenarioExecution.DecisionPolicy
|
||||
@@ -60,6 +60,41 @@ def configured_scenario_execution_environments(monkeypatch):
|
||||
# #endregion Test.ScenarioRegistry.Conftest.EnvironmentPolicy
|
||||
|
||||
|
||||
# #region Test.ScenarioRegistry.Conftest.PublishedBaselinePin [C:2] [TYPE Function] [SEMANTICS test,scenario,execution,baseline,pin,fixture]
|
||||
# @BRIEF Inject the published 044 catalog snapshot and a hardcoded BaselineSelectionPin for registry start_run.
|
||||
# @TEST_FIXTURE production-contract-refresh.json -> specs/044-dashboard-scenario-execution/fixtures/
|
||||
# @RATIONALE Registry tests exercise walker/lifecycle, not catalog publication. They share graph.json
|
||||
# which legally requires a pin; the [EXT] seam supplies the published fixture bytes.
|
||||
# @REJECTED Making absent baseline_set legal on graphs with compare_to_baseline — production
|
||||
# ScenarioBaselineResolver stays fail-closed (D11).
|
||||
@pytest.fixture(autouse=True)
|
||||
def published_baseline_pin_for_registry_starts(monkeypatch):
|
||||
refresh_path = (
|
||||
Path(__file__).resolve().parents[5]
|
||||
/ "specs" / "044-dashboard-scenario-execution" / "fixtures" / "production-contract-refresh.json"
|
||||
)
|
||||
refresh = json.loads(refresh_path.read_text(encoding="utf-8"))
|
||||
pin = refresh["baseline_pin"]
|
||||
snapshot = {
|
||||
"baseline_set_id": pin["baseline_set_id"],
|
||||
"baseline_set_version": pin["baseline_set_version"],
|
||||
"release_id": pin["release_id"],
|
||||
"baseline_family": pin["baseline_family"],
|
||||
"catalog_digest": pin["catalog_digest"],
|
||||
"catalog_revision": refresh["catalog_revision"],
|
||||
}
|
||||
monkeypatch.setattr(
|
||||
"src.services.dashboard_testing.execution.runner.load_published_catalog",
|
||||
lambda injected=None: injected if injected is not None else snapshot,
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"src.services.dashboard_testing.execution.runner.resolve_baseline_pin",
|
||||
lambda **_kwargs: pin,
|
||||
)
|
||||
return pin
|
||||
# #endregion Test.ScenarioRegistry.Conftest.PublishedBaselinePin
|
||||
|
||||
|
||||
def _make_request_with_capture(agent_run_id: str, capture_artifact_id: str) -> CandidateRequest:
|
||||
return CandidateRequest.model_construct(
|
||||
environment_id="ss-preprod",
|
||||
|
||||
@@ -0,0 +1,204 @@
|
||||
# #region Test.ScenarioExecution.AgentEvaluationStore [C:3] [TYPE Module] [SEMANTICS test,scenario,evaluation,store,evidence]
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.AgentEvaluation]
|
||||
# @TEST_INVARIANT ScenarioExecution.AgentEvaluation: append-only identity CAS, P0-1 evidence binding
|
||||
# (owner/run/step/attempt/digest/MIME/length, raw available+active).
|
||||
# @TEST_INVARIANT ScenarioExecution.AgentEvaluation.Evidence: manifest/findings may bind prior-step
|
||||
# same-run active artifacts; raw-response stays this step+attempt; foreign-run rejected.
|
||||
# -> VERIFIED_BY: prior_step_manifest_accepted, foreign_run_manifest_rejected, prior_step_raw_rejected
|
||||
from __future__ import annotations
|
||||
|
||||
import uuid
|
||||
from datetime import UTC, datetime
|
||||
|
||||
import pytest
|
||||
|
||||
from src.models.scenario_artifact import ScenarioArtifact
|
||||
from src.models.scenario_evaluation import AgentEvaluation as AgentEvaluationRow
|
||||
from src.models.scenario_run import ScenarioRun, ScenarioStepRun
|
||||
from src.services.dashboard_testing.execution.agent_evaluation import (
|
||||
AgentEvaluation,
|
||||
persist_agent_evaluation,
|
||||
validate_evaluation_evidence,
|
||||
)
|
||||
|
||||
_NOW = datetime(2026, 9, 10, 0, 0, 0, tzinfo=UTC)
|
||||
_SCENARIO = "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaa1"
|
||||
|
||||
|
||||
def _eval_dict(**overrides) -> dict:
|
||||
data: dict = {
|
||||
"schema_version": 1,
|
||||
"evaluation_id": str(uuid.uuid4()),
|
||||
"scenario_run_id": _SCENARIO,
|
||||
"logical_step_id": "step-eval-1",
|
||||
"attempt": 1,
|
||||
"operation_id": str(uuid.uuid4()),
|
||||
"evaluation_spec_hash": "a" * 64,
|
||||
"provider_id": "llm-provider-a",
|
||||
"provider_version": "1",
|
||||
"model_id": "vision-model",
|
||||
"model_version": "2026-08",
|
||||
"prompt_template_id": "agent-evaluation-prompt",
|
||||
"prompt_template_version": "1.0.0",
|
||||
"prompt_template_hash": "b" * 64,
|
||||
"output_schema_hash": "c" * 64,
|
||||
"input_manifest_hash": "d" * 64,
|
||||
"input_manifest": [{"artifact_id": "artifact-1", "sha256": "e" * 64, "content_type": "image/jpeg", "byte_length": 100, "role": "actual"}],
|
||||
"baseline_pin": {"catalog_revision_id": str(uuid.uuid4())},
|
||||
"comparison_ids": [str(uuid.uuid4())],
|
||||
"status": "succeeded",
|
||||
"verdict": "pass",
|
||||
"confidence": 0.93,
|
||||
"findings": [{"finding_id": "f-1", "severity": "info", "message": "matches", "evidence_artifact_ids": ["artifact-1"], "criterion_id": "crit-visual", "criterion_kind": "semantic"}],
|
||||
"reason_codes": ["EVALUATION_PASS"],
|
||||
"raw_response_artifact_ref": "artifact-1",
|
||||
"raw_response_sha256": "e" * 64,
|
||||
"trust_policy_hash": "f" * 64,
|
||||
"usage": {"input_tokens": 100, "output_tokens": 10, "cost_amount": "0.01", "currency": "USD", "pricing_version": "1"},
|
||||
"started_at": _NOW,
|
||||
"finished_at": _NOW,
|
||||
}
|
||||
data.update(overrides)
|
||||
return data
|
||||
|
||||
|
||||
def test_persist_and_duplicate_identity_cas(seeded_execution):
|
||||
db = seeded_execution
|
||||
record = AgentEvaluation.model_validate(_eval_dict())
|
||||
row = persist_agent_evaluation(db, record)
|
||||
assert isinstance(row, AgentEvaluationRow)
|
||||
db.flush()
|
||||
with pytest.raises(ValueError, match="EVALUATION_ALREADY_EXISTS"):
|
||||
persist_agent_evaluation(db, record)
|
||||
|
||||
|
||||
def _seed_evidence(db):
|
||||
run = ScenarioRun(
|
||||
id=str(uuid.uuid4()), scenario_id=_SCENARIO, scenario_revision_id="bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbb1",
|
||||
scenario_content_hash="a" * 64, environment_id="env-preprod-02", idempotency_key=f"ev-{uuid.uuid4()}",
|
||||
)
|
||||
db.add(run)
|
||||
db.flush()
|
||||
step = ScenarioStepRun(run_id=run.id, logical_step_id="step-eval-1", step_position=0, attempt=1)
|
||||
db.add(step)
|
||||
artifact = ScenarioArtifact(
|
||||
owner_type="scenario_run", owner_id=run.id, kind="screenshot", name="s",
|
||||
content_ref="artifact-1", sha256="e" * 64, content_type="image/jpeg", byte_length=100,
|
||||
logical_step_id="step-eval-1", attempt=1, is_active=True,
|
||||
)
|
||||
db.add(artifact)
|
||||
db.flush()
|
||||
return run
|
||||
|
||||
|
||||
def test_evidence_binding_accepts_matching_manifest_and_raw(seeded_execution):
|
||||
db = seeded_execution
|
||||
run = _seed_evidence(db)
|
||||
validate_evaluation_evidence(
|
||||
db, run_id=run.id, logical_step_id="step-eval-1", attempt=1,
|
||||
input_manifest=[{"artifact_id": "artifact-1", "sha256": "e" * 64, "content_type": "image/jpeg", "byte_length": 100}],
|
||||
raw_response_artifact_ref="artifact-1", raw_response_sha256="e" * 64, findings=[], succeeded=True,
|
||||
)
|
||||
|
||||
|
||||
def test_evidence_binding_rejects_unknown_artifact(seeded_execution):
|
||||
db = seeded_execution
|
||||
run = _seed_evidence(db)
|
||||
with pytest.raises(ValueError, match="EVALUATION_EVIDENCE_NOT_FOUND"):
|
||||
validate_evaluation_evidence(
|
||||
db, run_id=run.id, logical_step_id="step-eval-1", attempt=1,
|
||||
input_manifest=[{"artifact_id": "ghost", "sha256": "e" * 64, "content_type": "image/jpeg", "byte_length": 100}],
|
||||
raw_response_artifact_ref=None, raw_response_sha256=None, findings=[], succeeded=False,
|
||||
)
|
||||
|
||||
|
||||
def test_evidence_binding_rejects_metadata_mismatch(seeded_execution):
|
||||
db = seeded_execution
|
||||
run = _seed_evidence(db)
|
||||
with pytest.raises(ValueError, match="EVALUATION_EVIDENCE_METADATA_MISMATCH"):
|
||||
validate_evaluation_evidence(
|
||||
db, run_id=run.id, logical_step_id="step-eval-1", attempt=1,
|
||||
input_manifest=[{"artifact_id": "artifact-1", "sha256": "f" * 64, "content_type": "image/jpeg", "byte_length": 100}],
|
||||
raw_response_artifact_ref=None, raw_response_sha256=None, findings=[], succeeded=False,
|
||||
)
|
||||
|
||||
|
||||
def test_evidence_binding_rejects_raw_digest_mismatch(seeded_execution):
|
||||
db = seeded_execution
|
||||
run = _seed_evidence(db)
|
||||
with pytest.raises(ValueError, match="EVALUATION_EVIDENCE_RAW_DIGEST_MISMATCH"):
|
||||
validate_evaluation_evidence(
|
||||
db, run_id=run.id, logical_step_id="step-eval-1", attempt=1,
|
||||
input_manifest=[{"artifact_id": "artifact-1", "sha256": "e" * 64, "content_type": "image/jpeg", "byte_length": 100}],
|
||||
raw_response_artifact_ref="artifact-1", raw_response_sha256="f" * 64, findings=[], succeeded=True,
|
||||
)
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.AgentEvaluationStore.PriorStep [C:2] [TYPE Function]
|
||||
# @BRIEF D1: input_manifest may cite an active prior-step artifact of the same run.
|
||||
# @TEST_INVARIANT ScenarioExecution.AgentEvaluation.Evidence: prior-step same-run manifest is accepted. -> VERIFIED_BY: prior_step_manifest_accepted
|
||||
def test_evidence_binding_accepts_prior_step_manifest(seeded_execution):
|
||||
db = seeded_execution
|
||||
run = _seed_evidence(db)
|
||||
prior = ScenarioArtifact(
|
||||
owner_type="scenario_run", owner_id=run.id, kind="screenshot", name="prior",
|
||||
content_ref="prior-shot", sha256="a" * 64, content_type="image/jpeg", byte_length=40,
|
||||
logical_step_id="step-shot-1", attempt=1, is_active=True,
|
||||
)
|
||||
db.add(prior)
|
||||
db.flush()
|
||||
validate_evaluation_evidence(
|
||||
db, run_id=run.id, logical_step_id="step-eval-1", attempt=1,
|
||||
input_manifest=[{"artifact_id": "prior-shot", "sha256": "a" * 64, "content_type": "image/jpeg", "byte_length": 40}],
|
||||
raw_response_artifact_ref="artifact-1", raw_response_sha256="e" * 64, findings=[], succeeded=True,
|
||||
)
|
||||
# #endregion Test.ScenarioExecution.AgentEvaluationStore.PriorStep
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.AgentEvaluationStore.ForeignRun [C:2] [TYPE Function]
|
||||
# @BRIEF D1: a foreign-run artifact cannot satisfy this run's input_manifest.
|
||||
# @TEST_INVARIANT ScenarioExecution.AgentEvaluation.Evidence: foreign-run manifest is rejected. -> VERIFIED_BY: foreign_run_manifest_rejected
|
||||
def test_evidence_binding_rejects_foreign_run_manifest(seeded_execution):
|
||||
db = seeded_execution
|
||||
run = _seed_evidence(db)
|
||||
other = ScenarioRun(
|
||||
id=str(uuid.uuid4()), scenario_id=_SCENARIO, scenario_revision_id="bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbb1",
|
||||
scenario_content_hash="a" * 64, environment_id="env-preprod-02", idempotency_key=f"ev-{uuid.uuid4()}",
|
||||
)
|
||||
db.add(other)
|
||||
db.flush()
|
||||
db.add(ScenarioArtifact(
|
||||
owner_type="scenario_run", owner_id=other.id, kind="screenshot", name="foreign",
|
||||
content_ref="foreign-shot", sha256="a" * 64, content_type="image/jpeg", byte_length=40,
|
||||
logical_step_id="step-shot-1", attempt=1, is_active=True,
|
||||
))
|
||||
db.flush()
|
||||
with pytest.raises(ValueError, match="EVALUATION_EVIDENCE_NOT_FOUND"):
|
||||
validate_evaluation_evidence(
|
||||
db, run_id=run.id, logical_step_id="step-eval-1", attempt=1,
|
||||
input_manifest=[{"artifact_id": "foreign-shot", "sha256": "a" * 64, "content_type": "image/jpeg", "byte_length": 40}],
|
||||
raw_response_artifact_ref="artifact-1", raw_response_sha256="e" * 64, findings=[], succeeded=True,
|
||||
)
|
||||
# #endregion Test.ScenarioExecution.AgentEvaluationStore.ForeignRun
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.AgentEvaluationStore.PriorRaw [C:2] [TYPE Function]
|
||||
# @BRIEF D1: raw-response must belong to this evaluation step+attempt, not a prior step.
|
||||
# @TEST_INVARIANT ScenarioExecution.AgentEvaluation.Evidence: prior-step raw-response is rejected. -> VERIFIED_BY: prior_step_raw_rejected
|
||||
def test_evidence_binding_rejects_prior_step_raw_response(seeded_execution):
|
||||
db = seeded_execution
|
||||
run = _seed_evidence(db)
|
||||
db.add(ScenarioArtifact(
|
||||
owner_type="scenario_run", owner_id=run.id, kind="evidence", name="prior-raw",
|
||||
content_ref="prior-raw", sha256="a" * 64, content_type="application/json", byte_length=8,
|
||||
logical_step_id="step-shot-1", attempt=1, is_active=True,
|
||||
))
|
||||
db.flush()
|
||||
with pytest.raises(ValueError, match="EVALUATION_EVIDENCE_RAW_UNAVAILABLE"):
|
||||
validate_evaluation_evidence(
|
||||
db, run_id=run.id, logical_step_id="step-eval-1", attempt=1,
|
||||
input_manifest=[{"artifact_id": "artifact-1", "sha256": "e" * 64, "content_type": "image/jpeg", "byte_length": 100}],
|
||||
raw_response_artifact_ref="prior-raw", raw_response_sha256="a" * 64, findings=[], succeeded=True,
|
||||
)
|
||||
# #endregion Test.ScenarioExecution.AgentEvaluationStore.PriorRaw
|
||||
# #endregion Test.ScenarioExecution.AgentEvaluationStore
|
||||
@@ -17,6 +17,27 @@ def test_queue_and_case_are_idempotent(seeded_registry):
|
||||
assert open_case(seeded_registry, fingerprint="fp", scenario_id=first.scenario_id, actor_id="analyst").id == case.id
|
||||
|
||||
|
||||
# #region Test.ScenarioAnalytics.Investigation.AtomicDequeue [C:2] [TYPE Function]
|
||||
# @BRIEF DEF-04 regression: open dequeues to in_case, disposition closes, recurrence requeues a new row.
|
||||
def test_open_and_disposition_dequeue_atomically(seeded_registry):
|
||||
from src.models.scenario_investigation import InvestigationQueueItem
|
||||
|
||||
item = queue_signal(seeded_registry, fingerprint="fp-deq", scenario_id=_SCENARIO, run_id="run-deq")
|
||||
case = open_case(seeded_registry, fingerprint="fp-deq", scenario_id=_SCENARIO, actor_id="analyst")
|
||||
assert item.status == "in_case"
|
||||
queued = seeded_registry.query(InvestigationQueueItem).filter(
|
||||
InvestigationQueueItem.fingerprint == "fp-deq", InvestigationQueueItem.status == "queued"
|
||||
).count()
|
||||
assert queued == 0
|
||||
|
||||
set_disposition(seeded_registry, case.id, disposition="accepted", expected_version=1, actor_id="analyst", rationale="accepted risk")
|
||||
assert item.status == "closed"
|
||||
|
||||
recurrence = queue_signal(seeded_registry, fingerprint="fp-deq", scenario_id=_SCENARIO, run_id="run-deq-2")
|
||||
assert recurrence.id != item.id and recurrence.status == "queued"
|
||||
# #endregion Test.ScenarioAnalytics.Investigation.AtomicDequeue
|
||||
|
||||
|
||||
def test_disposition_uses_cas(seeded_registry):
|
||||
case = open_case(seeded_registry, fingerprint="fp-cas", scenario_id=_SCENARIO, actor_id="analyst")
|
||||
set_disposition(seeded_registry, case.id, disposition="resolved", expected_version=1, actor_id="analyst", verification_evidence={"reconciled": True, "artifact_id": "evidence-1"})
|
||||
|
||||
@@ -141,9 +141,9 @@ def test_start_run_pins_revision_content_hash_not_request_hash(seeded_registry):
|
||||
@pytest.mark.parametrize(
|
||||
("registry_version", "registry_hash", "action", "error_code"),
|
||||
[
|
||||
("038.2.0", action_registry_fingerprint(), "not_registered", "ACTION_DESCRIPTOR_UNKNOWN"),
|
||||
("038.4.0", action_registry_fingerprint(), "not_registered", "ACTION_DESCRIPTOR_UNKNOWN"),
|
||||
("038.0.0", action_registry_fingerprint(), "structural_assert", "ACTION_REGISTRY_VERSION_MISMATCH"),
|
||||
("038.2.0", "0" * 64, "structural_assert", "ACTION_REGISTRY_HASH_MISMATCH"),
|
||||
("038.4.0", "0" * 64, "structural_assert", "ACTION_REGISTRY_HASH_MISMATCH"),
|
||||
],
|
||||
)
|
||||
def test_action_preflight_rejects_before_run_creation(
|
||||
@@ -225,7 +225,7 @@ def test_prod_browser_mutation_rejects_before_run_creation(seeded_execution):
|
||||
revision_id="bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbb1"
|
||||
).one()
|
||||
revision.graph_snapshot = {
|
||||
"action_registry_version": "038.2.0",
|
||||
"action_registry_version": "038.4.0",
|
||||
"action_registry_hash": action_registry_fingerprint(),
|
||||
"steps": [{
|
||||
"logical_step_id": "prod-browser-write-044", "tool": "browser", "action": "row_edit",
|
||||
|
||||
@@ -48,7 +48,7 @@ def test_derived_plan_matches_hardcoded_fixture(seeded_execution):
|
||||
assert plan["resolved_params"] == expected["resolved_params"]
|
||||
assert plan["pinned_baselines"] == expected["pinned_baselines"]
|
||||
assert plan["topological_order"] == expected["topological_order"]
|
||||
assert plan["action_registry_version"] == "038.2.0"
|
||||
assert plan["action_registry_version"] == "038.4.0"
|
||||
assert plan["action_registry_hash"] == action_registry_fingerprint()
|
||||
assert [plan["executor_mapping"][step_id]["action"] for step_id in plan["topological_order"]] == [
|
||||
"apply_filters", "execute_metric", "parse_xlsx", "compare_to_baseline",
|
||||
@@ -85,3 +85,39 @@ def test_cycle_in_graph_is_rejected(seeded_registry):
|
||||
with pytest.raises(ValueError, match="cycle"):
|
||||
derive_runner_plan(seeded_registry, "11111111-1111-4111-8111-111111111111", "22222222-2222-4222-8222-222222222222")
|
||||
# #endregion Test.ScenarioExecution.RunnerPlan
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.RunnerPlan.Policy [C:3] [TYPE Module] [SEMANTICS test,scenario,execution,runnerplan,policy]
|
||||
# @ingroup Test
|
||||
# @BRIEF Slice-C: plans carry the pinned DecisionPolicy; legacy plans resolve to the v1 default.
|
||||
# @RELATION VERIFIES -> [ScenarioExecution.RunnerPlan.ResolvePolicy]
|
||||
|
||||
|
||||
def test_derived_plan_pins_baseline_semantic_disabled(seeded_registry):
|
||||
from src.services.dashboard_testing.execution.decision_policy import DecisionPolicy
|
||||
|
||||
plan = derive_runner_plan(seeded_registry, "11111111-1111-4111-8111-111111111111", "22222222-2222-4222-8222-222222222222")
|
||||
pinned = plan["decision_policy"]
|
||||
assert DecisionPolicy.model_validate(pinned).policy_id == "baseline-semantic"
|
||||
assert pinned["evaluation_mode"] == "disabled"
|
||||
|
||||
|
||||
def test_legacy_plan_without_policy_resolves_to_v1_default():
|
||||
from src.services.dashboard_testing.execution.runner_plan import resolve_pinned_policy
|
||||
|
||||
legacy = {"topological_order": []}
|
||||
assert resolve_pinned_policy(legacy) == resolve_pinned_policy({})
|
||||
assert resolve_pinned_policy(legacy).policy_id == "baseline-semantic"
|
||||
assert resolve_pinned_policy(legacy).evaluation_mode == "disabled"
|
||||
|
||||
|
||||
def test_malformed_pinned_policy_is_rejected():
|
||||
from src.services.dashboard_testing.execution.runner_plan import resolve_pinned_policy, validate_pinned_runner_plan
|
||||
|
||||
malformed = {"topological_order": [], "decision_policy": {"policy_id": "not-baseline"}}
|
||||
with pytest.raises(ValueError, match="DECISION_POLICY_INVALID"):
|
||||
resolve_pinned_policy(malformed)
|
||||
with pytest.raises(ValueError, match="DECISION_POLICY_INVALID"):
|
||||
validate_pinned_runner_plan(malformed)
|
||||
validate_pinned_runner_plan({"topological_order": [], "decision_policy": None})
|
||||
# #endregion Test.ScenarioExecution.RunnerPlan.Policy
|
||||
|
||||
@@ -5,12 +5,17 @@
|
||||
# @TEST_FIXTURE: xlsx_bytes_044 + artifact_ref_044 -> INLINE hardcoded evidence provenance
|
||||
# @TEST_INVARIANT ScenarioExecution.Runner.Walker: Scenario evidence artifacts persist only executor-supplied or locally computed real sha256 digests; missing or invalid digests never create placeholder artifacts. -> VERIFIED_BY: xlsx_digest_persisted, missing_digest_rejected, zero_digest_rejected
|
||||
# @TEST_INVARIANT ScenarioExecution.LiveAdapter: Untrusted runtime adapter statuses become typed inconclusive before the runner aggregates a result. -> VERIFIED_BY: invalid_adapter_status_remains_inconclusive
|
||||
# @TEST_INVARIANT ScenarioExecution.AgentEvaluation: walker mock-adapter success persists evaluation ids and DecisionPolicy row 14; required + missing adapter is EVALUATION_UNAVAILABLE. -> VERIFIED_BY: walker_evaluation_baseline_and_semantic_pass, walker_required_missing_adapter_unavailable
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import UTC, datetime
|
||||
from hashlib import sha256
|
||||
import uuid
|
||||
|
||||
from src.models.scenario_artifact import ScenarioArtifact
|
||||
from src.models.scenario_evaluation import AgentEvaluation as AgentEvaluationRow
|
||||
from src.models.scenario_run import ScenarioStepRun
|
||||
from src.services.dashboard_testing.execution.decision_policy import BASELINE_SEMANTIC_V1
|
||||
from src.services.dashboard_testing.execution.executor_registry import ScenarioExecutorRegistry
|
||||
from src.services.dashboard_testing.execution.executors import BrowserAdapterResult, _register_default_executors, assertion, xlsx
|
||||
from src.services.dashboard_testing.execution.runner import _advance_run, start_run
|
||||
@@ -20,6 +25,10 @@ from src.services.dashboard_testing.scenario.templates import (
|
||||
resolve_action_descriptor,
|
||||
)
|
||||
|
||||
_RAW_EVAL_BYTES = b'{"verdict":"pass","confidence":0.95}'
|
||||
_RAW_EVAL_DIGEST = sha256(_RAW_EVAL_BYTES).hexdigest()
|
||||
_EVAL_NOW = datetime(2026, 9, 10, 0, 0, 0, tzinfo=UTC)
|
||||
|
||||
|
||||
def _action_plan(step_id: str, tool: str, action: str) -> dict:
|
||||
descriptor = resolve_action_descriptor(
|
||||
@@ -42,6 +51,75 @@ def _action_plan(step_id: str, tool: str, action: str) -> dict:
|
||||
}
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.Walker.RequiredEvalPlan [C:1] [TYPE Function]
|
||||
def _required_eval_plan(step_id: str) -> dict:
|
||||
plan = _action_plan(step_id, "agent_evaluation", "evaluate_declared_spec")
|
||||
plan["decision_policy"] = BASELINE_SEMANTIC_V1.model_copy(
|
||||
update={"evaluation_mode": "required"}
|
||||
).model_dump()
|
||||
return plan
|
||||
# #endregion Test.ScenarioExecution.Walker.RequiredEvalPlan
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.Walker.MockEvalAdapter [C:1] [TYPE Function]
|
||||
def _mock_eval_adapter(step: dict, _completed: dict) -> dict:
|
||||
run_id = step["scenario_run_id"]
|
||||
step_id = step["logical_step_id"]
|
||||
raw_ref = f"draft:{run_id}:{_RAW_EVAL_DIGEST}"
|
||||
eval_id = str(uuid.uuid4())
|
||||
record = {
|
||||
"schema_version": 1,
|
||||
"evaluation_id": eval_id,
|
||||
"scenario_run_id": run_id,
|
||||
"logical_step_id": step_id,
|
||||
"attempt": 1,
|
||||
"operation_id": str(uuid.uuid4()),
|
||||
"evaluation_spec_hash": "a" * 64,
|
||||
"provider_id": "llm-provider-a",
|
||||
"provider_version": "1",
|
||||
"model_id": "vision-model",
|
||||
"model_version": "2026-08",
|
||||
"prompt_template_id": "agent-evaluation-prompt",
|
||||
"prompt_template_version": "1.0.0",
|
||||
"prompt_template_hash": "b" * 64,
|
||||
"output_schema_hash": "c" * 64,
|
||||
"input_manifest_hash": "d" * 64,
|
||||
"input_manifest": [{
|
||||
"artifact_id": raw_ref, "sha256": _RAW_EVAL_DIGEST,
|
||||
"content_type": "application/json", "byte_length": len(_RAW_EVAL_BYTES), "role": "context",
|
||||
}],
|
||||
"baseline_pin": {"catalog_revision_id": str(uuid.uuid4())},
|
||||
"comparison_ids": [str(uuid.uuid4())],
|
||||
"status": "succeeded",
|
||||
"verdict": "pass",
|
||||
"confidence": 0.95,
|
||||
"findings": [{
|
||||
"finding_id": "f-1", "severity": "info", "message": "matches",
|
||||
"evidence_artifact_ids": [raw_ref], "criterion_id": "crit-visual", "criterion_kind": "semantic",
|
||||
}],
|
||||
"reason_codes": ["EVALUATION_PASS"],
|
||||
"raw_response_artifact_ref": raw_ref,
|
||||
"raw_response_sha256": _RAW_EVAL_DIGEST,
|
||||
"trust_policy_hash": "f" * 64,
|
||||
"usage": {"input_tokens": 100, "output_tokens": 10, "cost_amount": "0.01", "currency": "USD", "pricing_version": "1"},
|
||||
"started_at": _EVAL_NOW.isoformat(),
|
||||
"finished_at": _EVAL_NOW.isoformat(),
|
||||
}
|
||||
return {
|
||||
"evaluation_input": {
|
||||
"evaluation_id": eval_id, "status": "succeeded", "verdict": "pass", "confidence": 0.95,
|
||||
"has_error_critical_findings": False, "conflicts_with_deterministic_criterion": False,
|
||||
"fail_supported_by_findings": False,
|
||||
},
|
||||
"evaluation_record": record,
|
||||
"artifact_refs": [raw_ref],
|
||||
"artifact_digests": {raw_ref: _RAW_EVAL_DIGEST},
|
||||
"content_type": "application/json",
|
||||
"byte_length": len(_RAW_EVAL_BYTES),
|
||||
}
|
||||
# #endregion Test.ScenarioExecution.Walker.MockEvalAdapter
|
||||
|
||||
|
||||
def test_walker_runs_fixture_to_completion(seeded_execution):
|
||||
run = start_run(
|
||||
seeded_execution,
|
||||
@@ -241,4 +319,150 @@ def test_invalid_adapter_status_remains_inconclusive(seeded_execution):
|
||||
assert result["status"] == "inconclusive"
|
||||
assert result["step_counts"] == {"passed": 0, "failed": 0, "blocked": 0, "inconclusive": 1}
|
||||
# #endregion Test.ScenarioExecution.Walker.InvalidAdapterStatus
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.Walker.PolicyDecision [C:3] [TYPE Function]
|
||||
# @BRIEF DEF/Slice-C: normative steps publish a DecisionPolicy stamp with reason codes into step_outcome.
|
||||
# @TEST_INVARIANT ScenarioExecution.DecisionPolicy: the walker routes comparison-bearing steps through the
|
||||
# sole mapper; a deterministic pass under disabled mode yields BASELINE_PASS.
|
||||
def test_assertion_pass_records_policy_decision(seeded_execution):
|
||||
run = start_run(
|
||||
seeded_execution,
|
||||
"aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaa1",
|
||||
"bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbb1",
|
||||
{},
|
||||
"env-preprod-02",
|
||||
actor="test-walker",
|
||||
idempotency_key="walker-policy-001",
|
||||
auto_advance=False,
|
||||
)
|
||||
run.runner_plan = _action_plan("policy-assert-044", "assertion", "compare_to_baseline")
|
||||
seeded_execution.flush()
|
||||
registry = ScenarioExecutorRegistry()
|
||||
registry.register(
|
||||
"assertion",
|
||||
lambda _step, _completed: {"status": "passed", "step_outcome": {"tool": "assertion", "actual": 1}},
|
||||
)
|
||||
|
||||
_advance_run(seeded_execution, run, registry, worker_id="test-walker")
|
||||
|
||||
step = seeded_execution.query(ScenarioStepRun).filter(
|
||||
ScenarioStepRun.run_id == run.id, ScenarioStepRun.logical_step_id == "policy-assert-044"
|
||||
).one()
|
||||
assert step.status == "passed"
|
||||
outcome = step.step_outcome
|
||||
assert outcome["decision_policy_id"] == "baseline-semantic"
|
||||
assert outcome["decision_policy_version"] == "1.0.0"
|
||||
assert outcome["reason_codes"] == ["BASELINE_PASS"]
|
||||
assert outcome["comparison_ids"] == ["policy-assert-044"]
|
||||
assert outcome["agent_evaluation_ids"] == []
|
||||
assert "decided_at" in outcome
|
||||
# #endregion Test.ScenarioExecution.Walker.PolicyDecision
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.Walker.RequiredEvidenceBlocked [C:3] [TYPE Function]
|
||||
# @BRIEF Slice-C deviation (decision-policy.md rows 2-3): a required-evidence step with corrupted artifacts
|
||||
# resolves through the sole mapper to blocked EVIDENCE_CORRUPT instead of ad-hoc inconclusive.
|
||||
# @TEST_INVARIANT ScenarioExecution.DecisionPolicy: required evidence corruption is blocked by the policy,
|
||||
# never downgraded by an executor payload.
|
||||
def test_required_evidence_integrity_becomes_blocked(seeded_execution):
|
||||
run = start_run(
|
||||
seeded_execution,
|
||||
"aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaa1",
|
||||
"bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbb1",
|
||||
{},
|
||||
"env-preprod-02",
|
||||
actor="test-walker",
|
||||
idempotency_key="walker-policy-002",
|
||||
auto_advance=False,
|
||||
)
|
||||
run.runner_plan = _action_plan("shot-integrity-044", "screenshot", "capture_screenshot")
|
||||
seeded_execution.flush()
|
||||
registry = ScenarioExecutorRegistry()
|
||||
registry.register(
|
||||
"screenshot",
|
||||
lambda _step, _completed: {
|
||||
"status": "passed",
|
||||
"step_outcome": {"tool": "screenshot"},
|
||||
"artifact_refs": ["shot-missing-digest-044"],
|
||||
},
|
||||
)
|
||||
|
||||
_advance_run(seeded_execution, run, registry, worker_id="test-walker")
|
||||
|
||||
step = seeded_execution.query(ScenarioStepRun).filter(
|
||||
ScenarioStepRun.run_id == run.id, ScenarioStepRun.logical_step_id == "shot-integrity-044"
|
||||
).one()
|
||||
assert step.status == "blocked"
|
||||
assert step.error_code == "ARTIFACT_DIGEST_MISSING"
|
||||
assert step.step_outcome["reason_codes"] == ["EVIDENCE_CORRUPT"]
|
||||
assert step.step_outcome["decision_policy_id"] == "baseline-semantic"
|
||||
# #endregion Test.ScenarioExecution.Walker.RequiredEvidenceBlocked
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.Walker.EvaluationPass [C:2] [TYPE Function]
|
||||
# @BRIEF Mock adapter success persists AgentEvaluation and DecisionPolicy row 14.
|
||||
# @TEST_INVARIANT ScenarioExecution.AgentEvaluation: mock-adapter success -> persisted ids + BASELINE_AND_SEMANTIC_PASS. -> VERIFIED_BY: walker_evaluation_baseline_and_semantic_pass
|
||||
def test_walker_evaluation_baseline_and_semantic_pass(seeded_execution):
|
||||
run = start_run(
|
||||
seeded_execution,
|
||||
"aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaa1",
|
||||
"bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbb1",
|
||||
{},
|
||||
"env-preprod-02",
|
||||
actor="test-walker",
|
||||
idempotency_key="walker-eval-pass-001",
|
||||
auto_advance=False,
|
||||
)
|
||||
run.runner_plan = _required_eval_plan("eval-pass-044")
|
||||
seeded_execution.flush()
|
||||
registry = ScenarioExecutorRegistry()
|
||||
_register_default_executors(registry, agent_evaluation_adapter=_mock_eval_adapter)
|
||||
|
||||
_advance_run(seeded_execution, run, registry, worker_id="test-walker")
|
||||
|
||||
step = seeded_execution.query(ScenarioStepRun).filter(
|
||||
ScenarioStepRun.run_id == run.id, ScenarioStepRun.logical_step_id == "eval-pass-044",
|
||||
).one()
|
||||
eval_ids = step.step_outcome.get("agent_evaluation_ids")
|
||||
row = seeded_execution.query(AgentEvaluationRow).filter_by(scenario_run_id=run.id).one()
|
||||
artifact = seeded_execution.query(ScenarioArtifact).filter_by(owner_id=run.id).one()
|
||||
assert step.status == "passed"
|
||||
assert step.step_outcome["reason_codes"] == ["BASELINE_AND_SEMANTIC_PASS"]
|
||||
assert eval_ids == [row.evaluation_id]
|
||||
assert artifact.sha256 == _RAW_EVAL_DIGEST
|
||||
assert artifact.content_type == "application/json"
|
||||
assert artifact.byte_length == len(_RAW_EVAL_BYTES)
|
||||
# #endregion Test.ScenarioExecution.Walker.EvaluationPass
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.Walker.EvaluationUnavailable [C:2] [TYPE Function]
|
||||
# @BRIEF Required evaluation with no adapter is inconclusive EVALUATION_UNAVAILABLE.
|
||||
# @TEST_INVARIANT ScenarioExecution.AgentEvaluation: required + missing adapter -> EVALUATION_UNAVAILABLE. -> VERIFIED_BY: walker_required_missing_adapter_unavailable
|
||||
def test_walker_required_missing_adapter_unavailable(seeded_execution):
|
||||
run = start_run(
|
||||
seeded_execution,
|
||||
"aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaa1",
|
||||
"bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbb1",
|
||||
{},
|
||||
"env-preprod-02",
|
||||
actor="test-walker",
|
||||
idempotency_key="walker-eval-missing-001",
|
||||
auto_advance=False,
|
||||
)
|
||||
run.runner_plan = _required_eval_plan("eval-missing-044")
|
||||
seeded_execution.flush()
|
||||
registry = ScenarioExecutorRegistry()
|
||||
_register_default_executors(registry)
|
||||
|
||||
_advance_run(seeded_execution, run, registry, worker_id="test-walker")
|
||||
|
||||
step = seeded_execution.query(ScenarioStepRun).filter(
|
||||
ScenarioStepRun.run_id == run.id, ScenarioStepRun.logical_step_id == "eval-missing-044",
|
||||
).one()
|
||||
assert step.status == "inconclusive"
|
||||
assert step.step_outcome["reason_codes"] == ["EVALUATION_UNAVAILABLE"]
|
||||
assert step.step_outcome.get("agent_evaluation_ids") == []
|
||||
assert seeded_execution.query(AgentEvaluationRow).filter_by(scenario_run_id=run.id).count() == 0
|
||||
# #endregion Test.ScenarioExecution.Walker.EvaluationUnavailable
|
||||
# #endregion Test.ScenarioExecution.Walker
|
||||
|
||||
@@ -0,0 +1,97 @@
|
||||
# #region Test.ScenarioGraph.AgentEvaluationModels [C:3] [TYPE Module] [SEMANTICS test,scenario,evaluation,models,spec]
|
||||
# @RELATION BINDS_TO -> [ScenarioGraph.Models.AgentEvaluationSpec]
|
||||
# @TEST_EDGE criterion truth table -> unique ids, deterministic binds comparison, semantic binds null
|
||||
# @TEST_EDGE drift -> ScenarioStep tool literal accepts agent_evaluation/sql_evidence/transform
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import UTC, datetime
|
||||
|
||||
import pytest
|
||||
|
||||
from src.services.dashboard_testing.execution.decision_policy import DecisionPolicy
|
||||
from src.services.dashboard_testing.scenario.models import (
|
||||
AgentEvaluationSpec,
|
||||
EvaluationCriterion,
|
||||
EvaluationLimits,
|
||||
ScenarioStep,
|
||||
)
|
||||
|
||||
_NOW = datetime(2026, 9, 10, 0, 0, 0, tzinfo=UTC)
|
||||
|
||||
|
||||
def _spec(**overrides) -> AgentEvaluationSpec:
|
||||
data: dict = {
|
||||
"schema_version": 1,
|
||||
"spec_id": "e5e5e5e5-e5e5-4e5e-8e5e-e5e5e5e5e5e5",
|
||||
"provider_id": "llm-provider-a",
|
||||
"provider_version": "1",
|
||||
"model_id": "vision-model",
|
||||
"model_version": "2026-08",
|
||||
"prompt_template_id": "agent-evaluation-prompt",
|
||||
"prompt_template_version": "1.0.0",
|
||||
"prompt_template_hash": "a" * 64,
|
||||
"evidence_refs": ["artifact-1"],
|
||||
"comparison_refs": ["44444444-4444-4444-8444-444444444444"],
|
||||
"output_schema": "agent-evaluation.schema.json",
|
||||
"decision_policy": DecisionPolicy(policy_id="baseline-semantic", version="1.0.0"),
|
||||
"limits": EvaluationLimits(timeout_ms=60000, max_images=4, max_input_tokens=32000, max_output_tokens=2000, max_cost="1.00", currency="USD"),
|
||||
"trust_policy_hash": "b" * 64,
|
||||
"criteria": [
|
||||
EvaluationCriterion(criterion_id="crit-visual", criterion_kind="semantic", description="visual layout matches", comparison_id=None),
|
||||
EvaluationCriterion(criterion_id="crit-metric", criterion_kind="deterministic_comparison", description="metric matches", comparison_id="44444444-4444-4444-8444-444444444444"),
|
||||
],
|
||||
}
|
||||
data.update(overrides)
|
||||
return AgentEvaluationSpec.model_validate(data)
|
||||
|
||||
|
||||
def test_spec_roundtrip_is_valid():
|
||||
spec = _spec()
|
||||
assert spec.provider_id == "llm-provider-a"
|
||||
assert spec.decision_policy.policy_id == "baseline-semantic"
|
||||
assert [c.criterion_id for c in spec.criteria] == ["crit-visual", "crit-metric"]
|
||||
|
||||
|
||||
def test_duplicate_criterion_id_rejected():
|
||||
with pytest.raises(ValueError):
|
||||
_spec(criteria=[
|
||||
EvaluationCriterion(criterion_id="dup", criterion_kind="semantic", description="a", comparison_id=None),
|
||||
EvaluationCriterion(criterion_id="dup", criterion_kind="semantic", description="b", comparison_id=None),
|
||||
])
|
||||
|
||||
|
||||
def test_semantic_criterion_forbids_comparison_binding():
|
||||
with pytest.raises(ValueError):
|
||||
_spec(criteria=[
|
||||
EvaluationCriterion(criterion_id="crit-visual", criterion_kind="semantic", description="a", comparison_id="44444444-4444-4444-8444-444444444444"),
|
||||
])
|
||||
|
||||
|
||||
def test_deterministic_criterion_requires_declared_comparison():
|
||||
with pytest.raises(ValueError):
|
||||
_spec(criteria=[
|
||||
EvaluationCriterion(criterion_id="crit-metric", criterion_kind="deterministic_comparison", description="b", comparison_id="not-declared"),
|
||||
])
|
||||
|
||||
|
||||
def test_tool_allowlist_is_closed():
|
||||
with pytest.raises(ValueError):
|
||||
_spec(tool_allowlist=["some_tool"])
|
||||
|
||||
|
||||
def test_scenario_step_tool_literal_includes_agent_evaluation_and_drift_fixes():
|
||||
base = {
|
||||
"id": "phase-1-b01-open",
|
||||
"phase": "setup",
|
||||
"title": "t",
|
||||
"tool": "agent_evaluation",
|
||||
"action": "evaluate_declared_spec",
|
||||
"expected": {"kind": "structural", "description": "d"},
|
||||
"automation_status": "ready",
|
||||
"risk": "read",
|
||||
}
|
||||
step = ScenarioStep.model_validate(base)
|
||||
assert step.tool == "agent_evaluation"
|
||||
for tool in ("sql_evidence", "transform", "agent_evaluation"):
|
||||
ScenarioStep.model_validate({**base, "tool": tool})
|
||||
# #endregion Test.ScenarioGraph.AgentEvaluationModels
|
||||
@@ -9,6 +9,7 @@ from __future__ import annotations
|
||||
|
||||
from src.services.dashboard_testing.scenario.compiler import CompileScenarioRequest, compile_scenario
|
||||
from src.services.dashboard_testing.scenario.templates import REGISTERED_ACTIONS
|
||||
from src.services.dashboard_testing.scenario.validator import validate_scenario
|
||||
|
||||
FULL = {
|
||||
"browser": True, # browser is base infrastructure, always available in test env
|
||||
@@ -89,7 +90,9 @@ def test_unsupported_case_warned_not_dropped() -> None:
|
||||
result = compile_scenario(_req(["C04"], capabilities=caps))
|
||||
covered = {c.case_id for c in result.scenario.checklist_coverage}
|
||||
assert "C04" in covered # still present in coverage with rationale
|
||||
assert any(w.code == "UNSUPPORTED_CASE" for w in result.warnings) or not result.scenario.steps
|
||||
assert any(b.code == "UNSUPPORTED_ACTION" for b in result.blockers)
|
||||
assert result.scenario.steps == []
|
||||
assert not any(w.code == "UNSUPPORTED_CASE" for w in result.warnings)
|
||||
|
||||
|
||||
def test_cyrillic_goal_produces_ascii_scenario_id() -> None:
|
||||
@@ -142,4 +145,143 @@ def test_scalar_parameter_spec() -> None:
|
||||
assert p.value == "default-value"
|
||||
|
||||
|
||||
def _selector_params() -> dict:
|
||||
return {
|
||||
"test_date": {"type": "date", "default": "2026-07-01"},
|
||||
"counterparty": {"type": "string", "default": "ACME"},
|
||||
"selector_hint": {"type": "selector_hint", "default": "#native-filter", "required": False},
|
||||
}
|
||||
|
||||
|
||||
def test_visual_case_emits_canonical_capture_compare_chain() -> None:
|
||||
caps = dict(FULL, baseline=True)
|
||||
result = compile_scenario(_req(["B01"], parameters=_selector_params(), capabilities=caps))
|
||||
steps = result.scenario.steps
|
||||
assert [s.action for s in steps] == ["apply_native_filter", "capture_screenshot", "compare_to_baseline"]
|
||||
assert [s.tool for s in steps] == ["browser", "screenshot", "assertion"]
|
||||
assert steps[0].depends_on == []
|
||||
assert steps[1].depends_on == [steps[0].id]
|
||||
assert steps[2].depends_on == [steps[1].id]
|
||||
assert "text_filter" not in {s.action for s in steps}
|
||||
assert "apply_filters" not in {s.action for s in steps}
|
||||
assert "register_artifact" not in {s.action for s in steps}
|
||||
validation = validate_scenario(result.scenario)
|
||||
assert validation.valid is True
|
||||
assert not validation.errors
|
||||
assert not validation.blockers
|
||||
|
||||
|
||||
def test_text_filter_alias_is_not_emitted() -> None:
|
||||
caps = dict(FULL, baseline=True)
|
||||
result = compile_scenario(_req(["B02"], parameters=_selector_params(), capabilities=caps))
|
||||
actions = [s.action for s in result.scenario.steps]
|
||||
assert "text_filter" not in actions
|
||||
assert actions[0] == "apply_native_filter"
|
||||
assert actions == ["apply_native_filter", "capture_screenshot", "compare_to_baseline"]
|
||||
|
||||
|
||||
def test_selected_unsupported_case_blocks_without_steps() -> None:
|
||||
caps = dict(FULL, xlsx_export=False, safe_test_data=True)
|
||||
result = compile_scenario(_req(["C04"], capabilities=caps))
|
||||
assert result.scenario.steps == []
|
||||
assert any(b.code == "UNSUPPORTED_ACTION" for b in result.blockers)
|
||||
assert {c.case_id for c in result.scenario.checklist_coverage} >= {"C04"}
|
||||
|
||||
|
||||
def test_unselected_unsupported_does_not_block_selected_chain() -> None:
|
||||
caps = dict(FULL, xlsx_export=False, baseline=True)
|
||||
result = compile_scenario(_req(["B01"], parameters=_selector_params(), capabilities=caps))
|
||||
assert not any(b.code == "UNSUPPORTED_ACTION" for b in result.blockers)
|
||||
covered = {c.case_id: c.classification for c in result.scenario.checklist_coverage}
|
||||
assert covered["C04"] == "unsupported"
|
||||
assert [s.action for s in result.scenario.steps] == [
|
||||
"apply_native_filter", "capture_screenshot", "compare_to_baseline",
|
||||
]
|
||||
|
||||
|
||||
def test_visual_compare_without_screenshot_is_blocker() -> None:
|
||||
caps = dict(FULL, screenshot=False, baseline=True)
|
||||
result = compile_scenario(_req(["B01"], parameters=_selector_params(), capabilities=caps))
|
||||
actions = [s.action for s in result.scenario.steps]
|
||||
assert actions == ["apply_native_filter"]
|
||||
assert "capture_screenshot" not in actions
|
||||
assert "compare_to_baseline" not in actions
|
||||
assert any(b.code == "MISSING_SCREENSHOT" for b in result.blockers)
|
||||
|
||||
|
||||
def test_metric_case_appends_compare_when_baseline_present() -> None:
|
||||
caps = dict(FULL, baseline=True)
|
||||
result = compile_scenario(_req(["T01"], capabilities=caps))
|
||||
steps = result.scenario.steps
|
||||
assert [s.action for s in steps] == ["dataset_field_assert", "compare_to_baseline"]
|
||||
assert steps[1].depends_on == [steps[0].id]
|
||||
assert "capture_screenshot" not in {s.action for s in steps}
|
||||
|
||||
|
||||
def test_human_checkpoint_case_unchanged() -> None:
|
||||
caps = dict(FULL, safe_test_data=False)
|
||||
result = compile_scenario(_req(["B05"], capabilities=caps))
|
||||
assert [s.action for s in result.scenario.steps] == ["human_checkpoint"]
|
||||
assert result.scenario.steps[0].automation_status == "manual"
|
||||
assert result.scenario.steps[0].depends_on == []
|
||||
assert "capture_screenshot" not in {s.action for s in result.scenario.steps}
|
||||
|
||||
|
||||
def _evaluation_spec():
|
||||
from src.services.dashboard_testing.execution.decision_policy import DecisionPolicy
|
||||
from src.services.dashboard_testing.scenario.models import (
|
||||
AgentEvaluationSpec,
|
||||
EvaluationCriterion,
|
||||
EvaluationLimits,
|
||||
)
|
||||
|
||||
return AgentEvaluationSpec.model_validate({
|
||||
"schema_version": 1,
|
||||
"spec_id": "e5e5e5e5-e5e5-4e5e-8e5e-e5e5e5e5e5e5",
|
||||
"provider_id": "llm-provider-a",
|
||||
"provider_version": "1",
|
||||
"model_id": "vision-model",
|
||||
"model_version": "2026-08",
|
||||
"prompt_template_id": "agent-evaluation-prompt",
|
||||
"prompt_template_version": "1.0.0",
|
||||
"prompt_template_hash": "a" * 64,
|
||||
"evidence_refs": ["artifact-1"],
|
||||
"comparison_refs": ["44444444-4444-4444-8444-444444444444"],
|
||||
"output_schema": "agent-evaluation.schema.json",
|
||||
"decision_policy": DecisionPolicy(policy_id="baseline-semantic", version="1.0.0"),
|
||||
"limits": EvaluationLimits(
|
||||
timeout_ms=60000, max_images=4, max_input_tokens=32000,
|
||||
max_output_tokens=2000, max_cost="1.00", currency="USD",
|
||||
),
|
||||
"trust_policy_hash": "b" * 64,
|
||||
"criteria": [
|
||||
EvaluationCriterion(
|
||||
criterion_id="crit-visual", criterion_kind="semantic",
|
||||
description="visual layout matches", comparison_id=None,
|
||||
),
|
||||
],
|
||||
})
|
||||
|
||||
|
||||
def test_evaluate_declared_spec_emitted_only_when_request_has_spec() -> None:
|
||||
from dataclasses import replace
|
||||
|
||||
caps = dict(FULL, baseline=True)
|
||||
without_spec = compile_scenario(_req(["B01"], parameters=_selector_params(), capabilities=caps))
|
||||
assert "evaluate_declared_spec" not in {s.action for s in without_spec.scenario.steps}
|
||||
|
||||
spec = _evaluation_spec()
|
||||
with_spec = compile_scenario(replace(
|
||||
_req(["B01"], parameters=_selector_params(), capabilities=caps),
|
||||
agent_evaluation_spec=spec,
|
||||
))
|
||||
steps = with_spec.scenario.steps
|
||||
assert [s.action for s in steps] == [
|
||||
"apply_native_filter", "capture_screenshot", "compare_to_baseline", "evaluate_declared_spec",
|
||||
]
|
||||
assert steps[3].depends_on == [steps[2].id]
|
||||
assert steps[3].agent_evaluation_spec == spec
|
||||
assert steps[3].decision_policy is None
|
||||
|
||||
|
||||
# #endregion Test.Scenario.Compiler.Branches
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
# #region Test.ScenarioGraph.SqlGuard [C:2] [TYPE Module] [SEMANTICS test,scenario,safety,sql]
|
||||
# @RELATION BINDS_TO -> [ScenarioGraph.SqlGuard]
|
||||
# @TEST_EDGE: prose_with_double_hyphen -> not flagged (DEF-01 false-positive regression)
|
||||
# @TEST_EDGE: stacked_statements -> flagged via recursive statement scan
|
||||
# @TEST_EDGE: path_traversal_free_text -> flagged independently of SQL
|
||||
from src.services.dashboard_testing.scenario.sql_guard import contains_sql_statement, contains_unsafe_free_text
|
||||
|
||||
|
||||
# #region Test.ScenarioGraph.SqlGuard.ProsePasses [C:2] [TYPE Function]
|
||||
# @BRIEF Ordinary prose and ru/en titles never flag, including "--" runs that broke the old token scan.
|
||||
def test_prose_passes():
|
||||
assert contains_sql_statement("Провести review окна 2026-09-08--2026-09-09") is False
|
||||
assert contains_sql_statement("well--done") is False
|
||||
assert contains_sql_statement("Join date filter applied") is False
|
||||
assert contains_sql_statement("Обновлено вчера") is False
|
||||
assert contains_sql_statement("Выбрать строки для экспорта") is False
|
||||
assert contains_sql_statement("") is False
|
||||
assert contains_sql_statement(None) is False
|
||||
# #endregion Test.ScenarioGraph.SqlGuard.ProsePasses
|
||||
|
||||
|
||||
# #region Test.ScenarioGraph.SqlGuard.SqlFlags [C:2] [TYPE Function]
|
||||
# @BRIEF Executable SQL shapes flag deterministically, including stacked statements.
|
||||
def test_sql_shapes_flag():
|
||||
assert contains_sql_statement("SELECT * FROM users WHERE 1=1") is True
|
||||
assert contains_sql_statement("drop table audit --") is True
|
||||
assert contains_sql_statement("1; DROP TABLE y") is True
|
||||
assert contains_sql_statement("UPDATE accounts SET balance = 0 WHERE id = 1") is True
|
||||
assert contains_sql_statement("INSERT INTO t VALUES (1)") is True
|
||||
# #endregion Test.ScenarioGraph.SqlGuard.SqlFlags
|
||||
|
||||
|
||||
# #region Test.ScenarioGraph.SqlGuard.FreeTextPaths [C:2] [TYPE Function]
|
||||
# @BRIEF Free-text guard adds path-traversal detection and ignores non-strings.
|
||||
def test_free_text_paths():
|
||||
assert contains_unsafe_free_text("../credentials.json") is True
|
||||
assert contains_unsafe_free_text("..\\windows\\payload") is True
|
||||
assert contains_unsafe_free_text("normal step title") is False
|
||||
assert contains_unsafe_free_text(None) is False
|
||||
assert contains_unsafe_free_text(42) is False
|
||||
# #endregion Test.ScenarioGraph.SqlGuard.FreeTextPaths
|
||||
|
||||
# #endregion Test.ScenarioGraph.SqlGuard
|
||||
@@ -248,6 +248,67 @@ class TestLifespan:
|
||||
mock_scheduler.stop.assert_called_once()
|
||||
# #endregion Test.AppModule.TestLifespanShutdownStopsScheduler
|
||||
|
||||
# #region Test.AppModule.TestProviderLoopOrdering [C:3] [TYPE Function]
|
||||
# @BRIEF Slice-C Q1: ProviderEventLoop starts before composition bootstrap and stops on shutdown.
|
||||
# @TEST_INVARIANT App.AppModule.Lifespan: exactly one provider loop start precedes composition; stop
|
||||
# always follows on shutdown (ordered startup per 044 ProviderRuntime).
|
||||
@pytest.mark.asyncio
|
||||
async def test_provider_loop_starts_before_composition_and_stops_on_shutdown(self):
|
||||
from src.app import lifespan
|
||||
mock_app = MagicMock(spec=FastAPI)
|
||||
order: list[str] = []
|
||||
mock_loop = MagicMock()
|
||||
mock_loop.start.side_effect = lambda: order.append("start")
|
||||
mock_composition = MagicMock(side_effect=lambda: order.append("composition"))
|
||||
mock_stop = MagicMock(side_effect=lambda: order.append("stop"))
|
||||
with (
|
||||
patch("src.app.seed_trace_id"), patch("src.app.ensure_encryption_key"),
|
||||
patch("src.app.init_db"), patch("src.app.ensure_initial_admin_user"),
|
||||
patch("src.app.get_scheduler_service", return_value=MagicMock()),
|
||||
patch(
|
||||
"src.services.dashboard_testing.execution.provider_runtime.get_provider_event_loop",
|
||||
return_value=mock_loop,
|
||||
),
|
||||
patch("src.app.initialize_live_execution_composition", new=mock_composition),
|
||||
patch(
|
||||
"src.services.dashboard_testing.execution.provider_runtime.stop_provider_event_loop",
|
||||
new=mock_stop,
|
||||
),
|
||||
):
|
||||
async with lifespan(mock_app):
|
||||
pass
|
||||
assert order == ["start", "composition", "stop"]
|
||||
|
||||
# #region Test.AppModule.TestProviderLoopRollback [C:2] [TYPE Function]
|
||||
# @BRIEF Composition bootstrap failure stops the loop (fail-closed rollback) and re-raises.
|
||||
@pytest.mark.asyncio
|
||||
async def test_provider_loop_stopped_when_composition_fails(self):
|
||||
from src.app import lifespan
|
||||
mock_app = MagicMock(spec=FastAPI)
|
||||
mock_loop = MagicMock()
|
||||
mock_stop = MagicMock()
|
||||
with (
|
||||
patch("src.app.seed_trace_id"), patch("src.app.ensure_encryption_key"),
|
||||
patch("src.app.init_db"), patch("src.app.ensure_initial_admin_user"),
|
||||
patch("src.app.get_scheduler_service", return_value=MagicMock()),
|
||||
patch(
|
||||
"src.services.dashboard_testing.execution.provider_runtime.get_provider_event_loop",
|
||||
return_value=mock_loop,
|
||||
),
|
||||
patch("src.app.initialize_live_execution_composition", side_effect=RuntimeError("bootstrap-failed")),
|
||||
patch(
|
||||
"src.services.dashboard_testing.execution.provider_runtime.stop_provider_event_loop",
|
||||
new=mock_stop,
|
||||
),
|
||||
):
|
||||
with pytest.raises(RuntimeError, match="bootstrap-failed"):
|
||||
async with lifespan(mock_app):
|
||||
pass
|
||||
mock_loop.start.assert_called_once()
|
||||
mock_stop.assert_called_once()
|
||||
# #endregion Test.AppModule.TestProviderLoopRollback
|
||||
# #endregion Test.AppModule.TestProviderLoopOrdering
|
||||
|
||||
|
||||
class TestLifespanStuckGeneralTasks:
|
||||
"""lifespan — reconciliation of stuck general tasks (RUNNING -> FAILED)."""
|
||||
|
||||
@@ -18,14 +18,14 @@ from fastapi import Request, HTTPException
|
||||
|
||||
|
||||
# #region _make_mock_request [C:1] [TYPE Function]
|
||||
def _make_mock_request(method="GET", path="/api/test", client_host="127.0.0.1"):
|
||||
def _make_mock_request(method="GET", path="/api/test", client_host="127.0.0.1", query_params=None):
|
||||
req = MagicMock(spec=Request)
|
||||
req.method = method
|
||||
req.url = MagicMock()
|
||||
req.url.path = path
|
||||
req.client = MagicMock()
|
||||
req.client.host = client_host
|
||||
req.query_params = {}
|
||||
req.query_params = query_params if query_params is not None else {}
|
||||
return req
|
||||
# #endregion _make_mock_request
|
||||
|
||||
@@ -119,7 +119,12 @@ class TestLogRequests:
|
||||
# @TEST_EDGE: session_activity_no_log -> POST /api/auth/session/activity is suppressed
|
||||
@pytest.mark.asyncio
|
||||
async def test_log_session_activity_polling_skipped(self):
|
||||
from src.app import log_requests
|
||||
"""POST heartbeat must actually be suppressed — a status-only assert let the
|
||||
GET-only suppression bug ship (prod log spam: 66 lines per 2h)."""
|
||||
from src.app import _is_suppressed_request, log_requests
|
||||
assert _is_suppressed_request(
|
||||
_make_mock_request(method="POST", path="/api/auth/session/activity")
|
||||
) is True
|
||||
async def call_next(_r):
|
||||
resp = MagicMock(); resp.status_code = 200; return resp
|
||||
resp = await log_requests(
|
||||
@@ -371,6 +376,29 @@ class TestPollingSuppressionMethods:
|
||||
assert _is_suppressed_request(_make_mock_request("POST", "/api/maintenance/events")) is False
|
||||
assert _is_suppressed_request(_make_mock_request("POST", "/api/reports")) is False
|
||||
assert _is_suppressed_request(_make_mock_request("POST", "/api/tasks/run")) is False
|
||||
|
||||
# #region Test.AppModule.TestPollingScenarioRunsBadgeSuppressed [C:2] [TYPE Function]
|
||||
# @TEST_EDGE: scenario_runs_badge_poll_no_log -> GET /api/scenario-runs?waiting_for_me=true
|
||||
# (sidebar badge poll every 60s) is suppressed; run-center list fetches are not.
|
||||
def test_scenario_runs_badge_poll_suppressed(self):
|
||||
from src.app import _is_suppressed_request
|
||||
assert _is_suppressed_request(
|
||||
_make_mock_request(
|
||||
"GET", "/api/scenario-runs",
|
||||
query_params={"waiting_for_me": "true", "page_size": "1"},
|
||||
)
|
||||
) is True
|
||||
|
||||
def test_scenario_runs_list_not_suppressed(self):
|
||||
from src.app import _is_suppressed_request
|
||||
# Run-center list fetch (no waiting_for_me) keeps full framing.
|
||||
assert _is_suppressed_request(
|
||||
_make_mock_request("GET", "/api/scenario-runs", query_params={"page_size": "20"})
|
||||
) is False
|
||||
assert _is_suppressed_request(_make_mock_request("GET", "/api/scenario-runs")) is False
|
||||
# POST (launch a run) never suppressed.
|
||||
assert _is_suppressed_request(_make_mock_request("POST", "/api/scenario-runs")) is False
|
||||
# #endregion Test.AppModule.TestPollingScenarioRunsBadgeSuppressed
|
||||
# #endregion Test.AppModule.TestPollingPostBatchSuppressed
|
||||
|
||||
|
||||
|
||||
@@ -60,6 +60,27 @@ def _scenario_fixture_graph() -> dict:
|
||||
return json.loads((fixtures / "graph.json").read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def _published_catalog_snapshot() -> dict:
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
refresh = json.loads(
|
||||
(
|
||||
Path(__file__).resolve().parents[2]
|
||||
/ "specs" / "044-dashboard-scenario-execution" / "fixtures" / "production-contract-refresh.json"
|
||||
).read_text(encoding="utf-8")
|
||||
)
|
||||
pin = refresh["baseline_pin"]
|
||||
return {
|
||||
"baseline_set_id": pin["baseline_set_id"],
|
||||
"baseline_set_version": pin["baseline_set_version"],
|
||||
"release_id": pin["release_id"],
|
||||
"baseline_family": pin["baseline_family"],
|
||||
"catalog_digest": pin["catalog_digest"],
|
||||
"catalog_revision": refresh["catalog_revision"],
|
||||
}
|
||||
|
||||
|
||||
_COMPILE_REQUEST = {
|
||||
"agent_run_id": "550e8400-e29b-41d4-a716-446655440000",
|
||||
"objective": {"goal": "verify filters metric xlsx", "selected_case_ids": ["B01", "C04", "C05", "T01"], "rationale": "rc"},
|
||||
@@ -216,7 +237,8 @@ async def test_external_client_creates_and_activates_scenario_revision_end_to_en
|
||||
resolved = _unwrap(await server.call_tool("scenario_resolve", {"request": {
|
||||
"scenario": compiled_graph,
|
||||
"changes": [
|
||||
{"kind": "selector", "target": "phase-2-C04-download_xlsx", "value": "#download-btn", "reason": "pin export control"},
|
||||
{"kind": "selector", "target": "phase-2-B01-apply_native_filter", "value": "#native-filter", "reason": "pin native filter"},
|
||||
{"kind": "selector", "target": "phase-2-C04-download", "value": "#download-btn", "reason": "pin export control"},
|
||||
{"kind": "parameter", "target": "test_date", "value": "2026-08-01", "reason": "pin release window"},
|
||||
{"kind": "parameter", "target": "counterparty", "value": "ACME", "reason": "pin counterparty"},
|
||||
],
|
||||
@@ -251,12 +273,19 @@ async def test_external_client_creates_and_activates_scenario_revision_end_to_en
|
||||
# (start_scenario_run) resolves config through tools_scenario since decomposition Phase D.
|
||||
monkeypatch.setattr(tools_scenario_module, "get_config_manager", lambda: SimpleNamespace(get_environment=environments.get))
|
||||
monkeypatch.setattr(rbac_server_module, "get_config_manager", lambda: SimpleNamespace(get_environment=environments.get))
|
||||
snapshot = _published_catalog_snapshot()
|
||||
monkeypatch.setattr(
|
||||
"src.services.dashboard_testing.execution.runner.load_published_catalog",
|
||||
lambda injected=None: injected if injected is not None else snapshot,
|
||||
)
|
||||
started = _unwrap(await server.call_tool("start_scenario_run", {"request": {
|
||||
"scenario_id": scenario_id,
|
||||
"revision_id": activated.get("revision_id") or candidate_revision_id,
|
||||
"environment_id": "env-dev",
|
||||
"params": {"region": "emea", "currency": "EUR"},
|
||||
"idempotency_key": f"e2e-run-{uuid4()}",
|
||||
"baseline_set": "ss-prod-visual",
|
||||
"baseline_set_version": "1",
|
||||
}}))
|
||||
assert started["status"] == "queued", started
|
||||
with SessionLocal() as db:
|
||||
|
||||
@@ -0,0 +1,142 @@
|
||||
# Промежуточный отчёт: agentic dashboard-testing runtime — handoff для планировщика
|
||||
|
||||
**Дата:** 2026-09-10
|
||||
**Статус:** три runtime-среза реализованы (A+B, C, F), в рабочем дереве, **не закоммичены**. Документальный долг spec-refresh (Slice D) не закрыт. Отчёт — durable decision memory для следующего агента.
|
||||
|
||||
---
|
||||
|
||||
## 1. Что уже реализовано (обзор срезами)
|
||||
|
||||
### Срез A+B — E2E-дефекты 2026-09-08 + provider loop
|
||||
- **DEF-01** (promotion safety + diff-коррупция): structure-first validation `Services.AgentAuthoringWorkspace.PromotionGraphValidation` (canonical → 038-validator, gate-коды `{FORBIDDEN_SQL, FORBIDDEN_QUERY_CONTEXT, PATH_TRAVERSAL, CYCLE}`; legacy → free-text-скан), typed `apply_ops` по `DashboardTestScenario` + legacy-compat path `_apply_legacy_ops`, sqlparse-guard `ScenarioGraph.SqlGuard` вместо naive token-скана.
|
||||
- **DEF-02**: shared eligibility `ScenarioAutomation.Eligibility.Assert` (REST+MCP, predicate идентичен runner `_reject_automated_human_plan`).
|
||||
- **DEF-03**: compiler эмитит `dashboard_context.query` (server-issued query model).
|
||||
- **DEF-04**: atomic dequeue investigation queue (`queued→in_case→closed`, FOR UPDATE, `synchronize_session="evaluate"`).
|
||||
- **SEC-01**: шесть automation GET требуют authenticated principal.
|
||||
- **Provider loop**: `ProviderEventLoop.start()` до composition bootstrap + fail-closed rollback + stop на shutdown (`app.py` lifespan).
|
||||
- Верификация: `validate_contract_refresh.py` → `PASS: 11 schemas / 5 positive / 49 negative`.
|
||||
|
||||
### Срез C — DecisionPolicy mapper (единственный MISSING-модуль цепи)
|
||||
- `ScenarioExecution.DecisionPolicy` (`execution/decision_policy.py`): 15-строчная first-match truth table, `DecisionPolicy`/`EvaluationInput`/`StepPolicyInputs`/`PolicyDecision`, `map_human_disposition`, `policy_digest`.
|
||||
- Пиннинг политики в `RunnerPlan` (`derive_runner_plan` + `resolve_pinned_policy` legacy fallback), integration в walker (`step_outcome.decision_policy_id/version/reason_codes/comparison_ids/agent_evaluation_ids/deterministic_evidence_refs/decided_at`), human checkpoint через единый `map_human_disposition`.
|
||||
- Верификация: строки 1–15 покрыты hardcoded unit-тестами (`execution/test_decision_policy.py`, 21 case).
|
||||
|
||||
### Срез F — immutable AgentEvaluation runtime (замкнул DecisionPolicy rows 9–14)
|
||||
- Модели: `AgentEvaluationSpec` (стремится к 038-схеме), `AgentEvaluation` (pydantic граница + ORM `models/scenario_evaluation.py`).
|
||||
- ORM + миграция `0022_agent_evaluation` (идемпотентная — из-за `0001_baseline` `Base.metadata.create_all` добивка колонок через `inspector` guard).
|
||||
- Store/parser/provider: `ScenarioExecution.AgentEvaluation` — append-only CAS, P0-1 evidence-binding `validate_evaluation_evidence`, строгий парсер `parse_evaluation_response`, provider-seam `submit_evaluation` (reuse `LLMClient`+`LLMProviderService`, capacity `agent_evaluation`).
|
||||
- Registry/tool-action: `agent_evaluation`/`evaluate_declared_spec` в `TOOLS`/`REGISTERED_ACTIONS`/`ScenarioStep.tool`, `_derive_evaluation_mode` по `agent_evaluation_spec`.
|
||||
- Executor `ScenarioExecution.Executors.AgentEvaluation` (fail-closed) + walker ветка `evaluation_record → persist → step_outcome.agent_evaluation_ids`.
|
||||
- Верификация: 18 новых тестов (5 model + 7 parser + 6 store), 292 passed в 044-срезе, 52 passed editor/workspace/promotion/checkpoints.
|
||||
|
||||
---
|
||||
|
||||
## 2. Текущее состояние рабочего дерева (всё uncommitted)
|
||||
|
||||
Ключевые изменённые пути (срезы A+B+C+F плюс существующие пользовательские изменения maintenance и пр., которые НЕ трогать):
|
||||
|
||||
```
|
||||
# runtime slices
|
||||
backend/src/services/dashboard_testing/execution/{decision_policy,agent_evaluation}.py (новые)
|
||||
backend/src/services/dashboard_testing/execution/{runner,runner_plan,lifecycle,executors,artifacts}.py
|
||||
backend/src/services/dashboard_testing/scenario/{models,validator,compiler,sql_guard}.py
|
||||
backend/src/services/dashboard_testing/scenario/templates/__init__.py
|
||||
backend/src/services/dashboard_testing/editor/apply.py
|
||||
backend/src/services/dashboard_testing/automation/eligibility.py (новый)
|
||||
backend/src/services/dashboard_testing/analytics/investigation.py
|
||||
backend/src/services/agent_authoring_workspace/service.py
|
||||
backend/src/api/routes/dashboard_testing/scenario_automation.py
|
||||
backend/src/models/scenario_evaluation.py (новый)
|
||||
backend/src/models/scenario_artifact.py (+content_type/byte_length)
|
||||
backend/alembic/versions/0022_agent_evaluation.py (новый)
|
||||
backend/src/app.py (provider loop lifecycle)
|
||||
# тесты
|
||||
backend/tests/services/dashboard_testing/execution/test_{decision_policy,agent_evaluation}.py (новые)
|
||||
backend/tests/services/dashboard_testing/registry/test_{agent_evaluation_store,scenario_runner_walker,scenario_runner_plan}.py
|
||||
backend/tests/services/dashboard_testing/scenario/test_{sql_guard,agent_evaluation_models}.py
|
||||
backend/tests/test_app_lifespan.py, test_agent_authoring_workspace.py
|
||||
backend/tests/fixtures/scenario_execution/graph.json (action_registry_hash регенерирован)
|
||||
```
|
||||
|
||||
**Запрещённые файлы** (не менять): `.kilo/agent-manager.json`, `frontend/src/lib/api.ts`, `frontend/src/lib/api/__tests__/api.test.ts`, `specs-036-050-20260907-111314.md`.
|
||||
|
||||
---
|
||||
|
||||
## 3. Фронты работы для планировщика
|
||||
|
||||
### Фронт 1 — Доделать production-цепочку AgentEvaluation (высший приоритет, незавершённый срез F)
|
||||
**Проблема:** executor принимает `adapter`-инъекцию, но production-adapter, который строит полный `evaluation_record`, **не существует**; raw-response artifact не персистится до публикации оценки.
|
||||
Что осталось:
|
||||
- `evaluation_adapter_from(...)` / composition-root adapter: собрать `input_manifest` из артефактов прошлых шагов (`completed`), вызвать `submit_evaluation`, спарсить, **персистить raw-response artifact** (owner=run, тот же step/attempt), построить `evaluation_record` и вернуть через `step_outcome.evaluation_input`+`evaluation_record`.
|
||||
- Полная runner-интеграция: сегодня walker ждёт `evaluation_record` в `step_outcome`; закрыть путь adapter→record→persist→`agent_evaluation_ids` end-to-end.
|
||||
- Тест walker-интеграции: `agent_evaluation` шаг с mock-adapter → персист оценки + `step_outcome.agent_evaluation_ids` + DecisionPolicy row 14 (`BASELINE_AND_SEMANTIC_PASS`) при `evaluation_mode=required`.
|
||||
|
||||
### Фронт 2 — Версия ActionRegistry + синхронизация спецификации 038
|
||||
**Проблема:** добавлен `agent_evaluation`/`evaluate_declared_spec`, но `ACTION_REGISTRY_VERSION` остаётся `"038.2.0"` (я сознательно НЕ бампил — версия проверяется строгим `!=` в `resolve_action_descriptor`, зашита в тестах).
|
||||
Что сделать:
|
||||
- Bump `ACTION_REGISTRY_VERSION` → `"038.3.0"` + обновить все hardcoded `"038.2.0"` (`templates/__init__.py:19`, `test_scenario_runner_plan.py:51`, `test_scenario_runner.py:144/146/228`, `graph.json:4`) + перегенерировать `action_registry_hash`.
|
||||
- Синхронизировать `specs/038` (data-model.md уже упоминает `agent_evaluation_spec`/`decision_policy`; проверить соответствие tool-enum и версии реестра).
|
||||
|
||||
### Фронт 3 — DecisionPolicy rows 9–14 достижимы через walker (интеграционный тест)
|
||||
**Проблема:** строки 9–14 покрыты unit-тестами (`execution/test_decision_policy.py`), но walker-path доказательства нет (недостижимы из-за отсутствия production-adapter из фронта 1).
|
||||
Что сделать: интеграционный тест `required + missing → EVALUATION_UNAVAILABLE`, `required + high-conf pass → BASELINE_AND_SEMANTIC_PASS` через реальный walker с `agent_evaluation_spec`.
|
||||
|
||||
### Фронт 4 — Спек-документация (Slice D, долг с 2026-09-08)
|
||||
**Проблема:** spec-refresh workstream не закрыт.
|
||||
Что сделать:
|
||||
- Quickstarts: переписать 043/044/045 (banner «Refresh 2026-09-08», offline-валидатор, удаление agent-driven формулировок), баннеры 042/046/047, **создать quickstart 050**.
|
||||
- `docs/reports/ss-prod-agentic-e2e-spec-refresh-report-2026-09-08.md` (refresh report, отсутствует).
|
||||
- Обновить `docs/reports/ss-prod-agentic-e2e-session-state-2026-09-08.md` (P0/P1 residual findings → Closed second-pass).
|
||||
|
||||
### Фронт 5 — Оставшаяся production-цепочка (последующие срезы)
|
||||
- **Slice E** — compiler visual-chain emission (browser→capture→artifact→comparison→evaluation→policy) + browser-action coverage (filter/pagination/download transport; DEF-01/03 gap #4/#5) + reject unsupported actions. Частично требует live Playwright.
|
||||
- **Slice G** — ScenarioArtifact authenticated content GET/HEAD API (`artifact-content.openapi.yaml`, digest/MIME/retention, typed Error-Code) (gap #6).
|
||||
- **BaselineResolver/пиннинг 037** — runtime-резолв `BaselineSelectionPin` из published-каталога (gap: browser/screenshot провайдеры не зарегистрированы, LLM providers = 0).
|
||||
- **`Expected.threshold` модель-гap** — типизированное поле threshold в `Expected` (038 schema + models) вместо текущего кодирования в description.
|
||||
- **Live deploy** — регистрация browser/screenshot провайдеров, LLM-провайдер (Admin→LLM Settings), provider-loop canary (gap #3/#10; требует окружения/credential's — не offline).
|
||||
|
||||
### Фронт 6 — Pre-existing структурный долг (семантический)
|
||||
- `runner.py` ~1142 строк, `lifecycle.py` ~611, `executors.py` ~466, `browser.py` Factory 191 (INV_7 violations, зафиксированы audit_contracts).
|
||||
- Orphan/unresolved relations: `BaselineEngine.Catalog.Materialization`, `McpServer.ScenarioTools.register_draft_pack`, `ScenarioGraph.Models.CanonicalBytes` (workspace-wide 409 unresolved).
|
||||
- Pydantic serializer warnings в `test_validator_branches` (pre-existing).
|
||||
|
||||
---
|
||||
|
||||
## 4. Верификация (реальные прогоны, актуальны на 2026-09-10)
|
||||
|
||||
```bash
|
||||
cd backend && source .venv/bin/activate
|
||||
|
||||
# регрессия 044-среза (включая новые tests)
|
||||
python -m pytest -q tests/services/dashboard_testing/registry/test_scenario_*.py \
|
||||
tests/api/test_scenario_runs_api.py tests/api/test_scenario_automation_api.py \
|
||||
tests/api/test_scenario_analytics_api.py # 292 passed
|
||||
|
||||
# editor/workspace/promotion/checkpoints
|
||||
python -m pytest -q tests/services/dashboard_testing/registry/test_scenario_editor_ops.py \
|
||||
tests/services/dashboard_testing/registry/test_scenario_editor_agent.py \
|
||||
tests/services/test_agent_authoring_workspace.py tests/test_mcp_authoring_promotion_e2e.py \
|
||||
tests/test_mcp_checkpoints.py # 52 passed
|
||||
|
||||
# decision policy + agent evaluation
|
||||
python -m pytest -q tests/services/dashboard_testing/execution \
|
||||
tests/services/dashboard_testing/registry/test_agent_evaluation_store.py \
|
||||
tests/services/dashboard_testing/scenario/test_agent_evaluation_models.py # 21+18
|
||||
|
||||
python -m ruff check . # All checks passed
|
||||
python -m compileall -q src # clean
|
||||
python -m alembic heads # 0022_agent_evaluation (one head)
|
||||
cd .. && backend/.venv/bin/python specs/044-dashboard-scenario-execution/prototype/validate_contract_refresh.py \
|
||||
specs/044-dashboard-scenario-execution/fixtures/production-contract-refresh.json # 11/5/49 PASS
|
||||
git diff --check -- backend # clean
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 5. Ограничения и инварианты (для следующего агента)
|
||||
|
||||
- **GRACE-Poly:** регионы `#region/#endregion` на всех новых функциях; `@`-теги в одну строку (indexer читает только первую); реальные решения — `@RATIONALE`/`@REJECTED`; INV_7 новые модули <400 строк; INV_9 без синтетики. После правок анкеров — `axiom_search rebuild` + `read_outline` (0 parse warnings).
|
||||
- **Без новых зависимостей** (только `sqlparse` уже pinned); **без новых migrations** кроме 0022; error-коды стабильны (`EVALUATION_*`, `AUTOMATION_INELIGIBLE_HUMAN_STEP`, `QUEUE_ITEM_INACTIVE`, `FORBIDDEN_SQL`, `*_LOOP_UNAVAILABLE`).
|
||||
- **Нормативные `.schema.json` в specs/ не менять** — они authority; runtime-зеркала P0-1/P0-2 должны совпадать с офлайн-валидатором.
|
||||
- **Среда примечания:** воркер-сессия могла видеть `DATABASE_URL=__MUST_SET_DATABASE_URL__` placeholder; pytest при этом работает через SQLite conftest (root `conftest.py` делает `alembic upgrade head` на file-based SQLite). Cross-directory комбинированные прогоны `tests/` rootdir (совместно с `test_app_lifespan`/`test_mcp_checkpoints`) дают pre-existing fixture-конфликт — выполнять суиты в их собственном контексте.
|
||||
- Реальную runtime-имплементацию не отмечать «завершена» в tasks.md/specs без executable trace (правило session-state).
|
||||
200
docs/reports/agentic-runtime-orchestration-plan-2026-09-10.md
Normal file
200
docs/reports/agentic-runtime-orchestration-plan-2026-09-10.md
Normal file
@@ -0,0 +1,200 @@
|
||||
# Orchestration plan: agentic runtime handoff 2026-09-10
|
||||
|
||||
**Source:** `docs/reports/agentic-runtime-implementation-handoff-2026-09-10.md`
|
||||
**Role:** architect (no implementation in this context)
|
||||
**Date:** 2026-09-10
|
||||
|
||||
## Goal
|
||||
|
||||
Close the unfinished production AgentEvaluation chain (Front 1), bump ActionRegistry to `038.3.0` (Front 2), prove DecisionPolicy rows 9–14 through the walker (Front 3), and close Slice D spec-refresh docs (Front 4). Fronts 5–6 stay out of this pass.
|
||||
|
||||
## Already landed (do not re-implement)
|
||||
|
||||
Slices A+B, C, F store/parser/executor/walker persist branch. See the handoff. Uncommitted working tree.
|
||||
|
||||
## Decomposition
|
||||
|
||||
```
|
||||
handoff
|
||||
├─ Worker A (Implement) Front 1 then Front 3 [code, sequential in one worker]
|
||||
├─ Worker B (Implement) Front 2 [registry version, parallel]
|
||||
└─ Worker C (Implement) Front 4 [docs only, parallel]
|
||||
```
|
||||
|
||||
Dispatched 2026-09-10 (shared workspace, disjoint file ownership D6):
|
||||
|
||||
| Worker | subagent_id | Front | Envelope |
|
||||
|---|---|---|---|
|
||||
| A | `01a08aa2-fe79-7662-bd5f-756d2d21187a` | 1+3 AgentEvaluation adapter + walker rows 9–14 | `done` (self-reported 68 + 261; 044-slice count vs handoff 292 is unverified) |
|
||||
| B | `01a08aa2-fe79-7662-bd5f-757ca5f33d82` | 2 ACTION_REGISTRY_VERSION 038.3.0 | `done` (self-reported 50; printed `038.3.0 4482c3c1834f9fdbafdcb93ea6b62d4c9c4c048ba24d3b46ae874adb6caab738`) |
|
||||
| C | `01a08aa2-fe79-7662-bd5f-758159e767dd` | 4 Slice D spec-refresh docs | `done` (self-reported validator 11/5/49; remaining still lists Fronts 1–3 as open — stale parallel) |
|
||||
| V | `01a08ab3-ca0c-7b72-817d-a16ef6d9ff1c` | independent verify of A+B+C | `done` — 044-slice glob 261 passed; execution+store+models+live_binding+walker 68 passed; 261+31 execution-only = handoff 292; ruff/compileall clean; validator 11/5/49; `038.3.0` fingerprint matches graph.json |
|
||||
|
||||
Worker A extra decision (accepted): fail-closed `agent_evaluation` outcomes emit `comparison_status=pass` so DecisionPolicy row 9 (`EVALUATION_UNAVAILABLE`) is reachable instead of row 7 (`COMPARISON_INCONCLUSIVE`). Storage resolved at invoke, not compose time.
|
||||
|
||||
Front 3 depends on Front 1 (same worker). Front 2 must not edit adapter/walker files. Front 4 must not edit runtime source.
|
||||
|
||||
## Architect decisions (durable)
|
||||
|
||||
### D1 — input_manifest ownership vs P0-1
|
||||
|
||||
`validate_evaluation_evidence` currently requires every manifest/finding/raw ref to be a `ScenarioArtifact` of the same `run + logical_step_id + attempt`. The production adapter must assemble `input_manifest` from **previous** steps (`completed`). Those rows will fail `EVALUATION_EVIDENCE_NOT_FOUND` as written.
|
||||
|
||||
**Chosen:** keep raw-response bound to the evaluation step/attempt; relax the lookup for `input_manifest` items and finding `evidence_artifact_ids` to same-run, `owner_type=scenario_run`, `is_active` artifacts (prior steps allowed). Foreign-run artifacts remain rejected.
|
||||
|
||||
**Rejected:** duplicating previous-step artifact rows onto the evaluation step — that forges a second provenance identity for the same digest.
|
||||
|
||||
### D2 — raw-response persist before publish
|
||||
|
||||
Adapter stores raw JSON bytes, returns `artifact_refs` + `artifact_digests` (+ MIME/byte_length). Walker `register_step_evidence` must persist the evaluation-step raw artifact **before** `validate_evaluation_evidence` + `persist_agent_evaluation`. Succeeded records still require bound raw-response (existing `@INVARIANT`).
|
||||
|
||||
If `register_step_evidence` currently omits `content_type`/`byte_length`, extend it from the outcome payload so metadata match does not fail.
|
||||
|
||||
### D3 — adapter shape
|
||||
|
||||
Executor stays sync `(step, completed) -> dict`. `evaluation_adapter_from(...)` closes over `db`/`run_async` the same way live Superset adapters close over `run_async`. `submit_evaluation` remains the provider seam; tests inject a mock adapter and never call a live LLM.
|
||||
|
||||
Composition root: add `agent_evaluation_adapter()` on `LiveExecutionCompositionRoot`. Runner already calls it via `hasattr`. Put the factory in a new module if `live_composition.py` would exceed INV_7 (400 LOC).
|
||||
|
||||
### D4 — out of scope this pass
|
||||
|
||||
Slice E (compiler visual-chain / live Playwright), Slice G (artifact content GET/HEAD), BaselineResolver 037, `Expected.threshold`, live provider registration, INV_7 splits of `runner.py`/`lifecycle.py`, orphan relations, Pydantic serializer warnings. No new Python dependencies. No new Alembic migrations besides existing `0022`. Do not mutate normative `*.schema.json`. Do not mark `tasks.md` complete without an executable verifier trace.
|
||||
|
||||
### D5 — forbidden files
|
||||
|
||||
Do not change: `.kilo/agent-manager.json`, `frontend/src/lib/api.ts`, `frontend/src/lib/api/__tests__/api.test.ts`, `specs-036-050-20260907-111314.md`. Do not touch unrelated user maintenance diffs.
|
||||
|
||||
### D6 — file ownership (parallel workers)
|
||||
|
||||
| Worker | Owns |
|
||||
|---|---|
|
||||
| A | `execution/agent_evaluation.py`, new adapter module, `live_composition.py`, `executors.py`, `artifacts.py` (only if MIME/length needed), `runner.py` only if persist order is broken, walker tests, composition-root tests |
|
||||
| B | `ACTION_REGISTRY_VERSION` + hardcoded `038.2.0` → `038.3.0`, `graph.json` hash, `specs/038` tool-enum/version sync (`action-registry.yaml` version, data-model/spec mentions). Not `*.schema.json` |
|
||||
| C | quickstarts 043/044/045 rewrite + 042/046/047 banners + **create** 050 quickstart; `docs/reports/ss-prod-agentic-e2e-spec-refresh-report-2026-09-08.md`; update `docs/reports/ss-prod-agentic-e2e-session-state-2026-09-08.md`. No `backend/` or `frontend/` |
|
||||
|
||||
## Acceptance
|
||||
|
||||
Worker A:
|
||||
- Production adapter builds `input_manifest` from `completed`, calls `submit_evaluation` (or injected client), parses, persists raw-response artifact, returns `evaluation_input` + `evaluation_record`.
|
||||
- Walker path: mock-adapter → persist evaluation → `step_outcome.agent_evaluation_ids` non-empty.
|
||||
- Walker: `required + missing → EVALUATION_UNAVAILABLE`; `required + high-conf pass → BASELINE_AND_SEMANTIC_PASS`.
|
||||
- Pytest: 044-slice + `tests/services/dashboard_testing/execution` + new walker cases. Ruff + compileall.
|
||||
|
||||
Worker B:
|
||||
- Runtime `ACTION_REGISTRY_VERSION == "038.3.0"`. All previously hardcoded `038.2.0` in listed files updated. `action_registry_hash` regenerated. `resolve_action_descriptor` still strict `!=`. Spec 038 yaml version matches.
|
||||
|
||||
Worker C:
|
||||
- Quickstarts carry «Refresh 2026-09-08», offline validator, no agent-driven product-UI flow. 050 quickstart exists. Refresh report exists. Session-state P0/P1 marked Closed second-pass only where contracts/validator already prove it (do not close unimplemented runtime gaps).
|
||||
|
||||
## Verification commands (workers run these)
|
||||
|
||||
```bash
|
||||
cd backend && source .venv/bin/activate
|
||||
python -m pytest -q tests/services/dashboard_testing/registry/test_scenario_*.py \
|
||||
tests/api/test_scenario_runs_api.py tests/api/test_scenario_automation_api.py \
|
||||
tests/api/test_scenario_analytics_api.py
|
||||
python -m pytest -q tests/services/dashboard_testing/execution \
|
||||
tests/services/dashboard_testing/registry/test_agent_evaluation_store.py \
|
||||
tests/services/dashboard_testing/scenario/test_agent_evaluation_models.py
|
||||
python -m ruff check src/services/dashboard_testing/execution \
|
||||
src/services/dashboard_testing/scenario src/models/scenario_evaluation.py
|
||||
python -m compileall -q src
|
||||
cd .. && backend/.venv/bin/python \
|
||||
specs/044-dashboard-scenario-execution/prototype/validate_contract_refresh.py \
|
||||
specs/044-dashboard-scenario-execution/fixtures/production-contract-refresh.json
|
||||
```
|
||||
|
||||
Cross-directory combined `tests/` rootdir with `test_app_lifespan`/`test_mcp_checkpoints` has a pre-existing fixture conflict — run suites in their own context.
|
||||
|
||||
## Closure (2026-09-10, after independent verify)
|
||||
|
||||
| Front | Status |
|
||||
|---|---|
|
||||
| 1 Production AgentEvaluation adapter | Applied. `evaluation_adapter_from` + `LiveExecutionCompositionRoot.agent_evaluation_adapter()`. D1 prior-step same-run manifest; raw-response this step/attempt. |
|
||||
| 2 ACTION_REGISTRY_VERSION 038.3.0 | Applied. Fingerprint `4482c3c1834f9fdbafdcb93ea6b62d4c9c4c048ba24d3b46ae874adb6caab738` matches `graph.json`. YAML 038.3.0. Schema.json still says 038.2.0 (D4). |
|
||||
| 3 Walker rows 9–14 | Applied. Hardcoded `EVALUATION_UNAVAILABLE` and `BASELINE_AND_SEMANTIC_PASS`. Fail-closed executor emits `comparison_status=pass` so row 9 is reachable. |
|
||||
| 4 Slice D docs | Applied. 050 quickstart created; refresh report created; stale Front 1–3 OPEN claims corrected after verify. |
|
||||
| 5–6 | Remaining. Slice E/G, BaselineResolver, live providers, INV_7 splits, orphan relations. |
|
||||
|
||||
**Verified (independent worker, not implementer self-report):** 044-slice glob 261 passed; execution+store+models+live_binding+walker 68 passed; 261+31 execution-only = handoff 292; ruff clean; compileall 0; validator 11/5/49 PASS.
|
||||
|
||||
**Decision memory:** this file + handoff + Worker A D1/D2/D3 + fail-closed `comparison_status=pass` for row 9.
|
||||
|
||||
**Next action:** Front 5 (Slice E compiler visual-chain, Slice G artifact content API, BaselineResolver, live deploy). Do not mark `tasks.md` complete without a live-stand trace. Working tree remains uncommitted.
|
||||
|
||||
---
|
||||
|
||||
## Front 5 pass (2026-09-10 continued)
|
||||
|
||||
### Goal
|
||||
|
||||
Offline production-chain remaining work: Slice G content API, Slice E compiler visual-chain (no Playwright), BaselineResolver from existing 037 catalog bytes. Live deploy and `Expected.threshold` stay out.
|
||||
|
||||
### Decomposition
|
||||
|
||||
```
|
||||
Front 5
|
||||
├─ Worker G (Implement) Slice G artifact GET/HEAD
|
||||
├─ Worker E (Implement) Slice E compiler chain + canonical actions + 038.4.0
|
||||
└─ Worker R (Implement) BaselineResolver pin before start_run I/O
|
||||
```
|
||||
|
||||
Dispatched:
|
||||
|
||||
| Worker | subagent_id | Envelope |
|
||||
|---|---|---|
|
||||
| G | `01a08ac0-d962-7783-9ca4-3af67526247b` | `done` — 23 content tests; dedicated service+router |
|
||||
| E | `01a08ac0-d962-7783-9ca4-3b03e80ab8e5` then `01a08ace-232c-7661-aced-9fa928c49ad2` | `done` — 038.4.0 `9839099380356b2ebf482eb3b3c34c64320dc88fcc4694288e8d4278291c7310`; MCP selector pins canonical |
|
||||
| R | `01a08ac0-d962-7783-9ca4-3b195dcde610` then `01a08ad2-d9e8-7013-b8dd-795555ffb00c` | first `done`; 044-slice/MCP start BASELINE_MISSING test seam in flight |
|
||||
| V | `01a08acd-f3c6-7e52-bbf6-6b35ad0545d6` then `01a08ad6-3ebc-7610-9a11-3cc0befa07ed` | first `blocked` on 044-slice/MCP seam; re-verify `done` — 273 044-slice, MCP e2e 2, resolver 8 fail-closed, G 23, compiler 19, ruff/compileall/validator PASS |
|
||||
|
||||
### Decisions
|
||||
|
||||
### D7 — Slice G module boundary
|
||||
New service `execution/artifact_content.py` + new router `api/routes/dashboard_testing/scenario_artifact_content.py` included from the dashboard-testing package. Do not grow `scenario_runs.py` (already INV_7). Path: `GET|HEAD /api/scenario-runs/{run_id}/artifacts/{artifact_id}/content`. Contract: `specs/044-dashboard-scenario-execution/contracts/artifact-content.openapi.yaml` and `production-chain.md` ScenarioArtifactContentService.
|
||||
|
||||
### D8 — Slice G ACL and errors
|
||||
Permission: `("scenario:result", "VIEW")` on every request (`scenario-result:view`). Unauthenticated → 401 `AUTHENTICATION_REQUIRED`. Known parent, no permission → 403 `PERMISSION_DENIED`. Foreign/nonexistent child → 404 `NOT_FOUND` (indistinguishable). Envelope `{code, message, retryable, correlation_id}` plus `Error-Code` header. HEAD: identical status/headers, empty body on success and error. Range after ACL → 416 `RANGE_NOT_SUPPORTED`. No storage paths in any body.
|
||||
|
||||
### D9 — Slice E canonical names + registry 038.4.0
|
||||
Compiler emits canonical `browser-actions.md` names (`apply_native_filter`, `download`, `edit_row`, …). Legacy aliases (`text_filter`, `apply_filters`, `download_xlsx`, `row_edit`, `table_filter`) are compile-time migrations only — never emitted on new graphs. Adding canonical `{tool,action}` pairs changes `action_registry_fingerprint` → bump `ACTION_REGISTRY_VERSION` `038.3.0` → `038.4.0` and regenerate `graph.json` hash. Worker E owns the bump. Old 038.3.0 plans remain historical; new compiles pin 038.4.0.
|
||||
|
||||
### D10 — Slice E visual chain shape
|
||||
For an automated visual/browser case the compiler emits a depends_on chain: canonical browser action → `capture_screenshot` → `compare_to_baseline`. Artifact bytes are provider output, not a compiled `artifact/register` step. `evaluate_declared_spec` is emitted only when `CompileScenarioRequest` carries a validated `AgentEvaluationSpec` (new optional field; omit rather than invent a spec). DecisionPolicy is the mapper, not a compiled step. Metric cases: observe/assert action already in template → `compare_to_baseline` when baseline capability is present. Selected unsupported cases become blockers (`UNSUPPORTED_ACTION`), not skip-warnings. Unselected catalog coverage may still warn. Live Playwright/driver canary is out of this worker.
|
||||
|
||||
### D11 — BaselineResolver, no new migration
|
||||
New `execution/baseline_resolver.py`. Resolve `BaselineSelectionPin` from **published 037 catalog bytes** (existing `baseline_catalog` loader / injected catalog snapshot), never from client digests and never “latest”. Errors: `BASELINE_MISSING`, `BASELINE_STALE`, `BASELINE_AMBIGUOUS`, `BASELINE_NOT_PUBLISHED`, `BASELINE_EVIDENCE_UNAVAILABLE` → start_run 422; pin/CAS change → 409. Absent `baseline_set` is legal only when the graph has no baseline refs (null pin). Pin is part of canonical request hash, RunnerPlan, ScenarioRun.target_snapshot. No new Alembic revision. Do not call live Git publication.
|
||||
|
||||
### D12 — still out
|
||||
Live browser/screenshot/LLM registration and provider-loop canary (needs credentials). `Expected.threshold` (would mutate normative `*.schema.json`). Front 6 INV_7 splits of `runner.py`/`lifecycle.py`. `tasks.md` completion marks. Forbidden files unchanged (D5).
|
||||
|
||||
### File ownership
|
||||
|
||||
| Worker | Owns |
|
||||
|---|---|
|
||||
| G | `execution/artifact_content.py`, `api/routes/dashboard_testing/scenario_artifact_content.py`, package include, `tests/api/test_scenario_artifact_content_api.py` (and/or service tests). Not compiler, not runner.py, not templates version. |
|
||||
| E | `scenario/compiler.py`, `scenario/templates/__init__.py`, compiler tests, `graph.json` hash + hardcoded `038.3.0`→`038.4.0` in tests that pin the current registry. Not artifact content API, not runner.py. |
|
||||
| R | `execution/baseline_resolver.py`, minimal `start_run`/`derive_runner_plan`/`_request_hash` wiring, resolver tests. Not compiler emission, not artifact content router. |
|
||||
|
||||
### Acceptance
|
||||
|
||||
Worker G: GET 200 real JPEG/PNG/JSON bytes with required headers; HEAD 200 no body same headers; 401/403/404/409/410/413/416/503 typed Error-Code; Range rejected after ACL; foreign child 404; integrity mismatch 409 no bytes.
|
||||
|
||||
Worker E: visual selected case emits ≥3 chained steps with canonical actions; unsupported selected case blocks compile; aliases not emitted; 038.4.0 fingerprint matches graph.json; existing compiler determinism tests still pass (updated expected step counts if needed).
|
||||
|
||||
Worker R: hardcoded published catalog fixture → pin on plan/run; missing/stale/unpublished fail before run row I/O of providers; null pin when no baseline refs; request_hash includes pin.
|
||||
|
||||
Live deploy remains blocked.
|
||||
|
||||
## Closure Front 5 offline (2026-09-10, after independent re-verify)
|
||||
|
||||
| Slice | Status |
|
||||
|---|---|
|
||||
| G content GET/HEAD | Applied. Service + dedicated router. 23 tests. VIEW ACL, typed Error-Code, HEAD empty body. |
|
||||
| E compiler chain | Applied. Canonical browser names, `UNSUPPORTED_ACTION` blockers, `038.4.0` fingerprint `9839099380356b2ebf482eb3b3c34c64320dc88fcc4694288e8d4278291c7310`. Live Playwright still OPEN. |
|
||||
| R BaselineResolver | Applied. Pin before ScenarioRun row; `BASELINE_*` → 422. Registry/MCP tests inject published 044 fixture; production stays fail-closed (8 resolver tests). |
|
||||
|
||||
**Verified:** 044-slice glob 273 passed; MCP e2e 2; resolver 8; G 23; compiler 19; ruff clean; compileall 0; validator 11/5/49 PASS.
|
||||
|
||||
**Still out:** live browser/screenshot/LLM registration, provider-loop canary, `Expected.threshold`, Front 6 INV_7 splits, Git CatalogRevision store. Working tree uncommitted. `tasks.md` not marked complete.
|
||||
|
||||
117
docs/reports/ss-prod-agentic-e2e-session-state-2026-09-08.md
Normal file
117
docs/reports/ss-prod-agentic-e2e-session-state-2026-09-08.md
Normal file
@@ -0,0 +1,117 @@
|
||||
# Session state — SS-Prod agentic dashboard E2E/spec refresh
|
||||
|
||||
Updated: 2026-09-10 (Slice D documentation pass). Original E2E/audit date remains 2026-09-08.
|
||||
|
||||
## User objective
|
||||
|
||||
Run and assess the complete SS-Prod dashboard-testing scenario lifecycle, compare actual execution with specifications, and refresh all related specifications for production-ready browser/screenshot/LLM-agentic E2E.
|
||||
|
||||
## Explicit product constraint
|
||||
|
||||
Frontend UI/UX must contain no agent interaction: no agent chat/prompt, agent-assisted editing, proposal generation, agent workspace, or agent-start controls. Agent interaction belongs to an external MCP client. Product UI may provide manual CRUD/editor, human approval/review, monitoring, and read-only screenshot/evaluation results. Any existing agent-editing panel is runtime drift and requires removal; runtime frontend code was not changed in this spec-refresh task.
|
||||
|
||||
## Completed evidence
|
||||
|
||||
- SS-Prod E2E executed with valid ss-tools credentials supplied by the user.
|
||||
- MCP: 35 tools, 28 PASS / 1 FAIL / 5 BLOCKED / 1 N/A.
|
||||
- REST: 61 operations, 58 PASS / 0 FAIL / 3 BLOCKED.
|
||||
- Durable create/read/reload/update/manual-run/analytics/automation/cancel/archive/cleanup lifecycle exercised.
|
||||
- Automation objects deleted; browser auth state cleared.
|
||||
- Browser/screenshot providers were unregistered; LLM provider was absent; provider-loop startup had an application wiring defect.
|
||||
- Three test scenarios remain archived because the API has no hard-delete for the relevant records; residual owned records are listed in the E2E report.
|
||||
- Main defects: MCP graph promotion safety rejection; MCP/REST automation eligibility mismatch; draft compiler missing `dashboard_context.query`; analytics queue case not atomically dequeued; unauthenticated automation reads.
|
||||
|
||||
Those five defects were **runtime** findings. Spec refresh does not close them. The 2026-09-10 runtime handoff records slices A+B (DEF-01..04, SEC-01, provider loop), C (DecisionPolicy), and F (AgentEvaluation store/parser/executor) as landed in the working tree and **uncommitted**. Production GO remains OPEN.
|
||||
|
||||
## Reports
|
||||
|
||||
- `docs/reports/ss-prod-dashboard-scenario-e2e-report-2026-09-08.md`
|
||||
- `docs/reports/ss-prod-agentic-e2e-production-gap-2026-09-08.md`
|
||||
- `docs/reports/ss-prod-agentic-e2e-spec-coverage-2026-09-08.md`
|
||||
- `docs/reports/ss-prod-agentic-e2e-baseline-gap-2026-09-08.md`
|
||||
- `docs/reports/ss-prod-agentic-e2e-spec-refresh-plan-2026-09-08.md`
|
||||
- `docs/reports/ss-prod-agentic-e2e-spec-refresh-report-2026-09-08.md` **(created Slice D)**
|
||||
- `docs/reports/agentic-runtime-implementation-handoff-2026-09-10.md`
|
||||
- `docs/reports/agentic-runtime-orchestration-plan-2026-09-10.md`
|
||||
|
||||
## Spec refresh scope
|
||||
|
||||
Twelve packages are being refreshed: `017`, `036`, `037`, `038`, `039`, `042`, `043`, `044`, `045`, `046`, `047`, `050`.
|
||||
|
||||
Required canonical chain:
|
||||
|
||||
`browser action → screenshot → artifact receipt/content → deterministic metric/visual baseline comparison → optional AgentEvaluation → DecisionPolicy → StepOutcome → ScenarioRun result/analytics`.
|
||||
|
||||
Required normative areas include browser action mapping, authenticated artifact bytes/MIME/digest/ownership/retention, provider lifecycle/cancel/reconcile, immutable evaluation and closed policy truth table, unified result/evidence DTO, baseline CatalogRevision/CAS/rebaseline/publishing, baseline pinning through request/idempotency/RunnerPlan/run/result/analytics, REST/MCP parity, automation/analytics transitions, and strict LLM parser/fallback/provenance.
|
||||
|
||||
Baseline is split into visual, numerical metric, and optional future performance/quality baselines. The optional performance baseline is not part of the current implementation scope.
|
||||
|
||||
## Changes made by the implementation worker so far
|
||||
|
||||
- All 12 `spec.md` files received production FR updates.
|
||||
- New/updated machine-readable contracts include baseline catalog revision/pin, AgentEvaluation/spec/policy, result/evidence, artifact content, provider/lifecycle, and MCP tool contracts.
|
||||
- `050` gained missing `plan.md`, `data-model.md`, `contracts/modules.md`, and strict tool-contract schema.
|
||||
- Active UI/UX requirements in `039/043/050` were rewritten toward manual CRUD/editor and external-MCP-only agent interaction; AgentEvaluationCard is read-only.
|
||||
- `044` contract refresh validator and fixtures were added:
|
||||
`specs/044-dashboard-scenario-execution/prototype/validate_contract_refresh.py`
|
||||
with `specs/044-dashboard-scenario-execution/fixtures/production-contract-refresh.json`.
|
||||
- Initial validator run passed: 11 schemas/refs, 3 positive fixtures, 15 hardcoded negative regressions.
|
||||
- All 12 `tasks.md` files were reported as updated; disputed completed marks such as `044` AgentEvaluation/provider bootstrap/tests were reopened where implementation trace was absent.
|
||||
|
||||
## Latest independent verification findings
|
||||
|
||||
The verifier confirmed the initial refresh validator and found residual issues. **Second-pass contract/validator status (Slice D, 2026-09-10):** Closed only where the offline validator or the OpenAPI contract already encodes the invariant. Runtime/live gaps stay OPEN.
|
||||
|
||||
P0:
|
||||
|
||||
1. **Closed second-pass (contracts/validator).** Evaluation evidence ownership: `validate_evaluation_evidence` requires manifest/raw-response/finding artifacts to exist in the result evidence index and match owner, run, digest, MIME, and length. Runtime D1 (2026-09-10): manifest/finding refs may be prior-step same-run active artifacts; raw-response stays this step+attempt; foreign-run stays rejected. Hardcoded negatives remain in the offline validator. This does **not** close live evidence bytes (Slice G).
|
||||
2. **Closed second-pass (contracts/validator).** Evaluation criteria executable validation: unique `criterion_id`, `comparison_id` membership in `comparison_refs`, kind-bound comparison (deterministic bound / semantic unbound), finding criterion existence/kind match. Hardcoded negatives: `duplicate-criterion-id`, `criterion-comparison-not-member`, `deterministic-criterion-unbound`, `semantic-criterion-bound`, `finding-unknown-criterion`, `finding-kind-mismatch`, `manifest-outside-spec`, `evaluation-comparison-outside-spec`, `deterministic-criterion-outside-evaluation`.
|
||||
|
||||
P1:
|
||||
|
||||
1. **Closed second-pass (contracts/validator).** Publication state truth table rejects invalid `materialized` commit/receipt combinations (`materialized-with-commit`, `materialized-with-receipt`, `committed-with-receipt`, `published-null-commit`, `published-no-receipt`, `failed-no-error`). Runtime explicit Git publication remains OPEN.
|
||||
2. **Closed second-pass (contracts/validator).** Numeric comparison tolerances: non-negative values, `min <= max`, row-width invariants (`negative-tolerance`, `inverted-range`, `row-width-mismatch`).
|
||||
3. **Closed second-pass (OpenAPI contract, not JSON-Schema fixtures).** HEAD artifact error-code headers reuse the same status-specific Error-Code components as GET in `specs/044-dashboard-scenario-execution/contracts/artifact-content.openapi.yaml`. Runtime GET/HEAD landed 2026-09-10 offline; live storage canary remains OPEN.
|
||||
4. **Closed second-pass (contracts/validator).** Positive catalog fixture includes a visual entry revision; visual review/status branches are exercised (`visual-review-not-confirmed`, `visual-review-no-disposition`, `metric-review-bound`).
|
||||
|
||||
Earlier issues reported as fixed include publication null commit/receipt, visual review conditional, wrapper identity/status, strict catalog revision entries, baseline pin discrimination/duplicates, typed comparison statuses, artifact owner/URL, non-empty `passed` closure, image size/provenance, canonical `$id`, DecisionPolicy precedence, and GET/HEAD response headers.
|
||||
|
||||
## Slice D documentation pass (2026-09-10)
|
||||
|
||||
Closed in this pass (docs only; no `backend/` or `frontend/` edits; `tasks.md` checkboxes not marked complete):
|
||||
|
||||
- Quickstarts 043/044/045 rewritten: banner «Refresh 2026-09-08 (production contract)», offline validator, existing pytest/vitest commands, agent-driven product-UI flow removed, target vs observed distinguished, no production GO claim.
|
||||
- Banners added to 042/046/047. 047 no longer states DEF-04 as OPEN: runtime path cited (`investigation.py` `open_case` FOR UPDATE + `set_disposition` `synchronize_session="evaluate"`); production GO still OPEN.
|
||||
- `specs/050-mcp-interface/quickstart.md` created (was missing). MCP-only agent interaction; catalog `2.0.0` / `PINNED_CATALOG_MAJOR=2`; no frontend agent workspace.
|
||||
- Refresh report created: `docs/reports/ss-prod-agentic-e2e-spec-refresh-report-2026-09-08.md`.
|
||||
|
||||
Runtime A+B/C/F plus Fronts 1–3 (production adapter, walker rows 9–14, `ACTION_REGISTRY_VERSION` `038.3.0`) landed 2026-09-10 in the working tree, independently verified, still **uncommitted**. Production GO remains OPEN.
|
||||
|
||||
## Remaining OPEN (do not mark Closed)
|
||||
|
||||
- Slice E live Playwright — driver/reconciler canary; compiler visual-chain offline landed 2026-09-10 (`038.4.0`).
|
||||
- Slice G live storage canary — GET/HEAD runtime landed 2026-09-10 offline.
|
||||
- BaselineResolver live Git publication — pin-from-injected-snapshot landed 2026-09-10; no CatalogRevision ORM.
|
||||
- Live providers — browser/screenshot registration, LLM Settings, provider-loop canary on a live stand.
|
||||
- 050 T029m live-stand replay; T044–T046 REST/MCP consume/publish parity; T030 negative UI acceptance.
|
||||
- Optional approved performance/quality baseline (out of scope).
|
||||
|
||||
## Next action
|
||||
|
||||
Fronts 1–5 offline of `docs/reports/agentic-runtime-orchestration-plan-2026-09-10.md` are closed in the working tree (independently verified). Next is live deploy (browser/screenshot/LLM registration, provider-loop canary) and Front 6 INV_7 splits — not this pass.
|
||||
|
||||
Offline contract gate (re-run anytime; does not require runtime edits):
|
||||
|
||||
```bash
|
||||
backend/.venv/bin/python specs/044-dashboard-scenario-execution/prototype/validate_contract_refresh.py \
|
||||
specs/044-dashboard-scenario-execution/fixtures/production-contract-refresh.json
|
||||
```
|
||||
|
||||
Runtime/frontend tests are not expected to pass merely because specs changed; keep implementation gates explicitly open. Do not mark `tasks.md` complete without an executable runtime trace.
|
||||
|
||||
## Important constraints
|
||||
|
||||
- Do not modify unrelated existing user changes: `.kilo/agent-manager.json`, `frontend/src/lib/api.ts`, `frontend/src/lib/api/__tests__/api.test.ts`, or `specs-036-050-20260907-111314.md`.
|
||||
- Do not mark runtime implementation complete from spec edits alone.
|
||||
- Do not add frontend agent controls; external MCP is the only agent interaction surface.
|
||||
- Preserve the reports and this state file as durable decision memory.
|
||||
@@ -0,0 +1,127 @@
|
||||
# SS-Prod Agentic E2E — specification refresh report (2026-09-08)
|
||||
|
||||
#region Report.SsProdAgenticE2ESpecRefresh [C:3] [TYPE Verification] [SEMANTICS spec-refresh,contracts,validator,production-gap]
|
||||
|
||||
**Slice D documentation pass closed 2026-09-10.** This report records what the 2026-09-08 production-contract refresh put into the twelve packages, plus the documentation residual closed in Slice D. It is **not** production GO. Runtime slices A+B/C/F (2026-09-10) are cited only as observed implementation; they do not close live browser/screenshot/LLM, Slice E, or Slice G.
|
||||
|
||||
Related durable memory:
|
||||
|
||||
- Plan: `docs/reports/ss-prod-agentic-e2e-spec-refresh-plan-2026-09-08.md`
|
||||
- Session state: `docs/reports/ss-prod-agentic-e2e-session-state-2026-09-08.md`
|
||||
- Runtime handoff: `docs/reports/agentic-runtime-implementation-handoff-2026-09-10.md`
|
||||
|
||||
## Scope
|
||||
|
||||
Twelve packages: 017, 036, 037, 038, 039, 042, 043, 044, 045, 046, 047, 050.
|
||||
|
||||
User constraint (unchanged): frontend must not expose agent prompts/chat/workspace. MCP owns agent interaction. Product UI = manual CRUD + human approvals + read-only evidence.
|
||||
|
||||
Normative files distinguish **target behavior** from **observed implementation**. Adding contracts does not close production acceptance.
|
||||
|
||||
## What was refreshed
|
||||
|
||||
| Package | Production FR | Quickstart status after Slice D | Target vs observed note |
|
||||
|---|---|---|---|
|
||||
| 017 | FR-060 capture/client reuse | Banner already present (prior pass) | ValidationRecord advisory; ScenarioRun authority is 044. Zero LLM providers on ss-prod 2026-09-08. |
|
||||
| 036 | AGSTAB-FR-014 evidence owner/promotion | Banner already present | Product agent workspace RETIRED; frontend `AgentRun*` suites are drift-removal candidates. |
|
||||
| 037 | AGBASE-FR-014 CatalogRevision/CAS/publication | Banner already present | Working-tree YAML is not a published baseline. Pin binds every ScenarioRun consumer. |
|
||||
| 038 | AGSCN-FR-015 canonical executable chain | Banner already present | `agent_evaluation`/`evaluate_declared_spec` supersedes `assertion`/`vlm_analyze`. Compiler visual-chain emission (Slice E) OPEN. |
|
||||
| 039 | AGUI-FR-017 evidence display ownership | Banner already present | Agent-driven dashboard→`/agent` workspace SUPERSEDED. Negative UI acceptance OPEN. |
|
||||
| 042 | SCREG-FR-014 revision baseline/activation | Banner added Slice D | Registry stores authoring constraints, never launch pin. |
|
||||
| 043 | SCEDIT-FR-011 safe proposal/semantic-chain editing | Rewritten Slice D | Manual editor + stored MCP diff review. Agent-driven UI formulations removed. |
|
||||
| 044 | SCEX-FR-028 production baseline-backed evaluation | Rewritten Slice D | Offline validator is the executable contract gate. Live providers / Slice E / Slice G OPEN. Runtime A+B/C/F landed 2026-09-10, production GO still OPEN. |
|
||||
| 045 | RUNMON-FR-014 unified result projection | Rewritten Slice D | EvidenceViewer/AgentEvaluationCard target; observed DTO still counts/failures. No auto-started agent investigation. |
|
||||
| 046 | SCAUTO-FR-019 parity/retention/rollout | Banner added Slice D | DEF-02/SEC-01 runtime landed 2026-09-10; live canary OPEN. |
|
||||
| 047 | SCAN-FR-015 atomic triage/exact context | Banner added Slice D | DEF-04 dequeue runtime landed 2026-09-10 (`investigation.py` FOR UPDATE + `synchronize_session="evaluate"`); production GO OPEN. |
|
||||
| 050 | MCPX-FR-030 contract-complete public parity | **Created** Slice D (was missing) | MCP-only agent interaction. Catalog `2.0.0` / `PINNED_CATALOG_MAJOR=2` (runtime additive `2.2.0`). T044–T046 and T029m OPEN. |
|
||||
|
||||
All twelve `spec.md` files already carried a `Production contract refresh — 2026-09-08` section from the first implementation pass. Slice D did not re-edit those normative bodies or `tasks.md` checkboxes.
|
||||
|
||||
## Contracts added (first refresh pass; not mutated in Slice D)
|
||||
|
||||
Machine-readable / module contracts introduced or promoted to production-refresh authority:
|
||||
|
||||
| Contract | Package | Role |
|
||||
|---|---|---|
|
||||
| `contracts/scenario-reuse.md` | 017 | Capture/client reuse and trust boundary vs ScenarioRun |
|
||||
| `contracts/catalog-lifecycle.md` | 037 | Publication truth table, CAS, explicit Git publish |
|
||||
| `contracts/catalog-revision.schema.json` | 037 | CatalogRevision wrapper + visual/metric entries |
|
||||
| `contracts/baseline-pin.schema.json` | 037 | `BaselineSelectionPin` identity |
|
||||
| `contracts/comparison-types.schema.json` | 037 | Typed comparison statuses/policies |
|
||||
| `contracts/agent-evaluation.schema.json` | 038 | Immutable AgentEvaluation |
|
||||
| `contracts/agent-evaluation-spec.schema.json` | 038 | Declared criteria / comparison_refs |
|
||||
| `contracts/decision-policy.schema.json` + `decision-policy.md` | 038 | Closed policy truth table |
|
||||
| `contracts/browser-actions.md` | 038 | Canonical browser mapping + editor chain rules |
|
||||
| `contracts/capture-profile.schema.json` | 038 | Capture profile identity |
|
||||
| `contracts/production-chain.md` | 044 | Baseline-backed evaluation chain, loop owner, artifact service |
|
||||
| `contracts/artifact-content.openapi.yaml` | 044 | Authenticated GET/HEAD; typed Error-Code on GET and HEAD |
|
||||
| `contracts/result-evidence.schema.json` | 044 | Unified result/evidence/evaluation projection |
|
||||
| `contracts/evidence-ui.md` | 045 | EvidenceViewer + read-only AgentEvaluationCard |
|
||||
| `contracts/production-operations.md` | 046 | REST/MCP parity, retention, canary |
|
||||
| `contracts/atomic-triage.md` | 047 | Queue/case CAS and exact AnalyticsContextKey |
|
||||
| `plan.md`, `data-model.md`, `contracts/modules.md`, `contracts/tool-contracts.schema.json` | 050 | Missing package artifacts completed in the first pass |
|
||||
|
||||
Normative `*.schema.json` files were not mutated in Slice D.
|
||||
|
||||
Eleven JSON Schema files are currently loaded by the offline validator (017–050 prefix scan):
|
||||
|
||||
1. `037/.../baseline-catalog.schema.json`
|
||||
2. `037/.../baseline-pin.schema.json`
|
||||
3. `037/.../catalog-revision.schema.json`
|
||||
4. `037/.../comparison-types.schema.json`
|
||||
5. `038/.../agent-evaluation.schema.json`
|
||||
6. `038/.../agent-evaluation-spec.schema.json`
|
||||
7. `038/.../capture-profile.schema.json`
|
||||
8. `038/.../dashboard-test-scenario.schema.json`
|
||||
9. `038/.../decision-policy.schema.json`
|
||||
10. `044/.../result-evidence.schema.json`
|
||||
11. `050/.../tool-contracts.schema.json`
|
||||
|
||||
## Validator evidence
|
||||
|
||||
Command:
|
||||
|
||||
```bash
|
||||
backend/.venv/bin/python specs/044-dashboard-scenario-execution/prototype/validate_contract_refresh.py \
|
||||
specs/044-dashboard-scenario-execution/fixtures/production-contract-refresh.json
|
||||
```
|
||||
|
||||
Historical first-pass result (session-state 2026-09-08): 11 schemas/refs, 3 positive fixtures, 15 hardcoded negative regressions.
|
||||
|
||||
Second-pass residual P0/P1 (verifier) were encoded as additional fixtures/invariants in the same validator. Expected current result:
|
||||
|
||||
```
|
||||
PASS: 11 schemas and all JSON references
|
||||
PASS: 5 positive fixture checks; 49 hardcoded negative regressions rejected
|
||||
```
|
||||
|
||||
Five positive checks: baseline pin, catalog revision (including visual entry), result projection, evaluation spec, result↔spec criteria pairing.
|
||||
|
||||
Forty-nine negatives include, among others:
|
||||
|
||||
- P0-1 evaluation evidence ownership (foreign run/manifest/raw-response, digest/MIME/length/attempt mismatch, unavailable raw).
|
||||
- P0-2 criterion truth table (duplicate `criterion_id`, comparison membership, kind match, finding existence).
|
||||
- P1-1 publication truth table (`materialized` with commit/receipt, `committed` with receipt, failed without error).
|
||||
- P1-2 numeric tolerances (negative, inverted range, table row-width).
|
||||
- P1-4 visual review branches (visual not-confirmed / missing disposition; metric wrongly bound).
|
||||
|
||||
P1-3 (HEAD artifact Error-Code typing) is closed at the **OpenAPI contract** (`artifact-content.openapi.yaml` reuses status-specific Error-Code header components on HEAD). It is not a JSON-Schema fixture in the 11-schema scan.
|
||||
|
||||
The validator does not execute providers, browsers, or LLM calls. PASS is contract-refresh evidence only.
|
||||
|
||||
## Remaining OPEN production gates
|
||||
|
||||
Do not treat any of the following as Closed by this documentation pass:
|
||||
|
||||
- **Live browser / screenshot / LLM.** ss-prod 2026-09-08: providers unregistered, bindings `0`, LLM providers `0`. Deployment + Chromium image + Admin LLM Settings remain required after code closure.
|
||||
- **Slice E live Playwright.** Compiler visual-chain emission (canonical browser → capture → compare_to_baseline) and `UNSUPPORTED_ACTION` blockers landed 2026-09-10 offline (`038.4.0`). Driver/reconciler canary and live filter/pagination/download transport remain OPEN.
|
||||
- **Slice G live storage canary.** Authenticated GET/HEAD runtime landed 2026-09-10 (`/api/scenario-runs/{run_id}/artifacts/{artifact_id}/content`, 23 offline tests). Live JPEG/PNG/WebP against deployment storage remains OPEN.
|
||||
- **BaselineResolver live publication.** Runtime pin from injected published catalog snapshot landed 2026-09-10 (fail-closed `BASELINE_*`, 422). Explicit Git publication / durable CatalogRevision store remains OPEN.
|
||||
- **046 live canary.** Cost/load/SLO, retention deletion proof, disabled-config REST/MCP identity on a live stand.
|
||||
- **047 production acceptance.** Exact `AnalyticsContextKey` grouping and live-stand triage after DEF-04 code landing.
|
||||
- **050 T029m / T044–T046 / T030.** Live-stand MCP replay; consume/publish parity; negative frontend agent-control acceptance.
|
||||
- **Optional ExecutionPerformanceBaseline.** Explicitly out of scope.
|
||||
|
||||
Runtime slices already landed (not GO): DEF-01 promotion safety, DEF-02 eligibility, DEF-03 `dashboard_context.query`, DEF-04 atomic dequeue, SEC-01 authenticated automation GETs, provider-loop lifespan start, DecisionPolicy mapper, AgentEvaluation store/parser/executor, production `evaluation_adapter_from` + walker rows 9–14 (`EVALUATION_UNAVAILABLE` / `BASELINE_AND_SEMANTIC_PASS`), `ACTION_REGISTRY_VERSION` `038.3.0` (fingerprint `4482c3c1834f9fdbafdcb93ea6b62d4c9c4c048ba24d3b46ae874adb6caab738`). Independent verify 2026-09-10: 044-slice glob 261 passed; execution+store+models+live_binding+walker 68 passed; 261+31 execution-only = the handoff 292. See `docs/reports/agentic-runtime-orchestration-plan-2026-09-10.md`.
|
||||
|
||||
#endregion Report.SsProdAgenticE2ESpecRefresh
|
||||
@@ -33,3 +33,14 @@
|
||||
- [x] CHK015 Traceability maps every functional requirement to contracts, tasks, and tests.
|
||||
- [x] CHK016 Tasks use exact repository paths, dependency order, and test-first sequencing.
|
||||
- [x] CHK017 Machine-readable contracts, external references, and semantic anchors pass validation.
|
||||
|
||||
## Production readiness checklist — 2026-09-08
|
||||
|
||||
AGSCN-FR-015: [Canonical executable chain](../contracts/decision-policy.md). Historical [x] marks do not close this new production gate; removed frontend components are not current evidence. All rows below implemented=false / OPEN.
|
||||
|
||||
- [ ] CHK018 Visual template expands complete browser/capture/artifact/compare/optional-evaluation/policy chain; missing producer or mandatory baseline rejects. Evidence: [T060](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK019 Every browser action/alias maps to one descriptor; sql_evidence/transform enum parity and disabled capability fail before I/O. Evidence: [T061](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK020 All 15 policy precedence rows, threshold equality, malformed output and criterion disagreement use hardcoded outcomes; no dynamic HumanCheckpoint. Evidence: [T062](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK021 Negative product UI test: no agent chat/prompt/assistant editing/proposal-generation/typical-operation-to-agent/workspace/start/handoff controls or agent invocation routes/requests; manual CRUD/editor/human review/read-only results remain usable.
|
||||
|
||||
Schema/static success alone is not runtime completion. Optional approved performance baseline is outside scope.
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
version: v1
|
||||
description: Canonical version-pinned executable action catalog. Runtime dispatches no action outside this file.
|
||||
version: 038.4.0
|
||||
implemented: false
|
||||
description: Normative production target. Canonical names and explicit import aliases are in browser-actions.md; each enabled runtime descriptor requires validated schemas and provider readiness. Runtime 038.1.0 is historical, not parity evidence.
|
||||
actions:
|
||||
- { tool: browser, action: open_dashboard, risk: READ_ONLY, timeout_ms: 30000, retry_safe: true, mutates: false }
|
||||
- { tool: browser, action: navigate_tab, risk: UI_INTERACTION, timeout_ms: 15000, retry_safe: true, mutates: false }
|
||||
@@ -16,8 +17,8 @@ actions:
|
||||
- { tool: browser, action: download, risk: UI_INTERACTION, timeout_ms: 60000, retry_safe: false, mutates: false }
|
||||
- { tool: browser, action: refresh, risk: UI_INTERACTION, timeout_ms: 30000, retry_safe: true, mutates: false }
|
||||
- { tool: browser, action: wait_for_state, risk: READ_ONLY, timeout_ms: 30000, retry_safe: true, mutates: false }
|
||||
- { tool: browser, action: apply_filters, risk: UI_INTERACTION, timeout_ms: 15000, retry_safe: true, mutates: false }
|
||||
- { tool: browser, action: download_xlsx, risk: UI_INTERACTION, timeout_ms: 60000, retry_safe: false, mutates: false }
|
||||
- { tool: browser, action: pagination, risk: UI_INTERACTION, timeout_ms: 15000, retry_safe: true, mutates: false }
|
||||
- { tool: browser, action: navigate_dashboard, risk: UI_INTERACTION, timeout_ms: 30000, retry_safe: true, mutates: false }
|
||||
- { tool: superset_api, action: execute_metric, risk: READ_ONLY, timeout_ms: 30000, retry_safe: true, mutates: false }
|
||||
- { tool: superset_api, action: dataset_field_assert, risk: READ_ONLY, timeout_ms: 30000, retry_safe: true, mutates: false }
|
||||
- { tool: sql_evidence, action: execute_pinned_sql, risk: READ_ONLY, timeout_ms: 30000, retry_safe: true, mutates: false }
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
#region DashboardScenarioModel.DataModel [C:5] [TYPE ADR] [SEMANTICS data-model,scenario,graph,step,validation]
|
||||
@BRIEF Canonical immutable Verification Program IR: scenario_key, content_hash, navigation/evidence/transform/assertion/semantic programs, typed steps, parameters, capability and authoring-validation models. Runtime state is owned by 044.
|
||||
@BRIEF Canonical immutable schema-v1 ActionRegistry DAG; future Verification Program IR: scenario_key, content_hash, navigation/evidence/transform/assertion/semantic programs, typed steps, parameters, capability and authoring-validation models. Runtime state is owned by 044.
|
||||
@RELATION DEPENDS_ON -> [DashboardScenarioModel.Research]
|
||||
@RELATION DEPENDS_ON -> [ScenarioExecution.DataModel]
|
||||
@RATIONALE 038 is the clean IR/compiler layer; identity/revision/runtime are reconciled with 042-047. Semantic identity (scenario_key) is separated from entity identity (scenario_id UUID, assigned by 042) and content identity (content_hash).
|
||||
@@ -18,7 +18,7 @@ Required:
|
||||
- **content_hash** (SHA-256 of the canonical executable graph only — timestamps/display-only excluded);
|
||||
- dashboard_context and objective;
|
||||
- input_fingerprints: query model, checklist catalog, baseline_version, parameters;
|
||||
- parameters, phases, steps, and required `verification_program` (`navigation_program`, `evidence_program`, `transformation_program`, `assertion_program`, `semantic_evaluation_program`);
|
||||
- parameters, phases and steps (schema-v1 ActionRegistry DAG); five-program `verification_program` is a deferred versioned IR, not a required v1 field;
|
||||
- outputs and artifact_plan;
|
||||
- checklist_coverage;
|
||||
- warnings, blockers, risk_summary.
|
||||
@@ -39,7 +39,7 @@ This immutable definition contains no resolved `value` or runtime `status`. Thos
|
||||
|
||||
## VerificationProgram — first-class executable IR
|
||||
|
||||
`VerificationProgram` is the canonical runtime program embedded in every saved scenario: navigation declares registered browser/API actions; evidence declares source/chart/XLSX observations; transformations declare bounded deterministic `TransformSpec` DSL; assertions declare `ComparisonSpec`/`AssertionSpec`; semantic evaluation declares only explicitly necessary `AgentEvaluationSpec`. All program entries are ref-addressed by `logical_step_id`, version-pinned and canonicalized into `content_hash`.
|
||||
`VerificationProgram` is the deferred five-program IR target, requiring an explicit schema/compiler migration before activation; current saved v1 scenarios execute their canonical ActionRegistry DAG. The target decomposition is: navigation declares registered browser/API actions; evidence declares source/chart/XLSX observations; transformations declare bounded deterministic `TransformSpec` DSL; assertions declare `ComparisonSpec`/`AssertionSpec`; semantic evaluation declares only explicitly necessary `AgentEvaluationSpec`. All program entries are ref-addressed by `logical_step_id`, version-pinned and canonicalized into `content_hash`.
|
||||
|
||||
`SqlEvidenceSpec` is a small source-evidence statement: `snippet_id, logical_step_id, connection_ref, database_identity, sql_template, sql_hash, parameter_definitions[], expected_output_schema, relation_refs[], execution_limits, generation_provenance, validation_result`. It is an authoring artifact, saved only after the SQL compilation gate. Runtime may bind declared parameters but MUST execute exactly the pinned template through the Superset SQL Lab adapter; it cannot rewrite SQL, relations, joins, projections or filters.
|
||||
|
||||
@@ -68,7 +68,10 @@ This immutable definition contains no resolved `value` or runtime `status`. Thos
|
||||
| risk | READ_ONLY, UI_INTERACTION, TEST_DATA_MUTATION, EXTERNAL_MUTATION, DANGEROUS_MUTATION, human |
|
||||
| mutation_contract | Mandatory for every mutating action: safe_test_fixture_id, scope, allowed environment, affected keys, cleanup/reconciliation, side-effect key |
|
||||
| capture_spec | Required when tool=screenshot; null otherwise |
|
||||
| vlm_analysis_spec | Required when tool=assertion and input is screenshot; null otherwise |
|
||||
| agent_evaluation_spec | Required only for tool=agent_evaluation/action=evaluate_declared_spec; strict versioned schema |
|
||||
| decision_policy | Pinned baseline-semantic/1.0.0 for baseline-backed semantic evaluation |
|
||||
| comparison_spec | Required for compare_to_baseline; actual ref and approved baseline ref |
|
||||
| vlm_analysis_spec | Legacy advisory profile only; never required for deterministic screenshot comparison |
|
||||
|
||||
The compiler assigns a UUID when it creates an initial graph. An editor/migration MUST carry an existing `logical_step_id` forward for the same logical step; it MUST mint a new UUID only for a genuinely new step. `step_key` and `position` are never identity inputs. Analytics (047) and comparison (045) key on it.
|
||||
|
||||
@@ -87,7 +90,7 @@ Mutating actions require `mutation_contract`. Policy: `DANGEROUS_MUTATION` is ne
|
||||
| viewport | {width, height} | Required; default 1920×1200 |
|
||||
| readiness | string | canvas_stabilized, network_idle, fixed_wait |
|
||||
| readiness_timeout_ms | integer | Default 15000 |
|
||||
| mask_selectors | string[] | OPTIONAL CSS selectors applied before capture; empty when no masking is requested; never a validation or VLM-submission prerequisite |
|
||||
| mask_selectors | string[] | OPTIONAL CSS selectors applied before capture; empty when no PII masking is requested; secret masking is mandatory and requested selectors must succeed before submission |
|
||||
| method | string | cdp, full_page, region |
|
||||
|
||||
This is the SPEC of capture; execution is owned by 044 (CaptureService over ScreenshotService, artifacts owner_type=scenario_run).
|
||||
@@ -104,7 +107,7 @@ This is the SPEC of capture; execution is owned by 044 (CaptureService over Scre
|
||||
| prompt_template_hash | sha256 | SHA-256 of prompt content |
|
||||
| confidence_threshold | float | Minimum confidence to auto-flag (default 0.7) |
|
||||
|
||||
This is the SPEC (what to analyze). Runtime `VlmFinding`/`HumanCheckpointDisposition` are 044 entities.
|
||||
This is a legacy advisory profile, not the authoritative execution contract. New semantic steps use [AgentEvaluationSpec](contracts/agent-evaluation-spec.schema.json) and [DecisionPolicy](contracts/decision-policy.md); immutable output uses [AgentEvaluation](contracts/agent-evaluation.schema.json). Runtime review/disposition is separate and never edits the evaluation.
|
||||
|
||||
## ScenarioRef
|
||||
|
||||
@@ -175,4 +178,17 @@ The schema/validator rejects:
|
||||
|
||||
## @} DashboardScenarioModel.AuthoringWorkspaceModel
|
||||
|
||||
## Production identity and frontend boundary — 2026-09-08
|
||||
|
||||
Capture→durable artifact→deterministic baseline comparison precedes optional declared semantic
|
||||
evaluation. Criteria have typed IDs/kinds; never infer disagreement from prose. Raw malformed
|
||||
model output is parser_error with stored provenance, not an empty success. Actual MIME/digests
|
||||
and trust profile follow [017 reuse](../017-llm-analysis-plugin/contracts/scenario-reuse.md).
|
||||
Authoring stores baseline constraints; 044 resolves/pins published catalog entries before I/O.
|
||||
Browser aliases are normalized only via [browser-actions](contracts/browser-actions.md) and
|
||||
versioned ActionRegistry; one-step placeholder compilation does not satisfy the template contract.
|
||||
AgentAuthoringWorkspace is external-MCP server state, never frontend interaction.
|
||||
Product UI contains manual CRUD/editor/review and read-only result evidence only.
|
||||
All refresh chain integration gates are OPEN; no approved performance baseline is specified.
|
||||
|
||||
#endregion DashboardScenarioModel.DataModel
|
||||
|
||||
@@ -160,3 +160,14 @@ No exception planned. Checklist data remains declarative and versioned; do not t
|
||||
- T059 — E2E: capture → VLM → findings → disposition с реальными байтами и provider-provenance.
|
||||
|
||||
**Exit rule**: evidence/VLM-ветка не демонстрирует реальную работу (пустые findings, синтетические hash) до мерджа T057–T059.
|
||||
|
||||
## Production delivery plan — 2026-09-08 (AGSCN-FR-015)
|
||||
|
||||
Status: specified, implemented=false; historical unit/prototype/transport results are not current production acceptance.
|
||||
|
||||
1. Pin [Canonical executable chain](contracts/decision-policy.md) and [data model](data-model.md); write negative fixtures before runtime changes.
|
||||
2. Implement existing domain boundaries for: Schema-v1 stores canonical ActionRegistry DAG; future five-program IR requires explicit migration. Step declares capture_spec, comparison_spec, optional agent_evaluation_spec and decision_policy. Criteria IDs/kinds distinguish semantic findings from deterministic comparison disagreement. Browser descriptor pins normalized alias, bounded typed inputs/outputs, capability/version and mutation policy; server query context survives compile→draft unchanged.
|
||||
3. Execute [tasks](tasks.md) T060, T061, T062 and retain reproducible evidence in [traceability](traceability.md), then close [checklist](checklists/requirements.md) individually.
|
||||
4. Run cross-spec canary only after 037 catalog publication, 038 chain and 044 provider/content/policy gates; use 046 versioned cost/load/SLO limits. Disable admission on rollback, retain pins/receipts/holds; no fallback to stale catalog or synthetic PASS.
|
||||
|
||||
Frontend agent prompts, chat, assistant editing, proposal-generation, workspace/start/handoff actions are prohibited. Only external MCP clients interact with agents; frontend provides ordinary manual CRUD/editor, human review/approval and read-only monitoring/evidence. Runtime removal tasks are not closed by this document. Optional approved performance baseline is outside this refresh.
|
||||
|
||||
@@ -1,5 +1,18 @@
|
||||
# Quickstart: Dashboard Scenario Model
|
||||
|
||||
> **Refresh 2026-09-08 (production contract):** the canonical executable chain (AGSCN-FR-015)
|
||||
> browser→capture→durable artifact→deterministic baseline comparison→optional declared AgentEvaluation→
|
||||
> DecisionPolicy→StepOutcome is normative in `contracts/browser-actions.md`, `agent-evaluation.schema.json`,
|
||||
> `agent-evaluation-spec.schema.json`, `decision-policy.schema.json`/`decision-policy.md` — implemented=false /
|
||||
> acceptance OPEN. Mandatory semantic evaluation uses `agent_evaluation`/`evaluate_declared_spec` — the old
|
||||
> assertion/vlm_analyze path is superseded. Criterion closed truth table (unique `criterion_id`;
|
||||
> deterministic_comparison binds a `comparison_id` member of `comparison_refs`; semantic binds null; findings
|
||||
> must match an existing criterion kind) is executable offline:
|
||||
>
|
||||
> ```bash
|
||||
> python specs/044-dashboard-scenario-execution/prototype/validate_contract_refresh.py specs/044-dashboard-scenario-execution/fixtures/production-contract-refresh.json
|
||||
> ```
|
||||
|
||||
## Prerequisites
|
||||
|
||||
036 draft registration and 037 query-model/baseline summary contracts must pass. `/speckit.validate` must report PASS before `/speckit.implement`.
|
||||
@@ -58,3 +71,7 @@ python3 -c "import yaml; d=yaml.safe_load(open('specs/038-dashboard-scenario-mod
|
||||
## Known Gap & Boundary (2026-08-07 reconciliation)
|
||||
|
||||
Шаги 1–9 проверяют детерминированный compiler/validator/resolver/pack контур — работает. **Runtime capture/VLM/disposition НЕ входят в 038** — перенесены в 044 (ScenarioExecution): реальный VLM submit через `LLMClient`/`LLMProviderService`, реальный capture через `ScreenshotService` с артефактами `owner_type=scenario_run`, HumanCheckpoint (confirm/false_positive/inconclusive). 038 определяет только VlmAnalysisSpec/ScreenshotCaptureSpec. Прежние T057–T059 — задачи 044.
|
||||
|
||||
## Refresh Gap (2026-09-08 production audit)
|
||||
|
||||
AgentEvaluationSpec/AgentEvaluation/DecisionPolicy контракты добавлены и офлайн-валидируемы (validate_contract_refresh: schemas/refs, criterion truth table, evaluation-evidence binding), но runtime отсутствует: в ss-prod не зарегистрирован ни один LLM-провайдер, declared-evaluation адаптер не развёрнут, а draft compiler теряет `dashboard_context.query` (дефект E2E). Все строки AGSCN-FR-015 остаются implemented=false / acceptance OPEN; отсутствие семантического провайдера не может отключать обязательную declared evaluation — deterministic-only прогоны LLM не требуют.
|
||||
|
||||
@@ -103,7 +103,7 @@
|
||||
| E8 | 403 forbidden role on scenario operations | auth | Permission denial rendered without approval gate | User contacts admin / RBAC test |
|
||||
| E9 | 429 rate limit on compile/validate | throttling | Retry-After honored; UI countdown | User waits; L2 UX test |
|
||||
| E10 | 5xx backend failure on compile | server-error | Error section + retry; partial graph not persisted | User retries; L2 UX test |
|
||||
| E11 | Malformed VLM response / empty findings | integration | Findings array empty; step inconclusive with reason; stale prompt blocked (422 STALE_PROMPT) | Re-run analysis; L1 VLM test |
|
||||
| E11 | Malformed VLM response / empty findings | integration | Malformed output is parser_error with raw provenance, never empty success; schema-valid empty findings require explicit verdict; stale prompt blocked (422 STALE_PROMPT) | Re-run analysis; L1 VLM test |
|
||||
| E12 | Missing runtime parameter at launch | data-quality | Saved pack remains save_eligible; 044 RunPreflight rejects only that launch | Supply a typed binding; L1 preflight test |
|
||||
| E13 | Unsafe path / executable code / SQL injection into pack | security | SQL compilation gate blocks unsafe SQL; arbitrary code/path validation blocks before draft registration | L1 security test; injected-code fixture |
|
||||
| E14 | Duplicate submit of draft-pack | idempotency | Idempotency key / revision hash prevents double registration | L1 API test; 409 on changed revision |
|
||||
@@ -115,7 +115,7 @@
|
||||
- **AGSCN-FR-001**: The system MUST define a `DashboardTestScenario` model with dashboard context, objective, parameters, steps, dependencies, outputs, artifacts, risks, and warnings.
|
||||
- **AGSCN-FR-002**: Every scenario step MUST declare tool category, action, inputs, outputs, expected result, dependencies, and automation status.
|
||||
- **AGSCN-FR-003**: Supported tool categories MUST include browser automation, Superset API execution, XLSX parsing, assertion, screenshot/evidence, report generation, artifact generation, and human checkpoint.
|
||||
- **AGSCN-FR-003a**: `DashboardTestScenario` MUST contain a first-class, content-hashed Verification Program with navigation, evidence, transformation, assertion and semantic-evaluation programs. It is immutable runtime input, not a runtime planning hint.
|
||||
- **AGSCN-FR-003a**: Current schema_version=1 MUST carry a canonical content-hashed ActionRegistry DAG. Five-program VerificationProgram IR is a deferred versioned migration, not an implicitly required v1 field. Runtime executes the exact saved DAG, never a planning hint.
|
||||
- **AGSCN-FR-004**: Assertions against reference values MUST use baseline references or baseline candidate references; raw expected numbers MUST NOT be embedded directly in executable steps.
|
||||
- **AGSCN-FR-005**: Authoring validation MUST detect missing refs, cycles, duplicate outputs, missing/stale baselines, unknown selectors, unsupported tools, and invalid ParameterDefinitions. Required runtime bindings are enforced only by 044 RunPreflight.
|
||||
- **AGSCN-FR-006**: Checklist mapping MUST use normalized checklist cases derived from the research PDF and capability tags, not hardcoded one-size-fits-all scripts. Mapping accepts the target release_version for baseline lookup.
|
||||
@@ -123,14 +123,14 @@
|
||||
- **AGSCN-FR-008**: Scenario output MUST be deterministic for the same dashboard query model, checklist template, baseline catalog, and user parameters.
|
||||
- **AGSCN-FR-009**: The scenario model MUST remain implementation-neutral and must not require the user to choose low-level artifacts such as Playwright, XLSX, or API output upfront.
|
||||
- **AGSCN-FR-010**: Screenshot steps MUST carry a capture SPECIFICATION: target (tab/viewport), viewport dimensions, readiness strategy, OPTIONAL masking selectors (relaxed 2026-08-24 — local-only providers; masking is display hygiene, not a gate), and max wait. **Execution of capture is owned by 044** (ScenarioExecution CaptureService delegating to `Plugin.Service.ScreenshotService`), with artifacts `owner_type=scenario_run`; 038 defines the spec, not the runtime path.
|
||||
- **AGSCN-FR-011**: Visual-analysis steps MUST carry a typed `VlmAnalysisSpec` (profile/provider/model/prompt template/hash/confidence). **Runtime VLM submission and `VlmFinding` production are owned by 044**, reusing `Plugin.Service.LLMClient` resolved through `Services.LlmProvider.LLMProviderService` (multimodal-required, encrypted-key handling, JSON mode); response redaction via `Plugin.Service.RedactionService` is OPTIONAL since 2026-08-24 (local-only providers). A stub/default submit returning empty findings without a real provider call is incomplete.
|
||||
- **AGSCN-FR-011**: Declared semantic steps MUST carry strict AgentEvaluationSpec and DecisionPolicy; VlmAnalysisSpec is legacy advisory input only, not mandatory for deterministic screenshot comparison. **Runtime VLM submission and `VlmFinding` production are owned by 044**, reusing `Plugin.Service.LLMClient` resolved through `Services.LlmProvider.LLMProviderService` (multimodal-required, encrypted-key handling, JSON mode); secret redaction is mandatory; optional local PII masking and requested-selector failure follow 017 scenario-reuse.md. A stub/default submit returning empty findings without a real provider call is incomplete.
|
||||
- **AGSCN-FR-011a**: Read-only `SqlEvidenceSpec` MAY be authored during creation/edit/revalidation or investigation proposal only. Save MUST require AST/policy/schema/preview validation. A ScenarioRun executes exactly the saved SQL template via the Superset SQL Lab adapter with typed bindings; runtime LLM SQL rewrite, relation/projection/join/filter mutation and credential handoff are forbidden.
|
||||
- **AGSCN-FR-011b**: `TransformSpec` MUST use only the bounded versioned DSL and `ComparisonSpec`/`AssertionSpec` MUST compare declared evidence refs. Arbitrary Python/code is forbidden.
|
||||
- **AGSCN-FR-011c**: `AgentEvaluationSpec` MAY cover only declared semantic/visual/ambiguous checks. It MUST pin model/prompt/evidence/input/tool access/output schema and DecisionPolicy; it cannot alter graph, SQL/DSL, orchestration, lifecycle or mutations.
|
||||
- **AGSCN-FR-011d**: Authoring MUST accept first-class `ChangeRequestContext`; the compiler must mark missing needed context as `needs_context`, never guess it.
|
||||
- **AGSCN-FR-012**: Human checkpoint steps MAY reference specific VLM finding ids. Resolution options (confirm, **false_positive**, inconclusive) MUST be typed and auditable — this is a 044 `HumanCheckpoint`, **distinct from** the 036 authorization `ActionApprovalGate`. Disposition changes finding status, not graph structure.
|
||||
- **AGSCN-FR-012**: Human checkpoint steps MAY reference specific VLM finding ids. Resolution options (confirm, **false_positive**, inconclusive) MUST be typed and auditable — this is a 044 `HumanCheckpoint`, **distinct from** the 036 authorization `ActionApprovalGate`. Disposition creates a separate immutable review record, never mutates AgentEvaluation, StepOutcome or graph structure.
|
||||
- **AGSCN-FR-013**: The agent MAY plan scenario coverage, resolve ambiguity, and create a validated WorkingDraft or executable revision when delegated policy permits. Every resulting graph MUST pass the deterministic 038 validator and retain canonical immutable provenance.
|
||||
- **AGSCN-FR-014**: Graph authoring and revision actions MUST be available in the persistent scenario workspace or agent thread; modal/dialog interaction MUST NOT be required for authoring, review, conflict recovery, or approval.
|
||||
- **AGSCN-FR-014**: Agent graph authoring occurs exclusively in external MCP clients. Product UI provides manual CRUD/editor and human review/approval without modal dependency; no agent thread/workspace/prompt/edit/proposal-generation/launch/handoff interaction may render.
|
||||
|
||||
### Key Entities
|
||||
|
||||
@@ -181,7 +181,7 @@
|
||||
- "Agent" throughout this spec now reads as any delegated actor — external MCP client or in-product surface — always behind the same deterministic validator and provenance rules.
|
||||
- **PII posture**: masking selectors in `ScreenshotCaptureSpec` and response redaction in the VLM path are OPTIONAL (2026-08-24) — all LLM/VLM providers run locally inside the trust perimeter.
|
||||
|
||||
**Status (2026-09-02): done** — реализовано в рамках 050: инструменты и гейты (`specs/050-mcp-interface/tasks.md` T012–T028 [x]), handoff-поверхность (050 T030–T033), демонтаж чата и сервиса `agent/` (050 T040–T041, чекпоинты `specs/WORKSTATE-043-047.md`).
|
||||
**Historical MCP transport/decommission status (2026-09-02): reported done; not production-readiness evidence** — реализовано в рамках 050: инструменты и гейты (`specs/050-mcp-interface/tasks.md` T012–T028 [x]), handoff-поверхность (050 T030–T033), демонтаж чата и сервиса `agent/` (050 T040–T041, чекпоинты `specs/WORKSTATE-043-047.md`).
|
||||
|
||||
## @{ DashboardScenarioModel.AgentAuthoringWorkspace [C:5] [TYPE ADR]
|
||||
@BRIEF Normative external-agent/user co-authoring workspace and promotion boundary for scenario authoring.
|
||||
@@ -190,7 +190,7 @@
|
||||
|
||||
`AgentAuthoringWorkspace` is a persistent, server-owned session in which an authenticated user and an external MCP agent may jointly author a scenario. It supports draft edits, exploratory Playwright/code-sandbox checks, result inspection, selector/wait/assertion corrections, and new graph proposals. The agent is a full co-author of intent and diagnostics, but never a production authority.
|
||||
|
||||
The workspace state enum is: `draft`, `exploring`, `exploration_failed`, `exploration_passed`, `proposal_ready`, `validation_blocked`, `awaiting_user_review`, `save_eligible`, `pending_approval`, `candidate`, `current`. State transitions are server-owned, audited, idempotent and CAS-protected; the workspace persists across MCP calls and client reconnects.
|
||||
This is external-MCP server state only, never a frontend agent workspace or interaction surface. The workspace state enum is: `draft`, `exploring`, `exploration_failed`, `exploration_passed`, `proposal_ready`, `validation_blocked`, `awaiting_user_review`, `save_eligible`, `pending_approval`, `candidate`, `current`. State transitions are server-owned, audited, idempotent and CAS-protected; the workspace persists across MCP calls and client reconnects.
|
||||
|
||||
`AuthoringArtifact` is distinct from `ExecutionProgram`. Authoring artifacts may be source snippets, patches, exploratory traces, screenshots, operation receipts, or diagnostics and retain provenance/ownership. `ExecutionProgram` is only the validated typed `ScenarioGraph`/`VerificationProgram`. A code-backed production provider is a separate future contract, unimplemented here, and cannot be inferred from an exploratory code artifact.
|
||||
|
||||
@@ -257,4 +257,12 @@ The sandbox contract is mandatory but its implementation is not claimed: isolate
|
||||
данные — только capability-факты из авторитетной модели; отсутствующий факт не может быть выдуман как
|
||||
`true` ради «успешной» классификации.
|
||||
|
||||
## Production contract refresh — 2026-09-08
|
||||
|
||||
**Frontend boundary (user decision 2026-09-08)**: All agent interaction is external MCP only. Product frontend MUST NOT contain agent chat, prompt/request textarea, assistant editing, typical-operation-to-agent selector, proposal-generation, agent workspace/start or handoff controls/routes. Ordinary manual CRUD/editor, human approval/review, monitoring and read-only evidence/evaluation are permitted. AgentEvaluationCard is read-only, with no prompt/retry-agent/provider controls. Existing agent proposal UI is runtime drift; removal/negative DOM-route-network acceptance remains OPEN in this spec-only change.
|
||||
|
||||
**AGSCN-FR-015 — Canonical executable chain**: Current schema_version=1 MUST compile the ActionRegistry DAG; five-program IR remains deferred and cannot be required implicitly. Visual template expansion MUST emit browser→capture→durable artifact→deterministic baseline comparison→optional declared AgentEvaluation→DecisionPolicy→StepOutcome with exact refs. Mandatory semantic evaluation uses agent_evaluation/evaluate_declared_spec, not assertion/vlm_analyze. Baseline refs resolve via 044 before I/O; raw expected literals and client-carried context authority are forbidden. Canonical browser mapping is contracts/browser-actions.md. Compile→draft registration MUST preserve the server-issued query model and canonical hash without client repair (DEF-03).
|
||||
|
||||
Normative contract: [Canonical executable chain](contracts/decision-policy.md). New requirements are specified, **implemented=false / acceptance OPEN** until executable evidence closes the linked tasks/checklist/traceability rows. Historical local tests and the manual inconclusive ss-prod run do not prove browser/capture/baseline/LLM production readiness. The refresh scope is the audited P0/P1/P2 agentic E2E and baseline gaps; an approved ExecutionPerformanceBaseline is not introduced.
|
||||
|
||||
#endregion DashboardScenarioModel.Spec
|
||||
|
||||
@@ -114,7 +114,7 @@
|
||||
@TEST_EDGE: capture_timeout→step inconclusive; no artifact registered, masking_applied→original + masked artifacts
|
||||
- [x] T040 [P] Write failing VLM analysis tests in `backend/tests/services/dashboard_testing/scenario/test_vlm.py` for typed findings, provenance, stale prompt rejection (5 passed)
|
||||
@TEST_EDGE: stale_prompt→422 STALE_PROMPT, vlm_timeout→inconclusive, empty_response→empty findings + inconclusive
|
||||
- [x] T041 Implement `backend/src/services/dashboard_testing/scenario/vlm.py`: submit the configured screenshot (masked only when requested) to the local VLM provider, parse typed VlmFinding[], validate model_provenance, and persist provider output under secret-safe handling
|
||||
- [ ] T041 Implement `backend/src/services/dashboard_testing/scenario/vlm.py`: submit the configured screenshot (masked only when requested) to the local VLM provider, parse typed VlmFinding[], validate model_provenance, and persist provider output under secret-safe handling
|
||||
@POST: returns typed VlmFinding[] with model/prompt provenance; raw response stored with credentials/cookies/tokens excluded
|
||||
@SIDE_EFFECT: enterprise-local VLM provider call; raw response persisted as separate DraftArtifact
|
||||
@INVARIANT: VLM findings advisory; never alter metric baseline truth; stale prompts block analysis
|
||||
@@ -127,11 +127,11 @@
|
||||
@SIDE_EFFECT: audit record of disposition decision
|
||||
@INVARIANT: disposition never alters scenario graph structure or step ordering
|
||||
- [x] T045 Add VlmFinding and HumanDisposition DTOs to `contracts/openapi.yaml` response schemas (verify round-trip)
|
||||
- [x] T046 Wire capture/VLM/disposition into the scenario step execution loop: screenshot step → capture → analysis step → VLM call → human step → disposition
|
||||
- [ ] T046 Wire capture/VLM/disposition into the scenario step execution loop: screenshot step → capture → analysis step → VLM call → human step → disposition
|
||||
@NOTE: runtime execution loop is OWNED by 044. This task is redefined as: verify the compiler emits valid ScreenshotCaptureSpec/VlmAnalysisSpec DTOs that 044 executors can consume; the 038 loop contracts are superseded by 044 ScenarioExecution.
|
||||
- [x] T047 Audit: VLM findings are advisory, never alter metric baseline truth; disposition never changes graph structure; stale prompts block analysis
|
||||
|
||||
**Checkpoint**: Capture/VLM/disposition flow verified end-to-end; typed findings auditable. ✅ (85 tests green, belief audit 0 errors)
|
||||
**Historical checkpoint**: 85 local tests reported; this did not prove production capture/VLM/content E2E.
|
||||
|
||||
## Phase 9 — Polish & Cross-Cutting Verification
|
||||
|
||||
@@ -156,10 +156,22 @@
|
||||
|
||||
**Context**: 044 ScenarioExecution owns all runtime capture/VLM/disposition/evidence. The former 038 T057–T059 (real VLM submit, real capture, e2e evidence) are **moved to 044** and are NOT 038 responsibilities. 038 keeps VlmAnalysisSpec/ScreenshotCaptureSpec definitions only.
|
||||
|
||||
- [x] (MOVED) Real VLM submit + real capture + e2e evidence → implemented under 044 ScenarioExecution (owner_type=scenario_run artifacts, HumanCheckpoint). 038 runtime stubs are out of scope.
|
||||
- [ ] (MOVED) Real capture/evaluation/content E2E is owned by 044 and remains OPEN. Screenshot adapter presence does not prove deployment registration, AgentEvaluation or content retrieval.
|
||||
|
||||
## Dependencies
|
||||
|
||||
T001–T005 → US1 → US2; US3 follows compiler; US4 follows validator; draft pack follows validation/resolution; Phase 8 depends on 036 Phase 8 (screenshot evidence artifacts) and 037 Phase 7 (visual baseline infrastructure). Phase 10 (T057–T059) is the **MVP runtime closure**: VLM/capture remain stubs until merged. 039 begins only after ScenarioResponse and DraftPack fixtures are stable. Frontend tasks N/A — 038 is DTO-only.
|
||||
|
||||
## Production readiness — 2026-09-08 (AGSCN-FR-015)
|
||||
|
||||
Historical [x] rows above retain only their dated local/transport evidence; they do not prove current production readiness. Reopened rows were contradicted by the audited gaps. Removed frontend/agent paths are historical, not implementation prerequisites. New acceptance is **implemented=false / OPEN**.
|
||||
|
||||
Contract: [Canonical executable chain](contracts/decision-policy.md).
|
||||
|
||||
- [ ] T060 [P0/P1/P2] Visual template expands complete browser/capture/artifact/compare/optional-evaluation/policy chain; missing producer or mandatory baseline rejects. Implement at the existing 038 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
|
||||
- [ ] T061 [P0/P1/P2] Every browser action/alias maps to one descriptor; sql_evidence/transform enum parity and disabled capability fail before I/O. Implement at the existing 038 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
|
||||
- [ ] T062 [P0/P1/P2] All 15 policy precedence rows, threshold equality, malformed output and criterion disagreement use hardcoded outcomes; no dynamic HumanCheckpoint. Implement at the existing 038 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
|
||||
|
||||
Frontend boundary for this package: manual CRUD/editor, human review/approval, monitoring and read-only evidence/evaluation only; all agent interaction is external MCP. No agent chat/prompt/assistant editing/proposal generation/workspace/start/handoff controls. Runtime removal is OPEN, not performed by this spec refresh. Optional approved performance baseline is outside scope.
|
||||
|
||||
#endregion DashboardScenarioModel.Tasks
|
||||
|
||||
@@ -90,4 +90,17 @@ The coordinated authoring path is `persistent workspace -> isolated exploration
|
||||
| ScenarioGraph.Capture.Dispatch | Plugin.Service.ScreenshotService, Services.AgentRuns.Evidence | One Playwright/CDP capture path; evidence auditability |
|
||||
| ScenarioGraph.Vlm.Analyze | Plugin.Service.LLMClient, Services.LlmProvider.LLMProviderService, Plugin.Service.RedactionService | One local provider client; multimodal-required validation; optional display redaction; secret-safe persistence |
|
||||
|
||||
## Production acceptance traceability — 2026-09-08
|
||||
|
||||
Historical rows above identify prior tests/code only; removed agent UI paths are retired. The following audited gates are **implemented=false / OPEN**, independent of local suite totals.
|
||||
|
||||
| Requirement | Domain contract / DTO | Task | Falsifiable acceptance | State |
|
||||
|---|---|---|---|---|
|
||||
| AGSCN-FR-015 | [Canonical executable chain](contracts/decision-policy.md); [data model](data-model.md) | [T060](tasks.md) | Visual template expands complete browser/capture/artifact/compare/optional-evaluation/policy chain; missing producer or mandatory baseline rejects. | OPEN |
|
||||
| AGSCN-FR-015 | [Canonical executable chain](contracts/decision-policy.md); [data model](data-model.md) | [T061](tasks.md) | Every browser action/alias maps to one descriptor; sql_evidence/transform enum parity and disabled capability fail before I/O. | OPEN |
|
||||
| AGSCN-FR-015 | [Canonical executable chain](contracts/decision-policy.md); [data model](data-model.md) | [T062](tasks.md) | All 15 policy precedence rows, threshold equality, malformed output and criterion disagreement use hardcoded outcomes; no dynamic HumanCheckpoint. | OPEN |
|
||||
| AGSCN-FR-015; external-MCP-only UI | manual editor/review; read-only evidence | [production tasks](tasks.md) | No frontend agent prompt/chat/assistant editing/proposal generation/workspace/start/handoff routes or requests; human approval remains usable. | OPEN |
|
||||
|
||||
Sources: [production gap](../../docs/reports/ss-prod-agentic-e2e-production-gap-2026-09-08.md), [coverage gap](../../docs/reports/ss-prod-agentic-e2e-spec-coverage-2026-09-08.md), [baseline gap](../../docs/reports/ss-prod-agentic-e2e-baseline-gap-2026-09-08.md). Spec schema/static checks prove contract structure only; live canary/runtime closure and optional approved performance baseline are not claimed.
|
||||
|
||||
#endregion DashboardScenarioModel.Traceability
|
||||
|
||||
@@ -42,3 +42,14 @@
|
||||
## Success Criteria
|
||||
|
||||
- [ ] CHK018 List/detail < 200ms; SC-001..005 verified
|
||||
|
||||
## Production readiness checklist — 2026-09-08
|
||||
|
||||
SCREG-FR-014: [Revision baseline requirements and activation](../../044-dashboard-scenario-execution/contracts/production-chain.md). Historical [x] marks do not close this new production gate; removed frontend components are not current evidence. All rows below implemented=false / OPEN.
|
||||
|
||||
- [ ] CHK019 Compile→register→bootstrap→save preserves server query context without client repair; unsafe/stale graph produces no partial registry rows. Evidence: [T030](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK020 Changing baseline constraint/capture/evaluation/policy creates a candidate revision and explicit diff; old runs retain exact original pins. Evidence: [T031](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK021 Activation rejects drifted materialized artifact or stale CAS; ordinary manual registry/editor routes contain no agent generation controls. Evidence: [T032](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK022 Negative product UI test: no agent chat/prompt/assistant editing/proposal-generation/typical-operation-to-agent/workspace/start/handoff controls or agent invocation routes/requests; manual CRUD/editor/human review/read-only results remain usable.
|
||||
|
||||
Schema/static success alone is not runtime completion. Optional approved performance baseline is outside scope.
|
||||
|
||||
@@ -122,4 +122,12 @@ Every registry/run/evidence decision additionally requires the intersection of s
|
||||
- Registry projection is authoritative for list/detail/status; git `scenario.yaml`/reference `runner.plan.json` are asynchronously materialized artifacts with explicit status.
|
||||
- Revisions are immutable and append-only; an executable save creates a new `candidate` row and does not advance `current_revision`. Only the separate `ActivateCurrentRevision` compare-and-set (CAS) operation may advance `current_revision` after its eligibility, materialization, policy, and approval checks. Initial scenario creation is the sole exception when the explicitly eligible initial revision is created as `current`. A `ScenarioRun` target is always selected from an explicit promoted `revision_id` plus its verified `content_hash`; it never resolves from the latest candidate or an implicit moving `current_revision`.
|
||||
|
||||
## Production record contract — 2026-09-08 (SCREG-FR-014)
|
||||
|
||||
ScenarioRevision stores baseline constraints and hashes for registry/capture/evaluation/policy, not runtime BaselineSelectionPin. WorkingDraft/save handles preserve server context authority and canonical content hash. Activation CAS validates capability/baseline constraints, materialization receipt and immutable provenance; old run snapshots remain unchanged.
|
||||
|
||||
Normative detail: [Revision baseline requirements and activation](../044-dashboard-scenario-execution/contracts/production-chain.md). New fields, CAS transitions and cross-record integrity checks are implemented=false until [production tasks](tasks.md) and [traceability](traceability.md) close with executable evidence. Existing shorter field lists are legacy compatibility projections, not permission to omit the production identity fields.
|
||||
|
||||
Frontend contains no agent interaction state, prompt, proposal-generation or workspace/start controls. Human manual editor/review state and read-only evaluation/result artifacts are separate from external MCP authoring. Approved performance baseline is outside scope.
|
||||
|
||||
#endregion ScenarioRegistry.DataModel
|
||||
|
||||
@@ -80,3 +80,14 @@ frontend/src/routes/dashboard-testing/scenarios/ # list + detail
|
||||
## Complexity Tracking
|
||||
|
||||
No exception planned. Registry/lifecycle/staleness are bounded C3-C4 contracts; do not collapse into one oversized controller.
|
||||
|
||||
## Production delivery plan — 2026-09-08 (SCREG-FR-014)
|
||||
|
||||
Status: specified, implemented=false; historical unit/prototype/transport results are not current production acceptance.
|
||||
|
||||
1. Pin [Revision baseline requirements and activation](../044-dashboard-scenario-execution/contracts/production-chain.md) and [data model](data-model.md); write negative fixtures before runtime changes.
|
||||
2. Implement existing domain boundaries for: ScenarioRevision stores baseline constraints and hashes for registry/capture/evaluation/policy, not runtime BaselineSelectionPin. WorkingDraft/save handles preserve server context authority and canonical content hash. Activation CAS validates capability/baseline constraints, materialization receipt and immutable provenance; old run snapshots remain unchanged.
|
||||
3. Execute [tasks](tasks.md) T030, T031, T032 and retain reproducible evidence in [traceability](traceability.md), then close [checklist](checklists/requirements.md) individually.
|
||||
4. Run cross-spec canary only after 037 catalog publication, 038 chain and 044 provider/content/policy gates; use 046 versioned cost/load/SLO limits. Disable admission on rollback, retain pins/receipts/holds; no fallback to stale catalog or synthetic PASS.
|
||||
|
||||
Frontend agent prompts, chat, assistant editing, proposal-generation, workspace/start/handoff actions are prohibited. Only external MCP clients interact with agents; frontend provides ordinary manual CRUD/editor, human review/approval and read-only monitoring/evidence. Runtime removal tasks are not closed by this document. Optional approved performance baseline is outside this refresh.
|
||||
|
||||
@@ -1,9 +1,14 @@
|
||||
# Quickstart: Scenario Registry & Lifecycle (042)
|
||||
|
||||
> **Factual audit 2026-08-20:** pending verification checklist only; no command result below is
|
||||
> currently asserted as evidence.
|
||||
> **Refresh 2026-09-08 (production contract):** revision baseline requirements and activation
|
||||
> (SCREG-FR-014) are normative in `../044-dashboard-scenario-execution/contracts/production-chain.md`
|
||||
> — implemented=false / acceptance OPEN. Registry stores authoring constraints, never a launch
|
||||
> `BaselineSelectionPin`. Activation/save cannot substitute latest catalog or candidate as
|
||||
> execution truth. Agent interaction is external MCP only; product UI is list/detail/lifecycle
|
||||
> CRUD. Commands below are local gates, not production GO.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
## Prereqs
|
||||
- Backend venv: `backend/.venv`
|
||||
- DB migrated (registry tables)
|
||||
|
||||
@@ -12,23 +17,43 @@
|
||||
```bash
|
||||
cd backend && source .venv/bin/activate
|
||||
|
||||
# Migration
|
||||
# Migration (requires a real DATABASE_URL / PostgreSQL; SQLite is not the production chain)
|
||||
alembic upgrade head
|
||||
|
||||
# Backend tests
|
||||
python -m pytest -v tests/services/dashboard_testing/registry/
|
||||
python -m pytest -v tests/services/dashboard_testing/registry/test_list.py \
|
||||
tests/services/dashboard_testing/registry/test_get.py \
|
||||
tests/services/dashboard_testing/registry/test_create.py \
|
||||
tests/services/dashboard_testing/registry/test_revisions.py \
|
||||
tests/services/dashboard_testing/registry/test_staleness.py \
|
||||
tests/services/dashboard_testing/registry/test_scenario_registry_health.py \
|
||||
tests/services/dashboard_testing/registry/test_scenario_registry_lifecycle.py
|
||||
python -m pytest -v tests/api/test_scenario_registry.py
|
||||
|
||||
# Lint
|
||||
python -m ruff check backend/src/services/dashboard_testing/registry/ backend/src/api/routes/dashboard_testing/scenarios.py
|
||||
|
||||
# Frontend tests
|
||||
cd frontend && npm run test -- ScenarioRegistry
|
||||
python -m ruff check src/services/dashboard_testing/registry/ src/api/routes/dashboard_testing/scenarios.py
|
||||
```
|
||||
|
||||
## Exit Gates
|
||||
- [ ] List/search/detail tests pass (incl. GET /scenarios/{id} not 404)
|
||||
- [ ] Revision chain immutability verified (stale base → 409)
|
||||
- [ ] Lifecycle transition + archive-not-delete verified
|
||||
- [ ] Staleness from 037/041 hooks verified
|
||||
- [ ] ruff clean; prototype state coverage passes
|
||||
```bash
|
||||
cd frontend
|
||||
npm run test -- --run src/lib/models/__tests__/ScenarioRegistryModel.test.ts
|
||||
npm run test -- --run src/routes/dashboard-testing/scenarios/__tests__/scenarios_index.ux.test.ts
|
||||
```
|
||||
|
||||
Offline contract check (baseline pin / catalog / result schemas used by activation constraints):
|
||||
|
||||
```bash
|
||||
backend/.venv/bin/python specs/044-dashboard-scenario-execution/prototype/validate_contract_refresh.py \
|
||||
specs/044-dashboard-scenario-execution/fixtures/production-contract-refresh.json
|
||||
```
|
||||
|
||||
## Exit gates (acceptance OPEN)
|
||||
|
||||
- List/search/detail tests pass, including `GET /dashboard-testing/scenarios/{id}`.
|
||||
- Revision chain immutability verified (stale base → 409).
|
||||
- Lifecycle transition + archive-not-delete verified.
|
||||
- Staleness from 037/041 hooks verified (upstream integration remains unproven).
|
||||
- Activation preserves canonical baseline refs, capture profile, optional AgentEvaluationSpec/DecisionPolicy hashes; original run snapshots stay unchanged after rebaseline.
|
||||
- ruff clean; prototype state coverage passes.
|
||||
|
||||
`GET /scenarios/{id}` was a historical 039 404; the route exists in the current registry API. That does not close SCREG-FR-014 or production E2E.
|
||||
|
||||
@@ -156,7 +156,7 @@ current tree. This establishes a partial implementation, not feature closure.
|
||||
|
||||
- Registry is the primary non-chat consumer of MCP-created scenarios: revisions saved through 050 tools land here with the same provenance, lifecycle and staleness semantics as any other actor.
|
||||
|
||||
**Status (2026-09-02): done** — реализовано в рамках 050: инструменты и гейты (`specs/050-mcp-interface/tasks.md` T012–T028 [x]), handoff-поверхность (050 T030–T033), демонтаж чата и сервиса `agent/` (050 T040–T041, чекпоинты `specs/WORKSTATE-043-047.md`).
|
||||
**Historical MCP transport/decommission status (2026-09-02): reported done; not production-readiness evidence** — реализовано в рамках 050: инструменты и гейты (`specs/050-mcp-interface/tasks.md` T012–T028 [x]), handoff-поверхность (050 T030–T033), демонтаж чата и сервиса `agent/` (050 T040–T041, чекпоинты `specs/WORKSTATE-043-047.md`).
|
||||
|
||||
## @{ ScenarioRegistry.AgentAuthoringPromotion [C:5] [TYPE ADR]
|
||||
@BRIEF Registry boundary for persistent AgentAuthoringWorkspace promotion.
|
||||
@@ -170,4 +170,12 @@ current tree. This establishes a partial implementation, not feature closure.
|
||||
The registry owns artifact/revision provenance and materialization; it does not execute exploratory code or infer graph authority from an artifact. 044 may launch only a promoted validated revision. This is a normative contract amendment, not an implementation-complete claim.
|
||||
## @} ScenarioRegistry.AgentAuthoringPromotion
|
||||
|
||||
## Production contract refresh — 2026-09-08
|
||||
|
||||
**Frontend boundary (user decision 2026-09-08)**: All agent interaction is external MCP only. Product frontend MUST NOT contain agent chat, prompt/request textarea, assistant editing, typical-operation-to-agent selector, proposal-generation, agent workspace/start or handoff controls/routes. Ordinary manual CRUD/editor, human approval/review, monitoring and read-only evidence/evaluation are permitted. AgentEvaluationCard is read-only, with no prompt/retry-agent/provider controls. Existing agent proposal UI is runtime drift; removal/negative DOM-route-network acceptance remains OPEN in this spec-only change.
|
||||
|
||||
**SCREG-FR-014 — Revision baseline requirements and activation**: Every immutable revision MUST preserve canonical baseline references, browser capability/registry fingerprints, capture profile, optional AgentEvaluationSpec and DecisionPolicy hashes. Registry stores authoring constraints, never launch BaselineSelectionPin. Activation/save MUST validate all refs and context authority through the same 038 service, preserve server-owned handles and user-reviewed diff/CAS, and cannot substitute latest catalog or candidate as execution truth. Original run snapshots remain unchanged after edits, activation or rebaseline.
|
||||
|
||||
Normative contract: [Revision baseline requirements and activation](../044-dashboard-scenario-execution/contracts/production-chain.md). New requirements are specified, **implemented=false / acceptance OPEN** until executable evidence closes the linked tasks/checklist/traceability rows. Historical local tests and the manual inconclusive ss-prod run do not prove browser/capture/baseline/LLM production readiness. The refresh scope is the audited P0/P1/P2 agentic E2E and baseline gaps; an approved ExecutionPerformanceBaseline is not introduced.
|
||||
|
||||
#endregion ScenarioRegistry.Spec
|
||||
|
||||
@@ -100,4 +100,16 @@
|
||||
|
||||
Setup → US1 → US2; revisions depend on models; staleness depends on 037/041 hooks; lifecycle depends on revisions. 043 editor and 044 runner consume this registry.
|
||||
|
||||
## Production readiness — 2026-09-08 (SCREG-FR-014)
|
||||
|
||||
Historical [x] rows above retain only their dated local/transport evidence; they do not prove current production readiness. Reopened rows were contradicted by the audited gaps. Removed frontend/agent paths are historical, not implementation prerequisites. New acceptance is **implemented=false / OPEN**.
|
||||
|
||||
Contract: [Revision baseline requirements and activation](../044-dashboard-scenario-execution/contracts/production-chain.md).
|
||||
|
||||
- [ ] T030 [P0/P1/P2] Compile→register→bootstrap→save preserves server query context without client repair; unsafe/stale graph produces no partial registry rows. Implement at the existing 042 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
|
||||
- [ ] T031 [P0/P1/P2] Changing baseline constraint/capture/evaluation/policy creates a candidate revision and explicit diff; old runs retain exact original pins. Implement at the existing 042 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
|
||||
- [ ] T032 [P0/P1/P2] Activation rejects drifted materialized artifact or stale CAS; ordinary manual registry/editor routes contain no agent generation controls. Implement at the existing 042 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
|
||||
|
||||
Frontend boundary for this package: manual CRUD/editor, human review/approval, monitoring and read-only evidence/evaluation only; all agent interaction is external MCP. No agent chat/prompt/assistant editing/proposal generation/workspace/start/handoff controls. Runtime removal is OPEN, not performed by this spec refresh. Optional approved performance baseline is outside scope.
|
||||
|
||||
#endregion ScenarioRegistry.Tasks
|
||||
|
||||
@@ -32,3 +32,16 @@ This matrix is conjunctive with the 038, 044 and 050 matrices. Open or partial
|
||||
evidence is a release `NO-GO`; unchecked implementation tasks remain unchanged.
|
||||
|
||||
Authoring E2E is a release gate: no GO unless sandbox outputs become typed 038 proposals, a user reviews the server-computed diff, 042 performs handle-based CAS/idempotent save, and 044 rejects every non-promoted revision or raw authoring artifact. This normative row does not imply implementation completion.
|
||||
|
||||
## Production acceptance traceability — 2026-09-08
|
||||
|
||||
Historical rows above identify prior tests/code only; removed agent UI paths are retired. The following audited gates are **implemented=false / OPEN**, independent of local suite totals.
|
||||
|
||||
| Requirement | Domain contract / DTO | Task | Falsifiable acceptance | State |
|
||||
|---|---|---|---|---|
|
||||
| SCREG-FR-014 | [Revision baseline requirements and activation](../044-dashboard-scenario-execution/contracts/production-chain.md); [data model](data-model.md) | [T030](tasks.md) | Compile→register→bootstrap→save preserves server query context without client repair; unsafe/stale graph produces no partial registry rows. | OPEN |
|
||||
| SCREG-FR-014 | [Revision baseline requirements and activation](../044-dashboard-scenario-execution/contracts/production-chain.md); [data model](data-model.md) | [T031](tasks.md) | Changing baseline constraint/capture/evaluation/policy creates a candidate revision and explicit diff; old runs retain exact original pins. | OPEN |
|
||||
| SCREG-FR-014 | [Revision baseline requirements and activation](../044-dashboard-scenario-execution/contracts/production-chain.md); [data model](data-model.md) | [T032](tasks.md) | Activation rejects drifted materialized artifact or stale CAS; ordinary manual registry/editor routes contain no agent generation controls. | OPEN |
|
||||
| SCREG-FR-014; external-MCP-only UI | manual editor/review; read-only evidence | [production tasks](tasks.md) | No frontend agent prompt/chat/assistant editing/proposal generation/workspace/start/handoff routes or requests; human approval remains usable. | OPEN |
|
||||
|
||||
Sources: [production gap](../../docs/reports/ss-prod-agentic-e2e-production-gap-2026-09-08.md), [coverage gap](../../docs/reports/ss-prod-agentic-e2e-spec-coverage-2026-09-08.md), [baseline gap](../../docs/reports/ss-prod-agentic-e2e-baseline-gap-2026-09-08.md). Spec schema/static checks prove contract structure only; live canary/runtime closure and optional approved performance baseline are not claimed.
|
||||
|
||||
@@ -27,9 +27,9 @@
|
||||
- [~] CHK009 Dependency editing via visual DAG
|
||||
- [~] CHK010 Cycle/duplicate-output validation on every change
|
||||
|
||||
## Agent Edit (FR-006)
|
||||
## External Proposal Human Review (FR-006)
|
||||
|
||||
- [ ] CHK011 "Edit with agent" proposes revision with diff
|
||||
- [ ] CHK011 External MCP creates typed stored proposal; frontend only reviews diff, with no agent prompt/generation control
|
||||
- [ ] CHK012 Proposal validated before display; any agent save is delegated, provenance-bearing and policy checked
|
||||
|
||||
## RBAC / UX (FR-007/008)
|
||||
@@ -40,3 +40,14 @@
|
||||
## Success Criteria
|
||||
|
||||
- [ ] CHK015 SC-001..006 verified (read-only default, revision, rejection, delegated policy, hybrid C)
|
||||
|
||||
## Production readiness checklist — 2026-09-08
|
||||
|
||||
SCEDIT-FR-011: [Safe proposal and semantic-chain editing](../../038-dashboard-scenario-model/contracts/browser-actions.md). Historical [x] marks do not close this new production gate; removed frontend components are not current evidence. All rows below implemented=false / OPEN.
|
||||
|
||||
- [ ] CHK016 DEF-01 proposal→validate→diff→save round-trips parameters/steps without unrelated changes; stale/unsafe proposal rejects before persistence. Evidence: [T021](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK017 Removal of producer/dependency updates exact closure while preserving unrelated logical IDs; changed capture/evaluation/policy/baseline constraints create a new candidate. Evidence: [T022](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK018 Remove AgentActionPanel/agent prompt/typical-operation/proposal-generation UI; route/DOM/network fixtures prove manual edit and human review make no agent request. Evidence: [T023](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK019 Negative product UI test: no agent chat/prompt/assistant editing/proposal-generation/typical-operation-to-agent/workspace/start/handoff controls or agent invocation routes/requests; manual CRUD/editor/human review/read-only results remain usable.
|
||||
|
||||
Schema/static success alone is not runtime completion. Optional approved performance baseline is outside scope.
|
||||
|
||||
@@ -48,12 +48,12 @@ def revalidate(db, scenario_id, base_revision_id): ...
|
||||
|
||||
# #region ScenarioEditor.AgentProposal [C:4] [TYPE Function] [SEMANTICS scenario,editor,agent,proposal,diff]
|
||||
# @ingroup ScenarioEditor
|
||||
# @BRIEF Generate an agent-assisted edit proposal as a full revision with diff.
|
||||
# @PRE request is bounded; proposal validated before display.
|
||||
# @BRIEF Accept typed externally authored graph operations over MCP and compute a server-owned diff.
|
||||
# @PRE external MCP principal is authorized; typed operations are bounded; base revision/digest match.
|
||||
# @POST returns proposed graph + diff + validation; a delegated agent may convert it into a saved immutable revision through SaveRevision.
|
||||
# @SIDE_EFFECT LLM call (agent tool); read-only draft.
|
||||
# @INVARIANT proposal passes 038 validation; never bypasses policy, immutable revision creation, or a required gate.
|
||||
def agent_propose(db, scenario_id, base_hash, request): ...
|
||||
# @SIDE_EFFECT server-owned proposal persistence and audit; no backend LLM call or frontend agent request.
|
||||
# @INVARIANT proposal passes 038 validation; never bypasses policy, immutable revision creation, or a required gate. Frontend can review stored proposals but cannot invoke this agent-authoring operation.
|
||||
def accept_external_proposal(db, scenario_id, base_hash, typed_ops, principal): ...
|
||||
# #endregion ScenarioEditor.AgentProposal
|
||||
|
||||
# #region ScenarioEditor.ValidateAssertion [C:3] [TYPE Function] [SEMANTICS scenario,editor,assertion,constrain]
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
@RELATION DEPENDS_ON -> [ScenarioEditor.Spec]
|
||||
|
||||
## Decision 1 — Hybrid edit model (C)
|
||||
Business fields/parameters manual, assertions constrained, dependencies visual, generated executable read-only, complex changes "Edit with agent". Compatible with 038 safety and mandatory human participation.
|
||||
Business fields/parameters manual, assertions constrained, dependencies visual, generated executable read-only, complex changes authored exclusively in an external MCP client. Compatible with 038 safety and mandatory human participation.
|
||||
|
||||
## Decision 2 — Read-only default
|
||||
Editor loads read-only; Edit mode is explicit. No revision is created until a durable edit passes delegated-action policy.
|
||||
@@ -14,6 +14,9 @@ Only registered operators (037 comparison) + baseline refs; free-form SQL/raw ba
|
||||
## Decision 4 — Visual DAG with live validation
|
||||
Dependency edits revalidate for cycles/duplicate outputs (038) on every change.
|
||||
|
||||
## Decision 5 — Agent edits never silent
|
||||
"Edit with agent" proposes a full revision + diff; a delegated agent may save it after validation, otherwise an inline ActionApprovalGate decides it.
|
||||
## Decision 5 — External authoring, human review only
|
||||
External MCP authors submit typed operations; the editor reviews the server-stored diff and provenance. Human review/approval and manual CRUD remain available. No agent request textarea, assistant editing, typical-operation selector, proposal-generation, chat, workspace or launch/handoff controls may render.
|
||||
|
||||
@RATIONALE Explicit user decision 2026-09-08 separates human product UX from agent interaction.
|
||||
@REJECTED Renaming AgentActionPanel as a handoff panel preserves the forbidden interaction and is not a compliant removal.
|
||||
#endregion ScenarioEditor.Ux.Decisions
|
||||
|
||||
@@ -45,6 +45,14 @@ Fields: new_revision_id, parent_revision_id, content_hash, activation_status=`ca
|
||||
|
||||
- Save requires `base_revision_id` match → 409 STALE_REVISION on mismatch.
|
||||
- Every edit operation validated by 038 validator before revision creation.
|
||||
- Agent proposals carry a full proposed graph which passes validation before display; an agent-saved revision retains AgentAction/InvestigationCase provenance. A save is not permission to activate it or change an automation target.
|
||||
- Externally authored MCP proposals carry a full proposed graph which passes validation before display; an agent-saved revision retains AgentAction/InvestigationCase provenance. A save is not permission to activate it or change an automation target.
|
||||
|
||||
## Production record contract — 2026-09-08 (SCEDIT-FR-011)
|
||||
|
||||
EditProposal and WorkingDraft are server-owned typed graphs with base_revision_id, digest, context authority, operation provenance and diff. External MCP authoring only; frontend holds ordinary edit buffer and read-only stored proposal diff, never agent request/prompt state. Save never accepts a client graph and never activates implicitly.
|
||||
|
||||
Normative detail: [Safe proposal and semantic-chain editing](../038-dashboard-scenario-model/contracts/browser-actions.md). New fields, CAS transitions and cross-record integrity checks are implemented=false until [production tasks](tasks.md) and [traceability](traceability.md) close with executable evidence. Existing shorter field lists are legacy compatibility projections, not permission to omit the production identity fields.
|
||||
|
||||
Frontend contains no agent interaction state, prompt, proposal-generation or workspace/start controls. Human manual editor/review state and read-only evaluation/result artifacts are separate from external MCP authoring. Approved performance baseline is outside scope.
|
||||
|
||||
#endregion ScenarioEditor.DataModel
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
|
||||
## Summary
|
||||
|
||||
A first-class editor for viewing and editing persisted scenarios (hybrid edit model C): manual business fields/parameters, constrained assertion editing, visual DAG dependency editing, read-only generated executable, and agent-assisted complex edits. Every durable edit produces a new immutable revision via 042 under delegated-action policy.
|
||||
A first-class editor for viewing and editing persisted scenarios (hybrid edit model C): manual business fields/parameters, constrained assertion editing, visual DAG dependency editing, read-only generated executable, and human review of external MCP-authored changes. Every durable edit produces a new immutable revision via 042 under delegated-action policy.
|
||||
|
||||
## Technical Context
|
||||
|
||||
@@ -42,7 +42,7 @@ specs/043-dashboard-scenario-editor/
|
||||
|
||||
backend/src/services/dashboard_testing/editor/ (ops.py, save.py, agent.py, assert_validate.py)
|
||||
frontend/src/lib/models/ScenarioEditorModel.svelte.ts
|
||||
frontend/src/lib/components/scenario-editor/ (StepCard, VisualDagCanvas, ConstrainedAssertionEditor, EditRevisionDiff, AgentActionPanel)
|
||||
frontend/src/lib/components/scenario-editor/ (StepCard, VisualDagCanvas, ConstrainedAssertionEditor, EditRevisionDiff, read-only external proposal diff)
|
||||
frontend/src/routes/dashboard-testing/scenarios/[id]/edit/+page.svelte
|
||||
```
|
||||
|
||||
@@ -53,7 +53,7 @@ frontend/src/routes/dashboard-testing/scenarios/[id]/edit/+page.svelte
|
||||
3. SaveRevision (delegated policy + 042 revision chain).
|
||||
4. Constrained assertion editor.
|
||||
5. Visual DAG dependency editor.
|
||||
6. Agent-assisted edit proposal.
|
||||
6. External MCP typed proposal intake; frontend human review only. Remove agent interaction controls.
|
||||
7. Frontend model/components, polish, regression gates.
|
||||
|
||||
## Traceability
|
||||
@@ -64,8 +64,19 @@ traceability.md maps Story → model → operationId → contract → task → t
|
||||
|
||||
- Reads persisted scenarios from 042; writes new revisions to 042.
|
||||
- Reuses 038 validator/resolver and 037 comparison operators.
|
||||
- "Edit with agent" reuses agent tooling from 036/038.
|
||||
- Agent authoring runs only in external MCP clients via 050; frontend has no agent request action.
|
||||
|
||||
## Complexity Tracking
|
||||
|
||||
No exception planned. Edit ops are bounded C3-C4; visual DAG and constrained assertion stay model-first.
|
||||
|
||||
## Production delivery plan — 2026-09-08 (SCEDIT-FR-011)
|
||||
|
||||
Status: specified, implemented=false; historical unit/prototype/transport results are not current production acceptance.
|
||||
|
||||
1. Pin [Safe proposal and semantic-chain editing](../038-dashboard-scenario-model/contracts/browser-actions.md) and [data model](data-model.md); write negative fixtures before runtime changes.
|
||||
2. Implement existing domain boundaries for: EditProposal and WorkingDraft are server-owned typed graphs with base_revision_id, digest, context authority, operation provenance and diff. External MCP authoring only; frontend holds ordinary edit buffer and read-only stored proposal diff, never agent request/prompt state. Save never accepts a client graph and never activates implicitly.
|
||||
3. Execute [tasks](tasks.md) T021, T022, T023 and retain reproducible evidence in [traceability](traceability.md), then close [checklist](checklists/requirements.md) individually.
|
||||
4. Run cross-spec canary only after 037 catalog publication, 038 chain and 044 provider/content/policy gates; use 046 versioned cost/load/SLO limits. Disable admission on rollback, retain pins/receipts/holds; no fallback to stale catalog or synthetic PASS.
|
||||
|
||||
Frontend agent prompts, chat, assistant editing, proposal-generation, workspace/start/handoff actions are prohibited. Only external MCP clients interact with agents; frontend provides ordinary manual CRUD/editor, human review/approval and read-only monitoring/evidence. Runtime removal tasks are not closed by this document. Optional approved performance baseline is outside this refresh.
|
||||
|
||||
@@ -1,30 +1,66 @@
|
||||
# Quickstart: Scenario Editor (043)
|
||||
|
||||
> **Factual audit 2026-08-20:** pending verification checklist only; no command result below is
|
||||
> currently asserted as evidence.
|
||||
> **Refresh 2026-09-08 (production contract):** safe proposal and semantic-chain editing
|
||||
> (SCEDIT-FR-011) is normative in `../038-dashboard-scenario-model/contracts/browser-actions.md`
|
||||
> — implemented=false / acceptance OPEN. Agent authoring is external MCP only (050); the product
|
||||
> editor is manual CRUD plus human review of a *stored* proposal diff. The former agent-driven
|
||||
> flow (dashboard → `/agent` workspace → agent-generated scenario) is SUPERSEDED. Existing
|
||||
> `AgentActionPanel` prompt/proposal-generation controls are runtime drift (T015); negative
|
||||
> DOM/route/network acceptance remains OPEN. Commands below are local gates, not production GO.
|
||||
|
||||
## Prereqs
|
||||
- 042 registry backend live (scenario + revisions)
|
||||
- Frontend deps installed
|
||||
## Target vs observed
|
||||
|
||||
## Commands
|
||||
| Layer | Target (refresh) | Observed (do not treat as GO) |
|
||||
|---|---|---|
|
||||
| Authoring | External MCP client creates a validated proposal; editor loads stored diff/provenance read-only | MCP proposal/save path exists; product UI still contains agent-panel drift |
|
||||
| Manual edit | Constrained assertion + visual DAG + metadata/params; SQL/raw-baseline rejected | Editor ops/save/revalidate unit tests exist |
|
||||
| Save | Server-stored WorkingDraft (`draft_id` + digest); no client graph upload | WorkingDraft save tests exist; route-level E2E/policy-gate evidence is not retained |
|
||||
| Chain | Preserve server context authority, stable logical IDs, AgentEvaluationSpec/DecisionPolicy in the diff | Compiler/editor still do not close the visual chain (Slice E) |
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- 042 registry backend reachable (scenario + revisions).
|
||||
- `backend/.venv` activated; frontend deps installed for UI suites only.
|
||||
|
||||
## Offline contract check
|
||||
|
||||
```bash
|
||||
# Backend editor op validation
|
||||
cd backend && source .venv/bin/activate && python -m pytest -v tests/services/dashboard_testing/editor/
|
||||
|
||||
# Frontend model + UX
|
||||
cd frontend && npm run test -- ScenarioEditor
|
||||
|
||||
# Lint
|
||||
cd backend && python -m ruff check src/services/dashboard_testing/editor/
|
||||
cd frontend && npm run lint
|
||||
backend/.venv/bin/python specs/044-dashboard-scenario-execution/prototype/validate_contract_refresh.py \
|
||||
specs/044-dashboard-scenario-execution/fixtures/production-contract-refresh.json
|
||||
```
|
||||
|
||||
## Exit Gates
|
||||
- [ ] Read-only view renders persisted scenario
|
||||
- [ ] SQL/raw-baseline injection rejected in assertion editor
|
||||
- [ ] Cycle/duplicate rejected in DAG editor
|
||||
- [ ] Save produces new immutable revision after delegated-policy evaluation or an inline ActionApprovalGate
|
||||
- [ ] Agent proposal shows diff and provenance; delegated agent save remains validator/policy checked
|
||||
- [ ] ruff clean; prototype states covered
|
||||
## Available local verification
|
||||
|
||||
Editor tests live under the registry suite, not `tests/services/dashboard_testing/editor/`.
|
||||
|
||||
```bash
|
||||
cd backend && source .venv/bin/activate
|
||||
python -m pytest -q \
|
||||
tests/services/dashboard_testing/registry/test_scenario_editor_ops.py \
|
||||
tests/services/dashboard_testing/registry/test_scenario_editor_save.py \
|
||||
tests/services/dashboard_testing/registry/test_scenario_editor_load.py \
|
||||
tests/services/dashboard_testing/registry/test_scenario_editor_revalidate.py \
|
||||
tests/services/dashboard_testing/registry/test_scenario_editor_agent.py \
|
||||
tests/api/test_scenario_editor_routes.py
|
||||
python -m ruff check src/services/dashboard_testing/editor/
|
||||
```
|
||||
|
||||
```bash
|
||||
cd frontend
|
||||
npm run test -- --run src/lib/models/__tests__/ScenarioEditorModel.test.ts
|
||||
npm run test -- --run src/lib/components/scenario-editor
|
||||
npm run lint
|
||||
```
|
||||
|
||||
`AgentActionPanel.test.ts` covers the retired prompt/proposal-generation panel. Keep it green only as a drift-removal regression until T015 deletes that surface.
|
||||
|
||||
## Exit gates (acceptance OPEN)
|
||||
|
||||
- Read-only view renders a persisted scenario; edit mode is explicit.
|
||||
- SQL/raw-baseline/path injection is rejected by the constrained assertion editor.
|
||||
- Cycle/duplicate-output is rejected by the DAG editor.
|
||||
- Save produces a new immutable revision after delegated-policy evaluation or an inline ActionApprovalGate; the client never uploads a full graph.
|
||||
- Stored MCP proposal shows diff and provenance; save remains validator/policy checked. No prompt/start/handoff control exists in the editor.
|
||||
- Prototype `@UX_STATE` coverage via `prototype/index.html` remains an open T018 check.
|
||||
|
||||
Local pytest/vitest success is historical runtime-gate evidence. Production GO additionally requires live browser/screenshot/LLM providers, Slice E compiler emission, and negative UI acceptance that the product frontend contains no agent interaction.
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
#region ScenarioEditor.Spec [C:3] [TYPE ADR] [SEMANTICS spec,requirements,ux,scenario,editor,visual,agent]
|
||||
@BRIEF User-facing Scenario Editor: view and edit a persisted scenario with a hybrid (C) editing model — manual business fields/parameters, constrained assertions, visual dependency editing, read-only generated executable, and agent-assisted complex changes.
|
||||
@BRIEF User-facing Scenario Editor: view and edit a persisted scenario with a hybrid (C) editing model — manual business fields/parameters, constrained assertions, visual dependency editing, read-only generated executable, and human review of externally authored changes.
|
||||
@RELATION DEPENDS_ON -> [Doc.Adr.ADR0001]
|
||||
@RELATION DEPENDS_ON -> [Doc.Adr.ADR0006]
|
||||
@RELATION DEPENDS_ON -> [ScenarioRegistry.Spec]
|
||||
@@ -7,14 +7,14 @@
|
||||
@RATIONALE 038 defines the DTOs but no lifecycle editing UX. A scenario must be viewable and editable outside the agent chat, with every durable edit producing a new immutable revision and delegated policy determining whether an inline approval is required.
|
||||
@REJECTED Agent-only editing (no visual surface) — rejected because users need to review and adjust a scenario without re-prompting the agent each time.
|
||||
@REJECTED Unconstrained free-form DAG/assertion editor — rejected because it could inject SQL/raw baselines/unsafe paths, violating 038 safety invariants; assertions use constrained editors and generated executable stays read-only.
|
||||
@REJECTED Chat-bound "Edit with agent" as the only proposal channel — superseded 2026-08-24: proposals are creatable through MCP tools per `specs/050-mcp-interface/spec.md`; server-stored WorkingDraft, digest binding and SCEDIT-FR-009 no-arbitrary-draft-save constraints stand unchanged.
|
||||
@REJECTED Any frontend agent interaction, including assistant editing and proposal-generation controls — prohibited by explicit user scope 2026-09-08. Proposals are creatable only through external MCP tools per `specs/050-mcp-interface/spec.md`; server-stored WorkingDraft, digest binding and SCEDIT-FR-009 no-arbitrary-draft-save constraints stand unchanged.
|
||||
|
||||
## Navigation (DSA Indexer keywords)
|
||||
@SEMANTICS: spec, requirements, feature, ux, scenario, editor, visual, revision, agent
|
||||
|
||||
**Feature Branch**: `043-dashboard-scenario-editor`
|
||||
**Created**: 2026-08-07 | **Status**: Partially implemented — factual audit pending remediation
|
||||
**Input**: "Provide a first-class Scenario Editor for viewing and editing persisted scenarios: business metadata and parameter definitions editable manually, assertions via constrained editors, dependencies via visual DAG editing, generated executable read-only, and complex changes delegated to 'Edit with agent'. Metadata changes use a registry metadata version; executable changes create immutable revisions."
|
||||
**Input**: "Provide a first-class Scenario Editor for viewing and editing persisted scenarios: business metadata and parameter definitions editable manually, assertions via constrained editors, dependencies via visual DAG editing, generated executable read-only, and complex changes authored only in an external MCP client. Metadata changes use a registry metadata version; executable changes create immutable revisions."
|
||||
|
||||
## User Scenarios
|
||||
|
||||
@@ -68,15 +68,16 @@
|
||||
|
||||
---
|
||||
|
||||
### Story 5 — Agent-Assisted Complex Edit (P3)
|
||||
### Story 5 — Review Externally Authored Changes (P3)
|
||||
|
||||
**Why P3**: Some changes (new checklist case, complex assertion) are easier described in natural language.
|
||||
External MCP clients may create validated EditProposals under delegated policy. The product editor only displays stored diff/provenance and permits human review or manual editing; it never collects an agent request or invokes an agent.
|
||||
|
||||
**Independent Test**: Request "add XLSX comparison" via Edit-with-agent and verify the agent creates a validated WorkingDraft and saves a revision when delegated policy permits.
|
||||
**Independent Test**: Submit a proposal through MCP, open its stored diff in the editor, and review it without any LLM call from the frontend.
|
||||
|
||||
**Acceptance**:
|
||||
1. **Given** a complex change request **When** "Edit with agent" runs **Then** the agent creates a server-stored `EditProposal`.
|
||||
2. **Given** the proposal is accepted **When** its base revision is still current **Then** it becomes a `WorkingDraft`; a policy-authorized analyst or agent save creates the revision. A stale proposal never saves; a non-delegated action waits at an inline gate.
|
||||
1. A current validated proposal may become a WorkingDraft; save uses its server digest and a fresh authorization decision.
|
||||
2. Stale or unsafe proposals cannot save; required human approval remains distinct from agent invocation.
|
||||
3. No prompt textarea, agent editing, operation-to-agent selector, proposal-generation button, agent workspace, or agent-start route exists in the editor.
|
||||
|
||||
---
|
||||
|
||||
@@ -113,7 +114,7 @@
|
||||
- **SCEDIT-FR-004**: Dependency editing MUST be visual (graph), with 038 cycle/duplicate-output validation on every change.
|
||||
- **SCEDIT-FR-005**: Every durable executable graph edit MUST produce a new immutable revision; metadata uses metadata_version concurrency. Deterministic policy decides whether the actor/agent may save immediately or must obtain an inline ActionApprovalGate.
|
||||
- **SCEDIT-FR-005a**: Every proposal/diff MUST expose Verification Program changes: SqlEvidenceSpec template/hash/relation refs, TransformSpec operations, assertions, AgentEvaluationSpec and DecisionPolicy. A runtime finding can only change this content through a new validated proposal and revision.
|
||||
- **SCEDIT-FR-006**: "Edit with agent" MUST create a server-stored validated proposal/draft with a diff. A delegated agent MAY save the revision; no agent edit may bypass 038 validation, 042 revision provenance, or a required gate.
|
||||
- **SCEDIT-FR-006**: Agent authoring MUST occur only in an external MCP client. Frontend MUST NOT provide agent prompts, assistant editing, proposal generation, typical-operation-to-agent controls or agent launch/handoff actions. Stored external proposals expose a read-only diff and human review; delegated saves still require 038 validation, 042 provenance and any required gate.
|
||||
- **SCEDIT-FR-009**: Every save MUST use a server-stored WorkingDraft (draft_id + digest); the client MUST NOT return the full graph to save (no arbitrary-draft bypass). Save re-validates/canonicalizes/hashes server-side.
|
||||
- **SCEDIT-FR-010**: A stale scenario MUST be revalidatable: affected refs mapped against the current dashboard, conflicts surfaced manually, a proposed revision + diff shown for approval (Scenario Migration workflow).
|
||||
- **SCEDIT-FR-007**: RBAC MUST enforce scenario:edit separately from scenario:run.
|
||||
@@ -139,9 +140,9 @@
|
||||
|
||||
### Session 2026-08-07
|
||||
|
||||
- Q: Which edit model? → A: **C (hybrid)** — business fields/params manual, assertions constrained, dependencies visual, generated executable read-only, complex changes "Edit with agent".
|
||||
- Q: Which edit model? → A: **C (hybrid)** — business fields/params manual, assertions constrained, dependencies visual, generated executable read-only, complex changes authored exclusively through external MCP clients.
|
||||
- Q: May the agent save a revision? → A: Yes, after deterministic validation when delegated policy permits. It always creates an immutable revision with agent/delegator/case provenance; policy-gated actions use an inline ActionApprovalGate.
|
||||
- Q: Does this replace 039 workspace? → A: No. 039 is the create flow in agent chat; 043 is the post-save edit surface over the registry.
|
||||
- Q: Does this replace 039 workspace? → A: 039 owns ordinary manual create/review surfaces; 043 owns post-save manual editing. Neither embeds agent interaction.
|
||||
|
||||
## Implementation Status & MVP Debt (factual audit 2026-08-20)
|
||||
|
||||
@@ -155,8 +156,16 @@ present. The hybrid editor therefore exists structurally but is not verified as
|
||||
|
||||
## Drift Amendment — MCP Interface (2026-08-24)
|
||||
|
||||
- "Edit with agent" (Story 5) continues with external MCP clients: the proposal/WorkingDraft flow, stale-proposal rejection and gate-bound saves are unchanged; only the conversation medium moves out of the product.
|
||||
- External proposal authoring (Story 5) occurs exclusively in external MCP clients: the proposal/WorkingDraft flow, stale-proposal rejection and gate-bound saves are unchanged; only the conversation medium moves out of the product.
|
||||
|
||||
**Status (2026-09-02): done** — реализовано в рамках 050: инструменты и гейты (`specs/050-mcp-interface/tasks.md` T012–T028 [x]), handoff-поверхность (050 T030–T033), демонтаж чата и сервиса `agent/` (050 T040–T041, чекпоинты `specs/WORKSTATE-043-047.md`).
|
||||
**Historical MCP transport/decommission status (2026-09-02): reported done; not production-readiness evidence** — реализовано в рамках 050: инструменты и гейты (`specs/050-mcp-interface/tasks.md` T012–T028 [x]), handoff-поверхность (050 T030–T033), демонтаж чата и сервиса `agent/` (050 T040–T041, чекпоинты `specs/WORKSTATE-043-047.md`).
|
||||
|
||||
## Production contract refresh — 2026-09-08
|
||||
|
||||
**Frontend boundary (user decision 2026-09-08)**: All agent interaction is external MCP only. Product frontend MUST NOT contain agent chat, prompt/request textarea, assistant editing, typical-operation-to-agent selector, proposal-generation, agent workspace/start or handoff controls/routes. Ordinary manual CRUD/editor, human approval/review, monitoring and read-only evidence/evaluation are permitted. AgentEvaluationCard is read-only, with no prompt/retry-agent/provider controls. Existing agent proposal UI is runtime drift; removal/negative DOM-route-network acceptance remains OPEN in this spec-only change.
|
||||
|
||||
**SCEDIT-FR-011 — Safe proposal and semantic-chain editing**: Proposal→validation→diff→save MUST round-trip the canonical typed graph without adding unrelated dependencies/assertions or coercing parameters between array/object (DEF-01). Preserve server context authority and stable logical IDs; changed baseline constraint/capture/evaluation/policy/registry creates a new candidate revision with explicit diff. A runtime finding proposes a new revision, never edits baseline or historical evaluation. Validate removal of a producer and all consumers, reject unsafe or stale proposal before save with zero partial rows.
|
||||
|
||||
Normative contract: [Safe proposal and semantic-chain editing](../038-dashboard-scenario-model/contracts/browser-actions.md). New requirements are specified, **implemented=false / acceptance OPEN** until executable evidence closes the linked tasks/checklist/traceability rows. Historical local tests and the manual inconclusive ss-prod run do not prove browser/capture/baseline/LLM production readiness. The refresh scope is the audited P0/P1/P2 agentic E2E and baseline gaps; an approved ExecutionPerformanceBaseline is not introduced.
|
||||
|
||||
#endregion ScenarioEditor.Spec
|
||||
|
||||
@@ -41,13 +41,12 @@
|
||||
- [x] T012 [US4] Write failing visual-dependency validation tests (cycle/duplicate) in `backend/tests/services/dashboard_testing/registry/test_scenario_editor_ops.py`
|
||||
- [x] T013 [US4] Build `VisualDagCanvas.svelte` with duplicate/self validation on change
|
||||
|
||||
## Phase 6 — US5 Agent-Assisted Edit
|
||||
## Phase 6 — US5 External Proposal Intake and Human Review
|
||||
|
||||
- [x] T014 [US5] Implement `agent_propose` in `backend/src/services/dashboard_testing/editor/agent.py`
|
||||
@INVARIANT: proposal passes validation; agent save is fully attributed and policy checked
|
||||
@TEST_INVARIANT: covered by `test_scenario_editor_agent.py` + route tests (agent-propose/proposal-save)
|
||||
- [x] T015 [US5] Build persistent `AgentActionPanel.svelte` + `EditRevisionDiff.svelte`; add `POST /scenarios/{id}/edits/agent-propose`
|
||||
Tests: `frontend/src/lib/components/scenario-editor/__tests__/AgentActionPanel.test.ts`
|
||||
- [ ] T015 [US5] Remove frontend AgentActionPanel prompt/typical-operation/proposal-generation controls and agent-propose client calls; retain EditRevisionDiff for stored external proposals and human review. Historical panel tests are retired and do not close the new negative UI requirement.
|
||||
|
||||
## Phase 6b — WorkingDraft + Revalidate (P0 #7 / #9)
|
||||
|
||||
@@ -74,4 +73,16 @@
|
||||
|
||||
Setup → US1; US2 depends on ops/validate; US3/4 build on ops; US5 depends on US2-4. Requires 042 registry for persistence.
|
||||
|
||||
## Production readiness — 2026-09-08 (SCEDIT-FR-011)
|
||||
|
||||
Historical [x] rows above retain only their dated local/transport evidence; they do not prove current production readiness. Reopened rows were contradicted by the audited gaps. Removed frontend/agent paths are historical, not implementation prerequisites. New acceptance is **implemented=false / OPEN**.
|
||||
|
||||
Contract: [Safe proposal and semantic-chain editing](../038-dashboard-scenario-model/contracts/browser-actions.md).
|
||||
|
||||
- [ ] T021 [P0/P1/P2] DEF-01 proposal→validate→diff→save round-trips parameters/steps without unrelated changes; stale/unsafe proposal rejects before persistence. Implement at the existing 043 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
|
||||
- [ ] T022 [P0/P1/P2] Removal of producer/dependency updates exact closure while preserving unrelated logical IDs; changed capture/evaluation/policy/baseline constraints create a new candidate. Implement at the existing 043 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
|
||||
- [ ] T023 [P0/P1/P2] Remove AgentActionPanel/agent prompt/typical-operation/proposal-generation UI; route/DOM/network fixtures prove manual edit and human review make no agent request. Implement at the existing 043 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
|
||||
|
||||
Frontend boundary for this package: manual CRUD/editor, human review/approval, monitoring and read-only evidence/evaluation only; all agent interaction is external MCP. No agent chat/prompt/assistant editing/proposal generation/workspace/start/handoff controls. Runtime removal is OPEN, not performed by this spec refresh. Optional approved performance baseline is outside scope.
|
||||
|
||||
#endregion ScenarioEditor.Tasks
|
||||
|
||||
@@ -9,7 +9,20 @@
|
||||
| US2 Manual | SCEDIT-FR-002/005 | EditOperation | editor.apply, editor.save | Editor.ApplyOps, Editor.SaveRevision | T006-T009 | test_ops, edit.ux.test |
|
||||
| US3 Assertion | SCEDIT-FR-003 | ConstrainedAssertionEdit | editor.validate-assertion | Editor.ValidateAssertion | T010-T011 | edit.ux.test |
|
||||
| US4 Deps | SCEDIT-FR-004 | VisualDependencyEdit | editor.apply | Editor.ApplyOps | T012-T013 | test_ops |
|
||||
| US5 Agent | SCEDIT-FR-006 | EditRevisionResult | editor.agent-propose | Editor.AgentProposal | T014-T015 | test_editor_agent |
|
||||
| US5 External proposal review | SCEDIT-FR-006 | EditRevisionResult | MCP typed proposal intake; editor stored diff | Editor.AgentProposal | T014-T015 | test_editor_agent |
|
||||
| RBAC/UX | SCEDIT-FR-007/008 | — | — | — | T016 | test_rbac |
|
||||
|
||||
N/A: Run Monitor (045), Automation (046), Analytics (047).
|
||||
|
||||
## Production acceptance traceability — 2026-09-08
|
||||
|
||||
Historical rows above identify prior tests/code only; removed agent UI paths are retired. The following audited gates are **implemented=false / OPEN**, independent of local suite totals.
|
||||
|
||||
| Requirement | Domain contract / DTO | Task | Falsifiable acceptance | State |
|
||||
|---|---|---|---|---|
|
||||
| SCEDIT-FR-011 | [Safe proposal and semantic-chain editing](../038-dashboard-scenario-model/contracts/browser-actions.md); [data model](data-model.md) | [T021](tasks.md) | DEF-01 proposal→validate→diff→save round-trips parameters/steps without unrelated changes; stale/unsafe proposal rejects before persistence. | OPEN |
|
||||
| SCEDIT-FR-011 | [Safe proposal and semantic-chain editing](../038-dashboard-scenario-model/contracts/browser-actions.md); [data model](data-model.md) | [T022](tasks.md) | Removal of producer/dependency updates exact closure while preserving unrelated logical IDs; changed capture/evaluation/policy/baseline constraints create a new candidate. | OPEN |
|
||||
| SCEDIT-FR-011 | [Safe proposal and semantic-chain editing](../038-dashboard-scenario-model/contracts/browser-actions.md); [data model](data-model.md) | [T023](tasks.md) | Remove AgentActionPanel/agent prompt/typical-operation/proposal-generation UI; route/DOM/network fixtures prove manual edit and human review make no agent request. | OPEN |
|
||||
| SCEDIT-FR-011; external-MCP-only UI | manual editor/review; read-only evidence | [production tasks](tasks.md) | No frontend agent prompt/chat/assistant editing/proposal generation/workspace/start/handoff routes or requests; human approval remains usable. | OPEN |
|
||||
|
||||
Sources: [production gap](../../docs/reports/ss-prod-agentic-e2e-production-gap-2026-09-08.md), [coverage gap](../../docs/reports/ss-prod-agentic-e2e-spec-coverage-2026-09-08.md), [baseline gap](../../docs/reports/ss-prod-agentic-e2e-baseline-gap-2026-09-08.md). Spec schema/static checks prove contract structure only; live canary/runtime closure and optional approved performance baseline are not claimed.
|
||||
|
||||
@@ -14,7 +14,7 @@ The analyst edits directly in the persistent editor or asks the agent for a comp
|
||||
## 3. Screens & States
|
||||
|
||||
### Screen: Scenario Editor
|
||||
- **Layout**: Left = graph canvas; right = inspector panel for selected step; top = toolbar (Edit mode toggle, Save, "Edit with agent"); footer = validation status + diff.
|
||||
- **Layout**: Left = graph canvas; right = inspector panel for selected step; top = toolbar (manual Edit mode toggle, Save; no agent request/generation controls); footer = validation status + diff.
|
||||
- **Read-only default**: view renders graph/params/assertions; Edit mode toggles editing.
|
||||
- **ConstrainedAssertionEditor**: operator select + threshold + baseline ref; no free-text expected value.
|
||||
- **VisualDagCanvas**: drag dependency edges; live cycle/duplicate validation.
|
||||
|
||||
@@ -76,3 +76,15 @@
|
||||
041, 042, 046 and 047.
|
||||
- [ ] CHK031 Browser and Screenshot providers perform authorized live I/O and produce owned durable evidence
|
||||
in a real deployment run.
|
||||
|
||||
## Production readiness checklist — 2026-09-08
|
||||
|
||||
SCEX-FR-028: [Production baseline-backed evaluation](../contracts/production-chain.md). Historical [x] marks do not close this new production gate; removed frontend components are not current evidence. All rows below implemented=false / OPEN.
|
||||
|
||||
- [ ] CHK032 Baseline set/version/catalog/release/commit/IDs/digests affect idempotency; moving catalog after admission cannot change plan/result; legacy unpinned result is ineligible. Evidence: [T043](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK033 GET/HEAD prove same ACL/status/headers; MIME/digest/length checked before bytes, cross-owner hidden, expired410, corrupt409, traversal/range/oversize rejected. Evidence: [T044](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK034 Startup/readiness/start-loop and shutdown/drain/cancel/reconcile survive fault injection; unknown effect quarantines capacity; late response cannot win. Evidence: [T045](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK035 End-to-end real browser→capture→durable artifact→deterministic comparison→optional immutable evaluation→policy→result; all required evidence present before PASS. Evidence: [T046](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK036 Negative product UI test: no agent chat/prompt/assistant editing/proposal-generation/typical-operation-to-agent/workspace/start/handoff controls or agent invocation routes/requests; manual CRUD/editor/human review/read-only results remain usable.
|
||||
|
||||
Schema/static success alone is not runtime completion. Optional approved performance baseline is outside scope.
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
openapi: 3.1.0
|
||||
info:
|
||||
title: Scenario Execution Engine API
|
||||
version: 0.1.0
|
||||
version: 0.2.0
|
||||
description: Start and manage execution of dashboard test scenario runs (044).
|
||||
paths:
|
||||
/api/scenario-runs/{run_id}/artifacts/{artifact_id}/content:
|
||||
$ref: 'artifact-content.openapi.yaml#/paths/~1api~1scenario-runs~1{run_id}~1artifacts~1{artifact_id}~1content'
|
||||
/api/scenario-runs:
|
||||
post:
|
||||
operationId: scenarioRun.start
|
||||
@@ -25,6 +27,7 @@ paths:
|
||||
params: { type: object, description: "Launch values; server validates against 038 ParameterDefinition and persists immutable ParameterBinding[]" }
|
||||
requested_target_reference: { type: object, description: "User-selected release/target; server verifies it matches the captured TargetSnapshot" }
|
||||
baseline_set: { type: string }
|
||||
baseline_set_version: { type: string, description: Immutable explicit catalog selection version; required with baseline_set, resolved before idempotency. }
|
||||
execution_toggles: { type: object, description: "Optional evidence only (diagnostic screenshots, verbose logs, optional VLM). Mandatory steps cannot be disabled." }
|
||||
responses:
|
||||
"201": { description: ScenarioRun created (queued or pending_approval), content: { application/json: { schema: { $ref: "#/components/schemas/ScenarioRun" } } } }
|
||||
@@ -150,6 +153,7 @@ components:
|
||||
target_snapshot: { $ref: "#/components/schemas/TargetSnapshot" }
|
||||
execution_principal_fingerprint: { type: string }
|
||||
analytics_context_key: { type: string, description: "Server-derived immutable analytics grouping key" }
|
||||
baseline_pin: { anyOf: [{ $ref: '../../037-superset-baseline-engine/contracts/baseline-pin.schema.json' }, { type: 'null' }] }
|
||||
verification_program_hash: { type: string }
|
||||
action_registry_version: { type: string }
|
||||
step_runs: { type: array, items: { $ref: "#/components/schemas/ScenarioStepRun" } }
|
||||
@@ -186,8 +190,12 @@ components:
|
||||
captured_at: { type: string, format: date-time }
|
||||
ScenarioExecutionResult:
|
||||
type: object
|
||||
required: [run_id, status, step_counts, provenance, failures]
|
||||
required: [run_id, status, step_counts, provenance, failures, baseline_pin, evidence, comparisons, evaluations, step_outcomes]
|
||||
properties:
|
||||
baseline_pin: { anyOf: [{ $ref: '../../037-superset-baseline-engine/contracts/baseline-pin.schema.json' }, { type: 'null' }] }
|
||||
evidence: { type: array, items: { $ref: 'result-evidence.schema.json#/$defs/artifact' } }
|
||||
comparisons: { type: array, items: { $ref: 'result-evidence.schema.json#/$defs/comparison' } }
|
||||
evaluations: { type: array, items: { $ref: '../../038-dashboard-scenario-model/contracts/agent-evaluation.schema.json' } }
|
||||
run_id: { type: string }
|
||||
status: { type: string }
|
||||
step_counts: { type: object, additionalProperties: { type: integer } }
|
||||
@@ -214,6 +222,7 @@ components:
|
||||
scenario_content_hash: { type: string, pattern: '^[0-9a-fA-F]{64}$' }
|
||||
verification_program_hash: { type: string, pattern: '^[0-9a-fA-F]{64}$' }
|
||||
runner_version: { type: string, minLength: 1 }
|
||||
baseline_pin: { anyOf: [{ $ref: '../../037-superset-baseline-engine/contracts/baseline-pin.schema.json' }, { type: 'null' }] }
|
||||
target_snapshot: { $ref: "#/components/schemas/TargetSnapshot" }
|
||||
parameter_bindings: { type: array, items: { $ref: "#/components/schemas/ParameterBinding" } }
|
||||
execution_principal_fingerprint: { type: string, minLength: 1 }
|
||||
@@ -237,44 +246,11 @@ components:
|
||||
byte_length: { type: integer, minimum: 1 }
|
||||
sha256: { type: string, pattern: '^[0-9a-fA-F]{64}$' }
|
||||
AgentEvaluationSummary:
|
||||
type: object
|
||||
required: [evaluation_id, logical_step_id, verdict, confidence, model_id, prompt_template_version]
|
||||
properties:
|
||||
evaluation_id: { type: string }
|
||||
logical_step_id: { type: string }
|
||||
verdict: { type: string, enum: [pass, fail, inconclusive] }
|
||||
confidence: { type: number, minimum: 0, maximum: 1 }
|
||||
model_id: { type: string }
|
||||
prompt_template_version: { type: string }
|
||||
reason_codes: { type: array, items: { type: string } }
|
||||
evidence_refs: { type: array, items: { type: string } }
|
||||
$ref: '../../038-dashboard-scenario-model/contracts/agent-evaluation.schema.json'
|
||||
AgentEvaluation:
|
||||
allOf:
|
||||
- $ref: "#/components/schemas/AgentEvaluationSummary"
|
||||
- type: object
|
||||
required: [scenario_run_id, attempt, provider_id, model_version, prompt_template_id, input_manifest_hash, findings, raw_response_artifact_ref, started_at, finished_at]
|
||||
properties:
|
||||
scenario_run_id: { type: string }
|
||||
attempt: { type: integer, minimum: 1 }
|
||||
provider_id: { type: string }
|
||||
model_version: { type: string }
|
||||
prompt_template_id: { type: string }
|
||||
input_manifest_hash: { type: string }
|
||||
findings: { type: array, items: { type: object } }
|
||||
raw_response_artifact_ref: { type: string }
|
||||
started_at: { type: string, format: date-time }
|
||||
finished_at: { type: string, format: date-time }
|
||||
$ref: '../../038-dashboard-scenario-model/contracts/agent-evaluation.schema.json'
|
||||
DecisionPolicy:
|
||||
type: object
|
||||
required: [policy_id, version, deterministic_hard_failure, high_confidence_failure, low_confidence, disagreement, missing_evidence]
|
||||
properties:
|
||||
policy_id: { type: string }
|
||||
version: { type: string }
|
||||
deterministic_hard_failure: { type: string, enum: [failed] }
|
||||
high_confidence_failure: { type: string, enum: [failed, inconclusive] }
|
||||
low_confidence: { type: string, enum: [inconclusive] }
|
||||
disagreement: { type: string, enum: [waiting_human, inconclusive] }
|
||||
missing_evidence: { type: string, enum: [blocked, inconclusive] }
|
||||
$ref: '../../038-dashboard-scenario-model/contracts/decision-policy.schema.json'
|
||||
RunComparison:
|
||||
type: object
|
||||
required: [run_a, run_b, step_deltas, compatibility]
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
|
||||
## ScenarioRun and PROD approval lifecycle
|
||||
|
||||
`ScenarioRun` is created before dispatch. Fields: id, scenario_id, scenario_revision_id (revision_id UUID), scenario_content_hash, verification_program_hash, action_registry_version, dashboard_id, environment_id, status (`pending_approval|queued|running|waiting_human|blocked|cancel_requested|cancelled|passed|failed|inconclusive`), phase (preflight|setup|executing|waiting_human|draining|terminal), parameter_bindings (immutable JSON), baselines_pinned (version), target_snapshot, execution_principal_fingerprint, execution_toggles (optional evidence only), trigger_source (server-owned), agent_run_id? (provenance), verification_run_id? (aggregation), idempotency_key (unique), started_at, finished_at, resume_token, error_code, runner_version.
|
||||
`ScenarioRun` is created before dispatch. Fields: id, scenario_id, scenario_revision_id (revision_id UUID), scenario_content_hash, verification_program_hash, action_registry_version, dashboard_id, environment_id, status (`pending_approval|queued|running|waiting_human|blocked|cancel_requested|cancelled|passed|failed|inconclusive`), phase (preflight|setup|executing|waiting_human|draining|terminal), parameter_bindings (immutable JSON), baseline_pin (exact 037 BaselineSelectionPin; null only for a revision declaring no baseline-backed checks), target_snapshot, execution_principal_fingerprint, execution_toggles (optional evidence only), trigger_source (server-owned), agent_run_id? (provenance), verification_run_id? (aggregation), idempotency_key (unique), started_at, finished_at, resume_token, error_code, runner_version.
|
||||
|
||||
`live_execution_binding_ref?` plus `live_execution_binding_snapshot?` are the only persisted live-I/O
|
||||
coordinates. The snapshot is an exact allowlist: binding ref; environment/release/query-model,
|
||||
@@ -59,11 +59,11 @@ Fields: id, run_id FK, logical_step_id (immutable UUID, from #8), step_position
|
||||
|
||||
`AgentEvaluation { evaluation_id, scenario_run_id, logical_step_id, attempt, provider_id, model_id, model_version, prompt_template_id, prompt_template_version, input_manifest_hash, evidence_refs, verdict, confidence, findings, reason_codes, raw_response_artifact_ref, started_at, finished_at }` is immutable evidence generated only for a declared 038 AgentEvaluationSpec. Its tool/evidence access is bounded by that spec; it cannot mutate program content, invoke mutation actions, change run state or choose downstream scheduling.
|
||||
|
||||
`DecisionPolicy { policy_id, version, deterministic_hard_failure, high_confidence_failure, low_confidence, disagreement, missing_evidence }` is a versioned deterministic mapper. Defaults: deterministic hard failure→failed; high-confidence policy-qualified agent failure→failed; low confidence→inconclusive; evaluation/evidence disagreement→waiting_human via a HumanCheckpoint only for manual runs; missing required evidence→blocked or inconclusive. Scenario aggregation consumes StepOutcome, not AgentEvaluation verdicts directly.
|
||||
`DecisionPolicy { policy_id, version, deterministic_hard_failure, high_confidence_failure, low_confidence, disagreement, missing_evidence }` is a versioned deterministic mapper. Defaults: deterministic hard failure→failed; high-confidence policy-qualified agent failure→failed; low confidence→inconclusive; evaluation/evidence disagreement→inconclusive; missing required evidence→blocked; provider error for required evaluation→inconclusive. Exact first-match baseline-semantic/1.0.0 cases and strict schemas are [038 DecisionPolicy](../038-dashboard-scenario-model/contracts/decision-policy.md); no dynamic checkpoint creation. Scenario aggregation consumes StepOutcome, not AgentEvaluation verdicts directly.
|
||||
|
||||
## RunnerPlan — deterministic derivation from revision (#2)
|
||||
|
||||
RunnerPlan is **derived deterministically at run start** from the selected immutable `ScenarioRevision`/Verification Program, NOT read from a stored `runner.plan.json`. Fields: scenario_revision_id, scenario_content_hash, verification_program_hash, `action_registry_version`, `action_registry_hash`, env targets, resolved params, pinned baselines, topological order, and one immutable `ActionExecutionDescriptor` per step. A descriptor contains the exact `{tool, action}`, typed input/output contracts, idempotency, retry safety, side-effect-key policy, timeout, mutation contract/risk. The runner persists the descriptor snapshot in both `steps` and `executor_mapping`; it derives leases/recovery/retry policy only from that snapshot. A missing, altered, unknown, version/hash-mismatched descriptor, invalid input/output shape, or mutating action without its required contract rejects before run/lease/I/O. Run refuses if its revision/program/action-registry hashes differ from the selected revision.
|
||||
RunnerPlan is **derived deterministically at run start** from the selected immutable `ScenarioRevision`/Verification Program, NOT read from a stored `runner.plan.json`. Fields: scenario_revision_id, scenario_content_hash, verification_program_hash, `action_registry_version`, `action_registry_hash`, env targets, resolved params, exact baseline_pin plus canonical pin digest, topological order, and one immutable `ActionExecutionDescriptor` per step. A descriptor contains the exact `{tool, action}`, typed input/output contracts, idempotency, retry safety, side-effect-key policy, timeout, mutation contract/risk. The runner persists the descriptor snapshot in both `steps` and `executor_mapping`; it derives leases/recovery/retry policy only from that snapshot. A missing, altered, unknown, version/hash-mismatched descriptor, invalid input/output shape, or mutating action without its required contract rejects before run/lease/I/O. Run refuses if its revision/program/action-registry hashes differ from the selected revision.
|
||||
|
||||
@INVARIANT RunnerPlan descriptor resolution is exact and version/hash pinned; `tool` is never a dispatch or retry-policy fallback.
|
||||
@REJECTED A universal `idempotent=true, retry_safe=true` claim based on a non-human tool was rejected because it can repeat unsafe effects.
|
||||
@@ -81,9 +81,9 @@ Artifacts use a **generic owner**: `Artifact { id, owner_type: agent_run|scenari
|
||||
## Decision gates (#3)
|
||||
|
||||
- `ActionApprovalGate` — authorization approval for PROD execution, baseline approval, repository mutation. Generalized 036 gate with `owner_type` + `owner_id`.
|
||||
- `HumanCheckpoint` — `checkpoint_id, run_id, logical_step_id, checkpoint_type, decision_policy, status, created_at, expires_at, eligible_role?, eligible_actor_ids?, assigned_to?, evidence_refs, decision_version, decided_by?, decided_at?, disposition?, comment?`. Status is `pending|decided|expired|cancelled`; decision is CAS on `decision_version`, stale/concurrent decision returns 409. `finding_review` maps confirm→failed, false_positive→passed, inconclusive→inconclusive; `manual_assertion` maps pass→passed, fail→failed, inconclusive→inconclusive. It is NOT a 036 ApprovalGate decision.
|
||||
- `HumanCheckpoint` — `checkpoint_id, run_id, logical_step_id, checkpoint_type, decision_policy, status, created_at, expires_at, eligible_role?, eligible_actor_ids?, assigned_to?, evidence_refs, decision_version, decided_by?, decided_at?, disposition?, comment?`. Status is `pending|decided|expired|cancelled`; decision is CAS on `decision_version`, stale/concurrent decision returns 409. v1 disposition maps confirm→passed, false_positive→inconclusive, inconclusive→inconclusive; `manual_assertion` maps pass→passed, fail→failed, inconclusive→inconclusive. It is NOT a 036 ApprovalGate decision.
|
||||
|
||||
A HumanCheckpoint is never delegated to the agent: it is a manual-run-only analyst decision inside a currently executing run. The agent may explain the evidence in its workspace but cannot consume the checkpoint or convert it into an automated result.
|
||||
A HumanCheckpoint is never delegated to the agent: it is a manual-run-only analyst decision inside a currently executing run. External MCP clients may display evidence; only authenticated human USER decision tools may consume the checkpoint under 050. Product frontend offers human review, never agent workspace or invocation.
|
||||
|
||||
## Worker semantics — at-least-once execution (#6)
|
||||
|
||||
@@ -120,7 +120,7 @@ The registry resolves `ActionExecutionDescriptor -> executor`; the following lis
|
||||
- artifact -> generic artifact register (owner_type=scenario_run)
|
||||
- human -> EXCLUDED (runner-lifecycle HumanCheckpoint control, not an executor)
|
||||
|
||||
`BrowserExecutor` implements only registered actions: `open_dashboard`, `navigate_tab`, `apply_native_filter`, `inspect_filter_state`, `apply_table_filter`, `extract_table`, `scroll_to`, `inspect_columns`, `click`, `select_rows`, `edit_row`, `bulk_edit`, `download`, `refresh`, `wait_for_state`. ScreenshotService is evidence infrastructure only, not the browser action executor. It validates the registry-declared inputs/outputs/risk/timeout/idempotency before dispatch.
|
||||
`BrowserExecutor` implements only registered actions: `open_dashboard`, `navigate_tab`, `apply_native_filter`, `inspect_filter_state`, `apply_table_filter`, `pagination`, `navigate_dashboard`, `extract_table`, `scroll_to`, `inspect_columns`, `click`, `select_rows`, `edit_row`, `bulk_edit`, `download`, `refresh`, `wait_for_state`. ScreenshotService is evidence infrastructure only, not the browser action executor. It validates the registry-declared inputs/outputs/risk/timeout/idempotency before dispatch.
|
||||
|
||||
## Provider runtime protocol
|
||||
|
||||
@@ -215,8 +215,7 @@ new leases, while active unknown operations retain their lease until reconciliat
|
||||
## Verification thresholds
|
||||
|
||||
- `scenario_content_hash`, `verification_program_hash`, `descriptor_fingerprint`, `query_model_fingerprint`,
|
||||
`execution_principal_fingerprint`, `rls_security_fingerprint` and evidence `sha256` are lowercase or
|
||||
uppercase hexadecimal SHA-256 values with exactly 64 characters.
|
||||
`execution_principal_fingerprint`, `rls_security_fingerprint` and evidence `sha256` are canonical lowercase hexadecimal SHA-256 values with exactly 64 characters.
|
||||
- `attempt` starts at 1 and increases by exactly 1 for each new logical-step attempt. Historical attempts
|
||||
are immutable; only one attempt may be the active projection for a logical step.
|
||||
- `byte_length` is a positive integer equal to the durable content length. A zero-byte evidence object is
|
||||
@@ -241,4 +240,26 @@ A ScenarioRun pins `scenario_revision_id` + `scenario_content_hash` at start. Re
|
||||
|
||||
All `AgentRun`, `VerificationRun`, `LoadRun`, and `ScenarioRun` claims pass through one environment-scoped capacity manager: `environment, workload_class, priority, quota, reserved_capacity`. A 046 scenario policy is a consumer of this global allocator, not an independent PROD/PREPROD concurrency limit.
|
||||
|
||||
## Production records and compatibility — 2026-09-08
|
||||
|
||||
[Result schema](contracts/result-evidence.schema.json) defines the unified run projection.
|
||||
[AgentEvaluation schema](../038-dashboard-scenario-model/contracts/agent-evaluation.schema.json)
|
||||
replaces the abbreviated field list above for new records: operation/spec identity,
|
||||
input artifact MIME/length/SHA manifest, provider/model/prompt/schema versions,
|
||||
criterion-bound findings, raw response provenance, usage/pricing and full baseline pin.
|
||||
Evaluations/comparisons/outcomes are append-only; unique active attempt is selected by CAS.
|
||||
Historical late attempts remain visible but never replace a winner.
|
||||
|
||||
Run request hash includes the server-resolved baseline set/version, catalog/release/publication
|
||||
commit and entry IDs/digests before run/gate creation. The exact pin is persisted in RunnerPlan,
|
||||
ScenarioRun, result and AnalyticsContextKey provenance; comparison never uses first-map-entry
|
||||
or moving latest fallback. Missing legacy pins are ineligible, not fabricated.
|
||||
|
||||
Artifact content uses [protected GET/HEAD](contracts/artifact-content.openapi.yaml), not public URLs.
|
||||
Metadata carries owner/run/step/attempt/operation, SHA-256, actual MIME, byte length,
|
||||
retention/expiry and availability. Required baseline bytes have retention holds.
|
||||
[Production chain](contracts/production-chain.md) owns provider startup/shutdown/cancel/reconcile.
|
||||
New records and lifecycle acceptance remain implemented=false. Frontend only renders read-only
|
||||
evaluation evidence; no provider/prompt/retry-agent controls. Approved performance baseline is outside scope.
|
||||
|
||||
#endregion ScenarioExecution.DataModel
|
||||
|
||||
@@ -96,3 +96,14 @@ traceability.md maps Story → model → operationId → contract → task → t
|
||||
No exception planned. Runner/dispatch/provider lifecycle are C5 but decomposed (runner, dispatch,
|
||||
provider protocol, operations, health, capacity, per-provider adapters). Do not collapse into one
|
||||
oversized orchestrator.
|
||||
|
||||
## Production delivery plan — 2026-09-08 (SCEX-FR-028)
|
||||
|
||||
Status: specified, implemented=false; historical unit/prototype/transport results are not current production acceptance.
|
||||
|
||||
1. Pin [Production baseline-backed evaluation](contracts/production-chain.md) and [data model](data-model.md); write negative fixtures before runtime changes.
|
||||
2. Implement existing domain boundaries for: ScenarioRun and RunnerPlan require exact server BaselineSelectionPin and pin digest in canonical request identity. Result DTO is contracts/result-evidence.schema.json; artifact bytes use protected GET/HEAD. Append-only comparison/evaluation/outcome receipts bind run/step/attempt/operation; only CAS-selected winning attempt can determine outcome. Provider loop and cleanup/reconcile receipts gate capacity release.
|
||||
3. Execute [tasks](tasks.md) T043, T044, T045, T046 and retain reproducible evidence in [traceability](traceability.md), then close [checklist](checklists/requirements.md) individually.
|
||||
4. Run cross-spec canary only after 037 catalog publication, 038 chain and 044 provider/content/policy gates; use 046 versioned cost/load/SLO limits. Disable admission on rollback, retain pins/receipts/holds; no fallback to stale catalog or synthetic PASS.
|
||||
|
||||
Frontend agent prompts, chat, assistant editing, proposal-generation, workspace/start/handoff actions are prohibited. Only external MCP clients interact with agents; frontend provides ordinary manual CRUD/editor, human review/approval and read-only monitoring/evidence. Runtime removal tasks are not closed by this document. Optional approved performance baseline is outside this refresh.
|
||||
|
||||
@@ -1,12 +1,36 @@
|
||||
# Quickstart: Scenario Execution Engine (044)
|
||||
|
||||
> **Verification status (2026-08-24):** the available SQLite/unit profile passes, but this is not
|
||||
> production evidence. Browser and Screenshot providers remain unavailable by default. Production GO
|
||||
> additionally requires provider contract tests, real PostgreSQL migration checks and live deployment proof.
|
||||
> **Refresh 2026-09-08 (production contract):** production baseline-backed evaluation (SCEX-FR-028)
|
||||
> is normative in `contracts/production-chain.md`, `contracts/artifact-content.openapi.yaml`,
|
||||
> `contracts/result-evidence.schema.json` — implemented=false / acceptance OPEN. Canonical chain:
|
||||
> browser→capture→durable artifact→deterministic baseline comparison→optional declared
|
||||
> AgentEvaluation→DecisionPolicy→StepOutcome. Offline executable check (schemas + truth tables +
|
||||
> evaluation-evidence binding):
|
||||
>
|
||||
> ```bash
|
||||
> backend/.venv/bin/python specs/044-dashboard-scenario-execution/prototype/validate_contract_refresh.py \
|
||||
> specs/044-dashboard-scenario-execution/fixtures/production-contract-refresh.json
|
||||
> ```
|
||||
>
|
||||
> Historical result: `PASS: 11 schemas / 5 positive / 49 negative`. Schema/fixture PASS is not
|
||||
> production GO.
|
||||
|
||||
## Target vs observed
|
||||
|
||||
| Layer | Target (refresh) | Observed (do not treat as GO) |
|
||||
|---|---|---|
|
||||
| Runner | Deterministic DAG walker, pinned ActionRegistry, human checkpoint, cancel/retry | Local SQLite/unit profile exists; fail-closed adapters exist |
|
||||
| Provider loop | FastAPI lifespan owns exactly one `ProviderEventLoop` | Runtime wiring landed 2026-09-10 (handoff slice A+B); ss-prod 2026-09-08 still recorded `not_started` |
|
||||
| AgentEvaluation | Immutable record + DecisionPolicy mapper; production adapter builds `input_manifest` from prior-step artifacts | Store/parser/executor plus production `evaluation_adapter_from` and walker rows 9–14 landed 2026-09-10 (uncommitted). Live LLM still unregistered |
|
||||
| Artifact bytes | Authenticated GET/HEAD with MIME/digest/length (Slice G) | Runtime GET/HEAD landed 2026-09-10 (uncommitted, offline tests). Live storage canary OPEN |
|
||||
| Visual chain | Compiler emits browser→capture→comparison→evaluation→policy (Slice E) | One-action templates; live Playwright canary OPEN |
|
||||
| Live providers | Registered browser/screenshot + multimodal LLM | ss-prod: unregistered, bindings `0`, LLM providers `0` |
|
||||
|
||||
The former agent-driven product-UI flow (dashboard → `/agent` workspace → agent-generated scenario) is SUPERSEDED. 044 is a headless execution engine; agent interaction is external MCP (050). Product UI (045) is read-only evidence plus human approvals.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- PostgreSQL reachable through a real `DATABASE_URL`.
|
||||
- PostgreSQL reachable through a real `DATABASE_URL` for migration/integration checks.
|
||||
- `backend/.venv` activated.
|
||||
- `SERVICE_JWT` set when using Docker Compose.
|
||||
- 042 registry and 044 migrations applied.
|
||||
@@ -29,20 +53,18 @@ python -m pytest -q \
|
||||
tests/services/dashboard_testing/registry/test_scenario_crash_recovery.py \
|
||||
tests/services/dashboard_testing/registry/test_scenario_worker.py \
|
||||
tests/services/dashboard_testing/registry/test_scenario_queued_dispatch.py
|
||||
python -m pytest -q tests/services/dashboard_testing/execution \
|
||||
tests/services/dashboard_testing/registry/test_agent_evaluation_store.py \
|
||||
tests/services/dashboard_testing/scenario/test_agent_evaluation_models.py
|
||||
python -m ruff check src/services/dashboard_testing/execution src/api/routes/dashboard_testing/scenario_runs.py
|
||||
python -m compileall -q src/services/dashboard_testing/execution src/api/routes/dashboard_testing/scenario_runs.py
|
||||
cd ..
|
||||
python3 specs/044-dashboard-scenario-execution/prototype/validate_static.py
|
||||
```
|
||||
|
||||
Expected current local evidence:
|
||||
2026-09-10 handoff recorded 292 passed on the first 044-slice command and 21+18 on the evaluation/policy suites. Those counts are historical runtime evidence for slices A+B/C/F, not this documentation pass and not production GO.
|
||||
|
||||
- full available 044 profile: `246 passed`;
|
||||
- provider/lifecycle edge profile: `59 passed`;
|
||||
- prototype static validation: `passed`;
|
||||
- scoped Ruff and compile: `passed`.
|
||||
|
||||
## Production verification
|
||||
## Production verification (still OPEN)
|
||||
|
||||
```bash
|
||||
cd backend
|
||||
@@ -52,14 +74,14 @@ alembic upgrade head
|
||||
python -m pytest -q --run-integration tests/integration/
|
||||
```
|
||||
|
||||
Run the provider contract profile only after T028-T042 exists:
|
||||
Provider contract profile (T028–T042; live Browser/Screenshot still required):
|
||||
|
||||
```bash
|
||||
python -m pytest -q tests/services/dashboard_testing/registry/test_provider_contract.py
|
||||
python -m pytest -q tests/services/dashboard_testing/registry/test_provider_*.py
|
||||
```
|
||||
|
||||
## Measurable exit gates
|
||||
## Measurable exit gates (acceptance OPEN)
|
||||
|
||||
1. SC-001..011 each has a passing named test or deployment evidence record.
|
||||
2. 100/100 cancellation trials terminate by `cancel_drain_deadline_at + 5 seconds`.
|
||||
@@ -68,12 +90,8 @@ python -m pytest -q tests/services/dashboard_testing/registry/test_provider_*.py
|
||||
5. 100% of unknown external effects are reconciled or terminalized non-pass before retry.
|
||||
6. Every enabled provider has passing liveness, readiness and dependency-health checks.
|
||||
7. Browser and Screenshot perform one real authorized deployment run with durable evidence.
|
||||
8. No unresolved P0/P1 traceability row remains in 044 or execution-critical dependencies 036, 037, 038,
|
||||
041, 042, 046 and 047.
|
||||
8. No unresolved P0/P1 traceability row remains in 044 or execution-critical dependencies 036, 037, 038, 041, 042, 046 and 047.
|
||||
|
||||
## Current boundary
|
||||
|
||||
The local profile proves fail-closed adapters, exact Superset binding behavior, lifecycle closure and
|
||||
prototype state coverage. It does not prove Browser/Screenshot live composition, shared provider capacity,
|
||||
AgentEvaluation runtime, real PostgreSQL migration validity, scheduler deployment behavior or 047 case
|
||||
ingestion.
|
||||
The local profile proves fail-closed adapters, exact Superset binding behavior, lifecycle closure, prototype state coverage, DecisionPolicy unit coverage, AgentEvaluation store/parser/executor, production evaluation adapter (mock provider), and walker `EVALUATION_UNAVAILABLE` / `BASELINE_AND_SEMANTIC_PASS`. It does not prove Browser/Screenshot live composition, shared provider capacity, live LLM evaluation, authenticated artifact GET/HEAD, real PostgreSQL migration validity, scheduler deployment behavior, or 047 case ingestion on a live stand.
|
||||
|
||||
@@ -114,10 +114,10 @@
|
||||
- **SCEX-FR-007**: A ScenarioRun MUST be recoverable by `scenario_run_id`; browser recovery MUST reconstruct state from a declared browser-safe checkpoint, not continue a dead Playwright context. Results MUST carry target and execution-principal provenance.
|
||||
- **SCEX-FR-008**: PROD-classified environments MUST require an ActionApprovalGate before execution.
|
||||
- **SCEX-FR-009**: Executors MUST reuse 037 (metric_executor/comparison), 038 (capture), 036 (evidence/artifacts/HITL), and existing browser/xlsx infra; a second Playwright/LLM/SQL stack is forbidden.
|
||||
- **SCEX-FR-010**: `human` is a runner-lifecycle control, not a dispatched executor; the executor registry covers the other seven tools.
|
||||
- **SCEX-FR-010**: `human` is a runner-lifecycle control, not a dispatched executor; the executor registry covers every enabled non-human descriptor in the version-pinned catalog.
|
||||
- **SCEX-FR-011**: Failed, inconclusive and blocked runs MUST emit idempotent 036 InvestigationSignals carrying immutable run/evidence provenance; 047 creates/updates the Queue/Episode. They MUST NOT automatically start a chat, an AgentRun, or a remediation action.
|
||||
- **SCEX-FR-012**: An opened InvestigationCase MAY use the agent to construct diagnostic runs and controlled experiments under delegated policy. Agent work cannot bypass executor contracts, runner lifecycle, capacity, mutation policy or a required ActionApprovalGate; declared AgentEvaluationSpec is the only permitted runtime reasoning boundary.
|
||||
- **SCEX-FR-013**: A HumanCheckpoint remains a manual-run-only analyst decision. The agent may summarize evidence but MUST NOT consume the checkpoint, choose its disposition, or turn it into scheduled automation.
|
||||
- **SCEX-FR-013**: A HumanCheckpoint remains a manual-run-only analyst decision. The external MCP agent may summarize evidence but MUST NOT consume the checkpoint, choose its disposition, or turn it into scheduled automation.
|
||||
- **SCEX-FR-014**: `SqlEvidenceExecutor` MUST execute exactly the immutable 038 SqlEvidenceSpec through Superset backend/SQL Lab with pinned database identity, ExecutionPrincipal and RLS/security fingerprint. Runtime only supplies typed ParameterBindings and may not alter SQL, relation, projection, JOIN or WHERE.
|
||||
- **SCEX-FR-015**: `AgentEvaluation` MUST be a separate immutable runtime record and DecisionPolicy MUST deterministically map it plus deterministic evidence to StepOutcome. A bare model verdict never directly sets ScenarioResult.
|
||||
- **SCEX-FR-016**: Browser actions and mutation safety MUST use the same versioned 038 ActionRegistry/mutation contract. Mutating browser steps in PROD are prohibited; test-data mutation needs fixture scope, record keys, side-effect/retry and cleanup policy independent of PROD approval.
|
||||
@@ -177,15 +177,11 @@ It marks browser and screenshot slots from an enabled binding as `BROWSER_BINDIN
|
||||
unavailable.
|
||||
|
||||
Providers return typed success/failure/timeout/cancellation outcomes. PASS additionally requires a
|
||||
durable evidence reference and verified non-zero SHA-256. Where a provider supports transport
|
||||
cancellation, the root passes the capability through; otherwise lifecycle cancellation remains
|
||||
database-authoritative and the result remains typed. Browser recovery is lawful only from the declared
|
||||
durable evidence reference and verified non-zero SHA-256. The root must cancel by operation_id. If a provider cannot acknowledge cancellation, its effect remains unknown and prevents retry/PASS until reconciliation; DB cancellation alone is not a provider acknowledgement. Browser recovery is lawful only from the declared
|
||||
safe checkpoint/reconstruction binding; a raw session is never revived. A mutating browser action must
|
||||
also satisfy the version-pinned 038 mutation contract and is rejected in PROD.
|
||||
|
||||
`ScreenshotService` currently captures paths for the LLM workflow but does not expose a 044
|
||||
principal/RLS/checkpoint-bound durable-evidence provider. It therefore remains typed unavailable until
|
||||
startup registers such a provider; no path or raw capture metadata is treated as evidence.
|
||||
The 044 ScreenshotProvider adapter already wraps ScreenshotService and writes durable receipts. It remains unregistered on the audited ss-prod deployment; protected content delivery, live lifecycle wiring and AgentEvaluation integration are still open. A raw path or capture metadata alone is never evidence.
|
||||
|
||||
### Key Entities
|
||||
|
||||
@@ -339,7 +335,7 @@ the BrowserProvider can be called production-ready.
|
||||
- Unaffected structurally: ScenarioRun, executors, capacity and gates never referenced the chat runtime. MCP clients author scenarios before runs (038 chain) and investigate after terminal signals (047 cases) through governed tools only; `manual_run_only` and PROD gating apply regardless of the actor.
|
||||
- **SCEX-FR-013 re-scoped for MCP (decision 2026-08-24)**: HumanCheckpoint disposition MAY be submitted through governed MCP decision tools (`decide_checkpoint`) when driven by an authenticated user principal — it remains a manual analyst decision, CAS-versioned and audited, equivalent to the 045 monitor path. The prohibition that stands unchanged: no automated origin (scheduled/deploy/release/ETL/API/service-principal) may create, consume or bypass a checkpoint, and the agent-as-autonomous-planner still cannot choose a disposition on its own.
|
||||
|
||||
**Status (2026-09-02): done** — реализовано в рамках 050: инструменты и гейты (`specs/050-mcp-interface/tasks.md` T012–T028 [x]), handoff-поверхность (050 T030–T033), демонтаж чата и сервиса `agent/` (050 T040–T041, чекпоинты `specs/WORKSTATE-043-047.md`).
|
||||
**Historical MCP transport/decommission status (2026-09-02): reported done; not production-readiness evidence** — реализовано в рамках 050: инструменты и гейты (`specs/050-mcp-interface/tasks.md` T012–T028 [x]), handoff-поверхность (050 T030–T033), демонтаж чата и сервиса `agent/` (050 T040–T041, чекпоинты `specs/WORKSTATE-043-047.md`).
|
||||
|
||||
## Field-run Amendment — HumanCheckpoint disposition clarity (2026-09-07)
|
||||
|
||||
@@ -375,4 +371,12 @@ the BrowserProvider can be called production-ready.
|
||||
Exploratory Playwright/code sandbox activity is authoring-only and must remain isolated, allowlisted, bounded, cancellable, receipt-backed and free of production side effects. Exploration results may inform typed graph proposals, but cannot create a run, lease, provider operation, evidence result or PASS. This boundary is normative; live sandbox/execution integration remains an explicit release gate.
|
||||
## @} ScenarioExecution.AuthoringPromotionBoundary
|
||||
|
||||
## Production contract refresh — 2026-09-08
|
||||
|
||||
**Frontend boundary (user decision 2026-09-08)**: All agent interaction is external MCP only. Product frontend MUST NOT contain agent chat, prompt/request textarea, assistant editing, typical-operation-to-agent selector, proposal-generation, agent workspace/start or handoff controls/routes. Ordinary manual CRUD/editor, human approval/review, monitoring and read-only evidence/evaluation are permitted. AgentEvaluationCard is read-only, with no prompt/retry-agent/provider controls. Existing agent proposal UI is runtime drift; removal/negative DOM-route-network acceptance remains OPEN in this spec-only change.
|
||||
|
||||
**SCEX-FR-028 — Production baseline-backed evaluation**: Run admission, canonical request hash, RunnerPlan, artifacts, immutable comparisons/evaluations, policy outcomes and result/SSE MUST satisfy production-chain.md, artifact-content.openapi.yaml and result-evidence.schema.json. Provider loop startup/shutdown, protected GET/HEAD, baseline resolution/pinning and cancellation/reconcile are mandatory; screenshots currently exist as reusable adapters but missing deployment/evaluation/content integration is not complete. 038 DecisionPolicy is sole semantic outcome mapper.
|
||||
|
||||
Normative contract: [Production baseline-backed evaluation](contracts/production-chain.md). New requirements are specified, **implemented=false / acceptance OPEN** until executable evidence closes the linked tasks/checklist/traceability rows. Historical local tests and the manual inconclusive ss-prod run do not prove browser/capture/baseline/LLM production readiness. The refresh scope is the audited P0/P1/P2 agentic E2E and baseline gaps; an approved ExecutionPerformanceBaseline is not introduced.
|
||||
|
||||
#endregion ScenarioExecution.Spec
|
||||
|
||||
@@ -99,11 +99,11 @@
|
||||
`ProviderExecutionResult` schemas in `backend/src/services/dashboard_testing/execution/provider_protocol.py`.
|
||||
- [x] T029 [P] [US2] Add provider ownership receipts and atomic evidence commit contract in
|
||||
`backend/src/services/dashboard_testing/execution/provider_evidence.py`.
|
||||
- [x] T030 [P] [US2] Add provider operation lifecycle, cancellation and reconciliation contracts in
|
||||
- [ ] T030 [P] [US2] Add provider operation lifecycle, cancellation and reconciliation contracts in
|
||||
`backend/src/services/dashboard_testing/execution/provider_operations.py`.
|
||||
- [x] T031 [P] [US2] Add provider liveness/readiness/dependency health and redacted telemetry contract in
|
||||
`backend/src/services/dashboard_testing/execution/provider_health.py`.
|
||||
- [x] T032 [US2] Integrate atomic shared `ExecutionCapacityManager` admission, heartbeat, expiry, release
|
||||
- [ ] T032 [US2] Integrate atomic shared `ExecutionCapacityManager` admission, heartbeat, expiry, release
|
||||
and reconciliation with dispatcher claims in `backend/src/services/dashboard_testing/execution/capacity.py`.
|
||||
@INVARIANT: no provider I/O without a capacity lease; retries claim a new lease.
|
||||
Include the `ProviderRuntime` contract (`provider_runtime.py`): one application-owned long-lived
|
||||
@@ -114,7 +114,7 @@
|
||||
SQL evidence, XLSX, assertion, transform, screenshot, report and artifact.
|
||||
Include bounded verification response bytes/rows/cells/canonicalization time: oversized output is
|
||||
`RESULT_TOO_LARGE` + inconclusive and can never be truncated into PASS or a baseline update.
|
||||
- [x] T034 [US2] Implement BrowserProvider resource ownership, safe-checkpoint replay, cancel and
|
||||
- [ ] T034 [US2] Implement BrowserProvider resource ownership, safe-checkpoint replay, cancel and
|
||||
reconciliation adapter in `backend/src/services/dashboard_testing/execution/providers/browser.py`.
|
||||
Required action catalog: open_dashboard, navigate_tab, apply_native_filter, inspect_filter_state,
|
||||
apply_table_filter, extract_table, scroll_to, inspect_columns, click, select_rows, edit_row,
|
||||
@@ -133,22 +133,22 @@
|
||||
`backend/src/services/dashboard_testing/execution/providers/xlsx.py`.
|
||||
- [x] T038 [P] [US2] Implement deterministic ReportProvider and ArtifactProvider durable registration
|
||||
with manifest/digest ownership in `backend/src/services/dashboard_testing/execution/providers/artifacts.py`.
|
||||
- [x] T039 [US2] Implement AgentEvaluationProvider and deterministic DecisionPolicy integration in
|
||||
- [ ] T039 [US2] Implement AgentEvaluationProvider and deterministic DecisionPolicy integration in
|
||||
`backend/src/services/dashboard_testing/execution/providers/agent_evaluation.py`.
|
||||
@INVARIANT: model verdict never directly sets ScenarioResult or consumes HumanCheckpoint.
|
||||
- [x] T040 [US2] Add startup deployment registration and readiness preflight for all provider capabilities
|
||||
- [ ] T040 [US2] Add startup deployment registration and readiness preflight for all provider capabilities
|
||||
in `backend/src/services/dashboard_testing/execution/provider_bootstrap.py`.
|
||||
- [x] T041 [US2] Add common provider contract tests for unavailable/dependency failure/capacity
|
||||
exhaustion/timeout/cancel/duplicate/late response/ownership mismatch/malformed result/cleanup and
|
||||
reconciliation in `backend/tests/services/dashboard_testing/registry/test_provider_contract.py`.
|
||||
- [x] T042 [US2] Add provider-specific contract tests in
|
||||
- [ ] T042 [US2] Add provider-specific contract tests in
|
||||
`backend/tests/services/dashboard_testing/registry/test_provider_*.py` and real deployment health
|
||||
checks for Superset, Browser and Screenshot bindings.
|
||||
- [x] T042c [US2] Add dispatcher policy/binding revalidation tests in
|
||||
`backend/tests/services/dashboard_testing/registry/test_provider_contract.py`: reclassification to PROD
|
||||
or a changed provider/security fingerprint after approval prevents all provider I/O and returns typed
|
||||
`POLICY_CHANGED` or `BINDING_CHANGED`.
|
||||
- [x] T042b [US2] Run BrowserProvider PREPROD canaries and retain evidence in
|
||||
- [ ] T042b [US2] Run BrowserProvider PREPROD canaries and retain evidence in
|
||||
`specs/044-dashboard-scenario-execution/evidence/browser-provider/`: read-only action canary,
|
||||
forced timeout/cleanup canary, safe-checkpoint reconstruction trace and readiness/health payload.
|
||||
`GO` requires all BrowserProvider acceptance vectors to pass and one owned evidence receipt.
|
||||
@@ -219,4 +219,17 @@
|
||||
|
||||
Setup → RunnerPlan; US1 (start+executors) → US2 (dispatch); US3 (human) depends on US2; US4 (lifecycle) depends on US2; US5 (snapshot/API/gate) depends on US3/4. 045 monitor consumes run/step results; 042 provides persisted scenario+revision.
|
||||
|
||||
## Production readiness — 2026-09-08 (SCEX-FR-028)
|
||||
|
||||
Historical [x] rows above retain only their dated local/transport evidence; they do not prove current production readiness. Reopened rows were contradicted by the audited gaps. Removed frontend/agent paths are historical, not implementation prerequisites. New acceptance is **implemented=false / OPEN**.
|
||||
|
||||
Contract: [Production baseline-backed evaluation](contracts/production-chain.md).
|
||||
|
||||
- [ ] T043 [P0/P1/P2] Baseline set/version/catalog/release/commit/IDs/digests affect idempotency; moving catalog after admission cannot change plan/result; legacy unpinned result is ineligible. Implement at the existing 044 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
|
||||
- [ ] T044 [P0/P1/P2] GET/HEAD prove same ACL/status/headers; MIME/digest/length checked before bytes, cross-owner hidden, expired410, corrupt409, traversal/range/oversize rejected. Implement at the existing 044 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
|
||||
- [ ] T045 [P0/P1/P2] Startup/readiness/start-loop and shutdown/drain/cancel/reconcile survive fault injection; unknown effect quarantines capacity; late response cannot win. Implement at the existing 044 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
|
||||
- [ ] T046 [P0/P1/P2] End-to-end real browser→capture→durable artifact→deterministic comparison→optional immutable evaluation→policy→result; all required evidence present before PASS. Implement at the existing 044 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
|
||||
|
||||
Frontend boundary for this package: manual CRUD/editor, human review/approval, monitoring and read-only evidence/evaluation only; all agent interaction is external MCP. No agent chat/prompt/assistant editing/proposal generation/workspace/start/handoff controls. Runtime removal is OPEN, not performed by this spec refresh. Optional approved performance baseline is outside scope.
|
||||
|
||||
#endregion ScenarioExecution.Tasks
|
||||
|
||||
@@ -70,3 +70,17 @@ human-containing automated revisions remain prohibited and PROD mutation remains
|
||||
lifecycle edge tests, scoped Ruff/compile and prototype validation. Real PostgreSQL migration checks,
|
||||
provider contract T028-T042, live Browser/Screenshot composition, scheduler deployment and Axiom index
|
||||
rebuild remain open.
|
||||
|
||||
## Production acceptance traceability — 2026-09-08
|
||||
|
||||
Historical rows above identify prior tests/code only; removed agent UI paths are retired. The following audited gates are **implemented=false / OPEN**, independent of local suite totals.
|
||||
|
||||
| Requirement | Domain contract / DTO | Task | Falsifiable acceptance | State |
|
||||
|---|---|---|---|---|
|
||||
| SCEX-FR-028 | [Production baseline-backed evaluation](contracts/production-chain.md); [data model](data-model.md) | [T043](tasks.md) | Baseline set/version/catalog/release/commit/IDs/digests affect idempotency; moving catalog after admission cannot change plan/result; legacy unpinned result is ineligible. | OPEN |
|
||||
| SCEX-FR-028 | [Production baseline-backed evaluation](contracts/production-chain.md); [data model](data-model.md) | [T044](tasks.md) | GET/HEAD prove same ACL/status/headers; MIME/digest/length checked before bytes, cross-owner hidden, expired410, corrupt409, traversal/range/oversize rejected. | OPEN |
|
||||
| SCEX-FR-028 | [Production baseline-backed evaluation](contracts/production-chain.md); [data model](data-model.md) | [T045](tasks.md) | Startup/readiness/start-loop and shutdown/drain/cancel/reconcile survive fault injection; unknown effect quarantines capacity; late response cannot win. | OPEN |
|
||||
| SCEX-FR-028 | [Production baseline-backed evaluation](contracts/production-chain.md); [data model](data-model.md) | [T046](tasks.md) | End-to-end real browser→capture→durable artifact→deterministic comparison→optional immutable evaluation→policy→result; all required evidence present before PASS. | OPEN |
|
||||
| SCEX-FR-028; external-MCP-only UI | manual editor/review; read-only evidence | [production tasks](tasks.md) | No frontend agent prompt/chat/assistant editing/proposal generation/workspace/start/handoff routes or requests; human approval remains usable. | OPEN |
|
||||
|
||||
Sources: [production gap](../../docs/reports/ss-prod-agentic-e2e-production-gap-2026-09-08.md), [coverage gap](../../docs/reports/ss-prod-agentic-e2e-spec-coverage-2026-09-08.md), [baseline gap](../../docs/reports/ss-prod-agentic-e2e-baseline-gap-2026-09-08.md). Spec schema/static checks prove contract structure only; live canary/runtime closure and optional approved performance baseline are not claimed.
|
||||
|
||||
@@ -40,3 +40,14 @@
|
||||
## Success Criteria
|
||||
|
||||
- [ ] CHK015 SC-001..006 verified
|
||||
|
||||
## Production readiness checklist — 2026-09-08
|
||||
|
||||
RUNMON-FR-014: [Unified result projection](../contracts/evidence-ui.md). Historical [x] marks do not close this new production gate; removed frontend components are not current evidence. All rows below implemented=false / OPEN.
|
||||
|
||||
- [ ] CHK016 All evidence states available/loading/forbidden/expired/corrupt/missing recover without unverified image or passing badge; Blob URLs revoked. Evidence: [T020](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK017 Run comparison checks full baseline/context provenance and exact logical IDs; deterministic value, model verdict and policy status remain distinct. Evidence: [T021](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK018 AgentEvaluationCard is read-only; remove investigate-with-agent/launch/prompt/retry-provider controls; human case open and approval remain ordinary actions. Evidence: [T022](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK019 Negative product UI test: no agent chat/prompt/assistant editing/proposal-generation/typical-operation-to-agent/workspace/start/handoff controls or agent invocation routes/requests; manual CRUD/editor/human review/read-only results remain usable.
|
||||
|
||||
Schema/static success alone is not runtime completion. Optional approved performance baseline is outside scope.
|
||||
|
||||
@@ -32,4 +32,12 @@ Failure/blocked/inconclusive results show a linked Investigation Queue item when
|
||||
|
||||
`AgentEvaluationPanel` is a result subprojection, not a chat surface: evaluation id, logical_step_id, model/prompt version, evidence manifest, typed verdict/confidence/reason codes and DecisionPolicy-derived StepOutcome. The UI reads this only from typed 044 result/SSE schemas.
|
||||
|
||||
## Production record contract — 2026-09-08 (RUNMON-FR-014)
|
||||
|
||||
RunConfiguration sends explicit baseline_set_id/version selectors; server pin is read-only. ScenarioResultView consumes the exact 044 evidence/comparisons/evaluations/step_outcomes DTO and full baseline provenance. Authorized Blob views own URL revoke/cancel lifecycle. RunComparison keys by logical_step_id and exact stored context, not first map entry.
|
||||
|
||||
Normative detail: [Unified result projection](contracts/evidence-ui.md). New fields, CAS transitions and cross-record integrity checks are implemented=false until [production tasks](tasks.md) and [traceability](traceability.md) close with executable evidence. Existing shorter field lists are legacy compatibility projections, not permission to omit the production identity fields.
|
||||
|
||||
Frontend contains no agent interaction state, prompt, proposal-generation or workspace/start controls. Human manual editor/review state and read-only evaluation/result artifacts are separate from external MCP authoring. Approved performance baseline is outside scope.
|
||||
|
||||
#endregion ScenarioRunMonitor.DataModel
|
||||
|
||||
@@ -66,3 +66,14 @@ traceability.md maps Story → model → action → contract → task → test.
|
||||
## Complexity Tracking
|
||||
|
||||
No exception planned. Monitor model is bounded C3-C4; comparison and result view model-first.
|
||||
|
||||
## Production delivery plan — 2026-09-08 (RUNMON-FR-014)
|
||||
|
||||
Status: specified, implemented=false; historical unit/prototype/transport results are not current production acceptance.
|
||||
|
||||
1. Pin [Unified result projection](contracts/evidence-ui.md) and [data model](data-model.md); write negative fixtures before runtime changes.
|
||||
2. Implement existing domain boundaries for: RunConfiguration sends explicit baseline_set_id/version selectors; server pin is read-only. ScenarioResultView consumes the exact 044 evidence/comparisons/evaluations/step_outcomes DTO and full baseline provenance. Authorized Blob views own URL revoke/cancel lifecycle. RunComparison keys by logical_step_id and exact stored context, not first map entry.
|
||||
3. Execute [tasks](tasks.md) T020, T021, T022 and retain reproducible evidence in [traceability](traceability.md), then close [checklist](checklists/requirements.md) individually.
|
||||
4. Run cross-spec canary only after 037 catalog publication, 038 chain and 044 provider/content/policy gates; use 046 versioned cost/load/SLO limits. Disable admission on rollback, retain pins/receipts/holds; no fallback to stale catalog or synthetic PASS.
|
||||
|
||||
Frontend agent prompts, chat, assistant editing, proposal-generation, workspace/start/handoff actions are prohibited. Only external MCP clients interact with agents; frontend provides ordinary manual CRUD/editor, human review/approval and read-only monitoring/evidence. Runtime removal tasks are not closed by this document. Optional approved performance baseline is outside this refresh.
|
||||
|
||||
@@ -1,25 +1,56 @@
|
||||
# Quickstart: Scenario Run Monitor & Results UX (045)
|
||||
|
||||
> **Factual audit 2026-08-20:** pending verification checklist only; launch-contract and investigation
|
||||
> handoff scenarios depend on remediation in 044 and 047.
|
||||
> **Refresh 2026-09-08 (production contract):** unified result projection (RUNMON-FR-014) is
|
||||
> normative in `contracts/evidence-ui.md` — implemented=false / acceptance OPEN. EvidenceViewer
|
||||
> and AgentEvaluationCard consume the typed 044 result/evidence/evaluation DTO and authorized
|
||||
> bytes; the card is read-only. The former agent-driven flow (dashboard → `/agent` workspace →
|
||||
> agent-generated scenario / auto-started investigation) is SUPERSEDED: product UI owns run
|
||||
> configuration, live timeline, human checkpoint disposition, and read-only evidence. Failed
|
||||
> results expose an Investigation Queue entry and a human “Open investigation case” action; they
|
||||
> never launch an agent. Offline contract check:
|
||||
>
|
||||
> ```bash
|
||||
> backend/.venv/bin/python specs/044-dashboard-scenario-execution/prototype/validate_contract_refresh.py \
|
||||
> specs/044-dashboard-scenario-execution/fixtures/production-contract-refresh.json
|
||||
> ```
|
||||
|
||||
## Prereqs
|
||||
- 044 execution API live (scenario-runs, events, human/decision)
|
||||
- Frontend deps installed
|
||||
## Target vs observed
|
||||
|
||||
## Commands
|
||||
| Layer | Target (refresh) | Observed (do not treat as GO) |
|
||||
|---|---|---|
|
||||
| Launch | Persistent run config: env/revision/baseline_set/params; PROD gate; mandatory steps not toggleable | Configuration panel exists; release/baseline_set still travel as `params.launch_config` (T014f/T018) |
|
||||
| Timeline | Typed 044 SSE events only; reconnect by `scenario_run_id` | Timeline/checkpoint/result components exist |
|
||||
| Human checkpoint | WAITING FOR HUMAN with confirm/false_positive/inconclusive mapped to persisted outcomes | Panel + vitest pins exist (DISP-001 labels 2026-09-07) |
|
||||
| Result | EvidenceViewer + read-only AgentEvaluationCard + full BaselineSelectionPin | `ScenarioResultView` still renders counts/failures/generic provenance; typed evidence/evaluation DTO is missing |
|
||||
| Investigation | Queue item + ordinary case open; no agent start/handoff | Queue exposure depends on 044/047; live E2E of this handoff is OPEN |
|
||||
|
||||
Launch-contract and investigation-handoff scenarios still depend on remaining 044/047 runtime work (Slice G artifact bytes, live providers). Local vitest is not production GO.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- 044 execution API live (scenario-runs, events, human/decision) for any live walkthrough.
|
||||
- Frontend deps installed.
|
||||
|
||||
## Available local verification
|
||||
|
||||
```bash
|
||||
cd frontend && npm run test -- RunMonitor
|
||||
cd frontend
|
||||
npm run test -- --run src/lib/models/__tests__/RunMonitorModel.test.ts
|
||||
npm run test -- --run src/lib/models/__tests__/RunCenterModel.test.ts
|
||||
npm run test -- --run src/lib/components/scenario-run
|
||||
npm run lint
|
||||
```
|
||||
|
||||
## Exit Gates
|
||||
- [ ] Persistent run configuration panel opens with env/revision/baseline/params/toggles; inline PROD gate appears
|
||||
- [ ] Live timeline renders from events only; step inspector shows inputs/outputs/evidence
|
||||
- [ ] Reconnect recovers run by scenario_run_id
|
||||
- [ ] WAITING FOR HUMAN panel resolves and resumes
|
||||
- [ ] Final result shows counts + failures + full provenance
|
||||
- [ ] Failed/blocked/inconclusive result exposes Investigation Queue entry without auto-starting agent work
|
||||
- [ ] Comparison surfaces per-step deltas + revision-diff warning
|
||||
- [ ] Scenario runs labeled distinctly from 037/040
|
||||
Prototype `@UX_STATE` coverage via `prototype/index.html` remains an open T017 check.
|
||||
|
||||
## Exit gates (acceptance OPEN)
|
||||
|
||||
- Persistent run configuration panel opens with env/revision/baseline/params/toggles; inline PROD gate appears; mandatory graph steps cannot be toggled off.
|
||||
- Live timeline renders from events only; step inspector shows inputs/outputs/evidence refs.
|
||||
- Reconnect recovers the run by `scenario_run_id`.
|
||||
- WAITING FOR HUMAN panel resolves and resumes; wording names the persisted outcome (`confirm`→passed).
|
||||
- Final result shows counts + failures + full provenance **and** typed evidence/evaluation/policy fields.
|
||||
- Failed/blocked/inconclusive result exposes an Investigation Queue entry without auto-starting agent work.
|
||||
- Comparison surfaces per-step deltas + revision-diff warning; baseline pin identity is preserved.
|
||||
- Scenario runs are labeled distinctly from 037/040.
|
||||
- Negative UI acceptance: no agent chat/prompt/workspace/start/handoff control, route, or network request (OPEN).
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
@RATIONALE After creating a scenario the user needs to run it and understand results; 042/044 provide the backend run/result model, 045 renders it live with recovery, evidence, human actions, history, and comparison (reusing 040 LoadRunComparison UX ideas). Without this, execution is a headless API.
|
||||
@REJECTED A tiny inline panel — rejected because live monitoring, evidence review, human checkpoint, and comparison each need dedicated surfaces; the run is a primary workflow, not a widget.
|
||||
@REJECTED Naming this "Verification history" — rejected because 037 VerificationRun is release-pipeline verification; Scenario runs must be labeled distinctly.
|
||||
@RATIONALE After the 050 MCP drift, RUNMON-FR-011 "Investigate with agent" transitions into an external MCP client session over the same InvestigationCase; the monitor keeps rendering typed events only.
|
||||
@RATIONALE RUNMON-FR-011 opens a normal human investigation case; external agents pull MCP independently. No product UI agent transition is permitted.
|
||||
|
||||
## Navigation (DSA Indexer keywords)
|
||||
@SEMANTICS: spec, requirements, feature, ux, scenario, run, monitor, result, history, compare
|
||||
@@ -122,7 +122,7 @@
|
||||
- **RUNMON-FR-008**: All UI MUST follow Svelte 5 runes/model-first conventions and be keyboard-accessible.
|
||||
- **RUNMON-FR-009**: A Global Run Operations Center MUST list all runs (active/queued/waiting-human/failed/recent) with filters and a "Waiting for me" view for pending human checkpoints.
|
||||
- **RUNMON-FR-010**: Run Configuration MUST match the 044 start contract (environment, revision, release, baseline_set, parameters, execution_toggles for optional evidence only); mandatory graph steps MUST NOT be toggleable off.
|
||||
- **RUNMON-FR-011**: Failed, blocked and inconclusive results MUST expose their Investigation Queue item and an explicit "Investigate with agent" transition into the persistent 047 case workspace. Opening it never mutates run truth or auto-executes tools.
|
||||
- **RUNMON-FR-011**: Failed, blocked and inconclusive results MUST expose their queue item and an ordinary “Open investigation case” human review action. It never launches/hands off to an agent or mutates run truth. Agent interaction occurs only in external MCP clients.
|
||||
- **RUNMON-FR-012**: Live Monitor MUST render only typed 044 ScenarioRunEvent payloads (including approval, evidence, agent-evaluation, checkpoint and terminal events), never parse assistant prose. Result and step inspection MUST render StepOutcome, EvidenceReference and AgentEvaluationSummary separately.
|
||||
- **RUNMON-FR-013**: An AgentEvaluation display MUST show declared prompt/model version, evidence manifest, verdict, confidence, reason codes and the DecisionPolicy-derived StepOutcome. It must never present the raw model verdict as ScenarioResult.
|
||||
- **RUNMON-FR-012**: Run configuration, gate decisions, conflict recovery and investigation entry MUST use persistent pages/panels or inline cards; modal/dialog interaction MUST NOT be required.
|
||||
@@ -164,9 +164,9 @@ Monitor model, timeline, checkpoint/result/history/compare components and scenar
|
||||
|
||||
## Drift Amendment — MCP Interface (2026-08-24)
|
||||
|
||||
- The monitor remains a pure web surface over typed events. Its "Investigate with agent" button targets a HandoffSurface (connection hint + case context) instead of an in-product chat route.
|
||||
- The monitor is a typed read-only result surface plus human case/approval actions. Investigate-with-agent, HandoffSurface, prompt and retry-agent/provider controls are prohibited.
|
||||
|
||||
**Status (2026-09-02): done** — реализовано в рамках 050: инструменты и гейты (`specs/050-mcp-interface/tasks.md` T012–T028 [x]), handoff-поверхность (050 T030–T033), демонтаж чата и сервиса `agent/` (050 T040–T041, чекпоинты `specs/WORKSTATE-043-047.md`).
|
||||
**Historical MCP transport/decommission status (2026-09-02): reported done; not production-readiness evidence** — реализовано в рамках 050: инструменты и гейты (`specs/050-mcp-interface/tasks.md` T012–T028 [x]), handoff-поверхность (050 T030–T033), демонтаж чата и сервиса `agent/` (050 T040–T041, чекпоинты `specs/WORKSTATE-043-047.md`).
|
||||
|
||||
## Field-run Amendment — disposition label alignment (2026-09-07)
|
||||
|
||||
@@ -185,4 +185,12 @@ RU «Подтвердить соответствие» / «Проблема не
|
||||
обновление vitest-пинов (`WaitingForMeView.test.ts` и связанные). Эквивалентность UI↔MCP сохраняется:
|
||||
обе поверхности рендерят одни и те же evidence/decision_version и одни и те же три исхода.
|
||||
|
||||
## Production contract refresh — 2026-09-08
|
||||
|
||||
**Frontend boundary (user decision 2026-09-08)**: All agent interaction is external MCP only. Product frontend MUST NOT contain agent chat, prompt/request textarea, assistant editing, typical-operation-to-agent selector, proposal-generation, agent workspace/start or handoff controls/routes. Ordinary manual CRUD/editor, human approval/review, monitoring and read-only evidence/evaluation are permitted. AgentEvaluationCard is read-only, with no prompt/retry-agent/provider controls. Existing agent proposal UI is runtime drift; removal/negative DOM-route-network acceptance remains OPEN in this spec-only change.
|
||||
|
||||
**RUNMON-FR-014 — Unified result projection**: EvidenceViewer and AgentEvaluationCard MUST consume the exact typed 044 result/evidence/evaluation projection, authorized bytes and full BaselineSelectionPin. UI MUST distinguish deterministic comparison, model evidence and policy outcome; state recovery, expired/corrupt/forbidden evidence and baseline-aware run comparison obey evidence-ui.md. Advisory findings cannot become a passing authoritative badge.
|
||||
|
||||
Normative contract: [Unified result projection](contracts/evidence-ui.md). New requirements are specified, **implemented=false / acceptance OPEN** until executable evidence closes the linked tasks/checklist/traceability rows. Historical local tests and the manual inconclusive ss-prod run do not prove browser/capture/baseline/LLM production readiness. The refresh scope is the audited P0/P1/P2 agentic E2E and baseline gaps; an approved ExecutionPerformanceBaseline is not introduced.
|
||||
|
||||
#endregion ScenarioRunMonitor.Spec
|
||||
|
||||
@@ -79,4 +79,16 @@
|
||||
|
||||
Setup → US1; US2 depends on 044 SSE; US3 depends on US2; US4 depends on US2; US5 depends on US4.
|
||||
|
||||
## Production readiness — 2026-09-08 (RUNMON-FR-014)
|
||||
|
||||
Historical [x] rows above retain only their dated local/transport evidence; they do not prove current production readiness. Reopened rows were contradicted by the audited gaps. Removed frontend/agent paths are historical, not implementation prerequisites. New acceptance is **implemented=false / OPEN**.
|
||||
|
||||
Contract: [Unified result projection](contracts/evidence-ui.md).
|
||||
|
||||
- [ ] T020 [P0/P1/P2] All evidence states available/loading/forbidden/expired/corrupt/missing recover without unverified image or passing badge; Blob URLs revoked. Implement at the existing 045 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
|
||||
- [ ] T021 [P0/P1/P2] Run comparison checks full baseline/context provenance and exact logical IDs; deterministic value, model verdict and policy status remain distinct. Implement at the existing 045 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
|
||||
- [ ] T022 [P0/P1/P2] AgentEvaluationCard is read-only; remove investigate-with-agent/launch/prompt/retry-provider controls; human case open and approval remain ordinary actions. Implement at the existing 045 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
|
||||
|
||||
Frontend boundary for this package: manual CRUD/editor, human review/approval, monitoring and read-only evidence/evaluation only; all agent interaction is external MCP. No agent chat/prompt/assistant editing/proposal generation/workspace/start/handoff controls. Runtime removal is OPEN, not performed by this spec refresh. Optional approved performance baseline is outside scope.
|
||||
|
||||
#endregion ScenarioRunMonitor.Tasks
|
||||
|
||||
@@ -13,3 +13,16 @@
|
||||
| Naming | RUNMON-FR-007/008 | — | — | — | T015 | run.ux.test |
|
||||
|
||||
N/A: Registry (042), Editor (043), Execution backend (044), Automation (046), Analytics (047).
|
||||
|
||||
## Production acceptance traceability — 2026-09-08
|
||||
|
||||
Historical rows above identify prior tests/code only; removed agent UI paths are retired. The following audited gates are **implemented=false / OPEN**, independent of local suite totals.
|
||||
|
||||
| Requirement | Domain contract / DTO | Task | Falsifiable acceptance | State |
|
||||
|---|---|---|---|---|
|
||||
| RUNMON-FR-014 | [Unified result projection](contracts/evidence-ui.md); [data model](data-model.md) | [T020](tasks.md) | All evidence states available/loading/forbidden/expired/corrupt/missing recover without unverified image or passing badge; Blob URLs revoked. | OPEN |
|
||||
| RUNMON-FR-014 | [Unified result projection](contracts/evidence-ui.md); [data model](data-model.md) | [T021](tasks.md) | Run comparison checks full baseline/context provenance and exact logical IDs; deterministic value, model verdict and policy status remain distinct. | OPEN |
|
||||
| RUNMON-FR-014 | [Unified result projection](contracts/evidence-ui.md); [data model](data-model.md) | [T022](tasks.md) | AgentEvaluationCard is read-only; remove investigate-with-agent/launch/prompt/retry-provider controls; human case open and approval remain ordinary actions. | OPEN |
|
||||
| RUNMON-FR-014; external-MCP-only UI | manual editor/review; read-only evidence | [production tasks](tasks.md) | No frontend agent prompt/chat/assistant editing/proposal generation/workspace/start/handoff routes or requests; human approval remains usable. | OPEN |
|
||||
|
||||
Sources: [production gap](../../docs/reports/ss-prod-agentic-e2e-production-gap-2026-09-08.md), [coverage gap](../../docs/reports/ss-prod-agentic-e2e-spec-coverage-2026-09-08.md), [baseline gap](../../docs/reports/ss-prod-agentic-e2e-baseline-gap-2026-09-08.md). Spec schema/static checks prove contract structure only; live canary/runtime closure and optional approved performance baseline are not claimed.
|
||||
|
||||
@@ -32,3 +32,14 @@
|
||||
## Success Criteria
|
||||
|
||||
- [ ] CHK011 SC-001..006 verified
|
||||
|
||||
## Production readiness checklist — 2026-09-08
|
||||
|
||||
SCAUTO-FR-019: [Parity, retention and rollout](../contracts/production-operations.md). Historical [x] marks do not close this new production gate; removed frontend components are not current evidence. All rows below implemented=false / OPEN.
|
||||
|
||||
- [ ] CHK012 Disabled and enabled HumanStep schedules reject equally via REST/MCP; all six automation reads reject anonymous/cross-owner principals. Evidence: [T019](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK013 Crash/replay/concurrent due events create one pinned run/gate/outbox and preserve approval before dispatcher CAS. Evidence: [T020](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK014 Retention honors baseline/case holds; deletion receipt proves bytes removed; 5/15/50-tab cost/load/timeout/cancel canaries meet versioned SLO thresholds. Evidence: [T021](../tasks.md), [traceability](../traceability.md).
|
||||
- [ ] CHK015 Negative product UI test: no agent chat/prompt/assistant editing/proposal-generation/typical-operation-to-agent/workspace/start/handoff controls or agent invocation routes/requests; manual CRUD/editor/human review/read-only results remain usable.
|
||||
|
||||
Schema/static success alone is not runtime completion. Optional approved performance baseline is outside scope.
|
||||
|
||||
@@ -36,4 +36,12 @@ Layered retention independent of the analytics minimum history window: run metad
|
||||
|
||||
A triggered run pins scenario_id + revision_id + immutable Verification Program/content hash + environment_id + target snapshot + execution principal + server-owned trigger source (from 044). Concurrency bucket (for capacity) may be environment+workload class; dedup identity is `canonical_execution_request_hash`, or for an event trigger `(source_type, source_event_id, scenario_id)`. External API triggers require an `Idempotency-Key`; same key/hash returns the same run, while a changed canonical request returns 409. Automation cannot supply or rewrite runtime SQL/DSL/program content.
|
||||
|
||||
## Production record contract — 2026-09-08 (SCAUTO-FR-019)
|
||||
|
||||
Schedule/TriggerRule pins explicit active revision/environment and baseline selector version. Atomic DueDispatchReceipt binds event/dedup key, resolved BaselineSelectionPin, request hash, run/gate/outbox. RetentionHold references baseline/investigation ownership and release criteria; deletion tombstones record proof. OperationalProfile versions cost/pricing/reservation, queue/dispatch/read latency and canary gates.
|
||||
|
||||
Normative detail: [Parity, retention and rollout](contracts/production-operations.md). New fields, CAS transitions and cross-record integrity checks are implemented=false until [production tasks](tasks.md) and [traceability](traceability.md) close with executable evidence. Existing shorter field lists are legacy compatibility projections, not permission to omit the production identity fields.
|
||||
|
||||
Frontend contains no agent interaction state, prompt, proposal-generation or workspace/start controls. Human manual editor/review state and read-only evaluation/result artifacts are separate from external MCP authoring. Approved performance baseline is outside scope.
|
||||
|
||||
#endregion ScenarioAutomation.DataModel
|
||||
|
||||
@@ -65,3 +65,14 @@ traceability.md maps Story → model → operationId → contract → task → t
|
||||
## Complexity Tracking
|
||||
|
||||
No exception planned. Automation is bounded C3-C4; policy/trigger decomposed.
|
||||
|
||||
## Production delivery plan — 2026-09-08 (SCAUTO-FR-019)
|
||||
|
||||
Status: specified, implemented=false; historical unit/prototype/transport results are not current production acceptance.
|
||||
|
||||
1. Pin [Parity, retention and rollout](contracts/production-operations.md) and [data model](data-model.md); write negative fixtures before runtime changes.
|
||||
2. Implement existing domain boundaries for: Schedule/TriggerRule pins explicit active revision/environment and baseline selector version. Atomic DueDispatchReceipt binds event/dedup key, resolved BaselineSelectionPin, request hash, run/gate/outbox. RetentionHold references baseline/investigation ownership and release criteria; deletion tombstones record proof. OperationalProfile versions cost/pricing/reservation, queue/dispatch/read latency and canary gates.
|
||||
3. Execute [tasks](tasks.md) T019, T020, T021 and retain reproducible evidence in [traceability](traceability.md), then close [checklist](checklists/requirements.md) individually.
|
||||
4. Run cross-spec canary only after 037 catalog publication, 038 chain and 044 provider/content/policy gates; use 046 versioned cost/load/SLO limits. Disable admission on rollback, retain pins/receipts/holds; no fallback to stale catalog or synthetic PASS.
|
||||
|
||||
Frontend agent prompts, chat, assistant editing, proposal-generation, workspace/start/handoff actions are prohibited. Only external MCP clients interact with agents; frontend provides ordinary manual CRUD/editor, human review/approval and read-only monitoring/evidence. Runtime removal tasks are not closed by this document. Optional approved performance baseline is outside this refresh.
|
||||
|
||||
@@ -1,30 +1,46 @@
|
||||
# Quickstart: Scenario Automation & Operations (046)
|
||||
|
||||
> **Factual audit 2026-08-20:** pending verification checklist only; scheduler/event/notification
|
||||
> integration is not yet production-complete.
|
||||
> **Refresh 2026-09-08 (production contract):** REST/MCP parity, authenticated operational reads,
|
||||
> and pinned baseline selection on due-event admission (SCAUTO-FR-019) are normative in
|
||||
> `contracts/production-operations.md` — implemented=false / acceptance OPEN. Shared eligibility
|
||||
> (`ScenarioAutomation.Eligibility.Assert`) and authenticated automation GETs (SEC-01) landed in
|
||||
> runtime 2026-09-10 (handoff slice A+B: DEF-02/SEC-01). Production GO still OPEN: live providers,
|
||||
> cost/load/SLO canary, and identical disabled-config validation across REST/MCP are not closed
|
||||
> by local pytest. Agent interaction is external MCP only; product UI is schedule CRUD + human
|
||||
> approvals.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
## Prereqs
|
||||
- 044 run API, 042 registry, DB migrated (scenario_automation tables)
|
||||
- APScheduler infrastructure (037) available
|
||||
- APScheduler infrastructure (037) available for scheduler-path tests
|
||||
|
||||
## Commands
|
||||
|
||||
```bash
|
||||
cd backend && source .venv/bin/activate
|
||||
alembic upgrade head
|
||||
python -m pytest -v tests/services/dashboard_testing/automation/
|
||||
python -m pytest -v tests/api/test_scenario_automation.py
|
||||
python -m pytest -v tests/services/dashboard_testing/registry/test_scenario_automation*.py
|
||||
python -m pytest -v tests/api/test_scenario_automation_api.py
|
||||
python -m ruff check src/services/dashboard_testing/automation/
|
||||
```
|
||||
|
||||
## Exit Gates
|
||||
- [ ] Each trigger type (deploy/release/ETL/schedule/api) starts an eligible run with pinned revision+env;
|
||||
a `manual_run_only=true` revision is rejected before any durable side effect
|
||||
- [ ] Notification events emitted for completed/failed/blocked/stale/flaky; no `human-action-required`
|
||||
event is emitted by automation
|
||||
- [ ] Concurrency/dedup prevents redundant parallel runs; overlap blocked/warned
|
||||
- [ ] Retention prunes preserving provenance/artifacts
|
||||
- [ ] Eligible PROD automated runs create ActionApprovalGate before dispatch; human-containing revisions
|
||||
are rejected before gate/run/queue/notification creation
|
||||
- [ ] Zero writes to 037 baseline catalog
|
||||
- [ ] ruff clean; prototype states covered
|
||||
Offline contract check:
|
||||
|
||||
```bash
|
||||
backend/.venv/bin/python specs/044-dashboard-scenario-execution/prototype/validate_contract_refresh.py \
|
||||
specs/044-dashboard-scenario-execution/fixtures/production-contract-refresh.json
|
||||
```
|
||||
|
||||
## Exit gates (acceptance OPEN)
|
||||
|
||||
- Each trigger type (deploy/release/ETL/schedule/api) starts an eligible run with pinned revision+env; a `manual_run_only=true` revision is rejected before any durable side effect.
|
||||
- REST and MCP return the same eligibility/validation outcome, including disabled configurations.
|
||||
- Notification events emitted for completed/failed/blocked/stale/flaky; no `human-action-required` event is emitted by automation.
|
||||
- Concurrency/dedup prevents redundant parallel runs; overlap blocked/warned.
|
||||
- Retention prunes while preserving provenance/artifacts.
|
||||
- Eligible PROD automated runs create ActionApprovalGate before dispatch; human-containing revisions are rejected before gate/run/queue/notification creation.
|
||||
- Six operational GET reads require an authenticated principal and object ACL.
|
||||
- Zero writes to 037 baseline catalog.
|
||||
- ruff clean; prototype states covered.
|
||||
|
||||
Local pytest success is historical runtime-gate evidence. Scheduler/event/notification integration on a live stand is not production-complete.
|
||||
|
||||
@@ -178,6 +178,14 @@ the automation workflow is not wired end-to-end.
|
||||
|
||||
- Automation management UI stays; MCP decision tools may drive the same CRUD under SCAUTO-FR-013 policy. Signals never auto-start client activity (pull-only).
|
||||
|
||||
**Status (2026-09-02): done** — реализовано в рамках 050: инструменты и гейты (`specs/050-mcp-interface/tasks.md` T012–T028 [x]), handoff-поверхность (050 T030–T033), демонтаж чата и сервиса `agent/` (050 T040–T041, чекпоинты `specs/WORKSTATE-043-047.md`).
|
||||
**Historical MCP transport/decommission status (2026-09-02): reported done; not production-readiness evidence** — реализовано в рамках 050: инструменты и гейты (`specs/050-mcp-interface/tasks.md` T012–T028 [x]), handoff-поверхность (050 T030–T033), демонтаж чата и сервиса `agent/` (050 T040–T041, чекпоинты `specs/WORKSTATE-043-047.md`).
|
||||
|
||||
## Production contract refresh — 2026-09-08
|
||||
|
||||
**Frontend boundary (user decision 2026-09-08)**: All agent interaction is external MCP only. Product frontend MUST NOT contain agent chat, prompt/request textarea, assistant editing, typical-operation-to-agent selector, proposal-generation, agent workspace/start or handoff controls/routes. Ordinary manual CRUD/editor, human approval/review, monitoring and read-only evidence/evaluation are permitted. AgentEvaluationCard is read-only, with no prompt/retry-agent/provider controls. Existing agent proposal UI is runtime drift; removal/negative DOM-route-network acceptance remains OPEN in this spec-only change.
|
||||
|
||||
**SCAUTO-FR-019 — Parity, retention and rollout**: REST/MCP automation validation MUST be identical even for disabled configurations, and all six operational reads require authentication and object ACL. Atomic due-event admission MUST pin baseline selection in idempotency/run/gate identity. Retention holds, deletion proof, cost/load/SLO metrics and staged canary gates obey production-operations.md. Baseline publishing/rebaselining is never an automation side effect.
|
||||
|
||||
Normative contract: [Parity, retention and rollout](contracts/production-operations.md). New requirements are specified, **implemented=false / acceptance OPEN** until executable evidence closes the linked tasks/checklist/traceability rows. Historical local tests and the manual inconclusive ss-prod run do not prove browser/capture/baseline/LLM production readiness. The refresh scope is the audited P0/P1/P2 agentic E2E and baseline gaps; an approved ExecutionPerformanceBaseline is not introduced.
|
||||
|
||||
#endregion ScenarioAutomation.Spec
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user