feat(scenario): complete dashboard testing UX flow

This commit is contained in:
2026-09-15 10:12:17 +03:00
parent 22ffdb5ba5
commit 8522a2ee3c
94 changed files with 11029 additions and 604 deletions

View File

@@ -176,6 +176,7 @@ superset-tools добавляет вокруг операций необходи
## Документация
- [Установка и настройка](INSTALL.md)
- [Подключение MCP-агента (гайд для BI-аналитика)](docs/mcp-client-setup.md)
- [Архитектура системы](docs/architecture.md)
- [Архитектурные решения](docs/adr/README.md)
- [Enterprise Clean Deployment](docs/enterprise-clean.md)

View File

@@ -0,0 +1,60 @@
# #region Migrations.ScenarioRetentionDeletion [C:3] [TYPE Module] [SEMANTICS migration,scenario,automation,retention,deletion]
# @BRIEF Create the durable retention deletion receipt table after publication operations.
# @RATIONALE The 046 T021 deletion pipeline (mark->eligible->delete-bytes->verify-absent->tombstone)
# requires durable receipts so a crash between byte removal and bookkeeping never loses
# the audit trail; idempotent retry is keyed by (target_type, target_id).
# @RATIONALE Idempotent inspector guards (as in 0023) are mandatory: 0001_baseline runs
# Base.metadata.create_all, which already creates every model table including this one on
# fresh databases; the guarded DDL only materializes on databases migrated before 0024.
# @REJECTED Unguarded create_table was rejected — it collides with the 0001 create_all bootstrap
# and breaks every fresh SQLite test schema.
"""scenario retention deletions"""
from alembic import op
import sqlalchemy as sa
revision = "0024_retention_deletions"
down_revision = "0023_publication_operations"
branch_labels = None
depends_on = None
def upgrade() -> None:
inspector = sa.inspect(op.get_bind())
tables = set(inspector.get_table_names())
if "scenario_retention_deletions" not in tables:
op.create_table(
"scenario_retention_deletions",
sa.Column("id", sa.String(36), primary_key=True),
sa.Column("target_type", sa.String(32), nullable=False),
sa.Column("target_id", sa.String(128), nullable=False),
sa.Column("content_ref", sa.String(255), nullable=True),
sa.Column("scenario_id", sa.String(36), nullable=True),
sa.Column("run_id", sa.String(36), nullable=True),
sa.Column("baseline_pin", sa.JSON(), nullable=True),
sa.Column("retention_class", sa.String(32), nullable=False),
sa.Column("state", sa.String(32), nullable=False),
sa.Column("holds_snapshot", sa.JSON(), nullable=False),
sa.Column("attempts", sa.Integer(), nullable=False),
sa.Column("error_code", sa.String(64), nullable=True),
sa.Column("verified_absent_at", sa.DateTime(), nullable=True),
sa.Column("tombstoned_at", sa.DateTime(), nullable=True),
sa.Column("created_at", sa.DateTime(), nullable=False),
sa.Column("updated_at", sa.DateTime(), nullable=False),
)
op.create_index("ix_scenario_retention_deletions_scenario_id", "scenario_retention_deletions", ["scenario_id"])
op.create_index("ix_scenario_retention_deletions_run_id", "scenario_retention_deletions", ["run_id"])
op.create_index("ix_scenario_retention_deletions_state", "scenario_retention_deletions", ["state"])
op.create_index(
"uq_scenario_retention_deletions_target",
"scenario_retention_deletions",
["target_type", "target_id"],
unique=True,
)
def downgrade() -> None:
inspector = sa.inspect(op.get_bind())
tables = set(inspector.get_table_names())
if "scenario_retention_deletions" in tables:
op.drop_table("scenario_retention_deletions")
# #endregion Migrations.ScenarioRetentionDeletion

View File

@@ -8,6 +8,8 @@
# @RELATION DEPENDS_ON -> [Api.DashboardTesting.StructureSnapshot]
# @RELATION DEPENDS_ON -> [Api.DashboardTesting.VerificationRuns]
# @RELATION DEPENDS_ON -> [Api.ScenarioArtifactContent]
# @RELATION DEPENDS_ON -> [Api.ScenarioLiveBindings]
# @RELATION DEPENDS_ON -> [Api.CatalogPublications]
# @INVARIANT The exported `router` includes all submodule routes under /api/dashboard-testing.
from __future__ import annotations

View File

@@ -13,9 +13,23 @@
# @INVARIANT 046 routes pass a server-owned automation origin to 044. A revision with a human
# step is rejected as manual-run-only before a ScenarioRun, gate, or notification exists.
# @INVARIANT All operational reads (schedules/trigger-rules/policies/notifications/metrics/retention)
# require an authenticated principal (SEC-01); parity with MCP reads whose permission is
# human-only. Schedule/trigger-rule persistence additionally runs the shared revision-bound
# require an authenticated principal holding scenario:automation READ plus per-object
# scenario-ownership ACL (SEC-01, DG-2 2026-09-12): anonymous -> 401 via the OAuth2
# bearer dependency, lacking READ -> 403, foreign rows are filtered out of every
# collection projection so existence never leaks (foreign object -> 404 semantics on
# a collection surface); admin roles bypass the ACL filter and see all rows.
# Schedule/trigger-rule persistence additionally runs the shared revision-bound
# eligibility guard before any row is written (DEF-02) — REST and MCP never diverge.
# @RATIONALE DG-2 (046 T019): reads admit an authenticated principal of ANY type holding
# scenario:automation READ with per-object ACL; mutations stay human-only. Policy
# objects carry no scenario ownership, so a policy is hidden from a non-admin only
# when it is referenced exclusively by foreign scenarios — unreferenced or own-
# referenced policies stay visible to keep the management surface functional.
# @REJECTED Authenticated-only reads without a READ grant were rejected (DG-2 supersedes the
# pre-2026-09-12 note) — operational config is not public-to-any-user data.
# @REJECTED Hiding unreferenced policies from non-admin READ holders was rejected — policies are
# shared named configuration; breaking the policy picker for legitimate READ holders
# buys no secrecy (a policy carries no scenario payload).
# @REJECTED A parallel scheduler was rejected — this surface persists configuration; APScheduler
# job registration stays server-owned via the 037 framework.
# @REJECTED str(environment_id).startswith("prod") PROD classification was rejected — a client-name
@@ -28,7 +42,8 @@ from fastapi import APIRouter, Depends, Header, HTTPException, status
from pydantic import BaseModel, Field
from src.dependencies import get_config_manager, get_current_user, get_db, get_scheduler_service, has_permission
from src.models.scenario_automation import AutomationPolicy, ScenarioNotificationEvent, ScenarioSchedule, ScenarioTriggerRule
from src.models.scenario_automation import AutomationPolicy, ScenarioNotificationEvent, ScenarioRetentionDeletion, ScenarioSchedule, ScenarioTriggerRule
from src.models.scenario_registry import ScenarioRegistryEntry
from src.services.dashboard_testing.automation.metrics import automation_metrics
from src.services.dashboard_testing.automation.eligibility import assert_automation_eligible
from src.services.dashboard_testing.automation.retention import tier_limits
@@ -41,6 +56,7 @@ from src.services.dashboard_testing.execution.runner import start_run
router = APIRouter(prefix="/api/scenario-automation", tags=["scenario-automation"])
_DB = Depends(get_db)
_USER = Depends(get_current_user)
_READ = Depends(has_permission("scenario:automation", "READ"))
_MANAGE = Depends(has_permission("scenario:automation", "MANAGE"))
_TRIGGER = Depends(has_permission("scenario:automation", "TRIGGER"))
_CONFIG_MANAGER = Depends(get_config_manager)
@@ -49,6 +65,62 @@ TRIGGER_TYPES = {"deploy_to_preprod", "release_created", "etl_completed", "api"}
MISSED_POLICIES = {"skip", "run_latest", "queue_all"}
# #region Api.ScenarioAutomation.ReadAcl [C:3] [TYPE Block] [SEMANTICS scenario,automation,api,rbac,acl]
# @ingroup Api
# @BRIEF Per-object read ACL for automation collections: admin sees all; any other READ holder
# sees only rows whose scenario registry entry they own (owner_id/owner_username).
# @POST _visible_scenario_ids returns None for admin (no filter) or a set of owned scenario ids;
# rows pointing at scenarios without a registry entry are hidden from non-admins (their
# ownership cannot be proven, so their existence must not leak).
def _is_admin(user) -> bool:
return any(getattr(role, "is_admin", False) for role in (getattr(user, "roles", None) or []))
def _visible_scenario_ids(db, user) -> set[str] | None:
if _is_admin(user):
return None
user_id = str(getattr(user, "id", "") or "")
username = str(getattr(user, "username", "") or "")
rows = (
db.query(ScenarioRegistryEntry.scenario_id)
.filter(
(ScenarioRegistryEntry.owner_id == user_id)
| (ScenarioRegistryEntry.owner_username == username)
)
.all()
)
return {row[0] for row in rows}
def _visible_policy_ids(db, visible: set[str] | None) -> set[str] | None:
"""None for admin; otherwise policies NOT referenced exclusively by foreign scenarios."""
if visible is not None:
referenced = {
row[0]
for row in db.query(ScenarioSchedule.policy_id).filter(ScenarioSchedule.policy_id.isnot(None)).all()
} | {
row[0]
for row in db.query(ScenarioTriggerRule.policy_id).filter(ScenarioTriggerRule.policy_id.isnot(None)).all()
}
own_referenced = {
row[0]
for row in db.query(ScenarioSchedule.policy_id)
.filter(ScenarioSchedule.policy_id.isnot(None), ScenarioSchedule.scenario_id.in_(visible))
.all()
} | {
row[0]
for row in db.query(ScenarioTriggerRule.policy_id)
.filter(ScenarioTriggerRule.policy_id.isnot(None), ScenarioTriggerRule.scenario_id.in_(visible))
.all()
}
foreign_only = referenced - own_referenced
return {
row[0] for row in db.query(AutomationPolicy.id).all()
} - foreign_only
return None
# #endregion Api.ScenarioAutomation.ReadAcl
# #region Api.ScenarioAutomation.ScheduleRequest [C:1] [TYPE Class] [SEMANTICS scenario,automation,api,schedule]
# @ingroup Api
class ScheduleRequest(BaseModel):
@@ -130,8 +202,12 @@ def _validate_missed_policy(policy: str) -> None:
# @INVARIANT Schedule CRUD may register work but never claims or executes a queued ScenarioRun;
# the separate 044 queued dispatcher remains the sole initial execution authority.
@router.get("/schedules")
def list_schedules(db=_DB, _user=_USER):
return db.query(ScenarioSchedule).order_by(ScenarioSchedule.created_at.desc()).all()
def list_schedules(db=_DB, current_user=_READ):
query = db.query(ScenarioSchedule).order_by(ScenarioSchedule.created_at.desc())
visible = _visible_scenario_ids(db, current_user)
if visible is not None:
query = query.filter(ScenarioSchedule.scenario_id.in_(visible))
return query.all()
@router.post("/schedules", status_code=status.HTTP_201_CREATED)
@@ -213,8 +289,12 @@ def delete_schedule(schedule_id: str, db=_DB, _perm=_MANAGE):
# #region Api.ScenarioAutomation.TriggerRules [C:3] [TYPE Block] [SEMANTICS scenario,automation,api,trigger,crud]
# @ingroup Api
@router.get("/trigger-rules")
def list_trigger_rules(db=_DB, _user=_USER):
return db.query(ScenarioTriggerRule).order_by(ScenarioTriggerRule.created_at.desc()).all()
def list_trigger_rules(db=_DB, current_user=_READ):
query = db.query(ScenarioTriggerRule).order_by(ScenarioTriggerRule.created_at.desc())
visible = _visible_scenario_ids(db, current_user)
if visible is not None:
query = query.filter(ScenarioTriggerRule.scenario_id.in_(visible))
return query.all()
@router.post("/trigger-rules", status_code=status.HTTP_201_CREATED)
@@ -262,8 +342,12 @@ def delete_trigger_rule(rule_id: str, db=_DB, _perm=_MANAGE):
# #region Api.ScenarioAutomation.Policies [C:3] [TYPE Block] [SEMANTICS scenario,automation,api,policy,crud]
# @ingroup Api
@router.get("/policies")
def list_policies(db=_DB, _user=_USER):
return db.query(AutomationPolicy).order_by(AutomationPolicy.created_at.desc()).all()
def list_policies(db=_DB, current_user=_READ):
query = db.query(AutomationPolicy).order_by(AutomationPolicy.created_at.desc())
visible = _visible_policy_ids(db, _visible_scenario_ids(db, current_user))
if visible is not None:
query = query.filter(AutomationPolicy.id.in_(visible))
return query.all()
@router.post("/policies", status_code=status.HTTP_201_CREATED)
@@ -301,8 +385,12 @@ def delete_policy(policy_id: str, db=_DB, _perm=_MANAGE):
# #region Api.ScenarioAutomation.Notifications [C:2] [TYPE Function] [SEMANTICS scenario,automation,api,notifications]
# @ingroup Api
@router.get("/notifications")
def list_notifications(limit: int = 100, db=_DB, _user=_USER):
return db.query(ScenarioNotificationEvent).order_by(ScenarioNotificationEvent.created_at.desc()).limit(limit).all()
def list_notifications(limit: int = 100, db=_DB, current_user=_READ):
query = db.query(ScenarioNotificationEvent).order_by(ScenarioNotificationEvent.created_at.desc())
visible = _visible_scenario_ids(db, current_user)
if visible is not None:
query = query.filter(ScenarioNotificationEvent.scenario_id.in_(visible))
return query.limit(limit).all()
# #endregion Api.ScenarioAutomation.Notifications
@@ -348,13 +436,23 @@ def api_dispatch_event(
# @ingroup Api
# @BRIEF Operational metrics over persisted schedules, trigger rules, runs and notifications.
@router.get("/metrics")
def metrics(db=_DB, _user=_USER):
schedules = db.query(ScenarioSchedule).all()
rules = db.query(ScenarioTriggerRule).all()
notifications = db.query(ScenarioNotificationEvent).all()
def metrics(db=_DB, current_user=_READ):
visible = _visible_scenario_ids(db, current_user)
schedules_q = db.query(ScenarioSchedule)
rules_q = db.query(ScenarioTriggerRule)
notifications_q = db.query(ScenarioNotificationEvent)
from src.models.scenario_run import ScenarioRun
runs = db.query(ScenarioRun).all()
runs_q = db.query(ScenarioRun)
if visible is not None:
schedules_q = schedules_q.filter(ScenarioSchedule.scenario_id.in_(visible))
rules_q = rules_q.filter(ScenarioTriggerRule.scenario_id.in_(visible))
notifications_q = notifications_q.filter(ScenarioNotificationEvent.scenario_id.in_(visible))
runs_q = runs_q.filter(ScenarioRun.scenario_id.in_(visible))
schedules = schedules_q.all()
rules = rules_q.all()
notifications = notifications_q.all()
runs = runs_q.all()
return automation_metrics(
schedules=[{"enabled": s.enabled} for s in schedules],
trigger_rules=[{"enabled": r.enabled} for r in rules],
@@ -370,12 +468,26 @@ def metrics(db=_DB, _user=_USER):
# #endregion Api.ScenarioAutomation.Metrics
# #region Api.ScenarioAutomation.RetentionDefaults [C:2] [TYPE Function] [SEMANTICS scenario,automation,api,retention,tiers]
# #region Api.ScenarioAutomation.RetentionDefaults [C:3] [TYPE Function] [SEMANTICS scenario,automation,api,retention,tiers,deletion,receipts]
# @ingroup Api
# @BRIEF Expose the canonical layered retention tier horizons for the management UI.
# @BRIEF Expose the canonical layered retention tier horizons plus per-object deletion receipts.
# @POST Returns tiers and the ACL-filtered deletion receipts; totals never leak foreign rows.
# @INVARIANT Receipts with a null scenario_id are admin-only (fail-closed ACL); a non-admin never
# sees a foreign receipt row nor its count.
@router.get("/retention")
def retention_defaults(_user=_USER):
return {"tiers": tier_limits()}
def retention_defaults(db=_DB, current_user=_READ):
from src.services.dashboard_testing.automation.deletions import retention_receipt_to_dict
query = db.query(ScenarioRetentionDeletion).order_by(ScenarioRetentionDeletion.created_at.desc())
visible = _visible_scenario_ids(db, current_user)
if visible is not None:
query = query.filter(ScenarioRetentionDeletion.scenario_id.in_(visible))
receipts = query.limit(200).all()
return {
"tiers": tier_limits(),
"deletions": [retention_receipt_to_dict(receipt) for receipt in receipts],
"deletions_total": len(receipts),
}
# #endregion Api.ScenarioAutomation.RetentionDefaults

View File

@@ -9,8 +9,18 @@
# @INVARIANT This module is read-only: no route mutates a run, checkpoint or scenario row.
# @INVARIANT waiting_for_me returns only runs with a pending HumanCheckpoint — the current user
# is the sole analyst decision-maker in this simplified model (no eligible_actor_ids yet).
# @INVARIANT pending_approval=true returns only runs in status=pending_approval (PROD gate awaiting
# approve, incl. scheduled runs); waiting_for_me semantics are unchanged.
# @RATIONALE The pending_approval projection is exposed at the listing visibility (scenario:RUN),
# not behind RUN_PROD: it is a read-only aggregate that grants no action; the approve
# decision itself stays behind scenario:RUN_PROD (mcp list_pending_approvals /
# decide_approval, service_allowed=False — rbac_server.py), so a RUN-holder seeing the
# count cannot mutate anything while the badge must not hide a pending gate.
# @REJECTED Putting the list on the 044 worker-owned route module was rejected — 045 owns the
# cross-scenario center surface; separate module keeps the two workstreams merge-safe.
# @REJECTED Reusing status=pending_approval as the badge projection was rejected — a dedicated
# query param keeps the store contract explicit and allows future permission tightening
# without overloading the generic status filter.
from __future__ import annotations
from typing import Any
@@ -29,12 +39,15 @@ _DB = Depends(get_db)
_RUN_READ_PERMISSION = Depends(has_permission("scenario", "RUN"))
# #region Api.ScenarioRun.Center.List [C:4] [TYPE Function] [SEMANTICS scenario,run,center,list,waiting]
# #region Api.ScenarioRun.Center.List [C:4] [TYPE Function] [SEMANTICS scenario,run,center,list,waiting,pending]
# @ingroup Api
# @BRIEF Return paginated run rows (newest first) with scenario identity and pending checkpoint.
# @BRIEF Return paginated run rows (newest first) with scenario identity and pending checkpoint;
# waiting_for_me and pending_approval are server-computed projections for the sidebar badge.
# @PRE Caller has scenario:RUN.
# @POST Returns {items, total}; each item carries scenario_key/name and pending_checkpoint for
# inline human actions; filters narrow the result set server-side.
# inline human actions; filters narrow the result set server-side. pending_approval=true
# narrows the same listing to runs in status=pending_approval (bounded badge poll uses
# page_size=1 and consumes only total).
def _run_row(db, run: ScenarioRun) -> dict[str, Any]:
entry = db.query(ScenarioRegistryEntry).filter(ScenarioRegistryEntry.scenario_id == run.scenario_id).first()
checkpoint = (
@@ -77,6 +90,7 @@ def api_run_center_list(
trigger_source: str | None = Query(default=None, alias="trigger_source"),
owner: str | None = Query(default=None),
waiting_for_me: bool = Query(default=False),
pending_approval: bool = Query(default=False),
page: int = Query(default=1, ge=1),
page_size: int = Query(default=50, ge=1, le=200),
db=_DB,
@@ -103,6 +117,10 @@ def api_run_center_list(
.all()
}
query = query.filter(ScenarioRun.id.in_(waiting_ids)) if waiting_ids else query.filter(False)
if pending_approval:
# Badge projection (plan UX-7): PROD gates awaiting approve; same listing visibility,
# read-only aggregate — approve itself requires scenario:RUN_PROD elsewhere.
query = query.filter(ScenarioRun.status == "pending_approval")
total = query.count()
rows = (
query.order_by(ScenarioRun.created_at.desc(), ScenarioRun.id.desc())

View File

@@ -461,6 +461,17 @@ class SchedulerService:
id="agent_lifecycle_retention",
replace_existing=True,
)
# T021 retention sweep: daily deletion pipeline tick, idempotent retries of
# deletion_pending receipts + verify-absent of tombstoned rows.
from ..services.dashboard_testing.automation.deletions import (
execute_scheduled_retention,
)
self.scheduler.add_job(
execute_scheduled_retention,
CronTrigger.from_crontab("45 3 * * *", timezone="UTC"),
id="scenario_retention_sweep",
replace_existing=True,
)
self.scheduler.add_job(
execute_scheduled_scenario_cancel_finalizer,
IntervalTrigger(seconds=5),

View File

@@ -77,7 +77,7 @@ def _scenario_start_permission(arguments: dict[str, Any]) -> tuple[str, str]:
# tool listed with deprecated=True for one minor cycle; additive changes bump MINOR.
# The discipline is pinned executable by tests/test_mcp_catalog_version.py (the pinned
# major in that test is the deliberate-bump ritual — it cannot change by accident).
MCP_CATALOG_VERSION = "2.2.0"
MCP_CATALOG_VERSION = "2.3.0"
# #endregion McpServer.CatalogVersion
@@ -160,17 +160,19 @@ _MCP_CATALOG = (
McpToolDefinition("promote_to_scenario", None, service_allowed=False),
McpToolDefinition("request_save", ("scenario", "EDIT"), service_allowed=False, risk_level="guarded"),
McpToolDefinition("activate_revision", ("scenario", "EDIT"), service_allowed=False, risk_level="guarded"),
# REST exposes automation reads to authenticated principals without a READ grant.
# The tool bodies require a human owner, so service principals must remain denied.
McpToolDefinition("list_scenario_schedules", None, service_allowed=False),
# DG-2 (046 T019): automation reads admit any authenticated principal type — humans with
# scenario:automation READ (live DB RBAC) and service principals with the mcp:read scope —
# with per-object scenario-ownership ACL enforced inside the tool bodies. Mutations stay
# human-only below.
McpToolDefinition("list_scenario_schedules", ("scenario:automation", "READ")),
McpToolDefinition("upsert_scenario_schedule", ("scenario:automation", "MANAGE"), service_allowed=False),
McpToolDefinition("delete_scenario_schedule", ("scenario:automation", "MANAGE"), service_allowed=False),
McpToolDefinition("list_scenario_trigger_rules", None, service_allowed=False),
McpToolDefinition("list_scenario_trigger_rules", ("scenario:automation", "READ")),
McpToolDefinition("upsert_scenario_trigger_rule", ("scenario:automation", "MANAGE"), service_allowed=False),
McpToolDefinition("delete_scenario_trigger_rule", ("scenario:automation", "MANAGE"), service_allowed=False),
McpToolDefinition("get_scenario_automation_policy", None, service_allowed=False),
McpToolDefinition("get_scenario_automation_policy", ("scenario:automation", "READ")),
McpToolDefinition("upsert_scenario_automation_policy", ("scenario:automation", "MANAGE"), service_allowed=False),
McpToolDefinition("get_scenario_automation_metrics", None, service_allowed=False),
McpToolDefinition("get_scenario_automation_metrics", ("scenario:automation", "READ")),
)
_MCP_CATALOG_BY_NAME = {definition.name: definition for definition in _MCP_CATALOG}

View File

@@ -1,10 +1,23 @@
# #region McpServer.ToolsAutomation [C:4] [TYPE Module] [SEMANTICS mcp,automation,tools,rbac]
"""Curated MCP parity surface for persisted 046 automation configuration."""
# @RATIONALE DG-2 (046 T019, 2026-09-12): reads admit an authenticated principal of ANY type —
# human users holding scenario:automation READ (live DB RBAC, admin bypass) and service
# principals carrying the mcp:read scope — plus per-object scenario-ownership ACL;
# mutations stay human-only via owner(). Semantic parity with REST: no principal ->
# PermissionError("unauthenticated") (401), lacking grant/scope ->
# PermissionError("permission_denied") (403), foreign scenario-addressed object ->
# {"status": "not_found"} identical to a missing one (404, no existence leak).
# @REJECTED Keeping owner() (human-only) on reads was rejected — DG-2 admits scoped service
# principals to operational reads while mutations remain human-only.
# @REJECTED Raising a distinct "forbidden" for foreign scenario-addressed reads was rejected —
# a typed difference from "not found" leaks the object's existence across owners.
from typing import Any
import hashlib
import json
from pydantic import BaseModel, ConfigDict, Field, StrictBool, StrictInt, StrictStr
from src.core.auth.permission_utils import user_has_permission
from src.core.auth.repository import AuthRepository
from src.core.database import SessionLocal
from src.dependencies import get_scheduler_service
from src.mcp_server.auth import _access_token_context
@@ -89,34 +102,83 @@ def register_automation_tools(server: Any) -> None:
def _projection(item, fields: tuple[str, ...]) -> dict[str, Any]:
return {key: getattr(item, key) for key in fields}
def owner() -> str:
access = _access_token_context.get()
if access is None or not access.subject or (access.claims and access.claims.get("principal_type") == "service"):
raise PermissionError("human_principal_required")
return access.subject
def _viewer(db) -> tuple[str, bool]:
"""DG-2 read admission: any authenticated principal type -> (subject, is_admin).
Human principals need scenario:automation READ in live DB RBAC (admin roles bypass);
service principals need the mcp:read scope. Raises PermissionError with the REST-parity
semantic code ("unauthenticated" 401 / "permission_denied" 403).
"""
access = _access_token_context.get()
if access is None or not access.subject:
raise PermissionError("unauthenticated")
if access.claims and access.claims.get("principal_type") == "service":
if "mcp:read" not in (access.scopes or []):
raise PermissionError("permission_denied")
return access.subject, False
user = AuthRepository(db).get_user_by_username(access.subject)
if user is None or not getattr(user, "is_active", False):
raise PermissionError("unauthenticated")
if not user_has_permission(user, "scenario:automation", "READ"):
raise PermissionError("permission_denied")
is_admin = any(getattr(role, "is_admin", False) for role in (getattr(user, "roles", None) or []))
return access.subject, is_admin
def _visible_scenario_ids(db, subject: str, is_admin: bool) -> set[str] | None:
"""None for admin (no ACL filter); otherwise the scenario ids owned by the subject."""
if is_admin:
return None
rows = (
db.query(ScenarioRegistryEntry.scenario_id)
.filter(
(ScenarioRegistryEntry.owner_id == subject)
| (ScenarioRegistryEntry.owner_username == subject)
)
.all()
)
return {row[0] for row in rows}
def _scenario_visible(scenario_id: str, visible: set[str] | None) -> bool:
return visible is None or scenario_id in visible
@server.tool(name="list_scenario_schedules", structured_output=True)
async def list_scenario_schedules(scenario_id: StrictStr | None = None) -> list[dict[str, Any]]:
owner()
with SessionLocal() as db:
subject, is_admin = _viewer(db)
visible = _visible_scenario_ids(db, subject, is_admin)
q = db.query(ScenarioSchedule)
if scenario_id:
q = q.filter(ScenarioSchedule.scenario_id == scenario_id)
if visible is not None:
q = q.filter(ScenarioSchedule.scenario_id.in_(visible))
return [_projection(x, ("id", "scenario_id", "revision_id", "environment_id", "cron_expr", "timezone", "enabled", "policy_id")) for x in q.all()]
@server.tool(name="list_scenario_trigger_rules", structured_output=True)
async def list_scenario_trigger_rules(scenario_id: StrictStr | None = None) -> list[dict[str, Any]]:
owner()
with SessionLocal() as db:
subject, is_admin = _viewer(db)
visible = _visible_scenario_ids(db, subject, is_admin)
q = db.query(ScenarioTriggerRule)
if scenario_id:
q = q.filter(ScenarioTriggerRule.scenario_id == scenario_id)
if visible is not None:
q = q.filter(ScenarioTriggerRule.scenario_id.in_(visible))
return [_projection(x, ("id", "scenario_id", "revision_id", "environment_id", "trigger", "enabled", "policy_id")) for x in q.all()]
@server.tool(name="get_scenario_automation_policy", structured_output=True)
async def get_scenario_automation_policy(scenario_id: StrictStr) -> dict[str, Any]:
owner()
with SessionLocal() as db:
subject, is_admin = _viewer(db)
visible = _visible_scenario_ids(db, subject, is_admin)
if not _scenario_visible(scenario_id, visible):
return {"status": "not_found", "scenario_id": scenario_id}
policy_id = db.query(ScenarioSchedule.policy_id).filter(ScenarioSchedule.scenario_id == scenario_id, ScenarioSchedule.policy_id.isnot(None)).first()
if policy_id is None:
policy_id = db.query(ScenarioTriggerRule.policy_id).filter(ScenarioTriggerRule.scenario_id == scenario_id, ScenarioTriggerRule.policy_id.isnot(None)).first()
@@ -125,8 +187,11 @@ def register_automation_tools(server: Any) -> None:
@server.tool(name="get_scenario_automation_metrics", structured_output=True)
async def get_scenario_automation_metrics(scenario_id: StrictStr) -> dict[str, Any]:
owner()
with SessionLocal() as db:
subject, is_admin = _viewer(db)
visible = _visible_scenario_ids(db, subject, is_admin)
if not _scenario_visible(scenario_id, visible):
return {"status": "not_found", "scenario_id": scenario_id}
schedules = db.query(ScenarioSchedule).filter_by(scenario_id=scenario_id).all()
rules = db.query(ScenarioTriggerRule).filter_by(scenario_id=scenario_id).all()
runs = db.query(ScenarioRun).filter_by(scenario_id=scenario_id).all()

View File

@@ -1,4 +1,4 @@
# #region Models.ScenarioAutomation [C:4] [TYPE Module] [SEMANTICS scenario,automation,schedule,trigger,notification,policy]
# #region Models.ScenarioAutomation [C:4] [TYPE Module] [SEMANTICS scenario,automation,schedule,trigger,notification,policy,retention,deletion]
# @defgroup Models Durable scenario schedules, trigger rules, policies and notifications.
# @BRIEF Persisted automation entities per data-model.md (046): ScenarioSchedule with explicit
# APScheduler semantics, ScenarioTriggerRule bound to the 037 trigger framework,
@@ -13,7 +13,7 @@ from __future__ import annotations
from datetime import UTC, datetime
import uuid
from sqlalchemy import JSON, Boolean, Column, DateTime, Integer, String
from sqlalchemy import JSON, Boolean, Column, DateTime, Index, Integer, String
from .mapping import Base
@@ -80,4 +80,42 @@ class ScenarioNotificationEvent(Base):
severity = Column(String(16), nullable=False)
payload = Column(JSON, nullable=False, default=dict)
created_at = Column(DateTime, nullable=False, default=_now)
# #region Models.ScenarioAutomation.RetentionDeletion [C:4] [TYPE Class] [SEMANTICS scenario,automation,retention,deletion,receipt,tombstone]
# @ingroup Models
# @BRIEF Durable deletion receipt (046 line 20): one row per mark->eligible->delete-bytes->
# verify-absent->tombstone lifecycle, with the holds snapshot, attempt count and the
# byte-verification timestamps serving as the tombstone/audit record.
# @DATA_CONTRACT state in {deletion_pending, eligible, deleted, tombstoned}; baseline_pin mirrors
# the server-stamped BaselineSelectionPin shape {baseline_set_id, baseline_set_version, ...}.
# @INVARIANT A receipt may never leave deletion_pending while its bytes survive: `deleted` is
# recorded only after verify-absent, and tombstoned only after the deleted checkpoint.
# @INVARIANT Retry of the same deletion id is idempotent: tombstoned rows are a no-op,
# deletion_pending/eligible rows re-attempt.
# @REJECTED An immediate (non-durable) delete in the prune sweep was rejected — a crash between
# bytes removal and bookkeeping would lose the audit trail the state machine exists for.
class ScenarioRetentionDeletion(Base):
__tablename__ = "scenario_retention_deletions"
id = Column(String(36), primary_key=True, default=_id)
target_type = Column(String(32), nullable=False) # artifact|run_metadata|step_metrics|notification
target_id = Column(String(128), nullable=False)
content_ref = Column(String(255), nullable=True) # opaque storage ref (never a raw FS path)
scenario_id = Column(String(36), nullable=True, index=True)
run_id = Column(String(36), nullable=True, index=True)
baseline_pin = Column(JSON, nullable=True)
retention_class = Column(String(32), nullable=False, default="artifacts")
state = Column(String(32), nullable=False, default="deletion_pending", index=True)
holds_snapshot = Column(JSON, nullable=False, default=dict)
attempts = Column(Integer, nullable=False, default=0)
error_code = Column(String(64), nullable=True)
verified_absent_at = Column(DateTime, nullable=True)
tombstoned_at = Column(DateTime, nullable=True)
created_at = Column(DateTime, nullable=False, default=_now)
updated_at = Column(DateTime, nullable=False, default=_now)
__table_args__ = (
Index("uq_scenario_retention_deletions_target", "target_type", "target_id", unique=True),
)
# #endregion Models.ScenarioAutomation.RetentionDeletion
# #endregion Models.ScenarioAutomation

View File

@@ -0,0 +1,331 @@
# #region ScenarioAutomation.Retention.DeletionPipeline [C:5] [TYPE Module] [SEMANTICS scenario,automation,retention,deletion,receipts,tombstone]
# @defgroup ScenarioAutomation Durable retention deletion receipts (046 T021).
# @BRIEF mark -> eligible-after-all-holds -> delete bytes -> verify absent -> tombstone/audit,
# with idempotent retry of the same deletion id. The durable ScenarioRetentionDeletion
# receipt is the tombstone/audit record; failures remain deletion_pending.
# @LAYER Service
# @RELATION DEPENDS_ON -> [Models.ScenarioAutomation.RetentionDeletion]
# @RELATION DEPENDS_ON -> [ScenarioAutomation.Retention.ApplyHolds]
# @RELATION DEPENDS_ON -> [ScenarioExecution.Artifacts]
# @RELATION CALLED_BY -> [Api.ScenarioAutomation.RetentionDefaults]
# @INVARIANT A receipt never leaves deletion_pending while its bytes survive: `deleted` is
# recorded only after verify-absent, `tombstoned` only after that checkpoint.
# @INVARIANT One bounded, idempotent sweep tick; overlapping ticks are safe (unique target index
# + state machine re-attempt from any non-tombstoned state).
# @RATIONALE The receipt row itself is the tombstone/audit record: holds_snapshot, attempts,
# error_code, verified_absent_at and tombstoned_at capture the full deletion history
# without a second audit table (SQLite-compatible, mirrors 0023 idempotent-guard style).
# @REJECTED In-place (non-durable) deletion inside the prune sweep was rejected — a crash between
# byte removal and bookkeeping would destroy the audit trail the state machine exists for.
# @REJECTED An eager published-catalog DB collector was rejected — 044 published catalogs live in
# Gitea, not the DB; the offline default keeps approved_baselines=None (fail-closed),
# and the live collector is wired by the E-wave canary package.
from __future__ import annotations
from datetime import UTC, datetime
from typing import Any, Protocol
from sqlalchemy.orm import Session
from src.core.cot_logger import seed_trace_id
from src.core.logger import logger
from src.models.scenario_automation import ScenarioRetentionDeletion
from src.services.dashboard_testing.automation.retention import apply_holds
_SRC = "ScenarioAutomation.Retention.DeletionPipeline"
REATTEMPT_STATES = ("deletion_pending", "eligible", "deleted")
ACTIVE_RUN_STATUSES = ("queued", "running")
RETENTION_DELETE_FAILED = "RETENTION_DELETE_FAILED"
RETENTION_BYTES_SURVIVED = "RETENTION_BYTES_SURVIVED"
RETENTION_REF_UNRESOLVABLE = "RETENTION_REF_UNRESOLVABLE"
RETENTION_DELETION_NOT_FOUND = "RETENTION_DELETION_NOT_FOUND"
# #region ScenarioAutomation.Retention.Deletion.ByteStore [C:3] [TYPE Class] [SEMANTICS scenario,retention,storage,delete,verify]
# @ingroup ScenarioAutomation
# @BRIEF Minimal delete-bytes adapter over the existing artifact storage (draft/handle refs).
# @POST delete() removes the bytes or raises; exists() reports whether bytes survive. Unknown
# opaque ref schemes fail closed via RETENTION_REF_UNRESOLVABLE (receipt stays pending).
class RetentionByteStore(Protocol):
def delete(self, content_ref: str) -> bool: ...
def exists(self, content_ref: str) -> bool: ...
class DraftByteStore:
"""Adapter over DraftStorage (draft:{run_id}:{sha256}) and HandleBlobStore (handle:{sha256})."""
def __init__(self) -> None:
from src.services.agent_runs.artifacts import get_draft_storage
self._drafts = get_draft_storage()
def delete(self, content_ref: str) -> bool:
if content_ref.startswith("handle:"):
from src.services.dashboard_testing.scenario.handles import get_handle_storage
return get_handle_storage().delete(content_ref)
return self._drafts.delete(content_ref)
def exists(self, content_ref: str) -> bool:
if content_ref.startswith("handle:"):
from src.services.dashboard_testing.scenario.handles import get_handle_storage
try:
return get_handle_storage().retrieve(content_ref) is not None
except ValueError:
return False
return self._drafts.retrieve(content_ref) is not None
# #endregion ScenarioAutomation.Retention.Deletion.ByteStore
# #region ScenarioAutomation.Retention.Deletion.LiveHolds [C:3] [TYPE Function] [SEMANTICS scenario,retention,holds,active,operations]
# @ingroup ScenarioAutomation
# @BRIEF Collect live holds for one sweep: active operations (queued/running ScenarioRuns) plus
# the injectable approved-baseline pin set (None = no catalog collector wired, fail-closed).
# @SIDE_EFFECT Read-only DB query over scenario_runs.
def _live_holds(db: Session, approved_baselines: set[str] | None) -> dict[str, Any]:
from src.models.scenario_run import ScenarioRun
active = {
row[0]
for row in db.query(ScenarioRun.id)
.filter(ScenarioRun.status.in_(ACTIVE_RUN_STATUSES))
.all()
}
return {"active_operations": active, "approved_baselines": approved_baselines}
def _receipt_item(receipt: ScenarioRetentionDeletion) -> dict[str, Any]:
return {
"id": receipt.target_id,
"retention_class": receipt.retention_class,
"run_id": receipt.run_id or (receipt.target_id if receipt.target_type == "run_metadata" else ""),
"baseline_pin": receipt.baseline_pin or None,
"created_at": receipt.created_at.isoformat() if receipt.created_at else None,
}
# #endregion ScenarioAutomation.Retention.Deletion.LiveHolds
# #region ScenarioAutomation.Retention.Deletion.Mark [C:3] [TYPE Function] [SEMANTICS scenario,retention,deletion,mark,idempotency]
# @ingroup ScenarioAutomation
# @BRIEF Idempotently mark a deletion target: the same (target_type, target_id) returns the
# existing receipt unchanged; a new receipt is created in deletion_pending with the
# holds snapshot taken at mark time.
# @POST Returns the durable receipt; never deletes anything.
# @SIDE_EFFECT DB insert (receipt row only, flushed inside the caller's transaction).
def mark_deletion(
db: Session,
*,
target_type: str,
target_id: str,
scenario_id: str | None = None,
content_ref: str | None = None,
run_id: str | None = None,
baseline_pin: dict[str, Any] | None = None,
retention_class: str = "artifacts",
) -> ScenarioRetentionDeletion:
existing = (
db.query(ScenarioRetentionDeletion)
.filter(
ScenarioRetentionDeletion.target_type == target_type,
ScenarioRetentionDeletion.target_id == target_id,
)
.first()
)
if existing is not None:
return existing
receipt = ScenarioRetentionDeletion(
target_type=target_type,
target_id=target_id,
scenario_id=scenario_id,
content_ref=content_ref,
run_id=run_id,
baseline_pin=baseline_pin,
retention_class=retention_class,
state="deletion_pending",
)
now = datetime.now(UTC)
reasons = apply_holds([_receipt_item(receipt)], _live_holds(db, None), now=now)[1]
receipt.holds_snapshot = {"reasons": reasons[0]["reasons"] if reasons else [], "at": now.isoformat()}
db.add(receipt)
db.flush()
return receipt
# #endregion ScenarioAutomation.Retention.Deletion.Mark
# #region ScenarioAutomation.Retention.Deletion.Sweep [C:4] [TYPE Function] [SEMANTICS scenario,retention,deletion,sweep,verify,tombstone]
# @ingroup ScenarioAutomation
# @BRIEF Advance one receipt through the state machine: holds clear -> eligible -> delete bytes
# -> verify absent -> deleted -> tombstoned/audit. Any byte-step failure (delete raised,
# bytes survived) leaves the receipt in deletion_pending and records the error code.
# @POST Returns "tombstoned" | "failed" | "held"; the receipt row reflects the new state.
# @SIDE_EFFECT DB updates on the receipt; external storage delete for byte targets.
def _delete_and_verify(receipt: ScenarioRetentionDeletion, store: RetentionByteStore) -> str | None:
ref = receipt.content_ref
if not ref:
return None
if not (ref.startswith("draft:") or ref.startswith("handle:")):
logger.explore("Retention content_ref scheme not resolvable offline", src=_SRC,
claim="POST: delete bytes over known storage adapters",
error_code=RETENTION_REF_UNRESOLVABLE, payload={"ref_prefix": ref[:8]})
return RETENTION_REF_UNRESOLVABLE
try:
store.delete(ref)
except Exception as exc: # noqa: BLE001 — every storage failure must keep the receipt pending
logger.explore("Retention byte delete failed", src=_SRC, error=str(exc),
error_code=RETENTION_DELETE_FAILED, payload={"deletion_id": receipt.id})
return RETENTION_DELETE_FAILED
if store.exists(ref):
logger.explore("Verify-absent failed: bytes survive deletion", src=_SRC,
claim="INV: never report deleted while bytes survive",
error_code=RETENTION_BYTES_SURVIVED, payload={"deletion_id": receipt.id})
return RETENTION_BYTES_SURVIVED
return None
def _sweep_receipt(
db: Session,
receipt: ScenarioRetentionDeletion,
holds: dict[str, Any],
store: RetentionByteStore,
now: datetime,
) -> str:
_, held = apply_holds([_receipt_item(receipt)], holds, now=now)
if held:
reasons = held[0]["reasons"]
logger.reason("Deletion receipt held", src=_SRC, level="DEBUG",
payload={"deletion_id": receipt.id, "reasons": reasons})
receipt.state = "deletion_pending"
receipt.holds_snapshot = {"reasons": reasons, "at": now.isoformat()}
receipt.updated_at = now
return "held"
receipt.state = "eligible"
receipt.updated_at = now
receipt.attempts += 1
if receipt.state != "deleted" and receipt.error_code is not None:
receipt.error_code = None
if receipt.state != "deleted":
error_code = _delete_and_verify(receipt, store)
if error_code is not None:
receipt.error_code = error_code
receipt.state = "deletion_pending"
receipt.updated_at = now
return "failed"
receipt.error_code = None
receipt.verified_absent_at = now
receipt.state = "deleted"
# Tombstone/audit: the receipt row is the durable audit record of the completed deletion.
receipt.state = "tombstoned"
receipt.tombstoned_at = now
receipt.updated_at = now
logger.reflect("Retention deletion tombstoned", src=_SRC,
claim="INV: bytes verified absent before tombstone",
payload={"deletion_id": receipt.id, "attempts": receipt.attempts})
return "tombstoned"
def advance_retention_deletions(
db: Session,
*,
byte_store: RetentionByteStore | None = None,
approved_baselines: set[str] | None = None,
limit: int = 100,
) -> dict[str, int]:
"""One bounded, idempotent sweep tick over re-attemptable receipts (caller commits)."""
store = byte_store or DraftByteStore()
now = datetime.now(UTC)
holds = _live_holds(db, approved_baselines)
summary = {"processed": 0, "tombstoned": 0, "failed": 0, "held": 0}
rows = (
db.query(ScenarioRetentionDeletion)
.filter(ScenarioRetentionDeletion.state.in_(REATTEMPT_STATES))
.order_by(ScenarioRetentionDeletion.created_at.asc())
.limit(limit)
.all()
)
for receipt in rows:
summary["processed"] += 1
summary[_sweep_receipt(db, receipt, holds, store, now)] += 1
return summary
# #endregion ScenarioAutomation.Retention.Deletion.Sweep
# #region ScenarioAutomation.Retention.Deletion.Retry [C:3] [TYPE Function] [SEMANTICS scenario,retention,deletion,retry,idempotency]
# @ingroup ScenarioAutomation
# @BRIEF Retry the same deletion id idempotently: tombstoned receipts are a no-op; pending,
# eligible or deleted receipts re-attempt the state machine from their current state.
# @POST Returns the receipt in its post-retry state; raises ValueError (typed code) when unknown.
def retry_deletion(
db: Session,
deletion_id: str,
*,
byte_store: RetentionByteStore | None = None,
) -> ScenarioRetentionDeletion:
receipt = db.query(ScenarioRetentionDeletion).filter(ScenarioRetentionDeletion.id == deletion_id).first()
if receipt is None:
raise ValueError(RETENTION_DELETION_NOT_FOUND)
if receipt.state == "tombstoned":
logger.reason("Retry of tombstoned deletion receipt is a no-op", src=_SRC, level="DEBUG",
payload={"deletion_id": receipt.id})
return receipt
store = byte_store or DraftByteStore()
holds = _live_holds(db, None)
_sweep_receipt(db, receipt, holds, store, datetime.now(UTC))
db.flush()
return receipt
# #endregion ScenarioAutomation.Retention.Deletion.Retry
# #region ScenarioAutomation.Retention.Deletion.ReceiptDict [C:2] [TYPE Function] [SEMANTICS scenario,retention,deletion,receipts,api]
# @ingroup ScenarioAutomation
# @BRIEF Serialize a deletion receipt for the retention API projection.
def retention_receipt_to_dict(receipt: ScenarioRetentionDeletion) -> dict[str, Any]:
return {
"id": receipt.id,
"target_type": receipt.target_type,
"target_id": receipt.target_id,
"scenario_id": receipt.scenario_id,
"run_id": receipt.run_id,
"retention_class": receipt.retention_class,
"state": receipt.state,
"holds_snapshot": receipt.holds_snapshot,
"attempts": receipt.attempts,
"error_code": receipt.error_code,
"verified_absent_at": receipt.verified_absent_at.isoformat() if receipt.verified_absent_at else None,
"tombstoned_at": receipt.tombstoned_at.isoformat() if receipt.tombstoned_at else None,
"created_at": receipt.created_at.isoformat() if receipt.created_at else None,
"updated_at": receipt.updated_at.isoformat() if receipt.updated_at else None,
}
# #endregion ScenarioAutomation.Retention.Deletion.ReceiptDict
# #region ScenarioAutomation.Retention.Deletion.Scheduled [C:3] [TYPE Function] [SEMANTICS scenario,retention,deletion,scheduler,callback]
# @ingroup ScenarioAutomation
# @BRIEF Module-level APScheduler callback: one bounded retention deletion sweep per tick.
# @RELATION CALLS -> [ScenarioAutomation.Retention.Deletion.Sweep]
# @POST Failures are logged and rolled back, never raised into the scheduler thread.
# @SIDE_EFFECT Commits the maintenance transaction; deletes artifact bytes whose receipts cleared holds.
# @RATIONALE Module-level callback follows the core/scheduler.py lifecycle-retention pattern so the
# persistent job store never serializes service graphs. Registration into
# SchedulerService.start is deferred: core/scheduler.py is owned by the parallel C1
# wave (B/E-wave remaining item).
def execute_scheduled_retention() -> None:
from src.core.database import SessionLocal
seed_trace_id()
db = SessionLocal()
try:
summary = advance_retention_deletions(db)
db.commit()
if summary["processed"]:
logger.reflect("Retention deletion sweep completed", src=_SRC, payload=summary)
except Exception as exc:
db.rollback()
logger.explore("Retention deletion sweep failed", src=_SRC, error=str(exc),
error_code="RETENTION_SWEEP_FAILED")
finally:
db.close()
# #endregion ScenarioAutomation.Retention.Deletion.Scheduled
# #endregion ScenarioAutomation.Retention.DeletionPipeline

View File

@@ -1,17 +1,26 @@
# #region ScenarioAutomation.Retention [C:4] [TYPE Module] [SEMANTICS scenario,automation,retention,prune,tiers]
# @defgroup ScenarioAutomation Layered retention decisions (SCAUTO-FR-006/016).
# #region ScenarioAutomation.Retention [C:4] [TYPE Module] [SEMANTICS scenario,automation,retention,prune,tiers,holds]
# @defgroup ScenarioAutomation Layered retention decisions (SCAUTO-FR-006/016, 046 T021).
# @BRIEF Prune automation artifacts by layered tiers independent of the analytics minimum
# history window: run metadata 180d, triage/audit 365d, step metrics 90d, heavy
# artifacts 30d, screenshots 30d, raw VLM 7d. Referenced items are always preserved.
# artifacts 30d, screenshots 30d, raw VLM 7d. Hold evaluation (approved-baseline
# references, active operations, analytics window) runs before any prune decision.
# @RELATION DEPENDS_ON -> [ScenarioAutomation.Retention.DeletionPipeline]
# @INVARIANT Referenced items (provenance/artifacts) are never pruned (spec E5).
# @INVARIANT The analytics minimum history window (047) is guaranteed independently of retention.
# @REJECTED Single-window retention was rejected — the data-model mandates layered tiers so
# aggressive run pruning never destroys audit or analytics provenance.
# @REJECTED Deleting items that carry a server-stamped baseline pin was rejected (046 ADR line 4):
# pins are resolved only from published/approved catalogs (ScenarioExecution.BaselineResolver),
# so a resolved pin is itself an approved-baseline reference and adds a hold.
from __future__ import annotations
from datetime import UTC, datetime
from typing import Any
HOLD_APPROVED_BASELINE = "approved_baseline"
HOLD_ACTIVE_OPERATION = "active_operation"
HOLD_ANALYTICS_WINDOW = "analytics_window"
DEFAULT_TIER_DAYS: dict[str, int] = {
"run_metadata": 180,
"triage_audit": 365,
@@ -62,20 +71,102 @@ def run_retention(items: list[dict[str, Any]], limits: dict[str, int]) -> list[d
# #endregion ScenarioAutomation.Retention.Run
# #region ScenarioAutomation.Retention.RunDays [C:3] [TYPE Function] [SEMANTICS scenario,automation,retention,run,days,tiers]
# #region ScenarioAutomation.Retention.ApplyHolds [C:3] [TYPE Function] [SEMANTICS scenario,automation,retention,holds,baseline,operations]
# @ingroup ScenarioAutomation
# @BRIEF Age-based layered retention — prune items older than their tier horizon.
# @POST Items older than the tier horizon are pruned unless referenced; unknown timestamps are
# always kept; limits are days-per-tier (see tier_limits for defaults).
def run_retention_days(items: list[dict[str, Any]], limits: dict[str, int]) -> list[dict[str, Any]]:
# @BRIEF Pure hold evaluation: an item is not deletable while it references an approved baseline
# (server-stamped BaselineSelectionPin), belongs to an active operation (running/queued
# run, active case), or falls inside the analytics minimum history window (047).
# @RELATION DEPENDS_ON -> [ScenarioExecution.BaselineResolver.Resolve]
# @DATA_CONTRACT holds -> {"approved_baselines": set[str] | None, "active_operations": set[str],
# "analytics_min_window_days": int}; items carry optional baseline_pin /
# run_id / created_at / retention_class keys.
# @POST Returns (deletable, held) where held = [{"id": ..., "reasons": [approved_baseline|
# active_operation|analytics_window]}]. approved_baselines=None means no live catalog
# collector is wired: any resolved pin holds (fail-closed). A provided set is authoritative:
# only pins present in it hold (a retired pin no longer blocks deletion).
# @REJECTED Treating an unknown pin catalog as deletable was rejected — it would silently delete
# baseline-held artifacts whenever the catalog collector is unwired (046 ADR line 4).
def _pin_key(pin: Any) -> str | None:
if not isinstance(pin, dict):
return None
set_id = str(pin.get("baseline_set_id") or "").strip()
version = str(pin.get("baseline_set_version") or "").strip()
if not set_id or not version:
return None
return f"{set_id}@{version}"
def _item_hold_reasons(item: dict[str, Any], holds: dict[str, Any], now: datetime) -> list[str]:
reasons: list[str] = []
pin = _pin_key(item.get("baseline_pin"))
if pin is not None:
approved = holds.get("approved_baselines")
if approved is None or pin in set(approved):
reasons.append(HOLD_APPROVED_BASELINE)
active = set(holds.get("active_operations") or ())
run_id = str(item.get("run_id") or "")
if not run_id and str(item.get("retention_class") or "") == "run_metadata":
run_id = str(item.get("id") or "")
if run_id and run_id in active:
reasons.append(HOLD_ACTIVE_OPERATION)
window_days = int(holds.get("analytics_min_window_days") or 0)
# The analytics minimum window protects run METADATA only (047 history); artifact/screenshot
# tiers keep their own horizons (production-operations.md line 13 — window independent of
# metadata pruning). Applying it to artifacts would void every tier below 365 days.
if window_days > 0 and str(item.get("retention_class") or "") == "run_metadata":
created = _parse_dt(item.get("created_at"))
# Fail-closed: an unknown timestamp cannot be proven outside the analytics window.
if created is None or (now - created).days < window_days:
reasons.append(HOLD_ANALYTICS_WINDOW)
return reasons
def apply_holds(
items: list[dict[str, Any]],
holds: dict[str, Any] | None,
now: datetime | None = None,
) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]:
holds = holds or {}
now = now or datetime.now(UTC)
deletable: list[dict[str, Any]] = []
held: list[dict[str, Any]] = []
for item in items:
reasons = _item_hold_reasons(item, holds, now)
if reasons:
held.append({"id": str(item.get("id") or ""), "reasons": reasons})
else:
deletable.append(item)
return deletable, held
# #endregion ScenarioAutomation.Retention.ApplyHolds
# #region ScenarioAutomation.Retention.RunDays [C:3] [TYPE Function] [SEMANTICS scenario,automation,retention,run,days,tiers,holds]
# @ingroup ScenarioAutomation
# @BRIEF Age-based layered retention — prune items older than their tier horizon, after holds.
# @POST Items older than the tier horizon are pruned unless referenced or held (see apply_holds);
# unknown timestamps are always kept; limits are days-per-tier (see tier_limits for defaults).
def run_retention_days(
items: list[dict[str, Any]],
limits: dict[str, int],
holds: dict[str, Any] | None = None,
now: datetime | None = None,
) -> list[dict[str, Any]]:
merged = tier_limits(limits)
now = datetime.now(UTC)
now = now or datetime.now(UTC)
_, held = apply_holds(items, holds, now=now)
held_ids = {entry["id"] for entry in held}
kept_items: list[dict[str, Any]] = []
for item in items:
tier = str(item.get("retention_class", "run_metadata"))
days = int(merged.get(tier, 0))
created = _parse_dt(item.get("created_at"))
if item.get("referenced") or created is None or days <= 0 or (now - created).days < days:
if (
str(item.get("id") or "") in held_ids
or item.get("referenced")
or created is None
or days <= 0
or (now - created).days < days
):
kept_items.append(item)
return kept_items
# #endregion ScenarioAutomation.Retention.RunDays

View File

@@ -13,8 +13,10 @@
# stale or forged payload could bypass the durable approval gate.
from __future__ import annotations
from datetime import UTC, datetime, timedelta
from typing import Any
from sqlalchemy import or_
from sqlalchemy.orm import Session
from .policy import apply_policy
@@ -50,6 +52,16 @@ def handle_trigger_event(event: dict[str, Any], rules: list[dict[str, Any]], act
# separately blocked by the 044 dispatcher before CAS/walker execution.
# @INVARIANT ConfigManager resolution occurs before each start; an unknown target raises before that
# rule creates a run, gate, queue, notification, or adapter side effect.
# @INVARIANT The policy supply honors the declared dedup window across active runs (including
# waiting_human) AND completed runs created inside the matched rules' maximum
# dedup_window_seconds; a run created earlier in the same dispatch consumes env
# capacity and its fingerprint immediately for the remaining rules.
# @REJECTED An unbounded completed-run supply for dedup_window_seconds=0 was rejected: the zero
# window is the legacy default (dedup among active runs only), and widening it to all
# history would convert the post-completion idempotency replay into a permanent dedup
# block.
# @REJECTED Re-reading the active-run snapshot per rule was rejected as needless churn; appending
# the just-created run to the same dispatch supply is the minimal exact capacity fix.
def dispatch_trigger_event(
db: Session,
event: dict[str, Any],
@@ -73,18 +85,36 @@ def dispatch_trigger_event(
ScenarioTriggerRule.enabled.is_(True),
ScenarioTriggerRule.trigger == event.get("type"),
).all()
active = [
policy_ids = {rule.policy_id for rule in rules if rule.policy_id}
policies_by_id = {
row.id: row
for row in db.query(AutomationPolicy).filter(AutomationPolicy.id.in_(policy_ids)).all()
} if policy_ids else {}
max_dedup_window = max(
(int(policies_by_id[rule.policy_id].dedup_window_seconds)
for rule in rules
if rule.policy_id in policies_by_id),
default=0,
)
active_statuses = ["queued", "running", "pending_approval", "waiting_human"]
supply_filter = ScenarioRun.status.in_(active_statuses)
if max_dedup_window > 0:
supply_filter = or_(
supply_filter,
ScenarioRun.created_at >= datetime.now(UTC) - timedelta(seconds=max_dedup_window),
)
supply = [
{
"environment_id": run.environment_id,
"dedup_fingerprint": (run.parameter_bindings or {}).get("automation_fingerprint"),
"status": run.status,
"created_at": run.created_at.isoformat() if run.created_at else None,
}
for run in db.query(ScenarioRun).filter(ScenarioRun.status.in_(["queued", "running", "pending_approval"])).all()
for run in db.query(ScenarioRun).filter(supply_filter).all()
]
created: list[str] = []
for rule in rules:
policy_row = db.query(AutomationPolicy).filter(AutomationPolicy.id == rule.policy_id).first() if rule.policy_id else None
policy_row = policies_by_id.get(rule.policy_id) if rule.policy_id else None
policy = {
"max_concurrent_per_env": policy_row.max_concurrent_per_env if policy_row else 1,
"dedup_window_seconds": policy_row.dedup_window_seconds if policy_row else 0,
@@ -101,7 +131,7 @@ def dispatch_trigger_event(
"dedup_fingerprint": event.get("fingerprint"),
"overlap": bool(event.get("overlap")),
}
decision = apply_policy(policy, candidate, active)
decision = apply_policy(policy, candidate, supply)
if not decision["allowed"] or not candidate["revision_id"]:
continue
run = start(
@@ -117,6 +147,16 @@ def dispatch_trigger_event(
trigger_source=str(event.get("type")),
)
created.append(run.id)
run_status = getattr(run, "status", None)
run_environment = getattr(run, "environment_id", None)
if run_status is not None and run_environment is not None:
run_created = getattr(run, "created_at", None)
supply.append({
"environment_id": run_environment,
"dedup_fingerprint": (getattr(run, "parameter_bindings", None) or {}).get("automation_fingerprint"),
"status": run_status,
"created_at": run_created.isoformat() if run_created else None,
})
return created
# #endregion ScenarioAutomation.Trigger.Dispatch
# #endregion ScenarioAutomation.Trigger

View File

@@ -14,6 +14,7 @@ from src.models.scenario_checkpoint import HumanCheckpoint
from src.models.scenario_run import ScenarioRun, ScenarioStepRun
from .lifecycle_helpers import _as_utc, _expire_step_leases, _retire_active_step_projection
from .providers.browser_session import close_run_sessions
_DEFAULT_CANCEL_DRAIN_SECONDS = 30
@@ -115,6 +116,9 @@ def cancel_run(
run.phase = "terminal"
run.finished_at = cancellation_at
db.flush()
# DG-1/B-wave finalizer: a cancelled run never keeps a run-scoped browser session alive;
# resume-after-cancel is impossible (retry refuses terminal runs), so replay is irrelevant.
close_run_sessions(run_id, reason="cancelled")
return run
# #endregion ScenarioExecution.Lifecycle.Cancel

View File

@@ -15,7 +15,12 @@
# @REJECTED Counting active lease rows at claim time was rejected — the count races concurrent claims
# and overbooks; a per-quota counter CAS is the single serialization point.
# @REJECTED Independent per-feature concurrency limits were rejected — they permit cross-workload
# starvation and bypass the environment-scoped allocator.
# starvation and bypass the environment-scoped allocator.
# @REJECTED A run-level lease class on top of per-step provider leases (T032 intake, 2026-09-13) was
# rejected — environment quotas per workload_class already pin concurrency (browser
# PROD=1/DEV=2); a parallel run-slot tier would double-count capacity and could block
# non-browser runs. Dispatcher integration is expiry-reconcile per tick + provider
# heartbeat before loop submission instead.
from __future__ import annotations
import uuid

View File

@@ -12,6 +12,7 @@ from typing import Any
from src.core.logger import logger
from .executor_helpers import _outcome
from .executor_registry import ScenarioExecutorRegistry
@@ -33,6 +34,19 @@ def _descendants(step_id: str, edges: list[dict[str, str]]) -> set[str]:
# @ingroup ScenarioExecution
# @BRIEF Dispatch one step: dependency gate -> human control -> descriptor-bound typed executor.
# @POST Returns a step outcome dict; invalid descriptors reject before any executor I/O.
# @POST An executor exception is contained at the step boundary as a typed inconclusive
# EXECUTOR_STEP_ERROR outcome; it never escapes dispatch, and the walker terminalizes the
# run honestly instead of losing the whole run to infrastructure closure.
# @RATIONALE Live UX-2 (ss-prod B01 run 6de8d0d9, 2026-09-12): an executor-level ValidationError on a
# compiled baseline_ref expectation propagated out of dispatch_step into the queued
# dispatcher, whose blanket except closed every step and the run as QUEUED_DISPATCH_ERROR
# with no diagnostics. Provider executors already contain adapter exceptions (typed
# *_ADAPTER_ERROR); the dispatch boundary applies the same fail-closed pattern once for
# every tool so one broken step cannot invalidate an otherwise healthy run.
# @REJECTED Letting executor exceptions reach the queued dispatcher's infrastructure closure was
# rejected — QUEUED_DISPATCH_ERROR is reserved for real infrastructure failures (its
# invariants stay untouched). Rolling back the transaction inside the boundary was
# rejected — it would discard the durable CAS claim and step progress of this cycle.
def dispatch_step(step: dict[str, Any], *, completed: dict[str, dict[str, Any]], registry: ScenarioExecutorRegistry, edges: list[dict[str, str]]) -> dict[str, Any]:
step_id = str(step["logical_step_id"])
dependencies = [str(edge["source"]) for edge in edges if str(edge["target"]) == step_id]
@@ -72,7 +86,23 @@ def dispatch_step(step: dict[str, Any], *, completed: dict[str, dict[str, Any]],
)
return {"status": "waiting_human", "logical_step_id": step_id}
executor = registry.resolve(descriptor)
result = executor(step, completed)
try:
result = executor(step, completed)
except Exception as exc:
logger.explore(
"Executor raised; step fails closed typed",
src="ScenarioExecution.Dispatch.dispatch_step",
claim="POST: executor invocation returns a typed outcome, never a dispatch-loop crash",
error_code="EXECUTOR_STEP_ERROR",
payload={"logical_step_id": step_id, "tool": descriptor.get("tool"), "action": descriptor.get("action")},
error=f"{type(exc).__name__}: {exc}",
)
return _outcome(
str(descriptor.get("tool") or step.get("tool") or ""),
"inconclusive",
reason="EXECUTOR_STEP_ERROR",
extra={"exception_type": type(exc).__name__},
)
logger.reflect(
"Executor returned step outcome", src="ScenarioExecution.Dispatch.dispatch_step",
payload={"logical_step_id": step_id, "status": result.get("status")},

View File

@@ -14,7 +14,9 @@ from sqlalchemy.orm import Session
from src.models.scenario_run import ScenarioRun
from .executor_registry import ScenarioExecutorRegistry
from .capacity import reconcile_expired_leases
from .live_binding import LiveExecutionBindingResolver
from .providers.browser_session import close_run_sessions
from .registry_builder import _build_default_registry
from .result import build_result
from .runner_plan import validate_pinned_runner_plan
@@ -22,6 +24,8 @@ from .start_run import TRIGGER_SOURCE_MANUAL
from .terminal_effects import _close_queued_dispatch_error, _record_terminal_side_effects, _reject_malformed_plan
from .walker import _advance_run, reconcile_capacity_blocked_runs
_TERMINAL_RUN_STATUSES = frozenset({"passed", "failed", "blocked", "inconclusive", "cancelled"})
# #region ScenarioExecution.Runner.QueuedDispatch [C:5] [TYPE Function] [SEMANTICS scenario,execution,dispatch,queue,cas,scheduler]
# @BRIEF Claim and advance eligible durable queued runs outside the HTTP request lifecycle.
@@ -37,6 +41,10 @@ from .walker import _advance_run, reconcile_capacity_blocked_runs
# queue a run but cannot invoke an adapter.
# @REJECTED Running a queued ScenarioRun in the API handler or relying on an in-memory worker flag was
# rejected — only the persisted status CAS is shared across server workers/processes.
# @INVARIANT (T032, 2026-09-13) Each tick reconciles expired capacity leases before admission, so a
# crashed worker's claimed leases free their quota; and every run reaching a terminal
# status on this tick closes its run-scoped browser sessions (DG-1 finalizer) — the close
# is best-effort and never masks the run outcome.
def dispatch_queued_runs(
db: Session,
*,
@@ -48,6 +56,9 @@ def dispatch_queued_runs(
if limit <= 0:
raise ValueError("dispatch limit must be positive")
reconcile_capacity_blocked_runs(db)
# T032: the dispatcher tick drives lease expiry — a crashed worker's claimed leases free
# their quota here so parked runs can be admitted on this or a later cycle.
reconcile_expired_leases(db)
candidates = (
db.query(ScenarioRun.id)
.filter(ScenarioRun.status == "queued")
@@ -101,6 +112,11 @@ def dispatch_queued_runs(
outcomes.append(_advance_run(db, run, selected_registry, worker_id=worker_id))
except Exception:
outcomes.append(_close_queued_dispatch_error(db, run))
# DG-1/B-wave finalizer: a terminal run never keeps a run-scoped browser session alive;
# the session registry close is best-effort and never masks the run outcome.
db.refresh(run)
if run.status in _TERMINAL_RUN_STATUSES:
close_run_sessions(run.id, reason=f"run_terminal:{run.status}")
return outcomes
# #endregion ScenarioExecution.Runner.QueuedDispatch

View File

@@ -14,11 +14,38 @@ from dataclasses import dataclass
from hashlib import sha256
from typing import Any, Protocol
from src.core.logger import logger
from src.schemas.dashboard_testing import DashboardQueryModel, ExecuteQueryRequest, NormalizedFilterContext, ValueKind
from src.services.dashboard_testing.query_executor import execute_dashboard_query_envelope
from .artifacts import is_valid_sha256
from .live_adapter import LiveAdapterResult
from .provider_operations import (
complete_provider_operation,
descriptor_fingerprint,
open_provider_operation,
)
_SRC = "ScenarioExecution.LiveBinding"
_SUPERSET_PROVIDER_VERSION = "superset/037"
# #region ScenarioExecution.LiveBinding.ReceiptFinalize [C:3] [TYPE Function] [SEMANTICS live-binding,superset,receipt,finalize]
# @ingroup ScenarioExecution
# @BRIEF Finalize the durable provider-operation receipt outside the run transaction; never masks the outcome.
# @RELATION DEPENDS_ON -> [ScenarioExecution.ProviderOperations.Service]
# @INVARIANT Receipt finalization failure is logged, not raised — the adapter result is never shadowed.
def _finalize_superset_receipt(operation_id: str | None, status: str, effect_state: str, summary: dict | None = None) -> None:
if operation_id is None:
return
from src.core.database import SessionLocal
try:
with SessionLocal() as db:
complete_provider_operation(db, operation_id, status=status, effect_state=effect_state, summary=summary)
db.commit()
except Exception as exc:
logger.explore("Superset receipt finalization failed", src=_SRC, payload={"operation_id": operation_id, "status": status}, error=repr(exc))
# #endregion ScenarioExecution.LiveBinding.ReceiptFinalize
_SNAPSHOT_FIELDS = frozenset({
"binding_ref", "environment_id", "dashboard_release_id", "release_fingerprint", "dashboard_id",
@@ -235,17 +262,65 @@ def _execute_bound_superset(
request = _request_from_step(step, binding)
if request is None:
return LiveAdapterResult(status="inconclusive", reason_code="SUPERSET_BINDING_REQUEST_REJECTED")
operation_id: str | None = None
step_meta = step.get("step_meta") if isinstance(step.get("step_meta"), dict) else {}
run_id = step.get("scenario_run_id")
logical_step_id = str(step.get("logical_step_id") or step_meta.get("logical_step_id") or "")
attempt = int(step_meta.get("attempt") or 1)
if isinstance(run_id, str) and run_id and logical_step_id:
from src.core.database import SessionLocal
try:
with SessionLocal() as db:
action_descriptor = step_meta.get("action_descriptor") if isinstance(step_meta.get("action_descriptor"), dict) else {}
receipt = open_provider_operation(
db,
run_id=run_id,
logical_step_id=logical_step_id,
attempt=attempt,
provider_id="superset",
provider_version=_SUPERSET_PROVIDER_VERSION,
action="execute_query",
descriptor_fingerprint=descriptor_fingerprint(action_descriptor or {
"binding_ref": binding.binding_ref,
"environment_id": binding.environment_id,
"dashboard_id": binding.dashboard_id,
}),
binding_ref=binding.binding_ref,
execution_principal_fingerprint=binding.execution_principal_fingerprint,
idempotency_key=f"{run_id}:{logical_step_id}:{attempt}",
effect_state="unknown",
summary={
"environment_id": binding.environment_id,
"dashboard_id": binding.dashboard_id,
},
)
operation_id = receipt["operation_id"]
db.commit()
except Exception as exc:
logger.explore("Superset receipt open failed", src=_SRC, payload={"run_id": run_id}, error=repr(exc))
operation_id = None
try:
envelope = resolved.run_async(execute_dashboard_query_envelope(
resolved.superset_client, request, resolved.query_model,
))
except TimeoutError:
_finalize_superset_receipt(operation_id, "failed", "unknown", summary={"phase": "timeout"})
raise
except Exception:
_finalize_superset_receipt(operation_id, "failed", "unknown", summary={"phase": "execution_error"})
return LiveAdapterResult(status="inconclusive", reason_code="SUPERSET_QUERY_EXECUTION_ERROR")
if envelope.normalized_value.kind == ValueKind.UNKNOWN:
_finalize_superset_receipt(operation_id, "failed", "none", summary={"phase": "query_failed"})
return LiveAdapterResult(status="failed", reason_code="SUPERSET_QUERY_FAILED")
return _evidence_result(step, resolved, envelope)
evidence = _evidence_result(step, resolved, envelope)
if evidence.status == "passed":
_finalize_superset_receipt(
operation_id, "completed", "none",
summary={"sha256": evidence.details.get("sha256"), "artifact_refs": list(evidence.artifact_refs or [])},
)
else:
_finalize_superset_receipt(operation_id, "failed", "none", summary={"phase": evidence.reason_code})
return evidence
# #endregion ScenarioExecution.LiveBinding.Execute

View File

@@ -11,20 +11,50 @@ from hashlib import sha256
from io import BytesIO
from typing import Any
from src.core.logger import logger
from src.services.dashboard_testing.comparison import compare_values
from src.services.dashboard_testing.normalization import normalize_result
from .artifacts import is_valid_sha256
from .executor_helpers import _STATUS, _as_normalized, _completed_value, _outcome, _policy_from, _step_payload
_SRC_ASSERTION = "ScenarioExecution.OfflineExecutors.Assertion"
# #region ScenarioExecution.OfflineExecutors.Assertion [C:3] [TYPE Function] [SEMANTICS scenario,execution,executor,assertion,comparison]
# @ingroup ScenarioExecution
# @BRIEF Literal-value assertion via 037 compare_values; a compiled baseline_ref expectation fails closed typed.
# @RELATION CALLS -> [BaselineEngine.Comparison.Compare]
# @RATIONALE compare_to_baseline compiles to Expected(kind=baseline_ref) (ScenarioGraph.Compiler.BuildStep),
# but no executable 037 comparison payload is materialized at the executor boundary, so a
# literal-value comparison is impossible here. Typed BASELINE_EVIDENCE_UNAVAILABLE (existing
# D11 taxonomy) keeps the queued dispatcher alive and the run honestly inconclusive
# (live UX-2 ss-prod B01 run 6de8d0d9, 2026-09-12).
# @REJECTED Normalizing a baseline_ref dict through NormalizedValue was rejected — kind is not a ValueKind
# and extra fields are forbidden, so pydantic ValidationError escaped the executor and the
# queued dispatcher closed the whole run as QUEUED_DISPATCH_ERROR. Faking a pass/fail from the
# bare ref was rejected — a reference is not baseline evidence.
def assertion(step: dict[str, Any], completed: dict[str, dict[str, Any]]) -> dict[str, Any]:
payload = _step_payload(step)
expected = payload.get("expected")
actual = payload.get("actual")
if actual is None:
actual = _completed_value(completed, payload.get("actual_from") or payload.get("source_step_id"))
if isinstance(expected, dict) and expected.get("kind") == "baseline_ref":
logger.explore(
"Compiled baseline_ref expectation cannot be materialized at the executor boundary",
src=_SRC_ASSERTION,
claim="POST: baseline_ref comparison is typed inconclusive, never a dispatch crash",
error_code="BASELINE_EVIDENCE_UNAVAILABLE",
payload={"logical_step_id": step.get("logical_step_id"), "baseline_ref": expected.get("ref")},
error="no executable 037 comparison payload accompanies the baseline reference",
)
return _outcome(
"assertion",
"inconclusive",
reason="BASELINE_EVIDENCE_UNAVAILABLE",
extra={"baseline_ref": expected.get("ref")},
)
if expected is None and actual is None:
return _outcome("assertion", "inconclusive", reason="ASSERTION_INPUTS_MISSING")
result = compare_values(

View File

@@ -6,16 +6,19 @@
# fingerprint and the claimed capacity lease; one receipt exists per attempt.
# @POST Every receipt reaches completed, failed, cancelled or reconciliation_required exactly once;
# late provider responses append to history without changing the terminal status.
# @INVARIANT Terminal statuses are write-once via CAS — no overwrite ever happens; reconciliation is
# only reachable from reconciliation_required and records its own resolution history.
# @INVARIANT Terminal statuses are write-once via CAS — completed/failed/cancelled are never
# overwritten; reconciliation_required is resolvable only through reconcile or a
# confirmed-stop cancel, each recording its own resolution history.
# @SIDE_EFFECT Durable provider_operation_receipts rows.
# @REJECTED Mutable in-place receipt updates were rejected — reconciliation evidence must be
# append-only and terminal states must survive racing workers.
from __future__ import annotations
import json
import threading
import uuid
from datetime import UTC, datetime, timedelta
from hashlib import sha256
from typing import Any, Callable
from src.core.logger import logger
@@ -37,10 +40,26 @@ _OPEN_REQUIRED_FIELDS = (
)
# #region ScenarioExecution.ProviderOperations.Service.DescriptorFingerprint [C:2] [TYPE Function] [SEMANTICS provider,operation,descriptor,fingerprint]
# @ingroup ScenarioExecution
# @BRIEF Canonical SHA-256 of an action descriptor for the operation receipt.
# @RATIONALE Hoisted to the receipt SSOT module so non-browser providers (screenshot, superset) can
# fingerprint descriptors without depending on a browser-specific helper module.
def descriptor_fingerprint(descriptor: dict[str, Any]) -> str:
canonical = json.dumps(descriptor, sort_keys=True, separators=(",", ":"))
return sha256(canonical.encode()).hexdigest()
# #endregion ScenarioExecution.ProviderOperations.Service.DescriptorFingerprint
def _now() -> datetime:
return datetime.now(UTC).replace(tzinfo=None)
_CANCEL_DEADLINE_SECONDS = 300
_ACKNOWLEDGES = frozenset({"stopped", "completed", "unknown"})
_TERMINAL_FOR_CANCEL = frozenset({"completed", "failed", "cancelled"})
# #region ScenarioExecution.ProviderOperations.Service.Open [C:4] [TYPE Function] [SEMANTICS provider,operation,open,receipt]
# @ingroup ScenarioExecution
# @BRIEF Open one durable receipt in status running before external provider I/O.
@@ -161,6 +180,67 @@ def record_late_response(db, operation_id: str, payload: dict[str, Any]) -> dict
# #endregion ScenarioExecution.ProviderOperations.Service.LateResponse
# #region ScenarioExecution.ProviderOperations.Service.Cancel [C:5] [TYPE Function] [SEMANTICS provider,operation,cancel,receipt,reconciliation]
# @ingroup ScenarioExecution
# @BRIEF Durable cancel request + acknowledge semantics per SCEX-FR-020 (operation-aware cancellation).
# @PRE operation_id exists; reason non-empty; acknowledge in {stopped, completed, unknown}.
# @POST Returns receipt projection with cancellation acknowledgement. Idempotent: repeated cancel of
# the same operation_id returns the same receipt without mutating history. Terminal receipts are
# never modified.
# @INVARIANT cancel only mutates running/reconciliation_required receipts; a terminal receipt is
# always returned unchanged. acknowledge=unknown always leaves reconciliation_required,
# blocking retry and PASS until the ReconcileWorker resolves it.
# @SIDE_EFFECT Mutates status/effect_state/cancellation_requested/cancellation_deadline_at/history
# on non-terminal receipts.
# @REJECTED Mutating a terminal receipt on late cancel was rejected — CAS write-once terminal
# semantics must survive racing cancel requests.
def cancel_provider_operation(db, operation_id: str, *, reason: str, acknowledge: str) -> dict[str, Any]:
if acknowledge not in _ACKNOWLEDGES:
logger.explore("Cancel rejected: unknown acknowledge", src=_SRC, payload={"acknowledge": acknowledge}, error_code="PROVIDER_OPERATION_CANCEL_INVALID")
raise ValueError("PROVIDER_OPERATION_CANCEL_INVALID")
if not str(reason or "").strip():
logger.explore("Cancel rejected: empty reason", src=_SRC, error_code="PROVIDER_OPERATION_CANCEL_INVALID")
raise ValueError("PROVIDER_OPERATION_CANCEL_INVALID")
receipt = db.query(ProviderOperationReceipt).filter(ProviderOperationReceipt.operation_id == operation_id).one_or_none()
if receipt is None:
logger.explore("Cancel for unknown receipt", src=_SRC, payload={"operation_id": operation_id}, error_code="PROVIDER_OPERATION_UNKNOWN")
raise ValueError("PROVIDER_OPERATION_UNKNOWN")
if receipt.status in _TERMINAL_FOR_CANCEL:
ack = "stopped" if receipt.status == "cancelled" else "completed"
return {**_project(receipt), "cancellation": ack}
# Idempotent re-cancel with the same acknowledge: return current projection, no duplicate history.
existing = [h for h in list(receipt.history or []) if isinstance(h, dict) and h.get("kind") == "cancel_request"]
if receipt.cancellation_requested and existing and existing[-1].get("acknowledge") == acknowledge:
return {**_project(receipt), "cancellation": acknowledge}
now = _now()
receipt.cancellation_requested = True
receipt.cancellation_deadline_at = now + timedelta(seconds=_CANCEL_DEADLINE_SECONDS)
receipt.history = [
*list(receipt.history or []),
{"at": now.isoformat(), "kind": "cancel_request", "acknowledge": acknowledge, "reason": reason},
]
if acknowledge == "stopped":
receipt.status = "cancelled"
receipt.effect_state = "not_started"
receipt.terminal_at = now
elif acknowledge == "completed":
receipt.status = "completed"
receipt.effect_state = "completed"
receipt.terminal_at = now
else: # unknown
receipt.status = "reconciliation_required"
receipt.effect_state = "unknown"
receipt.terminal_at = now
receipt.updated_at = now
db.flush()
logger.reflect(
"Provider operation cancel applied", src=_SRC,
payload={"operation_id": operation_id, "acknowledge": acknowledge, "status": receipt.status},
)
return {**_project(receipt), "cancellation": acknowledge}
# #endregion ScenarioExecution.ProviderOperations.Service.Cancel
def _project(receipt: ProviderOperationReceipt) -> dict[str, Any]:
return {
"operation_id": receipt.operation_id,

View File

@@ -1,36 +1,42 @@
# #region ScenarioExecution.BrowserProvider [C:5] [TYPE Module] [SEMANTICS scenario,execution,provider,browser,actions,checkpoint,evidence,capacity,mutation]
# #region ScenarioExecution.BrowserProvider [C:5] [TYPE Module] [SEMANTICS scenario,execution,provider,browser,actions,checkpoint,evidence,capacity,mutation,native-filter,session]
# @ingroup ScenarioExecution
# @BRIEF Server-owned isolated browser provider for 038 actions on the shared provider loop.
# @RELATION IMPLEMENTS -> [ScenarioExecution.ProviderCatalog]
# @RELATION DEPENDS_ON -> [ScenarioExecution.CapacityManager.Service]
# @RELATION DEPENDS_ON -> [ScenarioExecution.ProviderRuntime.Engine]
# @RELATION DEPENDS_ON -> [ScenarioExecution.BrowserProvider.Transport]
# @RELATION DEPENDS_ON -> [ScenarioExecution.BrowserProvider.Session]
# @RELATION DEPENDS_ON -> [ScenarioExecution.BrowserProvider.Admission]
# @RELATION DEPENDS_ON -> [ScenarioExecution.BrowserProvider.MutationSQL]
# @PRE The transport is built from the deployment-owned environment by startup composition; the step
# carries a valid binding snapshot, a scenario_run_id, a browser action descriptor whose tool is
# browser, and environment/dashboard identity matching the registered binding exactly.
# @POST A passed result carries the executed checkpoint(s), the final page URL, re-digested evidence
# refs within the size limit and, for mutations, an operation_id with a completed effect_state;
# unknown mutation effects are typed reconciliation-required and the capacity lease is always
# released.
# refs within the size limit, the stamped browser-safe checkpoint ({dashboard_id,
# native_filter_state, active_tab, wait_states, table_filter_state, filter_state_observed})
# when a run-scoped session executed the step and, for mutations, an operation_id with a
# completed effect_state; round-2 read-only actions also merge their structured details
# (tab/filter/table/columns/rows) and a download stores a bounded (25 MiB) server-owned
# artifact ref with sha256 beside the screenshot evidence; unknown mutation effects are typed
# reconciliation-required and the capacity lease is always released.
# @INVARIANT Mutations require a full mutation contract (fixture lease, target keys, field allowlist,
# precondition hash, cleanup policy, retry_safe=false) before any browser I/O; an unknown
# mutation effect can only resolve through reconciliation, never a retry or a PASS.
# @INVARIANT Authority comes only from the registered server-owned transport and the persisted binding;
# URLs, cookies and step metadata cannot select credentials or a principal. At most one
# isolated context exists per (run, lease); contexts never share across runs.
# isolated context exists per (run, lease); contexts never share across runs; a dead
# context is closed and never revived — recovery replays the persisted checkpoint.
# @SIDE_EFFECT Headless browser process with an isolated context, authenticated session, navigation,
# descriptor-authorized SQL-Lab-mediated mutation effects, durable draft evidence bytes,
# and provider capacity lease rows.
# @REJECTED Trusting a mutation contract without per-key validation was rejected — missing fixture
# lease or precondition hash would authorize untracked external effects. Process-global
# Playwright pages and per-call event loops were rejected — isolation is per (run, lease)
# on the single application-owned provider loop.
# on the single application-owned provider loop. Per-step context isolation as the norm
# was rejected by DG-1 (2026-09-12) — it breaks native-filter continuity across run steps.
from __future__ import annotations
import uuid
from collections.abc import Callable
from hashlib import sha256
from typing import Any
from src.core.database import SessionLocal
@@ -38,10 +44,10 @@ from src.core.logger import logger
from src.services.dashboard_testing.execution.capacity import (
CapacityUnavailable,
claim_capacity,
heartbeat_capacity,
release_capacity,
)
from src.services.dashboard_testing.execution.live_adapter import LiveAdapterResult
from src.services.dashboard_testing.execution.live_binding import binding_from_step
from src.services.dashboard_testing.execution.mime_sniff import sniff_mime
from src.services.dashboard_testing.execution.provider_operations import (
open_provider_operation,
@@ -51,15 +57,23 @@ from src.services.dashboard_testing.execution.provider_runtime import (
ProviderSubmissionOverflow,
get_provider_event_loop,
)
from src.services.dashboard_testing.execution.providers.browser_admission import (
admit_browser_action,
store_browser_evidence,
store_download_artifact,
)
from src.services.dashboard_testing.execution.providers.browser_session import (
BrowserCheckpointMissing,
BrowserSessionManager,
)
from src.services.dashboard_testing.execution.providers.browser_transport import ( # noqa: F401
BrowserActionTransport,
BrowserTransportCleanupFailed,
BrowserTransportOutcome,
BrowserTransportPreconditionMismatch,
BrowserTransportSelectorNotFound,
BrowserTransportUnsupported,
)
from src.services.dashboard_testing.execution.providers.browser_mutation import (
validate_mutation_contract,
validate_mutation_inputs,
is_session_capable_transport,
)
from src.services.dashboard_testing.execution.providers.browser_receipt import (
descriptor_fingerprint,
@@ -69,130 +83,23 @@ from src.services.dashboard_testing.execution.providers.browser_receipt import (
_SRC = "ScenarioExecution.BrowserProvider"
_DEFAULT_ACTION_TIMEOUT_SECONDS = 30
_DEFAULT_MAX_SCREENSHOT_BYTES = 10485760
_READ_ONLY_ACTIONS = frozenset({"open_dashboard", "wait_for_state", "refresh"})
_MUTATION_ACTIONS = frozenset({"row_edit", "bulk_edit"})
_DEFAULT_MAX_DOWNLOAD_BYTES = 26214400 # 25 MiB (T034 spec)
# #region ScenarioExecution.BrowserProvider.EnvironmentClass [C:2] [TYPE Function] [SEMANTICS provider,capacity,environment]
# @ingroup ScenarioExecution
# @BRIEF Resolve the capacity environment class from pinned step snapshots.
def _environment_class_from_step(step: dict[str, Any]) -> str:
target = step.get("target_snapshot") if isinstance(step.get("target_snapshot"), dict) else {}
metadata = step.get("step_meta") if isinstance(step.get("step_meta"), dict) else {}
return "PROD" if target.get("environment_class") == "PROD" or metadata.get("environment_class") == "PROD" else "DEV"
# #endregion ScenarioExecution.BrowserProvider.EnvironmentClass
# #region ScenarioExecution.BrowserProvider.Admission [C:5] [TYPE Function] [SEMANTICS provider,browser,admission,descriptor]
# @ingroup ScenarioExecution
# @BRIEF Fail-closed admission: loop, binding, run identity, target identity and descriptor gating.
# @POST Returns (None, admission_payload) when the step may proceed, or (typed_rejection, None);
# no branch performs external I/O.
def _admit_browser_action(
event_loop: ProviderEventLoop,
step: dict[str, Any],
) -> tuple[LiveAdapterResult | None, dict[str, Any] | None]:
if not event_loop.is_running:
logger.explore("Provider loop is not running; no I/O admitted", src=_SRC, error_code="BROWSER_LOOP_UNAVAILABLE")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_LOOP_UNAVAILABLE"), None
binding, error = binding_from_step(step, tool_code="BROWSER")
if binding is None:
logger.explore("Browser binding rejected before I/O", src=_SRC, error=error or "BROWSER_BINDING_INVALID")
return LiveAdapterResult(status="inconclusive", reason_code=error or "BROWSER_BINDING_INVALID"), None
metadata = step.get("step_meta") if isinstance(step.get("step_meta"), dict) else {}
run_id = step.get("scenario_run_id")
if not isinstance(run_id, str) or not run_id:
logger.explore("Browser step has no persisted run identity", src=_SRC, error_code="BROWSER_RUN_UNRESOLVED")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_RUN_UNRESOLVED"), None
if metadata.get("environment_id") != binding.environment_id or metadata.get("dashboard_id") != binding.dashboard_id:
logger.explore(
"Browser step identity mismatches the registered binding", src=_SRC,
payload={"binding_ref": binding.binding_ref},
error_code="BROWSER_TARGET_MISMATCH",
)
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_TARGET_MISMATCH"), None
descriptor = metadata.get("action_descriptor") if isinstance(metadata.get("action_descriptor"), dict) else {}
action = str(descriptor.get("action") or "")
if not action:
logger.explore("Browser action descriptor missing", src=_SRC, error_code="BROWSER_ACTION_DESCRIPTOR_REQUIRED")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_ACTION_DESCRIPTOR_REQUIRED"), None
if str(descriptor.get("tool") or "browser") != "browser":
logger.explore("Browser descriptor tool mismatch", src=_SRC, payload={"action": action}, error_code="BROWSER_ACTION_TOOL_MISMATCH")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_ACTION_TOOL_MISMATCH"), None
mutating = bool(descriptor.get("mutating"))
if mutating or action not in _READ_ONLY_ACTIONS:
if not mutating:
logger.explore("Unsupported browser action rejected before I/O", src=_SRC, payload={"action": action}, error_code="BROWSER_ACTION_NOT_SUPPORTED")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_ACTION_NOT_SUPPORTED"), None
contract_error = validate_mutation_contract(metadata)
if contract_error is not None:
return LiveAdapterResult(status="inconclusive", reason_code=contract_error), None
if action not in _MUTATION_ACTIONS:
logger.explore("Mutating action outside the supported catalog", src=_SRC, payload={"action": action}, error_code="BROWSER_ACTION_NOT_SUPPORTED")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_ACTION_NOT_SUPPORTED"), None
contract = metadata.get("mutation_contract") or {}
inputs_error = validate_mutation_inputs(descriptor.get("inputs"), contract)
if inputs_error is not None:
return LiveAdapterResult(status="inconclusive", reason_code=inputs_error), None
registry_fingerprint = metadata.get("action_registry_fingerprint")
if registry_fingerprint is not None:
from src.services.dashboard_testing.scenario.templates import action_registry_fingerprint
if registry_fingerprint != action_registry_fingerprint():
logger.explore("Browser registry fingerprint mismatch", src=_SRC, error_code="BROWSER_REGISTRY_MISMATCH")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_REGISTRY_MISMATCH"), None
if mutating and not str(metadata.get("logical_step_id") or "").strip():
logger.explore("Mutating action without a persisted step identity", src=_SRC, error_code="BROWSER_ACTION_DESCRIPTOR_REQUIRED")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_ACTION_DESCRIPTOR_REQUIRED"), None
return None, {
"binding": binding,
"metadata": metadata,
"run_id": run_id,
"action": action,
"mutating": mutating,
"descriptor": descriptor,
"environment_class": _environment_class_from_step(step),
}
# #endregion ScenarioExecution.BrowserProvider.Admission
# #region ScenarioExecution.BrowserProvider.Evidence [C:4] [TYPE Function] [SEMANTICS provider,browser,evidence,digest]
# @ingroup ScenarioExecution
# @BRIEF Re-digest, size-cap and store transport evidence as content-addressed durable refs.
# @POST Returns (None, refs, digests) on success or (typed_rejection, [], {}); oversize or unexpected
# storage refs never become PASS.
def _store_browser_evidence(
evidence: bytes,
storage: Any,
run_id: str,
*,
max_screenshot_bytes: int,
) -> tuple[LiveAdapterResult | None, list[str], dict[str, str]]:
if len(evidence) > max_screenshot_bytes:
logger.explore(
"Browser evidence exceeds the durable size limit", src=_SRC,
payload={"run_id": run_id, "bytes": len(evidence), "limit": max_screenshot_bytes},
error_code="BROWSER_EVIDENCE_TOO_LARGE",
)
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_EVIDENCE_TOO_LARGE"), [], {}
digest = sha256(evidence).hexdigest()
content_ref = storage.store(run_id, digest, evidence)
if content_ref != f"draft:{run_id}:{digest}":
logger.explore("Evidence storage returned an unexpected ref", src=_SRC, error_code="BROWSER_EVIDENCE_REF_INVALID")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_EVIDENCE_REF_INVALID"), [], {}
return None, [content_ref], {content_ref: digest}
# #endregion ScenarioExecution.BrowserProvider.Evidence
# #region ScenarioExecution.BrowserProvider.Factory [C:5] [TYPE Function] [SEMANTICS provider,browser,factory,capacity,evidence,mutation]
# #region ScenarioExecution.BrowserProvider.Factory [C:5] [TYPE Function] [SEMANTICS provider,browser,factory,capacity,evidence,mutation,session]
# @ingroup ScenarioExecution
# @BRIEF Build the registered LiveProvider executing browser actions with receipts and reconciliation.
# @PRE transport and storage are server-owned; loop defaults to the application provider event loop.
# @POST Admission happens before any I/O; mutations carry operation_id/effect_state and unknown effects
# are reconciliation-required; the lease is released on every path.
# @PRE transport and storage are server-owned; loop defaults to the application provider event loop;
# session_manager defaults to a per-provider-instance manager when the transport is
# session-capable (a run pins exactly one binding, hence one provider instance).
# @POST Admission happens before any I/O; a session-capable transport executes inside the run-scoped
# session (acquire/reuse/replay, checkpoint stamped on PASS); a legacy execute-only transport
# keeps per-step isolation as the recover-fallback; mutations carry operation_id/effect_state;
# unknown effects are reconciliation-required; the lease is released on every path.
# @RATIONALE DG-1 round 1 (2026-09-12): the capacity lease stays per step, so the provider has no
# terminal-run channel — the session survives per-step lease release and closes via the
# manager's leak-guards (dead-context close on timeout/crash/loop-loss, idle TTL reaper,
# explicit close). The B-wave dispatch finalizer will call close(run_id) on run terminal.
def build_browser_provider(
*,
transport: BrowserActionTransport,
@@ -200,11 +107,15 @@ def build_browser_provider(
loop: ProviderEventLoop | None = None,
action_timeout_seconds: int = _DEFAULT_ACTION_TIMEOUT_SECONDS,
max_screenshot_bytes: int = _DEFAULT_MAX_SCREENSHOT_BYTES,
max_download_bytes: int = _DEFAULT_MAX_DOWNLOAD_BYTES,
session_manager: BrowserSessionManager | None = None,
) -> Callable[[Any], LiveAdapterResult]:
event_loop = loop if loop is not None else get_provider_event_loop()
if session_manager is None and is_session_capable_transport(transport):
session_manager = BrowserSessionManager(transport=transport, event_loop=event_loop)
def provider(context: Any) -> LiveAdapterResult:
rejection, admission = _admit_browser_action(event_loop, context.step)
rejection, admission = admit_browser_action(event_loop, context.step)
if rejection is not None or admission is None:
return rejection or LiveAdapterResult(status="inconclusive", reason_code="BROWSER_ADMISSION_INVALID")
binding = admission["binding"]
@@ -263,14 +174,35 @@ def build_browser_provider(
operation_id = receipt["operation_id"]
db.commit()
lease_id = lease["lease_id"]
# T032: heartbeat refreshes the lease TTL across any pre-I/O admission work so the
# provider loop submission window stays covered; an already-expired/lost lease is a
# typed capacity refusal (walker parks the run), never I/O without a lease.
try:
with SessionLocal() as db:
heartbeat_capacity(db, lease_id)
db.commit()
except CapacityUnavailable as exc:
logger.explore("Browser lease lost before I/O", src=_SRC, payload={"run_id": run_id}, error=str(exc))
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_CAPACITY_UNAVAILABLE")
except CapacityUnavailable as exc:
logger.explore("Browser capacity unavailable", src=_SRC, payload={"run_id": run_id}, error=str(exc))
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_CAPACITY_UNAVAILABLE")
session_plan: Any = None
def transport_factory():
action_input = descriptor.get("inputs") if isinstance(descriptor.get("inputs"), dict) else {}
if not mutating and action == "apply_native_filter" and admission.get("filter_input") is not None:
action_input = dict(admission["filter_input"])
if mutating:
action_input = {**action_input, "mutation_contract": metadata.get("mutation_contract") or {}}
if session_manager is not None:
return session_manager.execute_prepared(
session_plan,
action,
action_input=action_input,
timeout_seconds=action_timeout_seconds,
)
return transport.execute(
binding.dashboard_id,
action,
@@ -278,9 +210,35 @@ def build_browser_provider(
timeout_seconds=action_timeout_seconds,
)
session_checkpoint: dict[str, Any] | None = None
try:
outcome = event_loop.submit(transport_factory, timeout=action_timeout_seconds * 3)
if session_manager is not None:
with session_manager.run_guard(run_id):
session_plan = session_manager.prepare_step(
run_id=run_id,
lease_id=lease_id,
dashboard_id=binding.dashboard_id,
)
outcome, session_checkpoint = event_loop.submit(transport_factory, timeout=action_timeout_seconds * 3)
else:
outcome = event_loop.submit(transport_factory, timeout=action_timeout_seconds * 3)
except BrowserCheckpointMissing:
logger.explore(
"Browser recovery blocked: no declared checkpoint", src=_SRC,
payload={"run_id": run_id, "action": action},
claim="PRE: declared checkpoint for recovery",
error_code="BROWSER_CHECKPOINT_MISSING",
)
if mutating:
finalize_provider_receipt(operation_id, "failed", "not_started", summary={"phase": "checkpoint_missing"})
return LiveAdapterResult(
status="inconclusive",
reason_code="BROWSER_CHECKPOINT_MISSING",
details={"action": action, "retry_disposition": "manual_only"},
)
except TimeoutError:
if session_manager is not None:
session_manager.close(run_id, reason="step_timeout")
if mutating:
logger.explore("Mutating action deadline expired with unknown effect", src=_SRC, payload={"run_id": run_id}, error_code="BROWSER_MUTATION_RECONCILE_REQUIRED")
finalize_provider_receipt(operation_id, "reconciliation_required", "unknown", summary={"phase": "timeout"})
@@ -302,6 +260,20 @@ def build_browser_provider(
reason_code="BROWSER_MUTATION_PRECONDITION_MISMATCH",
details={"effect_state": "not_started", "retry_disposition": "manual_only", "operation_id": operation_id},
)
except BrowserTransportCleanupFailed:
# T034 round 4: the mutation completed (known effect) but the fixture was not
# restored — the environment is dirty; reconciliation must confirm/restore state
# before any retry, and the run reports inconclusive, never pass.
logger.explore(
"Mutation completed but fixture restore failed", src=_SRC,
payload={"run_id": run_id}, error_code="BROWSER_MUTATION_CLEANUP_FAILED",
)
finalize_provider_receipt(operation_id, "reconciliation_required", "completed", summary={"phase": "cleanup_failed"})
return LiveAdapterResult(
status="inconclusive",
reason_code="BROWSER_MUTATION_CLEANUP_FAILED",
details={"effect_state": "completed", "reconciliation_required": True, "retry_disposition": "after_reconciliation", "operation_id": operation_id},
)
except BrowserTransportUnsupported:
logger.explore("Transport cannot execute the action; nothing started", src=_SRC, payload={"action": action}, error_code="BROWSER_ACTION_NOT_SUPPORTED")
finalize_provider_receipt(operation_id, "failed", "not_started", summary={"phase": "unsupported"})
@@ -310,11 +282,36 @@ def build_browser_provider(
reason_code="BROWSER_ACTION_NOT_SUPPORTED",
details={"effect_state": "not_started", "retry_disposition": "manual_only", "operation_id": operation_id},
)
except BrowserTransportSelectorNotFound:
if mutating:
raise
logger.explore(
"Filter-bar locator miss; typed inconclusive without retry", src=_SRC,
payload={"run_id": run_id, "action": action}, error_code="BROWSER_SELECTOR_NOT_FOUND",
)
return LiveAdapterResult(
status="inconclusive",
reason_code="BROWSER_SELECTOR_NOT_FOUND",
details={"action": action, "retry_disposition": "manual_only"},
)
except ValueError as exc:
code = str(exc)
if mutating or not code.startswith("BROWSER_"):
raise
logger.explore(
"Browser action input rejected by the transport", src=_SRC,
payload={"run_id": run_id, "action": action, "code": code}, error_code=code,
)
return LiveAdapterResult(status="inconclusive", reason_code=code)
except RuntimeError as exc:
if str(exc) == "PROVIDER_LOOP_NOT_RUNNING":
if session_manager is not None:
session_manager.close(run_id, reason="loop_unavailable")
logger.explore("Provider loop stopped mid-dispatch", src=_SRC, error_code="BROWSER_LOOP_UNAVAILABLE")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_LOOP_UNAVAILABLE")
if mutating:
if session_manager is not None:
session_manager.close(run_id, reason="step_crashed")
logger.explore("Mutating action failed with unknown effect", src=_SRC, payload={"run_id": run_id}, error_code="BROWSER_MUTATION_RECONCILE_REQUIRED")
finalize_provider_receipt(operation_id, "reconciliation_required", "unknown", summary={"phase": "runtime_error"})
return LiveAdapterResult(
@@ -329,12 +326,31 @@ def build_browser_provider(
logger.explore("Transport produced no evidence", src=_SRC, payload={"run_id": run_id}, error_code="BROWSER_EVIDENCE_REQUIRED")
finalize_provider_receipt(operation_id, "reconciliation_required" if mutating else "failed", "unknown" if mutating else "not_started", summary={"phase": "evidence_missing"})
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_EVIDENCE_REQUIRED")
rejection, artifact_refs, artifact_digests = _store_browser_evidence(
rejection, artifact_refs, artifact_digests = store_browser_evidence(
evidence, storage, run_id, max_screenshot_bytes=max_screenshot_bytes,
)
if rejection is not None:
finalize_provider_receipt(operation_id, "reconciliation_required" if mutating else "failed", "unknown" if mutating else "not_started", summary={"phase": "evidence_store"})
return rejection
ref_bytes = {ref: len(evidence) for ref in artifact_refs}
ref_types = {ref: sniff_mime(evidence) or "image/png" for ref in artifact_refs}
download_bytes = getattr(outcome, "download_bytes", None)
download_ref: str | None = None
if download_bytes is not None:
# Round-2 download action: the captured bytes become a second content-addressed
# artifact ref beside the screenshot evidence (25 MiB bound enforced twice — in the
# transport flow and here at the storage gate; read-only, never a mutation receipt).
rejection, download_refs, download_digests = store_download_artifact(
download_bytes, storage, run_id, max_download_bytes=max_download_bytes,
)
if rejection is not None:
finalize_provider_receipt(operation_id, "failed", "not_started", summary={"phase": "download_store"})
return rejection
download_ref = download_refs[0]
artifact_refs = [*artifact_refs, *download_refs]
artifact_digests = {**artifact_digests, **download_digests}
ref_bytes[download_ref] = len(download_bytes)
ref_types[download_ref] = sniff_mime(download_bytes) or "application/octet-stream"
effect_state = outcome.effect_state if mutating else "none"
if mutating:
finalize_provider_receipt(operation_id, "completed", effect_state, summary={"checkpoints": list(outcome.checkpoints), "page_url": outcome.page_url})
@@ -352,21 +368,21 @@ def build_browser_provider(
"action": action,
"effect_state": effect_state,
**({"operation_id": operation_id} if operation_id else {}),
# Evaluation manifest inputs: byte length + MIME let
# ScenarioExecution.EvaluationAdapter.Manifest admit browser evidence.
"artifact_byte_lengths": {
ref: len(evidence) for ref in artifact_refs
},
# Sniff MIME from the stored bytes (a WebP payload must not be labelled png).
"artifact_content_types": {
ref: sniff_mime(evidence) or "image/png" for ref in artifact_refs
},
# Evaluation manifest inputs: per-ref byte length + sniffed MIME let
# ScenarioExecution.EvaluationAdapter.Manifest admit browser/download evidence.
"artifact_byte_lengths": ref_bytes,
"artifact_content_types": ref_types,
**({"download_artifact_ref": download_ref} if download_ref else {}),
**outcome.details,
# DG-1 browser-safe checkpoint: reconstructible filter/tab/wait state slice.
**({"browser_checkpoint": session_checkpoint} if session_checkpoint else {}),
},
artifact_refs=artifact_refs,
artifact_digests=artifact_digests,
)
except Exception as exc:
if session_manager is not None:
session_manager.close(run_id, reason="step_error")
logger.explore("Browser provider failed", src=_SRC, payload={"run_id": run_id if isinstance(run_id, str) else None}, error=repr(exc))
if mutating:
finalize_provider_receipt(operation_id, "reconciliation_required", "unknown", summary={"phase": "provider_error"})

View File

@@ -0,0 +1,208 @@
# #region ScenarioExecution.BrowserProvider.Admission [C:5] [TYPE Module] [SEMANTICS scenario,execution,provider,browser,admission,descriptor,evidence,readonly]
# @ingroup ScenarioExecution
# @BRIEF Fail-closed pre-I/O gate for the browser provider: environment class resolution, binding/
# descriptor admission, round-2 read-only action typed input validation, and durable evidence
# re-digest + storage (extracted from BrowserProvider to keep both modules inside INV_7 after
# the DG-1 session integration).
# @RELATION DEPENDS_ON -> [ScenarioExecution.LiveBinding.Identity]
# @RELATION DEPENDS_ON -> [ScenarioExecution.BrowserProvider.MutationSQL]
# @RELATION DEPENDS_ON -> [ScenarioExecution.BrowserProvider.NativeFilter]
# @RELATION DEPENDS_ON -> [ScenarioExecution.BrowserProvider.ReadOnlyActions]
# @POST Admission returns a typed rejection or a fully-resolved payload before any browser I/O;
# evidence storage is content-addressed and size-capped; round-2 read-only actions are typed
# rejected at admission so malformed inputs never reach the transport.
from __future__ import annotations
from hashlib import sha256
from typing import Any
from src.core.logger import logger
from src.services.dashboard_testing.execution.live_adapter import LiveAdapterResult
from src.services.dashboard_testing.execution.live_binding import binding_from_step
from src.services.dashboard_testing.execution.providers.browser_mutation import (
validate_mutation_contract,
validate_mutation_inputs,
)
from src.services.dashboard_testing.execution.providers.browser_native_filter import (
resolve_native_filter_input,
validate_native_filter_input,
)
from src.services.dashboard_testing.execution.providers.browser_readonly_actions import (
validate_readonly_action_input,
)
_SRC = "ScenarioExecution.BrowserProvider.Admission"
_READ_ONLY_ACTIONS = frozenset({
"open_dashboard", "wait_for_state", "refresh", "apply_native_filter",
"navigate_tab", "inspect_filter_state", "apply_table_filter", "extract_table",
"scroll_to", "inspect_columns", "click", "select_rows", "download",
})
_MUTATION_ACTIONS = frozenset({"row_edit", "bulk_edit"})
# Round-2 read-only actions that carry typed inputs; validated at admission before any I/O.
_TYPED_READONLY_ACTIONS = frozenset({
"navigate_tab", "apply_table_filter", "extract_table", "scroll_to", "click",
"select_rows", "download",
})
_MAX_DOWNLOAD_BYTES = 25 * 1024 * 1024 # 25 MiB (T034 spec)
# #region ScenarioExecution.BrowserProvider.Admission.EnvironmentClass [C:2] [TYPE Function] [SEMANTICS provider,capacity,environment]
# @ingroup ScenarioExecution
# @BRIEF Resolve the capacity environment class from pinned step snapshots.
def environment_class_from_step(step: dict[str, Any]) -> str:
target = step.get("target_snapshot") if isinstance(step.get("target_snapshot"), dict) else {}
metadata = step.get("step_meta") if isinstance(step.get("step_meta"), dict) else {}
return "PROD" if target.get("environment_class") == "PROD" or metadata.get("environment_class") == "PROD" else "DEV"
# #endregion ScenarioExecution.BrowserProvider.Admission.EnvironmentClass
# #region ScenarioExecution.BrowserProvider.Admission.Gate [C:5] [TYPE Function] [SEMANTICS provider,browser,admission,descriptor]
# @ingroup ScenarioExecution
# @BRIEF Fail-closed admission: loop, binding, run identity, target identity and descriptor gating.
# @POST Returns (None, admission_payload) when the step may proceed, or (typed_rejection, None);
# no branch performs external I/O.
def admit_browser_action(
event_loop: Any,
step: dict[str, Any],
) -> tuple[LiveAdapterResult | None, dict[str, Any] | None]:
if not event_loop.is_running:
logger.explore("Provider loop is not running; no I/O admitted", src=_SRC, error_code="BROWSER_LOOP_UNAVAILABLE")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_LOOP_UNAVAILABLE"), None
binding, error = binding_from_step(step, tool_code="BROWSER")
if binding is None:
logger.explore("Browser binding rejected before I/O", src=_SRC, error=error or "BROWSER_BINDING_INVALID")
return LiveAdapterResult(status="inconclusive", reason_code=error or "BROWSER_BINDING_INVALID"), None
metadata = step.get("step_meta") if isinstance(step.get("step_meta"), dict) else {}
run_id = step.get("scenario_run_id")
if not isinstance(run_id, str) or not run_id:
logger.explore("Browser step has no persisted run identity", src=_SRC, error_code="BROWSER_RUN_UNRESOLVED")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_RUN_UNRESOLVED"), None
if metadata.get("environment_id") != binding.environment_id or metadata.get("dashboard_id") != binding.dashboard_id:
logger.explore(
"Browser step identity mismatches the registered binding", src=_SRC,
payload={"binding_ref": binding.binding_ref},
error_code="BROWSER_TARGET_MISMATCH",
)
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_TARGET_MISMATCH"), None
descriptor = metadata.get("action_descriptor") if isinstance(metadata.get("action_descriptor"), dict) else {}
action = str(descriptor.get("action") or "")
if not action:
logger.explore("Browser action descriptor missing", src=_SRC, error_code="BROWSER_ACTION_DESCRIPTOR_REQUIRED")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_ACTION_DESCRIPTOR_REQUIRED"), None
if str(descriptor.get("tool") or "browser") != "browser":
logger.explore("Browser descriptor tool mismatch", src=_SRC, payload={"action": action}, error_code="BROWSER_ACTION_TOOL_MISMATCH")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_ACTION_TOOL_MISMATCH"), None
mutating = bool(descriptor.get("mutating"))
if mutating or action not in _READ_ONLY_ACTIONS:
if not mutating:
logger.explore("Unsupported browser action rejected before I/O", src=_SRC, payload={"action": action}, error_code="BROWSER_ACTION_NOT_SUPPORTED")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_ACTION_NOT_SUPPORTED"), None
contract_error = validate_mutation_contract(metadata)
if contract_error is not None:
return LiveAdapterResult(status="inconclusive", reason_code=contract_error), None
if action not in _MUTATION_ACTIONS:
logger.explore("Mutating action outside the supported catalog", src=_SRC, payload={"action": action}, error_code="BROWSER_ACTION_NOT_SUPPORTED")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_ACTION_NOT_SUPPORTED"), None
contract = metadata.get("mutation_contract") or {}
inputs_error = validate_mutation_inputs(descriptor.get("inputs"), contract)
if inputs_error is not None:
return LiveAdapterResult(status="inconclusive", reason_code=inputs_error), None
filter_input: dict[str, Any] | None = None
if not mutating and action == "apply_native_filter":
filter_input = resolve_native_filter_input(metadata, descriptor.get("inputs"))
filter_error = validate_native_filter_input(filter_input)
if filter_error is not None:
logger.explore(
"Native filter input rejected before I/O", src=_SRC,
payload={"action": action}, error_code=filter_error,
)
return LiveAdapterResult(status="inconclusive", reason_code=filter_error), None
if not mutating and action in _TYPED_READONLY_ACTIONS:
readonly_inputs = descriptor.get("inputs") if isinstance(descriptor.get("inputs"), dict) else {}
readonly_error = validate_readonly_action_input(action, readonly_inputs)
if readonly_error is not None:
logger.explore(
"Read-only action input rejected before I/O", src=_SRC,
payload={"action": action}, error_code=readonly_error,
)
return LiveAdapterResult(status="inconclusive", reason_code=readonly_error), None
registry_fingerprint = metadata.get("action_registry_fingerprint")
if registry_fingerprint is not None:
from src.services.dashboard_testing.scenario.templates import action_registry_fingerprint
if registry_fingerprint != action_registry_fingerprint():
logger.explore("Browser registry fingerprint mismatch", src=_SRC, error_code="BROWSER_REGISTRY_MISMATCH")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_REGISTRY_MISMATCH"), None
if mutating and not str(metadata.get("logical_step_id") or "").strip():
logger.explore("Mutating action without a persisted step identity", src=_SRC, error_code="BROWSER_ACTION_DESCRIPTOR_REQUIRED")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_ACTION_DESCRIPTOR_REQUIRED"), None
return None, {
"binding": binding,
"metadata": metadata,
"run_id": run_id,
"action": action,
"mutating": mutating,
"descriptor": descriptor,
"filter_input": filter_input,
"environment_class": environment_class_from_step(step),
}
# #endregion ScenarioExecution.BrowserProvider.Admission.Gate
# #region ScenarioExecution.BrowserProvider.Admission.Evidence [C:4] [TYPE Function] [SEMANTICS provider,browser,evidence,digest]
# @ingroup ScenarioExecution
# @BRIEF Re-digest, size-cap and store transport evidence as content-addressed durable refs.
# @POST Returns (None, refs, digests) on success or (typed_rejection, [], {}); oversize or unexpected
# storage refs never become PASS.
def store_browser_evidence(
evidence: bytes,
storage: Any,
run_id: str,
*,
max_screenshot_bytes: int,
) -> tuple[LiveAdapterResult | None, list[str], dict[str, str]]:
if len(evidence) > max_screenshot_bytes:
logger.explore(
"Browser evidence exceeds the durable size limit", src=_SRC,
payload={"run_id": run_id, "bytes": len(evidence), "limit": max_screenshot_bytes},
error_code="BROWSER_EVIDENCE_TOO_LARGE",
)
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_EVIDENCE_TOO_LARGE"), [], {}
digest = sha256(evidence).hexdigest()
content_ref = storage.store(run_id, digest, evidence)
if content_ref != f"draft:{run_id}:{digest}":
logger.explore("Evidence storage returned an unexpected ref", src=_SRC, error_code="BROWSER_EVIDENCE_REF_INVALID")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_EVIDENCE_REF_INVALID"), [], {}
return None, [content_ref], {content_ref: digest}
# #endregion ScenarioExecution.BrowserProvider.Admission.Evidence
# #region ScenarioExecution.BrowserProvider.Admission.DownloadArtifact [C:4] [TYPE Function] [SEMANTICS provider,browser,evidence,download,artifact,bounded]
# @ingroup ScenarioExecution
# @BRIEF Re-digest, bound (25 MiB, T034) and store a captured download as a server-owned artifact ref.
# @POST Returns (None, refs, digests) on success or (typed_rejection, [], {}); an oversize download
# is BROWSER_DOWNLOAD_TOO_LARGE and an unexpected storage ref never becomes PASS.
def store_download_artifact(
download_bytes: bytes,
storage: Any,
run_id: str,
*,
max_download_bytes: int | None = None,
) -> tuple[LiveAdapterResult | None, list[str], dict[str, str]]:
limit = max_download_bytes if max_download_bytes is not None else _MAX_DOWNLOAD_BYTES
if len(download_bytes) > limit:
logger.explore(
"Downloaded artifact exceeds the 25 MiB limit", src=_SRC,
payload={"run_id": run_id, "bytes": len(download_bytes), "limit": limit},
error_code="BROWSER_DOWNLOAD_TOO_LARGE",
)
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_DOWNLOAD_TOO_LARGE"), [], {}
digest = sha256(download_bytes).hexdigest()
content_ref = storage.store(run_id, digest, download_bytes)
if content_ref != f"draft:{run_id}:{digest}":
logger.explore("Download storage returned an unexpected ref", src=_SRC, error_code="BROWSER_EVIDENCE_REF_INVALID")
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_EVIDENCE_REF_INVALID"), [], {}
return None, [content_ref], {content_ref: digest}
# #endregion ScenarioExecution.BrowserProvider.Admission.DownloadArtifact
# #endregion ScenarioExecution.BrowserProvider.Admission

View File

@@ -25,6 +25,10 @@ from src.core.logger import logger
_SRC = "ScenarioExecution.BrowserProvider.MutationSQL"
_MAX_MUTATION_TARGET_KEYS = 100
_MUTATION_CONTRACT_REQUIRED_KEYS = ("fixture_lease_id", "target_keys", "field_allowlist", "precondition_hash", "cleanup_policy")
# T034 round 4: the cleanup policy vocabulary is closed — "restore_fixture" reverts the mutation
# from the captured pre-image inside the same session; "retain" is an explicit declaration that the
# fixture stays mutated (never a silent default).
_MUTATION_CLEANUP_POLICIES = frozenset({"restore_fixture", "retain"})
_MUTATION_PAGE_BODY = r"""
const ident = (name) => {
@@ -131,6 +135,80 @@ def build_mutation_script(inputs: dict[str, Any], contract: dict[str, Any]) -> s
}
return "async () => {" + f"const p = {json.dumps(payload)};" + _MUTATION_PAGE_BODY + "}"
# #endregion ScenarioExecution.BrowserProvider.MutationSQL.Script
# #region ScenarioExecution.BrowserProvider.MutationSQL.Cleanup [C:4] [TYPE Function] [SEMANTICS provider,browser,mutation,cleanup,restore]
# @ingroup ScenarioExecution
# @BRIEF Build the revert script that restores the captured pre-image rows for restore_fixture.
# @PRE The mutation flow returned precondition_ok with pre_rows; the contract cleanup_policy is
# restore_fixture (the closed vocabulary — anything else is rejected at validation).
# @POST The script reverts every allowed field to its pre-image value per target row and returns
# {restored_hash}; the transport verifies restored_hash == precondition_hash, so a failed
# restore is never reported as clean.
# @REJECTED Cleanup-as-a-later-step was rejected — the fixture must be restored inside the same
# session and lease window, before the step outcome is finalized.
_CLEANUP_PAGE_BODY = r"""
const ident = (name) => {
if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(name)) { throw new Error('BROWSER_MUTATION_IDENTIFIER_INVALID'); }
return name;
};
const lit = (value) => {
if (typeof value === 'number' && Number.isFinite(value)) { return String(value); }
if (typeof value === 'string') { return "'" + value.replace(/'/g, "''") + "'"; }
throw new Error('BROWSER_MUTATION_VALUE_INVALID');
};
const fetchCsrf = async () => {
const res = await fetch('/api/v1/security/csrf_token/', {credentials: 'same-origin'});
if (!res.ok) { throw new Error('BROWSER_MUTATION_CSRF_UNAVAILABLE'); }
const body = await res.json();
return body.result;
};
const exec = async (sql) => {
const csrf = await fetchCsrf();
const res = await fetch('/api/v1/sqllab/execute/', {
method: 'POST',
credentials: 'same-origin',
headers: {'Content-Type': 'application/json', 'X-CSRFToken': csrf},
body: JSON.stringify({database_id: p.database_id, sql, select_as_cta: false, client_id: ('cl' + Math.random().toString(36).slice(2, 11)), runAsync: false}),
});
if (!res.ok) { throw new Error('BROWSER_MUTATION_TRANSPORT_HTTP_' + res.status); }
const body = await res.json();
if (body.status && body.status !== 'success') { throw new Error(body.errors?.[0]?.message || 'BROWSER_MUTATION_SQL_FAILED'); }
return body;
};
const digest = async (text) => {
const bytes = new TextEncoder().encode(text);
const hash = await crypto.subtle.digest('SHA-256', bytes);
return Array.from(new Uint8Array(hash)).map((b) => b.toString(16).padStart(2, '0')).join('');
};
const keyCols = p.key_columns;
for (const row of p.pre_rows) {
const where = keyCols.map((col) => ident(col) + ' = ' + lit(row[col])).join(' AND ');
const setClause = p.allowed_fields.map((field) => ident(field) + ' = ' + lit(row[field])).join(', ');
if (setClause) { await exec('UPDATE ' + ident(p.table) + ' SET ' + setClause + ' WHERE ' + where); }
}
const selectSql = 'SELECT ' + p.allowed_fields.concat(keyCols).map(ident).join(', ') + ' FROM ' + ident(p.table)
+ ' WHERE ' + keyCols.map((col, index) => ident(col) + ' = ' + lit(p.target_keys[index])).join(' AND ');
const post = await exec(selectSql);
return {restored_hash: await digest(JSON.stringify(post.data)), rows_restored: p.pre_rows.length};
"""
def build_cleanup_script(inputs: dict[str, Any], contract: dict[str, Any], pre_rows: list[dict[str, Any]]) -> str:
payload = {
"database_id": inputs["database_id"],
"table": str(inputs["table"]),
"key_columns": [str(key) for key in inputs["key_columns"]],
"allowed_fields": [str(field) for field in inputs["assignments"] if str(field) in contract.get("field_allowlist", [])],
"target_keys": [str(key) for key in contract["target_keys"]],
"pre_rows": [
{str(key): value for key, value in row.items() if isinstance(value, (int, float, str))}
for row in pre_rows
if isinstance(row, dict)
],
}
return "async () => {" + f"const p = {json.dumps(payload)};" + _CLEANUP_PAGE_BODY + "}"
# #endregion ScenarioExecution.BrowserProvider.MutationSQL.Cleanup
# #region ScenarioExecution.BrowserProvider.MutationSQL.Contract [C:4] [TYPE Function] [SEMANTICS provider,browser,mutation,contract]
# @ingroup ScenarioExecution
# @BRIEF Validate the mandatory mutation contract keys before a mutating browser action.
@@ -164,6 +242,13 @@ def validate_mutation_contract(metadata: dict[str, Any]) -> str | None:
if not isinstance(allowlist, list) or not allowlist or any(not isinstance(key, str) or not key.strip() for key in allowlist):
logger.explore("Mutation field allowlist invalid", src=_SRC, error_code="BROWSER_MUTATION_CONTRACT_INVALID")
return "BROWSER_MUTATION_CONTRACT_INVALID"
if str(contract.get("cleanup_policy") or "") not in _MUTATION_CLEANUP_POLICIES:
logger.explore(
"Mutation cleanup policy outside the closed vocabulary", src=_SRC,
payload={"cleanup_policy": str(contract.get("cleanup_policy") or "")},
error_code="BROWSER_MUTATION_CONTRACT_INVALID",
)
return "BROWSER_MUTATION_CONTRACT_INVALID"
return None
# #endregion ScenarioExecution.BrowserProvider.MutationSQL.Contract

View File

@@ -0,0 +1,275 @@
# #region ScenarioExecution.BrowserProvider.NativeFilter [C:4] [TYPE Module] [SEMANTICS scenario,execution,provider,browser,native-filter,selector]
# @ingroup ScenarioExecution
# @BRIEF Read-only native-filter application through the dashboard filter bar UI for the isolated
# browser transport (038 action apply_native_filter).
# @RELATION IMPLEMENTS -> [ScenarioExecution.BrowserProvider.Transport]
# @RELATION DEPENDS_ON -> [Plugin.Service.ScreenshotService]
# @PRE The page holds an authenticated Superset dashboard session opened by _launch_and_login; the
# filter identity, values and selector_hint come only from the pinned descriptor inputs and the
# pinned step description — never from URL native_filters state.
# @POST A successful apply returns typed details (filter_target, applied_mode, applied_values,
# chart_data_observed) after a bounded chart settle; any locator miss raises
# BrowserTransportSelectorNotFound before evidence is produced; invalid typed input raises a
# BROWSER_* ValueError before any locator work.
# @INVARIANT Read-only: the flow clicks the filter bar UI and reads bounded rendered state only;
# no SQL, no API writes, no URL native_filters state, no retry of locator misses.
# @SIDE_EFFECT Clicks inside the dashboard filter bar within the isolated context; no server-side
# data mutation.
# @RATIONALE UI automation of the filter bar is the only honest read-only path: descriptor-pinned
# selectors identify the control, and the observed settle/evidence proves the dashboard
# actually reacted — a fabricated apply cannot produce the charts_settled checkpoint.
# @REJECTED Driving native filters via the URL native_filters state was rejected — the ephemeral
# native_filters_key is not importable by design and would fabricate a filter state the
# dashboard never rendered. Extending the mutation catalog for filter application was
# rejected — applying a native filter changes only the session UI state, never data.
from __future__ import annotations
import re
from typing import Any
from src.core.logger import logger
_SRC = "ScenarioExecution.BrowserProvider.NativeFilter"
_ALLOWED_WAIT_STATES = frozenset({"load", "domcontentloaded", "networkidle"})
_MAX_FILTER_VALUES = 100
_MAX_IDENTITY_LENGTH = 128
_CHART_SETTLE_TIMEOUT_MS = 15000
_DROPDOWN_MOUNT_TIMEOUT_MS = 5000
_SELECTOR_HINT_MARKER = "selector_hint:"
_ATTRIBUTE_IDENTITY_RE = re.compile(r"[A-Za-z0-9_][A-Za-z0-9_.:-]{0,127}\Z")
# Multi-strategy locator candidates (Superset version drift is expected; a total miss is a typed
# failure, never a retry — see ux-flow-improvement-plan UX-1 risk note). 2026-09-12 live DOM facts
# (ss-prod dashboard 11): prod builds strip data-test attributes; the filter bar is the antd form
# that owns .select-container controls and the .filter-apply-button sibling ("Apply filters").
_FILTER_BAR_SELECTORS = (
'[data-test="dashboard-filters-panel"]',
"#dashboard-filters-panel",
".dashboard-filters-panel",
'[data-test="filter-bar"]',
"form.ant-form:has(.select-container)",
"div:has(.filter-apply-button):has(.select-container)",
)
_FILTER_CONTROL_SELECTORS = (
".ant-select-selector",
'[data-test="filter-control"]',
".filter-item",
"input.ant-input",
)
_APPLY_BUTTON_SELECTORS = (
'button[data-test="filter-bar-apply-button"]',
"button.filter-apply-button",
'button.ant-btn-primary:has-text("Apply")',
'button:has-text("Apply")',
)
_SELECTED_VALUE_SELECTOR = ".ant-select-selection-item"
class BrowserTransportSelectorNotFound(RuntimeError):
"""Raised when the filter bar/control/option/apply control cannot be located; no retry."""
# #region ScenarioExecution.BrowserProvider.NativeFilter.Resolve [C:3] [TYPE Function] [SEMANTICS provider,browser,native-filter,input]
# @ingroup ScenarioExecution
# @BRIEF Merge explicit descriptor inputs, the UX-6 run-parameter binding and the pinned selector_hint.
# @PRE metadata is the pinned plan step; descriptor_inputs is the descriptor's typed inputs value.
# @POST Returns a new dict; a run-param filter_values binding (param_binding, stamped at RunnerPlan
# derivation) wins over descriptor values, and an explicit descriptor selector_hint always wins
# over the pinned description hint.
# @RATIONALE scenario_resolve pins selector hints onto the step description; honoring that pin as the
# first locator candidate keeps the human/agent-pinned authority instead of ignoring it.
# UX-6 (2026-09-12): param_binding.filter_values comes from the launch params via
# RunnerPlan derivation — the per-run param is the most specific intent, so it overrides
# graph/registry-pinned values. The raw binding value passes through uncoerced: a malformed
# persisted shape is typed-rejected by validate_native_filter_input in admission before any
# browser I/O, never silently coerced into plausible values.
def resolve_native_filter_input(metadata: dict[str, Any], descriptor_inputs: Any) -> dict[str, Any]:
merged = dict(descriptor_inputs) if isinstance(descriptor_inputs, dict) else {}
binding = metadata.get("param_binding")
if isinstance(binding, dict) and binding.get("filter_values") is not None:
merged["values"] = binding["filter_values"]
if not str(merged.get("selector_hint") or "").strip():
description = str(metadata.get("description") or "")
if _SELECTOR_HINT_MARKER in description:
hint = description.rsplit(_SELECTOR_HINT_MARKER, 1)[1].strip()
if hint:
merged["selector_hint"] = hint
return merged
# #endregion ScenarioExecution.BrowserProvider.NativeFilter.Resolve
# #region ScenarioExecution.BrowserProvider.NativeFilter.Validate [C:3] [TYPE Function] [SEMANTICS provider,browser,native-filter,validate]
# @ingroup ScenarioExecution
# @BRIEF Provider-side typed validation of the merged filter input before any browser I/O.
# @POST Returns None for a well-formed input (empty dict = current-state apply mode), otherwise a
# stable typed rejection code.
def validate_native_filter_input(filter_input: dict[str, Any]) -> str | None:
for key in ("filter_id", "filter_name", "column"):
value = filter_input.get(key)
if value is not None and (
not isinstance(value, str) or not value.strip() or len(value) > _MAX_IDENTITY_LENGTH
):
logger.explore("Native filter identity invalid", src=_SRC, payload={"key": key}, error_code="BROWSER_FILTER_INPUT_INVALID")
return "BROWSER_FILTER_INPUT_INVALID"
selector_hint = filter_input.get("selector_hint")
if selector_hint is not None and (not isinstance(selector_hint, str) or not selector_hint.strip()):
logger.explore("Native filter selector hint invalid", src=_SRC, error_code="BROWSER_FILTER_INPUT_INVALID")
return "BROWSER_FILTER_INPUT_INVALID"
values = filter_input.get("values")
if values is not None:
if (
not isinstance(values, list)
or len(values) > _MAX_FILTER_VALUES
or any(not isinstance(item, str) or not item.strip() for item in values)
):
logger.explore("Native filter values invalid", src=_SRC, error_code="BROWSER_FILTER_VALUES_INVALID")
return "BROWSER_FILTER_VALUES_INVALID"
wait_state = filter_input.get("wait_state")
if wait_state is not None and str(wait_state) not in _ALLOWED_WAIT_STATES:
logger.explore("Native filter wait state invalid", src=_SRC, payload={"wait_state": str(wait_state)}, error_code="BROWSER_WAIT_STATE_INVALID")
return "BROWSER_WAIT_STATE_INVALID"
return None
# #endregion ScenarioExecution.BrowserProvider.NativeFilter.Validate
# #region ScenarioExecution.BrowserProvider.NativeFilter.Parse [C:3] [TYPE Function] [SEMANTICS provider,browser,native-filter,input]
# @ingroup ScenarioExecution
# @BRIEF Transport-side normalization of the typed input; raises typed ValueError on invalid shape.
# @POST Returns the normalized dict with all contract keys; unknown wait_state raises
# ValueError("BROWSER_WAIT_STATE_INVALID") like the wait_for_state branch.
def parse_native_filter_input(action_input: dict[str, Any]) -> dict[str, Any]:
parsed = {
"filter_id": action_input.get("filter_id"),
"filter_name": action_input.get("filter_name"),
"column": action_input.get("column"),
"selector_hint": action_input.get("selector_hint"),
"values": [str(item) for item in action_input.get("values") or []],
"wait_state": action_input.get("wait_state"),
}
code = validate_native_filter_input({key: value for key, value in parsed.items() if value is not None})
if code is not None:
raise ValueError(code)
return parsed
# #endregion ScenarioExecution.BrowserProvider.NativeFilter.Parse
# #region ScenarioExecution.BrowserProvider.NativeFilter.ResolveControl [C:3] [TYPE Function] [SEMANTICS provider,browser,native-filter,selector]
# @ingroup ScenarioExecution
# @BRIEF Resolve the target filter control: pinned selector_hint, then identity, then first visible.
# @POST Returns a visible locator or None; the caller maps None to BrowserTransportSelectorNotFound.
async def _resolve_filter_control(service: Any, page: Any, bar: Any, filter_input: dict[str, Any]) -> Any:
candidates: list[Any] = []
hint = filter_input.get("selector_hint")
if hint:
candidates.append(page.locator(str(hint)))
identity = filter_input.get("filter_id") or filter_input.get("filter_name") or filter_input.get("column")
if identity:
identity_text = str(identity).strip()
if _ATTRIBUTE_IDENTITY_RE.fullmatch(identity_text):
candidates.append(bar.locator(f'[data-test*="{identity_text}"]'))
candidates.append(bar.get_by_text(identity_text, exact=False))
candidates.extend(bar.locator(selector) for selector in _FILTER_CONTROL_SELECTORS)
return await service._find_first_visible_locator(candidates)
# #endregion ScenarioExecution.BrowserProvider.NativeFilter.ResolveControl
# #region ScenarioExecution.BrowserProvider.NativeFilter.ApplyValues [C:3] [TYPE Function] [SEMANTICS provider,browser,native-filter,values]
# @ingroup ScenarioExecution
# @BRIEF Open the control, select each typed value option and click the filter bar Apply control.
# @POST Returns the applied values; a missing option or Apply control raises
# BrowserTransportSelectorNotFound (typed, no retry).
async def _apply_filter_values(service: Any, page: Any, control: Any, values: list[str], timeout_ms: int) -> list[str]:
await control.click(timeout=timeout_ms)
for value in values:
option_candidates = [page.get_by_text(value, exact=True)]
if _ATTRIBUTE_IDENTITY_RE.fullmatch(value):
option_candidates.insert(0, page.locator(f'.ant-select-item-option[title="{value}"]'))
# The dropdown panel mounts asynchronously after the control click; synchronize on the
# first bounded render before the fail-closed visibility snapshot (a genuine miss still
# types BROWSER_SELECTOR_NOT_FOUND after the bound — this wait is not a retry loop).
try:
await option_candidates[0].first.wait_for(state="visible", timeout=min(timeout_ms, _DROPDOWN_MOUNT_TIMEOUT_MS))
except Exception:
pass
option = await service._find_first_visible_locator(option_candidates)
if option is None:
logger.explore("Native filter option not found", src=_SRC, payload={"value": value}, error_code="BROWSER_SELECTOR_NOT_FOUND")
raise BrowserTransportSelectorNotFound("BROWSER_SELECTOR_NOT_FOUND")
await option.click(timeout=timeout_ms)
apply_button = await service._find_first_visible_locator([page.locator(selector) for selector in _APPLY_BUTTON_SELECTORS])
if apply_button is None:
logger.explore("Native filter apply control not found", src=_SRC, error_code="BROWSER_SELECTOR_NOT_FOUND")
raise BrowserTransportSelectorNotFound("BROWSER_SELECTOR_NOT_FOUND")
await apply_button.click(timeout=timeout_ms)
return list(values)
# #endregion ScenarioExecution.BrowserProvider.NativeFilter.ApplyValues
# #region ScenarioExecution.BrowserProvider.NativeFilter.ObserveCurrent [C:2] [TYPE Function] [SEMANTICS provider,browser,native-filter,observe]
# @ingroup ScenarioExecution
# @BRIEF Current-state mode: open the control and read its bounded rendered selection without changes.
# @POST Returns up to _MAX_FILTER_VALUES selected labels; no filter state is modified.
async def _observe_current_selection(control: Any) -> list[str]:
selected = control.locator(_SELECTED_VALUE_SELECTOR)
count = min(await selected.count(), _MAX_FILTER_VALUES)
labels: list[str] = []
for index in range(count):
labels.append(str(await selected.nth(index).text_content() or "").strip())
return [label for label in labels if label]
# #endregion ScenarioExecution.BrowserProvider.NativeFilter.ObserveCurrent
# #region ScenarioExecution.BrowserProvider.NativeFilter.Apply [C:4] [TYPE Function] [SEMANTICS provider,browser,native-filter,apply,settle]
# @ingroup ScenarioExecution
# @BRIEF Apply the native filter through the filter bar UI and wait for bounded chart settle.
# @PRE filter_input is normalized by parse_native_filter_input; the dashboard page is open.
# @POST Returns typed details for the transport outcome; raises BrowserTransportSelectorNotFound on
# any locator miss (bar, control, option, apply control) before evidence exists.
# @SIDE_EFFECT Filter bar clicks; optional wait_state load-state wait; chart settle polling.
async def apply_native_filter_via_ui(
service: Any,
page: Any,
filter_input: dict[str, Any],
*,
timeout_seconds: float,
) -> dict[str, Any]:
timeout_ms = int(timeout_seconds * 1000)
bar = await service._find_first_visible_locator([page.locator(selector) for selector in _FILTER_BAR_SELECTORS])
if bar is None:
logger.explore("Dashboard filter bar not found", src=_SRC, error_code="BROWSER_SELECTOR_NOT_FOUND")
raise BrowserTransportSelectorNotFound("BROWSER_SELECTOR_NOT_FOUND")
control = await _resolve_filter_control(service, page, bar, filter_input)
if control is None:
logger.explore(
"Native filter control not found", src=_SRC,
payload={"identity": filter_input.get("filter_id") or filter_input.get("filter_name") or filter_input.get("column")},
error_code="BROWSER_SELECTOR_NOT_FOUND",
)
raise BrowserTransportSelectorNotFound("BROWSER_SELECTOR_NOT_FOUND")
values = list(filter_input.get("values") or [])
if values:
applied_values = await _apply_filter_values(service, page, control, values, timeout_ms)
applied_mode = "values"
else:
applied_values = await _observe_current_selection(control)
applied_mode = "current_state"
wait_state = filter_input.get("wait_state")
if wait_state is not None:
await page.wait_for_load_state(str(wait_state), timeout=timeout_ms)
await service._wait_for_charts_stabilized(page, timeout_ms=min(timeout_ms, _CHART_SETTLE_TIMEOUT_MS))
rendered_charts = await page.evaluate(
"() => document.querySelectorAll('.chart-container canvas, .slice_container svg, .grid-content canvas').length"
)
logger.reflect(
"Native filter applied through the filter bar UI", src=_SRC,
payload={"applied_mode": applied_mode, "values": len(applied_values), "rendered_charts": rendered_charts},
)
return {
"applied": True,
"applied_mode": applied_mode,
"applied_values": applied_values,
"filter_target": filter_input.get("filter_id") or filter_input.get("filter_name") or filter_input.get("column"),
"chart_data_observed": bool(rendered_charts),
}
# #endregion ScenarioExecution.BrowserProvider.NativeFilter.Apply
# #endregion ScenarioExecution.BrowserProvider.NativeFilter

View File

@@ -0,0 +1,534 @@
# #region ScenarioExecution.BrowserProvider.ReadOnlyActions [C:5] [TYPE Module] [SEMANTICS scenario,execution,provider,browser,readonly,transport,navigate,extract,download,table-filter]
# @ingroup ScenarioExecution
# @BRIEF Read-only browser action flows for the T034 round-2 catalog: navigate_tab,
# inspect_filter_state, apply_table_filter, extract_table, scroll_to, inspect_columns,
# click, select_rows, download. Each action carries typed input validation (BROWSER_*_INVALID
# before any I/O), fail-closed selector miss (BROWSER_SELECTOR_NOT_FOUND, no retry),
# bounded output (extract: 10 MiB / 10 000 rows / 100 columns; download: 25 MiB), and
# evidence (PNG for UI actions, structured rows for extract/inspect, artifact ref for
# download) following the browser_native_filter.py pattern.
# @RELATION IMPLEMENTS -> [ScenarioExecution.BrowserProvider.Transport]
# @RELATION DEPENDS_ON -> [Plugin.Service.ScreenshotService]
# @PRE The page holds an authenticated Superset dashboard session opened by _launch_and_login; all
# inputs are admission-validated before reaching these flows.
# @POST Every action returns typed details dict for the transport outcome; invalid input raises
# ValueError("BROWSER_*_INVALID") before any locator work; a locator miss raises
# BrowserTransportSelectorNotFound (fail-closed, no retry).
# @INVARIANT Read-only: no SQL, no API writes; download is bounded to 25 MiB and stored as a
# server-owned artifact ref with sha256; extract_table is bounded to 10 000 rows, 100
# columns and 10 MiB output.
# @SIDE_EFFECT Clicks, scrolls and DOM reads inside the isolated browser context; no server-side
# data mutation; download captures bytes for provider-side storage.
# @RATIONALE Factoring these flows into a dedicated module keeps browser_transport.py below INV_7
# and mirrors the browser_native_filter.py isolation pattern — each action's selectors,
# validation and UI flow are co-located for review and test.
# @REJECTED Inlining all nine actions into browser_transport.py was rejected — the module would
# exceed INV_7 and lose the per-action selector/timeout constants that make each flow
# independently testable.
from __future__ import annotations
import json
from typing import Any
from src.core.logger import logger
from src.services.dashboard_testing.execution.providers.browser_native_filter import (
BrowserTransportSelectorNotFound,
)
_SRC = "ScenarioExecution.BrowserProvider.ReadOnlyActions"
# -- Limits (T034 spec) ---------------------------------------------------
_MAX_EXTRACT_ROWS = 10_000
_MAX_EXTRACT_COLUMNS = 100
_MAX_EXTRACT_OUTPUT_BYTES = 10 * 1024 * 1024 # 10 MiB
_MAX_DOWNLOAD_BYTES = 25 * 1024 * 1024 # 25 MiB
_MAX_CLICK_SELECTOR_LENGTH = 512
_MAX_SCROLL_SELECTOR_LENGTH = 512
_MAX_SELECT_ROWS = 100
_MAX_TAB_LENGTH = 128
_MAX_COLUMN_LENGTH = 256
_MAX_TABLE_FILTER_VALUE_LENGTH = 256
# -- Allowlists -----------------------------------------------------------
_ALLOWED_TABS = frozenset({
"overview", "charts", "data", "filters", "settings", "annotations",
"css", "properties", "json", "code", "sql", "table", "dashboard",
})
_ALLOWED_SCROLL_DIRECTIONS = frozenset({"up", "down", "left", "right", "top", "bottom"})
_ALLOWED_ARTIFACT_TYPES = frozenset({"xlsx", "csv", "pdf", "png", "json"})
# -- Multi-strategy locator candidates (Superset version drift expected) ---
_TAB_SELECTORS = (
'[data-test="tab-{tab}"]',
'.nav-item a[data-tab="{tab}"]',
'button[role="tab"][aria-controls*="{tab}"]',
'a[href*="{tab}"]',
)
_TABLE_FILTER_SELECTORS = (
'[data-test="table-filter"]',
".table-filter-container",
"th .ant-table-filter-column",
)
_TABLE_CONTAINER_SELECTORS = (
'[data-test="table-container"]',
".ant-table-wrapper",
".table-container",
".grid-content table",
)
_COLUMN_HEADER_SELECTORS = (
"th .ant-table-column-title",
"th span",
"th",
)
_ROW_CHECKBOX_SELECTORS = (
'td .ant-checkbox-input',
'td input[type="checkbox"]',
"td .row-select",
)
_DOWNLOAD_TRIGGER_SELECTORS = (
'[data-test="download-button"]',
'button:has-text("Download")',
'button:has-text("Export")',
'a[download]',
)
# =========================================================================
# Input validation (BROWSER_*_INVALID before any I/O)
# =========================================================================
# #region ScenarioExecution.BrowserProvider.ReadOnlyActions.Validate [C:3] [TYPE Function] [SEMANTICS provider,browser,readonly,validate]
# @ingroup ScenarioExecution
# @BRIEF Typed input validation for every round-2 read-only action; returns a stable code or None.
# @POST Returns None for well-formed input, or a BROWSER_*_INVALID code; never performs I/O.
def validate_readonly_action_input(action: str, action_input: dict[str, Any]) -> str | None:
if action == "navigate_tab":
tab = action_input.get("tab")
if not isinstance(tab, str) or not tab.strip() or len(tab) > _MAX_TAB_LENGTH:
return "BROWSER_NAVIGATE_TAB_INVALID"
hint = action_input.get("selector_hint")
if hint is not None and (not isinstance(hint, str) or not hint.strip()):
return "BROWSER_NAVIGATE_TAB_INVALID"
elif action == "apply_table_filter":
column = action_input.get("column")
if not isinstance(column, str) or not column.strip() or len(column) > _MAX_COLUMN_LENGTH:
return "BROWSER_TABLE_FILTER_INVALID"
value = action_input.get("value")
if value is not None and (not isinstance(value, str) or len(value) > _MAX_TABLE_FILTER_VALUE_LENGTH):
return "BROWSER_TABLE_FILTER_INVALID"
elif action == "extract_table":
max_rows = action_input.get("max_rows")
if max_rows is not None:
if not isinstance(max_rows, int) or max_rows <= 0 or max_rows > _MAX_EXTRACT_ROWS:
return "BROWSER_EXTRACT_BOUNDS_INVALID"
max_cols = action_input.get("max_columns")
if max_cols is not None:
if not isinstance(max_cols, int) or max_cols <= 0 or max_cols > _MAX_EXTRACT_COLUMNS:
return "BROWSER_EXTRACT_BOUNDS_INVALID"
elif action == "scroll_to":
selector = action_input.get("selector")
if not isinstance(selector, str) or not selector.strip() or len(selector) > _MAX_SCROLL_SELECTOR_LENGTH:
return "BROWSER_SCROLL_INVALID"
direction = action_input.get("direction")
if direction is not None and str(direction) not in _ALLOWED_SCROLL_DIRECTIONS:
return "BROWSER_SCROLL_INVALID"
elif action == "click":
selector = action_input.get("selector")
if not isinstance(selector, str) or not selector.strip() or len(selector) > _MAX_CLICK_SELECTOR_LENGTH:
return "BROWSER_CLICK_INVALID"
elif action == "select_rows":
keys = action_input.get("row_keys")
if (
not isinstance(keys, list)
or len(keys) == 0
or len(keys) > _MAX_SELECT_ROWS
or any(not isinstance(k, str) or not k.strip() for k in keys)
):
return "BROWSER_SELECT_ROWS_INVALID"
elif action == "download":
artifact_type = action_input.get("artifact_type")
if artifact_type is not None and str(artifact_type) not in _ALLOWED_ARTIFACT_TYPES:
return "BROWSER_DOWNLOAD_INVALID"
# inspect_filter_state, inspect_columns: no typed input required.
return None
# #endregion ScenarioExecution.BrowserProvider.ReadOnlyActions.Validate
# =========================================================================
# Selector resolution helpers
# =========================================================================
# #region ScenarioExecution.BrowserProvider.ReadOnlyActions.ResolveSelector [C:2] [TYPE Function] [SEMANTICS provider,browser,readonly,selector]
# @ingroup ScenarioExecution
# @BRIEF Resolve the first visible locator from a list of selector templates.
# @POST Returns a visible locator or None (caller maps to BrowserTransportSelectorNotFound).
async def _resolve_first_visible(service: Any, page: Any, selectors: tuple[str, ...], **fmt: str) -> Any:
candidates = [page.locator(selector.format(**fmt)) for selector in selectors]
return await service._find_first_visible_locator(candidates)
# #endregion ScenarioExecution.BrowserProvider.ReadOnlyActions.ResolveSelector
# =========================================================================
# Per-action UI flows
# =========================================================================
# #region ScenarioExecution.BrowserProvider.ReadOnlyActions.NavigateTab [C:3] [TYPE Function] [SEMANTICS provider,browser,navigate-tab]
# @ingroup ScenarioExecution
# @BRIEF Click the dashboard tab identified by typed input or selector_hint.
# @POST Returns typed details {tab, tab_navigated}; raises BrowserTransportSelectorNotFound on miss.
async def navigate_tab_flow(
service: Any, page: Any, action_input: dict[str, Any], *, timeout_seconds: float,
) -> dict[str, Any]:
tab = str(action_input.get("tab") or "").strip()
hint = action_input.get("selector_hint")
timeout_ms = int(timeout_seconds * 1000)
locator = None
if isinstance(hint, str) and hint.strip():
locator = await service._find_first_visible_locator([page.locator(str(hint))])
if locator is None:
locator = await _resolve_first_visible(service, page, _TAB_SELECTORS, tab=tab)
if locator is None:
logger.explore("Tab locator not found", src=_SRC, payload={"tab": tab}, error_code="BROWSER_SELECTOR_NOT_FOUND")
raise BrowserTransportSelectorNotFound("BROWSER_SELECTOR_NOT_FOUND")
await locator.click(timeout=timeout_ms)
await page.wait_for_load_state("domcontentloaded", timeout=timeout_ms)
logger.reflect("Dashboard tab navigated", src=_SRC, payload={"tab": tab})
return {"tab": tab, "tab_navigated": True}
# #endregion ScenarioExecution.BrowserProvider.ReadOnlyActions.NavigateTab
# #region ScenarioExecution.BrowserProvider.ReadOnlyActions.InspectFilterState [C:3] [TYPE Function] [SEMANTICS provider,browser,inspect-filter]
# @ingroup ScenarioExecution
# @BRIEF Read the current filter-bar state without modifying it (observe-only).
# @POST Returns typed details {filter_controls, filter_count}; no filter state is modified.
async def inspect_filter_state_flow(
service: Any, page: Any, *, timeout_seconds: float,
) -> dict[str, Any]:
from src.services.dashboard_testing.execution.providers.browser_native_filter import (
_FILTER_BAR_SELECTORS,
_FILTER_CONTROL_SELECTORS,
_SELECTED_VALUE_SELECTOR,
)
bar = await service._find_first_visible_locator([page.locator(s) for s in _FILTER_BAR_SELECTORS])
if bar is None:
logger.explore("Filter bar not found for inspect_filter_state", src=_SRC, error_code="BROWSER_SELECTOR_NOT_FOUND")
raise BrowserTransportSelectorNotFound("BROWSER_SELECTOR_NOT_FOUND")
controls: list[dict[str, Any]] = []
for selector in _FILTER_CONTROL_SELECTORS:
loc = bar.locator(selector)
count = min(await loc.count(), 50)
for i in range(count):
item = loc.nth(i)
if not await item.is_visible():
continue
text = str(await item.text_content() or "").strip()
selected = item.locator(_SELECTED_VALUE_SELECTOR)
sel_count = min(await selected.count(), 20)
values = []
for j in range(sel_count):
val = str(await selected.nth(j).text_content() or "").strip()
if val:
values.append(val)
controls.append({"text": text, "selected_values": values})
if controls:
break
logger.reflect("Filter state inspected", src=_SRC, payload={"filter_count": len(controls)})
return {"filter_controls": controls, "filter_count": len(controls)}
# #endregion ScenarioExecution.BrowserProvider.ReadOnlyActions.InspectFilterState
# #region ScenarioExecution.BrowserProvider.ReadOnlyActions.ApplyTableFilter [C:3] [TYPE Function] [SEMANTICS provider,browser,table-filter]
# @ingroup ScenarioExecution
# @BRIEF Apply a typed column/value filter to the dashboard table UI.
# @POST Returns typed details {column, value, table_filter_applied}; selector miss is typed.
async def apply_table_filter_flow(
service: Any, page: Any, action_input: dict[str, Any], *, timeout_seconds: float,
) -> dict[str, Any]:
column = str(action_input.get("column") or "").strip()
value = action_input.get("value")
timeout_ms = int(timeout_seconds * 1000)
hint = action_input.get("selector_hint")
locator = None
if isinstance(hint, str) and hint.strip():
locator = await service._find_first_visible_locator([page.locator(str(hint))])
if locator is None:
locator = await _resolve_first_visible(service, page, _TABLE_FILTER_SELECTORS)
if locator is None:
logger.explore("Table filter control not found", src=_SRC, payload={"column": column}, error_code="BROWSER_SELECTOR_NOT_FOUND")
raise BrowserTransportSelectorNotFound("BROWSER_SELECTOR_NOT_FOUND")
await locator.click(timeout=timeout_ms)
if isinstance(value, str) and value.strip():
input_loc = await service._find_first_visible_locator([
page.locator(".ant-table-filter-dropdown input"),
page.locator(".table-filter-input input"),
page.locator('input[type="search"]'),
])
if input_loc is not None:
await input_loc.fill(value, timeout=timeout_ms)
confirm = await service._find_first_visible_locator([
page.locator(".ant-table-filter-dropdown .ant-btn-primary"),
page.locator('button:has-text("OK")'),
page.locator('button:has-text("Apply")'),
])
if confirm is not None:
await confirm.click(timeout=timeout_ms)
await page.wait_for_load_state("domcontentloaded", timeout=timeout_ms)
logger.reflect("Table filter applied", src=_SRC, payload={"column": column, "value": value})
return {"column": column, "value": value, "table_filter_applied": True}
# #endregion ScenarioExecution.BrowserProvider.ReadOnlyActions.ApplyTableFilter
# #region ScenarioExecution.BrowserProvider.ReadOnlyActions.ExtractTable [C:4] [TYPE Function] [SEMANTICS provider,browser,extract,table,bounded]
# @ingroup ScenarioExecution
# @BRIEF Extract bounded table data from the dashboard DOM (10 000 rows, 100 columns, 10 MiB).
# @POST Returns typed details {columns, rows, row_count, column_count}; oversized output raises
# ValueError("BROWSER_EXTRACT_TOO_LARGE") before evidence is produced.
async def extract_table_flow(
service: Any, page: Any, action_input: dict[str, Any], *, timeout_seconds: float,
) -> dict[str, Any]:
max_rows = int(action_input.get("max_rows") or _MAX_EXTRACT_ROWS)
max_cols = int(action_input.get("max_columns") or _MAX_EXTRACT_COLUMNS)
hint = action_input.get("selector_hint")
table_loc = None
if isinstance(hint, str) and hint.strip():
table_loc = await service._find_first_visible_locator([page.locator(str(hint))])
if table_loc is None:
table_loc = await _resolve_first_visible(service, page, _TABLE_CONTAINER_SELECTORS)
if table_loc is None:
logger.explore("Table container not found for extract_table", src=_SRC, error_code="BROWSER_SELECTOR_NOT_FOUND")
raise BrowserTransportSelectorNotFound("BROWSER_SELECTOR_NOT_FOUND")
raw = await table_loc.evaluate(
"""(el, opts) => {
const maxRows = opts.maxRows; const maxCols = opts.maxCols;
const headers = Array.from(el.querySelectorAll('thead th, tr:first-child th, th'))
.slice(0, maxCols)
.map(th => (th.innerText || '').trim());
const bodyRows = Array.from(el.querySelectorAll('tbody tr, tr'))
.slice(0, maxRows);
const rows = bodyRows.map(tr =>
Array.from(tr.querySelectorAll('td')).slice(0, maxCols).map(td => (td.innerText || '').trim())
);
return { columns: headers, rows: rows };
}""",
{"maxRows": max_rows, "maxCols": max_cols},
)
columns = list(raw.get("columns") or [])[:max_cols]
rows = list(raw.get("rows") or [])[:max_rows]
payload = {"columns": columns, "rows": rows}
serialized = json.dumps(payload, separators=(",", ":"), ensure_ascii=False)
byte_size = len(serialized.encode("utf-8"))
if byte_size > _MAX_EXTRACT_OUTPUT_BYTES:
logger.explore(
"Extract table output exceeds the 10 MiB limit", src=_SRC,
payload={"bytes": byte_size, "rows": len(rows), "columns": len(columns)},
error_code="BROWSER_EXTRACT_TOO_LARGE",
)
raise ValueError("BROWSER_EXTRACT_TOO_LARGE")
logger.reflect(
"Table extracted", src=_SRC,
payload={"row_count": len(rows), "column_count": len(columns), "bytes": byte_size},
)
return {
"columns": columns,
"rows": rows,
"row_count": len(rows),
"column_count": len(columns),
"byte_size": byte_size,
}
# #endregion ScenarioExecution.BrowserProvider.ReadOnlyActions.ExtractTable
# #region ScenarioExecution.BrowserProvider.ReadOnlyActions.ScrollTo [C:2] [TYPE Function] [SEMANTICS provider,browser,scroll]
# @ingroup ScenarioExecution
# @BRIEF Scroll the page or a specific element into view.
# @POST Returns typed details {selector, direction, scrolled}; selector miss is typed.
async def scroll_to_flow(
service: Any, page: Any, action_input: dict[str, Any], *, timeout_seconds: float,
) -> dict[str, Any]:
selector = str(action_input.get("selector") or "").strip()
direction = str(action_input.get("direction") or "down")
timeout_ms = int(timeout_seconds * 1000)
target = await service._find_first_visible_locator([page.locator(selector)])
if target is None:
logger.explore("Scroll target not found", src=_SRC, payload={"selector": selector}, error_code="BROWSER_SELECTOR_NOT_FOUND")
raise BrowserTransportSelectorNotFound("BROWSER_SELECTOR_NOT_FOUND")
await target.scroll_into_view_if_needed(timeout=timeout_ms)
logger.reflect("Page scrolled to target", src=_SRC, payload={"selector": selector, "direction": direction})
return {"selector": selector, "direction": direction, "scrolled": True}
# #endregion ScenarioExecution.BrowserProvider.ReadOnlyActions.ScrollTo
# #region ScenarioExecution.BrowserProvider.ReadOnlyActions.InspectColumns [C:3] [TYPE Function] [SEMANTICS provider,browser,inspect-columns]
# @ingroup ScenarioExecution
# @BRIEF Read visible table column headers (observe-only, bounded).
# @POST Returns typed details {columns, column_count}; selector miss is typed.
async def inspect_columns_flow(
service: Any, page: Any, action_input: dict[str, Any], *, timeout_seconds: float,
) -> dict[str, Any]:
hint = action_input.get("selector_hint")
table_loc = None
if isinstance(hint, str) and hint.strip():
table_loc = await service._find_first_visible_locator([page.locator(str(hint))])
if table_loc is None:
table_loc = await _resolve_first_visible(service, page, _TABLE_CONTAINER_SELECTORS)
if table_loc is None:
logger.explore("Table container not found for inspect_columns", src=_SRC, error_code="BROWSER_SELECTOR_NOT_FOUND")
raise BrowserTransportSelectorNotFound("BROWSER_SELECTOR_NOT_FOUND")
columns: list[str] = []
for selector in _COLUMN_HEADER_SELECTORS:
loc = table_loc.locator(selector)
count = min(await loc.count(), _MAX_EXTRACT_COLUMNS)
for i in range(count):
text = str(await loc.nth(i).text_content() or "").strip()
if text:
columns.append(text)
if columns:
break
logger.reflect("Columns inspected", src=_SRC, payload={"column_count": len(columns)})
return {"columns": columns, "column_count": len(columns)}
# #endregion ScenarioExecution.BrowserProvider.ReadOnlyActions.InspectColumns
# #region ScenarioExecution.BrowserProvider.ReadOnlyActions.Click [C:2] [TYPE Function] [SEMANTICS provider,browser,click]
# @ingroup ScenarioExecution
# @BRIEF Click a typed selector target (read-only interaction).
# @POST Returns typed details {selector, clicked}; selector miss is typed.
async def click_flow(
service: Any, page: Any, action_input: dict[str, Any], *, timeout_seconds: float,
) -> dict[str, Any]:
selector = str(action_input.get("selector") or "").strip()
hint = action_input.get("selector_hint")
timeout_ms = int(timeout_seconds * 1000)
locator = None
if isinstance(hint, str) and hint.strip():
locator = await service._find_first_visible_locator([page.locator(str(hint))])
if locator is None:
locator = await service._find_first_visible_locator([page.locator(selector)])
if locator is None:
logger.explore("Click target not found", src=_SRC, payload={"selector": selector}, error_code="BROWSER_SELECTOR_NOT_FOUND")
raise BrowserTransportSelectorNotFound("BROWSER_SELECTOR_NOT_FOUND")
await locator.click(timeout=timeout_ms)
logger.reflect("Click target activated", src=_SRC, payload={"selector": selector})
return {"selector": selector, "clicked": True}
# #endregion ScenarioExecution.BrowserProvider.ReadOnlyActions.Click
# #region ScenarioExecution.BrowserProvider.ReadOnlyActions.SelectRows [C:3] [TYPE Function] [SEMANTICS provider,browser,select-rows]
# @ingroup ScenarioExecution
# @BRIEF Select table rows by typed row_keys (selection-only, never proves mutation).
# @POST Returns typed details {row_keys, selected_count, rows_selected}; selector miss is typed.
async def select_rows_flow(
service: Any, page: Any, action_input: dict[str, Any], *, timeout_seconds: float,
) -> dict[str, Any]:
row_keys = [str(k).strip() for k in action_input.get("row_keys") or []]
timeout_ms = int(timeout_seconds * 1000)
selected = 0
for key in row_keys:
candidates = [
page.get_by_text(key, exact=True),
page.locator(f'tr:has-text("{key}")'),
]
row_loc = await service._find_first_visible_locator(candidates)
if row_loc is None:
logger.explore("Row key not found for select_rows", src=_SRC, payload={"key": key}, error_code="BROWSER_SELECTOR_NOT_FOUND")
raise BrowserTransportSelectorNotFound("BROWSER_SELECTOR_NOT_FOUND")
checkbox = await service._find_first_visible_locator([
row_loc.locator(s) for s in _ROW_CHECKBOX_SELECTORS
])
if checkbox is None:
await row_loc.click(timeout=timeout_ms)
else:
await checkbox.click(timeout=timeout_ms)
selected += 1
logger.reflect("Rows selected", src=_SRC, payload={"selected_count": selected})
return {"row_keys": row_keys, "selected_count": selected, "rows_selected": True}
# #endregion ScenarioExecution.BrowserProvider.ReadOnlyActions.SelectRows
# #region ScenarioExecution.BrowserProvider.ReadOnlyActions.Download [C:4] [TYPE Function] [SEMANTICS provider,browser,download,artifact,bounded]
# @ingroup ScenarioExecution
# @BRIEF Bounded download (25 MiB limit): trigger and return bytes for provider-side storage.
# @POST Returns (details, download_bytes); oversized downloads raise
# ValueError("BROWSER_DOWNLOAD_TOO_LARGE") before returning; the transport stays
# storage-agnostic — the provider stores the artifact ref + sha256 separately.
async def download_flow(
service: Any,
page: Any,
action_input: dict[str, Any],
*,
timeout_seconds: float,
) -> tuple[dict[str, Any], bytes]:
artifact_type = str(action_input.get("artifact_type") or "xlsx")
hint = action_input.get("selector_hint")
timeout_ms = int(timeout_seconds * 1000)
trigger = None
if isinstance(hint, str) and hint.strip():
trigger = await service._find_first_visible_locator([page.locator(str(hint))])
if trigger is None:
trigger = await _resolve_first_visible(service, page, _DOWNLOAD_TRIGGER_SELECTORS)
if trigger is None:
logger.explore("Download trigger not found", src=_SRC, error_code="BROWSER_SELECTOR_NOT_FOUND")
raise BrowserTransportSelectorNotFound("BROWSER_SELECTOR_NOT_FOUND")
async with page.expect_download(timeout=timeout_ms) as download_info:
await trigger.click(timeout=timeout_ms)
download = await download_info.value
data = b""
path = await download.path()
if path is not None:
with open(str(path), "rb") as fh:
data = fh.read()
if len(data) > _MAX_DOWNLOAD_BYTES:
logger.explore(
"Download exceeds the 25 MiB limit", src=_SRC,
payload={"bytes": len(data), "artifact_type": artifact_type},
error_code="BROWSER_DOWNLOAD_TOO_LARGE",
)
raise ValueError("BROWSER_DOWNLOAD_TOO_LARGE")
logger.reflect(
"Download captured (storage is the provider's)", src=_SRC,
payload={"artifact_type": artifact_type, "bytes": len(data)},
)
details = {
"artifact_type": artifact_type,
"byte_size": len(data),
"downloaded": True,
}
return details, data
# #endregion ScenarioExecution.BrowserProvider.ReadOnlyActions.Download
# #region ScenarioExecution.BrowserProvider.ReadOnlyActions.Dispatch [C:3] [TYPE Function] [SEMANTICS provider,browser,readonly,dispatch]
# @ingroup ScenarioExecution
# @BRIEF Dispatch a round-2 read-only action to its typed flow; returns (details, download_bytes).
# @POST Returns a (details_dict, None) pair for all actions except download, which returns
# (details_dict, bytes); unsupported actions raise ValueError("BROWSER_ACTION_NOT_SUPPORTED").
async def run_readonly_action_flow(
service: Any,
page: Any,
action: str,
action_input: dict[str, Any],
*,
timeout_seconds: float,
) -> tuple[dict[str, Any], bytes | None]:
if action == "navigate_tab":
return await navigate_tab_flow(service, page, action_input, timeout_seconds=timeout_seconds), None
if action == "inspect_filter_state":
return await inspect_filter_state_flow(service, page, timeout_seconds=timeout_seconds), None
if action == "apply_table_filter":
return await apply_table_filter_flow(service, page, action_input, timeout_seconds=timeout_seconds), None
if action == "extract_table":
return await extract_table_flow(service, page, action_input, timeout_seconds=timeout_seconds), None
if action == "scroll_to":
return await scroll_to_flow(service, page, action_input, timeout_seconds=timeout_seconds), None
if action == "inspect_columns":
return await inspect_columns_flow(service, page, action_input, timeout_seconds=timeout_seconds), None
if action == "click":
return await click_flow(service, page, action_input, timeout_seconds=timeout_seconds), None
if action == "select_rows":
return await select_rows_flow(service, page, action_input, timeout_seconds=timeout_seconds), None
if action == "download":
return await download_flow(service, page, action_input, timeout_seconds=timeout_seconds)
raise ValueError("BROWSER_ACTION_NOT_SUPPORTED")
# #endregion ScenarioExecution.BrowserProvider.ReadOnlyActions.Dispatch
# #endregion ScenarioExecution.BrowserProvider.ReadOnlyActions

View File

@@ -0,0 +1,501 @@
# #region ScenarioExecution.BrowserProvider.Session [C:5] [TYPE Module] [SEMANTICS scenario,execution,provider,browser,session,checkpoint,replay,leak-guard]
# @ingroup ScenarioExecution
# @BRIEF Run-scoped browser session registry on the shared provider loop: one isolated context per
# run lives across all browser steps, a browser-safe checkpoint is folded after every step,
# cross-process recovery replays the persisted checkpoint, and close paths are guaranteed
# (DG-1 amendment 2026-09-12, 044 T034 round 1).
# @RELATION IMPLEMENTS -> [ScenarioExecution.BrowserProvider]
# @RELATION DEPENDS_ON -> [ScenarioExecution.ProviderRuntime.Engine]
# @RELATION DEPENDS_ON -> [ScenarioExecution.BrowserProvider.Transport]
# @RELATION DEPENDS_ON -> [Models.ScenarioExecution.StepRun]
# @PRE The manager is bound to a session-capable transport (open_session / execute_in_session /
# close_session) and the shared provider event loop; the provider holds the per-run guard
# across prepare + submit so concurrent steps of one run serialize on the run context (DG-1 §4).
# @POST prepare_step returns exactly one live session per run_id: an in-memory hit is reused, a
# persisted checkpoint replays open + re-apply before the step, and browser history without
# any declared checkpoint raises BrowserCheckpointMissing (fail-closed, never a silent
# stateless session). close() is idempotent and best-effort on every path.
# @INVARIANT A context is never shared across runs; a dead context is never revived — recovery is a
# new context plus checkpoint replay from scenario_step_runs.step_outcome; the registry
# lives in process memory, so process death kills every context and only replay recovers.
# @SIDE_EFFECT Live browser/context handles bound to the provider loop, an in-process registry, and
# read-only checkpoint queries against scenario_step_runs.
# @RATIONALE DG-1 (044 spec 2026-09-12): native filter state applied by any run step must persist
# for all later steps — per-step isolation broke filter continuity and risked false-PASS.
# Round 1 keeps the capacity lease per step, so the provider has no terminal-run channel:
# the session survives per-step lease release and closes via explicit close(run_id)
# (B-wave dispatch finalizer), the dead-context guard, the idle TTL reaper and close_all.
# @REJECTED Closing the session at per-step lease release was rejected — it recreates the per-step
# isolation DG-1 removed (launch+login per step, filter state lost). Reviving a dead
# context object was rejected — process death kills the loop-bound context; only
# checkpoint replay reconstructs state. A warm cross-run context pool was rejected by
# T034 (state leakage between runs).
from __future__ import annotations
import copy
import threading
import time
import weakref
from collections.abc import Callable, Iterator
from contextlib import contextmanager
from typing import Any
from src.core.database import SessionLocal
from src.core.logger import logger
from src.models.scenario_run import ScenarioStepRun
from src.services.dashboard_testing.execution.provider_runtime import ProviderEventLoop
_SRC = "ScenarioExecution.BrowserProvider.Session"
_DEFAULT_IDLE_TIMEOUT_SECONDS = 900.0
_CLOSE_TIMEOUT_SECONDS = 10.0
_FILTER_REPLAY_KEYS = frozenset({"filter_id", "filter_name", "column", "selector_hint", "values", "wait_state"})
# #region ScenarioExecution.BrowserProvider.Session.ManagerRegistry [C:3] [TYPE Module] [SEMANTICS provider,browser,session,registry,finalizer]
# @ingroup ScenarioExecution
# @BRIEF Weak process registry of live session managers so server lifecycle paths (dispatch
# terminal finalizer, cancellation) can close a run's session without holding a provider
# reference (B-wave finalizer hook deferred from C1 round 1).
# @INVARIANT The registry holds weak references only: it never extends a manager's lifetime, and
# close_run_sessions never raises — session cleanup is best-effort and fail-closed.
# @POST close_run_sessions returns the number of managers that dropped a session for the run.
_MANAGERS: weakref.WeakSet[BrowserSessionManager] = weakref.WeakSet()
_MANAGERS_LOCK = threading.Lock()
def _register_manager(manager: BrowserSessionManager) -> None:
with _MANAGERS_LOCK:
_MANAGERS.add(manager)
def close_run_sessions(run_id: str, *, reason: str) -> int:
closed = 0
with _MANAGERS_LOCK:
managers = list(_MANAGERS)
for manager in managers:
try:
if manager.close(run_id, reason=reason):
closed += 1
except Exception as exc: # pragma: no cover - close() is designed not to raise
logger.explore(
"Session manager close raised; continuing fan-out", src=_SRC,
error_code="BROWSER_SESSION_CLOSE_FAILED", payload={"run_id": run_id, "reason": reason}, error=repr(exc),
)
if closed:
logger.reflect(
"Run-scoped browser sessions closed by lifecycle finalizer", src=_SRC,
payload={"run_id": run_id, "reason": reason, "closed": closed},
)
return closed
# #endregion ScenarioExecution.BrowserProvider.Session.ManagerRegistry
# #region ScenarioExecution.BrowserProvider.Session.Errors [C:2] [TYPE Class] [SEMANTICS provider,browser,session,checkpoint,error]
# @ingroup ScenarioExecution
# @BRIEF Typed fail-closed recovery refusal: browser history exists but no declared checkpoint.
class BrowserCheckpointMissing(Exception):
"""Raised when a run needs browser-state recovery but no declared checkpoint exists."""
# #endregion ScenarioExecution.BrowserProvider.Session.Errors
# #region ScenarioExecution.BrowserProvider.Session.Handle [C:3] [TYPE Class] [SEMANTICS provider,browser,session,handle]
# @ingroup ScenarioExecution
# @BRIEF Runtime handle: run identity, transport-owned context, and the folded browser checkpoint.
class BrowserSession:
def __init__(
self,
*,
run_id: str,
lease_id: str,
dashboard_id: int,
transport_session: Any,
checkpoint: dict[str, Any],
created_at: float,
last_used_at: float,
replayed: bool = False,
) -> None:
self.run_id = run_id
self.lease_id = lease_id
self.dashboard_id = dashboard_id
self.transport_session = transport_session
self.checkpoint = checkpoint
self.created_at = created_at
self.last_used_at = last_used_at
self.replayed = replayed
class _PreparedStep:
__slots__ = ("kind", "run_id", "lease_id", "dashboard_id", "session", "replay_state")
def __init__(
self,
*,
kind: str,
run_id: str,
lease_id: str,
dashboard_id: int,
session: BrowserSession | None = None,
replay_state: dict[str, Any] | None = None,
) -> None:
self.kind = kind
self.run_id = run_id
self.lease_id = lease_id
self.dashboard_id = dashboard_id
self.session = session
self.replay_state = replay_state
def _initial_checkpoint(dashboard_id: int) -> dict[str, Any]:
return {
"dashboard_id": dashboard_id,
"checkpoint_seq": 0,
"native_filter_state": [],
"active_tab": None,
"wait_states": [],
"table_filter_state": [],
"filter_state_observed": None,
}
# #endregion ScenarioExecution.BrowserProvider.Session.Handle
# #region ScenarioExecution.BrowserProvider.Session.Checkpoint [C:4] [TYPE Function] [SEMANTICS provider,browser,session,checkpoint,native-filter,tab,table-filter]
# @ingroup ScenarioExecution
# @BRIEF Fold one executed step into the browser-safe checkpoint (extendable dict per DG-1 §3).
# @PRE checkpoint is the session's current state (or None for a fresh context); details is the
# transport outcome details of the just-executed action.
# @POST Returns a NEW dict with checkpoint_seq incremented; apply_native_filter outcomes upsert a
# native_filter_state entry carrying the UX-1 summary (filter_target, applied_values,
# applied_mode) plus the exact replay_input; wait_for_state appends to wait_states;
# navigate_tab stamps active_tab; apply_table_filter upserts table_filter_state;
# inspect_filter_state stamps filter_state_observed (round 2, diagnostic — not replayed).
# @INVARIANT Pure function: the input checkpoint is never mutated; unknown actions only bump the
# sequence so the persisted slice always reflects the latest executed step.
def fold_checkpoint_state(
checkpoint: dict[str, Any] | None,
*,
dashboard_id: int,
action: str,
action_input: dict[str, Any],
details: dict[str, Any],
) -> dict[str, Any]:
state = copy.deepcopy(checkpoint) if isinstance(checkpoint, dict) and checkpoint else _initial_checkpoint(dashboard_id)
state["dashboard_id"] = dashboard_id
state["checkpoint_seq"] = int(state.get("checkpoint_seq") or 0) + 1
inputs = action_input if isinstance(action_input, dict) else {}
if action == "apply_native_filter" and isinstance(details, dict) and details.get("applied"):
entry = {
"filter_target": details.get("filter_target"),
"applied_values": [str(value) for value in details.get("applied_values") or []],
"applied_mode": str(details.get("applied_mode") or ""),
"replay_input": {key: copy.deepcopy(value) for key, value in inputs.items() if key in _FILTER_REPLAY_KEYS and value is not None},
}
filters = [
existing
for existing in state.get("native_filter_state") or []
if isinstance(existing, dict) and existing.get("filter_target") != entry["filter_target"]
]
filters.append(entry)
state["native_filter_state"] = filters
elif action == "wait_for_state":
state["wait_states"] = [*(state.get("wait_states") or []), str(inputs.get("state") or "load")]
elif action == "navigate_tab" and isinstance(details, dict) and details.get("tab_navigated"):
tab = str(details.get("tab") or "").strip()
if tab:
state["active_tab"] = tab
elif action == "apply_table_filter" and isinstance(details, dict) and details.get("table_filter_applied"):
entry = {"column": str(details.get("column") or ""), "value": details.get("value")}
table_filters = [
existing
for existing in state.get("table_filter_state") or []
if isinstance(existing, dict) and existing.get("column") != entry["column"]
]
table_filters.append(entry)
state["table_filter_state"] = table_filters
elif action == "inspect_filter_state" and isinstance(details, dict) and "filter_count" in details:
state["filter_state_observed"] = {
"filter_count": int(details.get("filter_count") or 0),
"controls": copy.deepcopy(details.get("filter_controls") or [])[:50],
}
return state
# #endregion ScenarioExecution.BrowserProvider.Session.Checkpoint
# #region ScenarioExecution.BrowserProvider.Session.ReplayLookup [C:4] [TYPE Function] [SEMANTICS provider,browser,session,checkpoint,replay,recovery]
# @ingroup ScenarioExecution
# @BRIEF Read the latest persisted browser checkpoint of one run from scenario_step_runs.
# @PRE run_id identifies persisted step rows; step_outcome carries tool="browser" and, for passed
# session steps, the stamped browser_checkpoint dict.
# @POST Returns ("none", None) when the run has no browser steps (fresh context is honest),
# ("checkpoint", state) from the latest checkpointed step, or ("missing", None) when browser
# history exists without any declared checkpoint (fail-closed per DG-1 §3).
# @INVARIANT Read-only; a checkpoint is accepted only from a row whose step_outcome carries the
# stamped dict — a crashed step that never stamped yields no recovery authority.
def load_persisted_browser_checkpoint(
run_id: str,
*,
session_factory: Callable[[], Any] = SessionLocal,
) -> tuple[str, dict[str, Any] | None]:
with session_factory() as db:
rows = (
db.query(ScenarioStepRun)
.filter(ScenarioStepRun.run_id == run_id)
.order_by(ScenarioStepRun.step_position.desc(), ScenarioStepRun.attempt.desc())
.all()
)
saw_browser = False
for row in rows:
outcome = row.step_outcome if isinstance(row.step_outcome, dict) else {}
if outcome.get("tool") != "browser":
continue
saw_browser = True
checkpoint = outcome.get("browser_checkpoint")
if isinstance(checkpoint, dict) and checkpoint.get("dashboard_id") is not None:
logger.reflect(
"Persisted browser checkpoint selected for replay", src=_SRC,
payload={"run_id": run_id, "logical_step_id": row.logical_step_id, "checkpoint_seq": checkpoint.get("checkpoint_seq")},
)
return "checkpoint", checkpoint
if saw_browser:
logger.explore(
"Run has browser history without a declared checkpoint", src=_SRC,
claim="POST: replay or typed missing", error_code="BROWSER_CHECKPOINT_MISSING",
payload={"run_id": run_id},
)
return "missing", None
return "none", None
# #endregion ScenarioExecution.BrowserProvider.Session.ReplayLookup
# #region ScenarioExecution.BrowserProvider.Session.Registry [C:5] [TYPE Class] [SEMANTICS provider,browser,session,registry,leak-guard]
# @ingroup ScenarioExecution
# @BRIEF Own the run-scoped sessions: acquire/reuse, replay recovery, fold, close, idle reaping.
# @RELATION DEPENDS_ON -> [ScenarioExecution.ProviderRuntime.Engine]
# @PRE Constructed with a session-capable transport and the shared provider event loop.
# @POST At most one live session per run_id; every close path is idempotent and never raises.
# @INVARIANT Registry mutations are serialized by the registry lock; concurrent steps of one run
# are serialized by the per-run guard held across prepare + loop execution.
class BrowserSessionManager:
def __init__(
self,
*,
transport: Any,
event_loop: ProviderEventLoop,
checkpoint_loader: Callable[..., tuple[str, dict[str, Any] | None]] | None = None,
idle_timeout_seconds: float = _DEFAULT_IDLE_TIMEOUT_SECONDS,
clock: Callable[[], float] = time.monotonic,
) -> None:
if idle_timeout_seconds <= 0:
raise ValueError("BROWSER_SESSION_IDLE_TIMEOUT_INVALID")
self._transport = transport
self._event_loop = event_loop
self._checkpoint_loader = checkpoint_loader or load_persisted_browser_checkpoint
self._idle_timeout = float(idle_timeout_seconds)
self._clock = clock
self._sessions: dict[str, BrowserSession] = {}
self._registry_lock = threading.Lock()
self._run_locks: dict[str, threading.Lock] = {}
_register_manager(self)
def active_run_ids(self) -> list[str]:
with self._registry_lock:
return sorted(self._sessions)
# #region ScenarioExecution.BrowserProvider.Session.Registry.Guard [C:3] [TYPE Function] [SEMANTICS provider,browser,session,serialization]
# @ingroup ScenarioExecution
# @BRIEF Per-run serialization guard (DG-1 §4): concurrent steps of one run are ordered.
@contextmanager
def run_guard(self, run_id: str) -> Iterator[None]:
with self._registry_lock:
lock = self._run_locks.setdefault(run_id, threading.Lock())
lock.acquire()
try:
yield
finally:
lock.release()
# #endregion ScenarioExecution.BrowserProvider.Session.Registry.Guard
# #region ScenarioExecution.BrowserProvider.Session.Registry.Acquire [C:5] [TYPE Function] [SEMANTICS provider,browser,session,acquire,replay]
# @ingroup ScenarioExecution
# @BRIEF Resolve the step plan: reuse the live session, replay the persisted checkpoint, or open fresh.
# @PRE The caller holds run_guard(run_id); lease_id is the current step's capacity lease
# (provenance only — the session outlives the per-step lease).
# @POST A reuse plan carries the live session; a replay/fresh plan opens on the loop inside
# execute_prepared; browser history without a checkpoint or a checkpoint pinned to another
# dashboard raises BrowserCheckpointMissing before any I/O.
def prepare_step(self, *, run_id: str, lease_id: str, dashboard_id: int) -> _PreparedStep:
self.reap_idle()
with self._registry_lock:
session = self._sessions.get(run_id)
if session is not None:
session.lease_id = lease_id
logger.reason(
"Reusing the run-scoped browser session", src=_SRC,
payload={"run_id": run_id, "dashboard_id": session.dashboard_id, "replayed": session.replayed},
)
return _PreparedStep(kind="reuse", run_id=run_id, lease_id=lease_id, dashboard_id=dashboard_id, session=session)
kind, state = self._checkpoint_loader(run_id)
if kind == "missing":
logger.explore(
"Browser recovery requires a declared checkpoint", src=_SRC,
claim="POST: replay or typed missing", error_code="BROWSER_CHECKPOINT_MISSING",
payload={"run_id": run_id},
)
raise BrowserCheckpointMissing("BROWSER_CHECKPOINT_MISSING")
if kind == "checkpoint":
filters = state.get("native_filter_state") if isinstance(state, dict) else None
if (
not isinstance(state, dict)
or state.get("dashboard_id") != dashboard_id
or any(not isinstance(entry, dict) or not isinstance(entry.get("replay_input"), dict) for entry in filters or [])
):
logger.explore(
"Persisted checkpoint cannot authorize replay for this binding", src=_SRC,
claim="POST: replay or typed missing", error_code="BROWSER_CHECKPOINT_MISSING",
payload={"run_id": run_id, "dashboard_id": dashboard_id},
)
raise BrowserCheckpointMissing("BROWSER_CHECKPOINT_MISSING")
return _PreparedStep(kind="replay", run_id=run_id, lease_id=lease_id, dashboard_id=dashboard_id, replay_state=state)
return _PreparedStep(kind="fresh", run_id=run_id, lease_id=lease_id, dashboard_id=dashboard_id)
# #endregion ScenarioExecution.BrowserProvider.Session.Registry.Acquire
# #region ScenarioExecution.BrowserProvider.Session.Registry.Execute [C:5] [TYPE Function] [SEMANTICS provider,browser,session,execute,checkpoint]
# @ingroup ScenarioExecution
# @BRIEF Coroutine (provider loop): open/replay on first use, execute the action, fold the checkpoint.
# @PRE prepared comes from prepare_step under the run guard; runs entirely on the provider loop.
# @POST Returns (transport outcome, stamped checkpoint copy); the session is registered before
# replay so a mid-step cancellation is always closable by the leak-guard; a failed replay
# discards and closes the fresh context before re-raising.
async def execute_prepared(
self,
prepared: _PreparedStep,
action: str,
*,
action_input: dict[str, Any],
timeout_seconds: float,
) -> tuple[Any, dict[str, Any]]:
session = prepared.session or await self._open_session(prepared, timeout_seconds=timeout_seconds)
outcome = await self._transport.execute_in_session(
session.transport_session,
action,
action_input=action_input,
timeout_seconds=timeout_seconds,
)
session.checkpoint = fold_checkpoint_state(
session.checkpoint,
dashboard_id=session.dashboard_id,
action=action,
action_input=action_input,
details=outcome.details if isinstance(outcome.details, dict) else {},
)
session.last_used_at = self._clock()
logger.reflect(
"Browser step folded into the run checkpoint", src=_SRC,
payload={"run_id": session.run_id, "action": action, "checkpoint_seq": session.checkpoint.get("checkpoint_seq"), "filters": len(session.checkpoint.get("native_filter_state") or [])},
)
return outcome, copy.deepcopy(session.checkpoint)
# #endregion ScenarioExecution.BrowserProvider.Session.Registry.Execute
# #region ScenarioExecution.BrowserProvider.Session.Registry.Open [C:4] [TYPE Function] [SEMANTICS provider,browser,session,open,replay]
# @ingroup ScenarioExecution
# @BRIEF Open the transport context, register early (leak-guard visibility), then replay state.
async def _open_session(self, prepared: _PreparedStep, *, timeout_seconds: float) -> BrowserSession:
handle = await self._transport.open_session(prepared.dashboard_id, timeout_seconds=timeout_seconds)
now = self._clock()
session = BrowserSession(
run_id=prepared.run_id,
lease_id=prepared.lease_id,
dashboard_id=prepared.dashboard_id,
transport_session=handle,
checkpoint=copy.deepcopy(prepared.replay_state) if prepared.replay_state is not None else _initial_checkpoint(prepared.dashboard_id),
created_at=now,
last_used_at=now,
)
with self._registry_lock:
self._sessions[prepared.run_id] = session
logger.reflect(
"Run-scoped browser session opened", src=_SRC,
payload={"run_id": prepared.run_id, "dashboard_id": prepared.dashboard_id, "kind": prepared.kind},
)
if prepared.replay_state is not None:
try:
for entry in prepared.replay_state.get("native_filter_state") or []:
await self._transport.execute_in_session(
handle,
"apply_native_filter",
action_input=copy.deepcopy(entry["replay_input"]),
timeout_seconds=timeout_seconds,
)
session.replayed = True
logger.reflect(
"Browser checkpoint replayed into a fresh context", src=_SRC,
payload={"run_id": prepared.run_id, "filters": len(prepared.replay_state.get("native_filter_state") or [])},
)
except Exception as exc:
with self._registry_lock:
self._sessions.pop(prepared.run_id, None)
logger.explore(
"Checkpoint replay failed; fresh context abandoned", src=_SRC,
error_code="BROWSER_CHECKPOINT_REPLAY_FAILED",
payload={"run_id": prepared.run_id}, error=repr(exc),
)
await self._close_handle(handle)
raise
return session
# #endregion ScenarioExecution.BrowserProvider.Session.Registry.Open
# #region ScenarioExecution.BrowserProvider.Session.Registry.Close [C:4] [TYPE Function] [SEMANTICS provider,browser,session,close,leak-guard]
# @ingroup ScenarioExecution
# @BRIEF Leak-guard close paths: explicit terminal close, dead-context close, idle reap, close_all.
# @POST close() is idempotent and never raises: the registry entry is dropped first and the
# transport close is best-effort — a dead loop abandons the loop-bound context by design.
def close(self, run_id: str, *, reason: str = "terminal") -> bool:
with self._registry_lock:
session = self._sessions.pop(run_id, None)
if session is None:
return False
try:
self._event_loop.submit(lambda: self._close_handle(session.transport_session), timeout=_CLOSE_TIMEOUT_SECONDS)
except Exception as exc:
logger.explore(
"Browser session close could not reach the loop; context dies with the loop", src=_SRC,
error_code="BROWSER_SESSION_CLOSE_FAILED",
payload={"run_id": run_id, "reason": reason}, error=repr(exc),
)
return True
logger.reflect("Run-scoped browser session closed", src=_SRC, payload={"run_id": run_id, "reason": reason})
return True
def reap_idle(self) -> int:
now = self._clock()
with self._registry_lock:
stale = [
run_id
for run_id, session in self._sessions.items()
if now - session.last_used_at > self._idle_timeout
]
closed = 0
for run_id in stale:
if self.close(run_id, reason="idle_timeout"):
closed += 1
return closed
def close_all(self, *, reason: str = "shutdown") -> int:
with self._registry_lock:
run_ids = list(self._sessions)
closed = 0
for run_id in run_ids:
if self.close(run_id, reason=reason):
closed += 1
return closed
async def _close_handle(self, handle: Any) -> None:
try:
await self._transport.close_session(handle)
except Exception as exc:
logger.explore(
"Transport session close raised; handle abandoned", src=_SRC,
error_code="BROWSER_SESSION_CLOSE_FAILED", error=repr(exc),
)
# #endregion ScenarioExecution.BrowserProvider.Session.Registry.Close
# #endregion ScenarioExecution.BrowserProvider.Session.Registry
# #endregion ScenarioExecution.BrowserProvider.Session

View File

@@ -1,34 +1,74 @@
# #region ScenarioExecution.BrowserProvider.Transport [C:5] [TYPE Module] [SEMANTICS scenario,execution,provider,browser,transport,playwright,mutation]
# #region ScenarioExecution.BrowserProvider.Transport [C:5] [TYPE Module] [SEMANTICS scenario,execution,provider,browser,transport,playwright,mutation,native-filter,readonly]
# @ingroup ScenarioExecution
# @BRIEF Isolated-context Playwright transport: server-owned auth, read-only actions and bounded
# SQL-Lab-mediated mutations executed by the authenticated session.
# @BRIEF Isolated-context Playwright transport: server-owned auth, read-only actions (the round-1
# filter-bar apply_native_filter flow plus the T034 round-2 catalog: navigate_tab,
# inspect_filter_state, apply_table_filter, extract_table, scroll_to, inspect_columns,
# click, select_rows, download) and bounded SQL-Lab-mediated mutations executed by the
# authenticated session. Two ownership modes share one action core: per-step execute (legacy
# fallback) and the run-scoped session flow owned by BrowserProvider.Session (DG-1).
# @RELATION IMPLEMENTS -> [ScenarioExecution.BrowserProvider]
# @RELATION DEPENDS_ON -> [ScenarioExecution.BrowserProvider.MutationSQL]
# @RELATION DEPENDS_ON -> [ScenarioExecution.BrowserProvider.NativeFilter]
# @RELATION DEPENDS_ON -> [ScenarioExecution.BrowserProvider.ReadOnlyActions]
# @RELATION DEPENDS_ON -> [ScenarioExecution.BrowserProvider.Session]
# @RELATION DEPENDS_ON -> [Plugin.Service.ScreenshotService]
# @PRE The service is constructed from the deployment environment by startup composition; mutating
# actions arrive only after provider admission validated the contract and inputs.
# @POST Returns the action outcome with checkpoints and evidence; browser/context are closed on every
# path; unsupported actions raise BrowserTransportUnsupported before any I/O; a precondition
# mismatch aborts before the UPDATE.
# @INVARIANT No context, page or cookie outlives this call; secrets stay inside the transport; the
# mutation WHERE clause is bound by the contract's target keys.
# @SIDE_EFFECT One isolated headless Chromium context per execution; for mutations, one SELECT/UPDATE/
# SELECT triple through the session's SQL Lab endpoint.
# @POST Returns the action outcome with checkpoints and evidence; the per-step execute path closes
# browser/context on every path while the session flow keeps exactly one context per run alive
# until Session close; unsupported actions raise BrowserTransportUnsupported before any I/O; a
# precondition mismatch aborts before the UPDATE; download outcomes carry the captured bytes
# for provider-side artifact storage (bounded to 25 MiB by the flow).
# @INVARIANT No context, page or cookie outlives one per-step execute call; a run-scoped session
# context is owned and closed only by BrowserProvider.Session (never shared across runs,
# never revived after death); secrets stay inside the transport; the mutation WHERE clause
# is bound by the contract's target keys.
# @SIDE_EFFECT One isolated headless Chromium context per execution (per-step) or per run (session
# flow); for mutations, one SELECT/UPDATE/SELECT triple through the session's SQL Lab
# endpoint; read-only flows click/scroll/read the DOM and capture bounded downloads.
# @REJECTED SQL execution from a server-side client for browser mutations was rejected — the effect
# must be performed by the authenticated browser session to stay inside the provider boundary.
from __future__ import annotations
import asyncio
from typing import Any, Protocol
from src.core.logger import logger
from src.services.dashboard_testing.execution.providers.browser_mutation import (
build_cleanup_script,
build_mutation_script,
)
from src.services.dashboard_testing.execution.providers.browser_native_filter import ( # noqa: F401
BrowserTransportSelectorNotFound,
apply_native_filter_via_ui,
parse_native_filter_input,
)
from src.services.dashboard_testing.execution.providers.browser_readonly_actions import (
run_readonly_action_flow,
validate_readonly_action_input,
)
_SRC = "ScenarioExecution.BrowserProvider.Transport"
_READ_ONLY_ACTIONS = frozenset({"open_dashboard", "wait_for_state", "refresh"})
_CONTEXT_AUTH_TIMEOUT_SECONDS = 120.0
_MAX_PAGES_PER_SESSION = 3
_READ_ONLY_ACTIONS = frozenset({
"open_dashboard", "wait_for_state", "refresh", "apply_native_filter",
"navigate_tab", "inspect_filter_state", "apply_table_filter", "extract_table",
"scroll_to", "inspect_columns", "click", "select_rows", "download",
})
_MUTATION_ACTIONS = frozenset({"row_edit", "bulk_edit"})
_ALLOWED_WAIT_STATES = frozenset({"load", "domcontentloaded", "networkidle"})
_ROUND2_CHECKPOINTS = {
"navigate_tab": "tab_navigated",
"inspect_filter_state": "filter_state_inspected",
"apply_table_filter": "table_filter_applied",
"extract_table": "table_extracted",
"scroll_to": "scrolled",
"inspect_columns": "columns_inspected",
"click": "clicked",
"select_rows": "rows_selected",
"download": "downloaded",
}
# #region ScenarioExecution.BrowserProvider.Transport.Seam [C:2] [TYPE Class] [SEMANTICS provider,browser,transport,protocol]
@@ -42,6 +82,10 @@ class BrowserTransportPreconditionMismatch(RuntimeError):
"""Raised when the pre-mutation row state does not match the contract precondition hash."""
class BrowserTransportCleanupFailed(RuntimeError):
"""Raised when the restore_fixture cleanup failed to return the fixture to its pre-image."""
class BrowserTransportOutcome:
def __init__(
self,
@@ -51,12 +95,14 @@ class BrowserTransportOutcome:
evidence_png: bytes | None = None,
details: dict[str, Any] | None = None,
effect_state: str = "completed",
download_bytes: bytes | None = None,
) -> None:
self.checkpoints = checkpoints
self.page_url = page_url
self.evidence_png = evidence_png
self.details = details or {}
self.effect_state = effect_state
self.download_bytes = download_bytes
class BrowserActionTransport(Protocol):
@@ -68,18 +114,179 @@ class BrowserActionTransport(Protocol):
action_input: dict[str, Any],
timeout_seconds: float,
) -> BrowserTransportOutcome: ...
# #region ScenarioExecution.BrowserProvider.Transport.SessionSeam [C:3] [TYPE Class] [SEMANTICS provider,browser,transport,session,protocol]
# @ingroup ScenarioExecution
# @BRIEF Optional run-scoped session capability: the context lives across steps under Session ownership.
# @POST open_session returns a transport-owned handle; execute_in_session runs the shared action core
# without closing; close_session is idempotent and never raises (leak-guard).
class BrowserSessionHandle:
def __init__(self, *, playwright_cm: Any, browser: Any, context: Any, page: Any, dashboard_id: int) -> None:
self.playwright_cm = playwright_cm
self.browser = browser
self.context = context
self.page = page
self.dashboard_id = dashboard_id
class BrowserSessionCapableTransport(Protocol):
async def open_session(self, dashboard_id: int, *, timeout_seconds: float) -> Any: ...
async def execute_in_session(
self,
session: Any,
action: str,
*,
action_input: dict[str, Any],
timeout_seconds: float,
) -> BrowserTransportOutcome: ...
async def close_session(self, session: Any) -> None: ...
def is_session_capable_transport(transport: Any) -> bool:
return all(
callable(getattr(transport, attr, None))
for attr in ("open_session", "execute_in_session", "close_session")
)
# #endregion ScenarioExecution.BrowserProvider.Transport.SessionSeam
# #endregion ScenarioExecution.BrowserProvider.Transport.Seam
# #region ScenarioExecution.BrowserProvider.Transport.Playwright [C:5] [TYPE Function] [SEMANTICS provider,browser,playwright,auth,mutation]
# #region ScenarioExecution.BrowserProvider.Transport.PageBound [C:3] [TYPE Function] [SEMANTICS provider,browser,transport,limits,pages]
# @ingroup ScenarioExecution
# @BRIEF Real transport reusing the server-owned login/navigation flow for isolated contexts.
# @BRIEF T034 round 3: a session never keeps more than _MAX_PAGES_PER_SESSION pages; extras close.
# @POST The driving page survives; overflow pages close oldest-first; a lost driving page is typed
# BROWSER_PAGE_LOST (fail-closed, the run reports honestly instead of screenshotting nothing).
async def _enforce_page_bound(page: Any) -> None:
context = getattr(page, "context", None)
if context is None:
return
if page.is_closed():
raise ValueError("BROWSER_PAGE_LOST")
pages = [p for p in (getattr(context, "pages", None) or []) if not p.is_closed()]
if len(pages) <= _MAX_PAGES_PER_SESSION:
return
extras = [p for p in pages if p is not page]
while len([p for p in context.pages if not p.is_closed()]) > _MAX_PAGES_PER_SESSION and extras:
extra = extras.pop(0)
logger.explore(
"Page overflow; closing a stray page", src=_SRC,
payload={"max_pages": _MAX_PAGES_PER_SESSION}, error_code="BROWSER_PAGE_OVERFLOW",
)
await extra.close()
# #endregion ScenarioExecution.BrowserProvider.Transport.PageBound
# #region ScenarioExecution.BrowserProvider.Transport.PageActions [C:5] [TYPE Function] [SEMANTICS provider,browser,playwright,actions,mutation,readonly]
# @ingroup ScenarioExecution
# @BRIEF Shared action core executed on an authenticated page for both ownership modes.
# @PRE page holds an authenticated dashboard session opened by service._launch_and_login; the action
# is in the read-only or mutation catalog; mutating inputs carry the admission-validated contract.
# @POST Returns the action outcome with checkpoints and evidence; the page/context lifecycle is the
# caller's mode (per-step close or run-scoped session); unsupported actions raise
# BrowserTransportUnsupported before any I/O; a precondition mismatch aborts before the UPDATE;
# round-2 read-only flows append their typed checkpoint and merge structured details, and
# download carries the captured bytes for provider-side artifact storage.
# @SIDE_EFFECT Filter-bar clicks, load-state waits, one screenshot; for mutations the bounded
# SELECT/UPDATE/SELECT triple through the session's SQL Lab endpoint; round-2 flows
# click/scroll/read the DOM and capture bounded downloads.
async def _execute_on_page(
service: Any,
page: Any,
action: str,
*,
action_input: dict[str, Any],
timeout_seconds: float,
) -> BrowserTransportOutcome:
if action not in _READ_ONLY_ACTIONS and action not in _MUTATION_ACTIONS:
raise BrowserTransportUnsupported("BROWSER_ACTION_NOT_SUPPORTED")
checkpoints: list[str] = ["dashboard_open"]
extra_details: dict[str, Any] = {}
download_bytes: bytes | None = None
if action == "wait_for_state":
state = str(action_input.get("state") or "load")
if state not in _ALLOWED_WAIT_STATES:
raise ValueError("BROWSER_WAIT_STATE_INVALID")
await page.wait_for_load_state(state, timeout=int(timeout_seconds * 1000))
checkpoints.append("wait_for_state")
elif action == "refresh":
await page.reload(timeout=int(timeout_seconds * 1000), wait_until="domcontentloaded")
checkpoints.append("refreshed")
elif action == "apply_native_filter":
filter_input = parse_native_filter_input(action_input)
extra_details = await apply_native_filter_via_ui(
service, page, filter_input, timeout_seconds=timeout_seconds,
)
checkpoints.extend(("filter_applied", "charts_settled"))
elif action in _ROUND2_CHECKPOINTS:
# T034 round 2: typed input validation is defense-in-depth — admission already rejected
# malformed inputs; a code here means the transport boundary was reached directly.
input_error = validate_readonly_action_input(action, action_input)
if input_error is not None:
raise ValueError(input_error)
extra_details, download_bytes = await run_readonly_action_flow(
service, page, action, action_input, timeout_seconds=timeout_seconds,
)
checkpoints.append(_ROUND2_CHECKPOINTS[action])
elif action in _MUTATION_ACTIONS:
contract = action_input.get("mutation_contract") or {}
flow = await page.evaluate(build_mutation_script(action_input, contract))
if not flow.get("precondition_ok"):
logger.explore(
"Precondition hash mismatch; mutation aborted before the UPDATE", src=_SRC,
payload={"pre_hash": flow.get("pre_hash")},
error_code="BROWSER_MUTATION_PRECONDITION_MISMATCH",
)
raise BrowserTransportPreconditionMismatch("BROWSER_MUTATION_PRECONDITION_MISMATCH")
checkpoints.append("row_edited" if action == "row_edit" else "bulk_edited")
# T034 round 4: restore_fixture cleanup executes inside the same session before the outcome
# is finalized; the restored state must hash to the precondition digest, never declared clean.
if str(contract.get("cleanup_policy") or "") == "restore_fixture":
cleanup = await page.evaluate(build_cleanup_script(action_input, contract, flow.get("pre_rows") or []))
if cleanup.get("restored_hash") != flow.get("pre_hash"):
logger.explore(
"Fixture restore failed; environment left mutated", src=_SRC,
payload={"restored_hash": cleanup.get("restored_hash"), "rows_restored": cleanup.get("rows_restored")},
error_code="BROWSER_MUTATION_CLEANUP_FAILED",
)
raise BrowserTransportCleanupFailed("BROWSER_MUTATION_CLEANUP_FAILED")
checkpoints.append("fixture_restored")
await _enforce_page_bound(page)
evidence = await page.screenshot(full_page=False, timeout=int(timeout_seconds * 1000))
return BrowserTransportOutcome(
checkpoints=tuple(checkpoints),
page_url=page.url,
evidence_png=evidence,
details={
"post_rows": flow.get("post_rows"),
"precondition_hash": flow.get("pre_hash"),
"post_hash": flow.get("post_hash"),
"title": await page.title(),
},
effect_state="completed",
)
await _enforce_page_bound(page)
evidence = await page.screenshot(full_page=False, timeout=int(timeout_seconds * 1000))
return BrowserTransportOutcome(
checkpoints=tuple(checkpoints),
page_url=page.url,
evidence_png=evidence,
details={"title": await page.title(), **extra_details},
download_bytes=download_bytes,
)
# #endregion ScenarioExecution.BrowserProvider.Transport.PageActions
# #region ScenarioExecution.BrowserProvider.Transport.Playwright [C:5] [TYPE Function] [SEMANTICS provider,browser,playwright,auth,mutation,session]
# @ingroup ScenarioExecution
# @BRIEF Real transport reusing the server-owned login/navigation flow for both ownership modes.
# @PRE service is the deployment-owned ScreenshotService whose _launch_and_login owns credentials.
# @POST Returns the action outcome with checkpoints and evidence; browser/context are closed on every
# path; unsupported actions raise BrowserTransportUnsupported before any I/O; a precondition
# mismatch aborts before the UPDATE.
# @SIDE_EFFECT Launches one isolated headless Chromium context per execution; for mutations executes
# the bounded SQL flow through the session's SQL Lab endpoint.
# @POST execute (legacy per-step fallback) opens and closes one context per call on every path;
# open_session returns a live handle owned by BrowserProvider.Session until close_session;
# execute_in_session runs the shared action core without closing; close_session is best-effort
# and never raises.
# @SIDE_EFFECT Per-step mode launches one isolated headless Chromium context per call; session mode
# keeps one context per run alive on the shared provider loop.
def build_playwright_browser_transport(service: Any) -> BrowserActionTransport:
async def transport(
dashboard_id: int,
@@ -90,61 +297,69 @@ def build_playwright_browser_transport(service: Any) -> BrowserActionTransport:
) -> BrowserTransportOutcome:
if action not in _READ_ONLY_ACTIONS and action not in _MUTATION_ACTIONS:
raise BrowserTransportUnsupported("BROWSER_ACTION_NOT_SUPPORTED")
session = await open_session(dashboard_id, timeout_seconds=timeout_seconds)
try:
return await execute_in_session(session, action, action_input=action_input, timeout_seconds=timeout_seconds)
finally:
await close_session(session)
async def open_session(dashboard_id: int, *, timeout_seconds: float) -> BrowserSessionHandle:
from playwright.async_api import async_playwright
checkpoints: list[str] = ["dashboard_open"]
async with async_playwright() as playwright:
browser, context, page = await service._launch_and_login(playwright, str(dashboard_id))
try:
if action == "wait_for_state":
state = str(action_input.get("state") or "load")
if state not in _ALLOWED_WAIT_STATES:
raise ValueError("BROWSER_WAIT_STATE_INVALID")
await page.wait_for_load_state(state, timeout=int(timeout_seconds * 1000))
checkpoints.append("wait_for_state")
elif action == "refresh":
await page.reload(timeout=int(timeout_seconds * 1000), wait_until="domcontentloaded")
checkpoints.append("refreshed")
elif action in _MUTATION_ACTIONS:
contract = action_input.get("mutation_contract") or {}
flow = await page.evaluate(build_mutation_script(action_input, contract))
if not flow.get("precondition_ok"):
logger.explore(
"Precondition hash mismatch; mutation aborted before the UPDATE", src=_SRC,
payload={"pre_hash": flow.get("pre_hash")},
error_code="BROWSER_MUTATION_PRECONDITION_MISMATCH",
)
raise BrowserTransportPreconditionMismatch("BROWSER_MUTATION_PRECONDITION_MISMATCH")
checkpoints.append("row_edited" if action == "row_edit" else "bulk_edited")
evidence = await page.screenshot(full_page=False, timeout=int(timeout_seconds * 1000))
return BrowserTransportOutcome(
checkpoints=tuple(checkpoints),
page_url=page.url,
evidence_png=evidence,
details={
"post_rows": flow.get("post_rows"),
"precondition_hash": flow.get("pre_hash"),
"post_hash": flow.get("post_hash"),
"title": await page.title(),
},
effect_state="completed",
)
evidence = await page.screenshot(full_page=False, timeout=int(timeout_seconds * 1000))
return BrowserTransportOutcome(
checkpoints=tuple(checkpoints),
page_url=page.url,
evidence_png=evidence,
details={"title": await page.title()},
)
finally:
try:
await context.close()
finally:
await browser.close()
playwright_cm = async_playwright()
playwright = await playwright_cm.__aenter__()
try:
# T034 round 3: context/auth is bounded at 120s independent of the action timeout;
# asyncio.TimeoutError (== TimeoutError) maps to typed BROWSER_ACTION_TIMEOUT upstream.
browser, context, page = await asyncio.wait_for(
service._launch_and_login(playwright, str(dashboard_id)),
timeout=_CONTEXT_AUTH_TIMEOUT_SECONDS,
)
except Exception:
await playwright_cm.__aexit__(None, None, None)
raise
return BrowserSessionHandle(
playwright_cm=playwright_cm,
browser=browser,
context=context,
page=page,
dashboard_id=dashboard_id,
)
async def execute_in_session(
session: Any,
action: str,
*,
action_input: dict[str, Any],
timeout_seconds: float,
) -> BrowserTransportOutcome:
if action not in _READ_ONLY_ACTIONS and action not in _MUTATION_ACTIONS:
raise BrowserTransportUnsupported("BROWSER_ACTION_NOT_SUPPORTED")
return await _execute_on_page(service, session.page, action, action_input=action_input, timeout_seconds=timeout_seconds)
async def close_session(session: Any) -> None:
try:
await session.context.close()
except Exception as exc:
logger.explore("Session context close failed", src=_SRC, error_code="BROWSER_SESSION_CLOSE_FAILED", error=repr(exc))
try:
await session.browser.close()
except Exception as exc:
logger.explore("Session browser close failed", src=_SRC, error_code="BROWSER_SESSION_CLOSE_FAILED", error=repr(exc))
try:
await session.playwright_cm.__aexit__(None, None, None)
except Exception as exc:
logger.explore("Session playwright stop failed", src=_SRC, error_code="BROWSER_SESSION_CLOSE_FAILED", error=repr(exc))
class _Transport:
execute = staticmethod(transport)
# Assigned post-class: `attr = staticmethod(attr)` inside the class body would shadow the
# enclosing function names and raise NameError (class-body name resolution skips closures).
_Transport.open_session = staticmethod(open_session)
_Transport.execute_in_session = staticmethod(execute_in_session)
_Transport.close_session = staticmethod(close_session)
return _Transport()
# #endregion ScenarioExecution.BrowserProvider.Transport.Playwright
# #endregion ScenarioExecution.BrowserProvider.Transport

View File

@@ -1,16 +1,31 @@
# #region ScenarioExecution.ProviderPreflight [C:4] [TYPE Module] [SEMANTICS scenario,execution,provider,health,readiness,preflight]
# @ingroup ScenarioExecution
# @BRIEF Startup readiness snapshot for registered provider capabilities; advisory and I/O-free.
# @BRIEF Startup readiness snapshot for registered provider capabilities: state plus pinned provider
# identity (name, registry version, capability fingerprint) and redacted dependency
# diagnostics (SCEX-FR-022 / SC-010); advisory and I/O-free.
# @RELATION DEPENDS_ON -> [ScenarioExecution.LiveCompositionRoot]
# @RELATION DEPENDS_ON -> [ScenarioExecution.ProviderRuntime.Engine]
# @RELATION DEPENDS_ON -> [ScenarioGraph.Templates.ActionRegistry]
# @PRE Called by application composition after provider registration; never by request paths.
# @POST Returns a JSON-safe readiness mapping per capability (loop, storage, screenshot, browser,
# bindings); every check failure becomes a typed degraded reason and never an exception or a
# startup blocker. No provider I/O occurs (no browser launch, no capture, no storage writes).
# @POST Returns a JSON-safe readiness mapping per capability (loop, storage, superset, screenshot,
# browser, bindings); every check failure becomes a typed degraded reason and never an
# exception or a startup blocker. No provider I/O occurs (no browser launch, no capture, no
# storage writes). Every provider entry carries provider/version/capabilities/dependencies;
# dependency diagnostics are booleans and binding counts only — no credentials, cookies, raw
# SQL, executable paths, or capture bytes (SCEX-FR-022 redaction).
# @INVARIANT Readiness is advisory health evidence, never admission authority; runtime dispatch keeps
# its own fail-closed checks before any external I/O.
# @RATIONALE Provider version and capability fingerprint are sourced from the existing version-pinned
# 038 ActionRegistry (ACTION_REGISTRY_VERSION + action_registry_fingerprint) that every
# registered provider executes against; no provider module defines its own version
# constant, and the registry is the only pinned descriptor authority.
# @REJECTED Blocking application startup on provider readiness was rejected — typed-unavailable
# registration already fails closed at dispatch time.
# @REJECTED Inventing per-provider version constants was rejected — a fabricated constant would drift
# from the descriptor registry that dispatch actually enforces.
# @REJECTED Exposing resolved executable paths, environment URLs, or binding credentials in the
# diagnostics was rejected — /api/ready is unauthenticated, so only resolved true/false and
# counts cross the boundary.
from __future__ import annotations
import asyncio
@@ -20,6 +35,11 @@ from src.core.logger import logger
from src.services.dashboard_testing.execution.provider_runtime import (
get_provider_event_loop,
)
from src.services.dashboard_testing.scenario.templates import (
ACTION_REGISTRY_VERSION,
REGISTERED_ACTIONS,
action_registry_fingerprint,
)
_SRC = "ScenarioExecution.ProviderPreflight"
_BROWSER_PROBE_TIMEOUT_SECONDS = 10.0
@@ -95,7 +115,16 @@ def check_evidence_storage() -> str | None:
# @BRIEF Build the JSON-safe readiness snapshot from the composition root and capability checks.
# @PRE root is the application composition root after bootstrap registration.
# @POST Capabilities report ready/unregistered/degraded with stable reason codes; check crashes are
# reported as degraded, never raised.
# reported as degraded, never raised. Each provider entry carries the pinned registry identity
# (provider/version/capabilities) and redacted dependency diagnostics (booleans/counts only).
# Provider -> registered tool codes whose action capabilities identify the provider feature set.
_PROVIDER_TOOLS: dict[str, tuple[str, ...]] = {
"superset": ("superset_api", "sql_evidence"),
"screenshot": ("screenshot",),
"browser": ("browser",),
}
def build_provider_readiness(
root: Any,
*,
@@ -106,15 +135,49 @@ def build_provider_readiness(
readiness: dict[str, Any] = {}
loop = get_provider_event_loop()
readiness["provider_loop"] = {"state": "ready" if loop.is_running else "not_started"}
loop_running = bool(loop.is_running)
readiness["provider_loop"] = {"state": "ready" if loop_running else "not_started"}
storage_reason = _safe(storage_check)
storage_ready = storage_reason is None
readiness["evidence_storage"] = (
{"state": "ready"} if storage_reason is None else {"state": "degraded", "reason_code": storage_reason}
{"state": "ready"} if storage_ready else {"state": "degraded", "reason_code": storage_reason}
)
readiness["screenshot"] = _capability_state(root.has_screenshot_provider, "screenshot")
readiness["browser"] = _capability_state(root.has_browser_provider, "browser", browser_check=browser_check, run_async=run_async)
readiness["superset"] = _provider_entry(
"superset",
registered=bool(root.has_superset_provider),
dependencies={
"evidence_storage_ready": storage_ready,
"registered_bindings": len(getattr(root, "_superset", {})),
"unavailable_bindings": len(getattr(root, "_superset_unavailable", set())),
},
)
readiness["screenshot"] = _provider_entry(
"screenshot",
registered=bool(root.has_screenshot_provider),
dependencies={
"evidence_storage_ready": storage_ready,
"provider_loop_running": loop_running,
"registered_bindings": len(getattr(root, "_screenshot", {})),
"unavailable_bindings": len(getattr(root, "_screenshot_unavailable", set())),
},
)
browser_registered = bool(root.has_browser_provider)
browser_check_ran = browser_registered and browser_check is not None
browser_reason = _safe(browser_check, run_async) if browser_check_ran else None
readiness["browser"] = _provider_entry(
"browser",
registered=browser_registered,
check_reason=browser_reason,
check_ran=browser_check_ran,
dependencies={
"chromium_executable_resolved": browser_registered and browser_reason is None,
"provider_loop_running": loop_running,
"registered_bindings": len(getattr(root, "_browser", {})),
"unavailable_bindings": len(getattr(root, "_browser_unavailable", set())),
},
)
readiness["bindings"] = {
"registered": len(getattr(root, "_superset", {})),
"screenshot_unavailable": sorted(getattr(root, "_screenshot_unavailable", set())),
@@ -132,19 +195,53 @@ def _safe(check: Callable[..., str | None], *args: Any) -> str | None:
return "PROVIDER_CHECK_CRASHED"
# #region ScenarioExecution.ProviderPreflight.ProviderEntry [C:3] [TYPE Function] [SEMANTICS provider,health,identity,fingerprint,redaction]
# @ingroup ScenarioExecution
# @BRIEF One provider readiness entry: capability state plus pinned identity and redacted diagnostics.
# @POST The entry always carries provider/version/capabilities (registry-pinned, static) and a
# dependencies mapping of redacted booleans/counts; version is ACTION_REGISTRY_VERSION and the
# capability fingerprint is action_registry_fingerprint(), never a fabricated constant.
def _provider_entry(
name: str,
*,
registered: bool,
dependencies: dict[str, Any],
check_reason: str | None = None,
check_ran: bool = False,
) -> dict[str, Any]:
entry = _capability_state(registered, name, check_reason=check_reason, check_ran=check_ran)
entry["provider"] = name
entry["version"] = ACTION_REGISTRY_VERSION
entry["capabilities"] = {
"features": _provider_features(name),
"registry_fingerprint": action_registry_fingerprint(),
}
entry["dependencies"] = dependencies
return entry
def _provider_features(provider: str) -> list[str]:
tools = _PROVIDER_TOOLS[provider]
return sorted({
registered["capability"]
for tool in tools
for registered in REGISTERED_ACTIONS.values()
if registered["tool"] == tool and registered["capability"]
})
# #endregion ScenarioExecution.ProviderPreflight.ProviderEntry
def _capability_state(
registered: bool,
name: str,
*,
browser_check: Callable[..., str | None] | None = None,
run_async: Callable[[Any], Any] | None = None,
check_reason: str | None = None,
check_ran: bool = False,
) -> dict[str, Any]:
if not registered:
return {"state": "unregistered", "reason_code": f"{name.upper()}_PROVIDER_UNREGISTERED"}
if browser_check is not None and run_async is not None:
reason = _safe(browser_check, run_async)
if reason is not None:
return {"state": "degraded", "reason_code": reason}
if check_ran and check_reason is not None:
return {"state": "degraded", "reason_code": check_reason}
return {"state": "ready"}
# #endregion ScenarioExecution.ProviderPreflight.Snapshot
# #endregion ScenarioExecution.ProviderPreflight

View File

@@ -32,6 +32,7 @@ from src.core.logger import logger
from src.services.dashboard_testing.execution.capacity import (
CapacityUnavailable,
claim_capacity,
heartbeat_capacity,
release_capacity,
)
from src.services.dashboard_testing.execution.live_adapter import LiveAdapterResult
@@ -44,12 +45,35 @@ from src.services.dashboard_testing.execution.provider_runtime import (
ProviderSubmissionOverflow,
get_provider_event_loop,
)
from src.services.dashboard_testing.execution.provider_operations import (
complete_provider_operation,
descriptor_fingerprint,
open_provider_operation,
)
from src.core.database import SessionLocal
from hashlib import sha256
_SRC = "ScenarioExecution.ScreenshotProvider"
_DEFAULT_CAPTURE_TIMEOUT_SECONDS = 120
_DEFAULT_MAX_SCREENSHOT_BYTES = 10485760
_PROVIDER_VERSION = "screenshot/1"
# #region ScenarioExecution.ScreenshotProvider.ReceiptFinalize [C:3] [TYPE Function] [SEMANTICS provider,screenshot,receipt,finalize]
# @ingroup ScenarioExecution
# @BRIEF Finalize the durable provider-operation receipt outside the run transaction; never masks the outcome.
# @RELATION DEPENDS_ON -> [ScenarioExecution.ProviderOperations.Service]
# @INVARIANT Receipt finalization failure is logged, not raised — the provider result is never shadowed.
def _finalize_screenshot_receipt(operation_id: str | None, status: str, effect_state: str, summary: dict | None = None) -> None:
if operation_id is None:
return
try:
with SessionLocal() as db:
complete_provider_operation(db, operation_id, status=status, effect_state=effect_state, summary=summary)
db.commit()
except Exception as exc:
logger.explore("Screenshot receipt finalization failed", src=_SRC, payload={"operation_id": operation_id, "status": status}, error=repr(exc))
# #endregion ScenarioExecution.ScreenshotProvider.ReceiptFinalize
# #region ScenarioExecution.ScreenshotProvider.Protocol [C:2] [TYPE Class] [SEMANTICS provider,screenshot,transport]
@@ -121,6 +145,7 @@ def build_screenshot_provider(
payload={"run_id": run_id, "dashboard_id": binding.dashboard_id, "environment_class": environment_class},
)
lease_id: str | None = None
operation_id: str | None = None
workdir = tempfile.mkdtemp(prefix="screenshot-provider-")
try:
try:
@@ -134,8 +159,46 @@ def build_screenshot_provider(
run_id=run_id,
logical_step_id=metadata.get("logical_step_id"),
)
logical_step_id = str(metadata.get("logical_step_id") or "")
attempt = int(metadata.get("attempt") or 1)
if logical_step_id:
receipt = open_provider_operation(
db,
run_id=run_id,
logical_step_id=logical_step_id,
attempt=attempt,
provider_id="screenshot",
provider_version=_PROVIDER_VERSION,
action="capture_dashboard",
descriptor_fingerprint=descriptor_fingerprint({
"dashboard_id": binding.dashboard_id,
"environment_id": binding.environment_id,
"binding_ref": binding.binding_ref,
}),
binding_ref=binding.binding_ref,
execution_principal_fingerprint=binding.execution_principal_fingerprint,
idempotency_key=f"{run_id}:{logical_step_id}:{attempt}",
capacity_lease_id=lease["lease_id"],
effect_state="unknown",
summary={
"environment_class": environment_class,
"environment_id": binding.environment_id,
"dashboard_id": binding.dashboard_id,
},
)
operation_id = receipt["operation_id"]
db.commit()
lease_id = lease["lease_id"]
# T032: heartbeat refreshes the lease TTL before the capture submission window;
# a lost/expired lease is a typed capacity refusal (walker parks the run), never
# I/O without a live lease.
try:
with SessionLocal() as db:
heartbeat_capacity(db, lease_id)
db.commit()
except CapacityUnavailable as exc:
logger.explore("Screenshot lease lost before I/O", src=_SRC, payload={"run_id": run_id}, error=str(exc))
return LiveAdapterResult(status="inconclusive", reason_code="SCREENSHOT_CAPACITY_UNAVAILABLE")
except CapacityUnavailable as exc:
logger.explore("Screenshot capacity unavailable", src=_SRC, payload={"run_id": run_id}, error=str(exc))
return LiveAdapterResult(status="inconclusive", reason_code="SCREENSHOT_CAPACITY_UNAVAILABLE")
@@ -149,18 +212,22 @@ def build_screenshot_provider(
jpeg_paths, _archive = event_loop.submit(capture_factory, timeout=capture_timeout_seconds)
except TimeoutError:
logger.explore("Screenshot capture exceeded the deadline", src=_SRC, payload={"run_id": run_id}, error_code="SCREENSHOT_CAPTURE_TIMEOUT")
_finalize_screenshot_receipt(operation_id, "failed", "unknown", summary={"phase": "timeout"})
return LiveAdapterResult(status="inconclusive", reason_code="SCREENSHOT_CAPTURE_TIMEOUT")
except ProviderSubmissionOverflow:
logger.explore("Screenshot submission overflowed the bounded queue", src=_SRC, payload={"run_id": run_id}, error_code="SCREENSHOT_LOOP_OVERFLOW")
_finalize_screenshot_receipt(operation_id, "failed", "not_started", summary={"phase": "loop_overflow"})
return LiveAdapterResult(status="inconclusive", reason_code="SCREENSHOT_LOOP_OVERFLOW")
except RuntimeError as exc:
if str(exc) == "PROVIDER_LOOP_NOT_RUNNING":
logger.explore("Provider loop is not running", src=_SRC, error_code="SCREENSHOT_LOOP_UNAVAILABLE")
_finalize_screenshot_receipt(operation_id, "failed", "not_started", summary={"phase": "loop_unavailable"})
return LiveAdapterResult(status="inconclusive", reason_code="SCREENSHOT_LOOP_UNAVAILABLE")
raise
if not jpeg_paths:
logger.explore("Capture service produced no screenshots", src=_SRC, payload={"run_id": run_id}, error_code="SCREENSHOT_CAPTURE_EMPTY")
_finalize_screenshot_receipt(operation_id, "failed", "none", summary={"phase": "empty"})
return LiveAdapterResult(status="inconclusive", reason_code="SCREENSHOT_CAPTURE_EMPTY")
artifact_refs: list[str] = []
@@ -175,11 +242,13 @@ def build_screenshot_provider(
payload={"run_id": run_id, "bytes": len(raw), "limit": max_screenshot_bytes},
error_code="SCREENSHOT_TOO_LARGE",
)
_finalize_screenshot_receipt(operation_id, "failed", "none", summary={"phase": "too_large", "bytes": len(raw)})
return LiveAdapterResult(status="inconclusive", reason_code="SCREENSHOT_TOO_LARGE")
digest = sha256(raw).hexdigest()
content_ref = storage.store(run_id, digest, raw)
if content_ref != f"draft:{run_id}:{digest}":
logger.explore("Evidence storage returned an unexpected ref", src=_SRC, error_code="SCREENSHOT_EVIDENCE_REF_INVALID")
_finalize_screenshot_receipt(operation_id, "failed", "none", summary={"phase": "evidence_ref_invalid"})
return LiveAdapterResult(status="inconclusive", reason_code="SCREENSHOT_EVIDENCE_REF_INVALID")
artifact_refs.append(content_ref)
artifact_digests[content_ref] = digest
@@ -191,6 +260,10 @@ def build_screenshot_provider(
payload={"run_id": run_id, "artifact_index": index, "bytes": len(raw)},
)
_finalize_screenshot_receipt(
operation_id, "completed", "none",
summary={"screenshot_count": len(artifact_refs), "artifact_refs": artifact_refs},
)
return LiveAdapterResult(
status="passed",
reason_code="SCREENSHOT_CAPTURED",
@@ -204,12 +277,14 @@ def build_screenshot_provider(
# evidence (the default byte_length is absent for capture outcomes).
"artifact_byte_lengths": artifact_byte_lengths,
"artifact_content_types": artifact_content_types,
**({"operation_id": operation_id} if operation_id else {}),
},
artifact_refs=artifact_refs,
artifact_digests=artifact_digests,
)
except Exception as exc:
logger.explore("Screenshot provider failed", src=_SRC, payload={"run_id": run_id if isinstance(run_id, str) else None}, error=repr(exc))
_finalize_screenshot_receipt(operation_id, "failed", "unknown", summary={"phase": "provider_error"})
return LiveAdapterResult(status="inconclusive", reason_code="SCREENSHOT_CAPTURE_FAILED")
finally:
shutil.rmtree(workdir, ignore_errors=True)

View File

@@ -2,7 +2,8 @@
# @defgroup ScenarioExecution Deterministically derive runtime plans from immutable revisions.
# @BRIEF Build the runtime source of truth from a pinned revision, never from runner.plan.json.
# @RELATION DEPENDS_ON -> [ScenarioRegistry.Revisions.Checkout]
# @INVARIANT Same revision snapshot always yields byte-equivalent plan.
# @INVARIANT Same revision snapshot and same launch params always yield a byte-equivalent plan;
# launch params are part of the derived program (UX-6 filter_values binding), not run metadata.
# @INVARIANT runner.plan.json in git is a reference artifact, never the runtime source of truth.
# @INVARIANT Derivation refuses any revision that is not the scenario's current revision (stale/unsafe).
# @INVARIANT A plan containing a human checkpoint derives manual_run_only=true; the start
@@ -32,6 +33,9 @@ from src.services.dashboard_testing.execution.decision_policy import (
BASELINE_SEMANTIC_V1,
DecisionPolicy,
)
from src.services.dashboard_testing.execution.providers.browser_native_filter import (
validate_native_filter_input,
)
from src.services.dashboard_testing.scenario.templates import validate_action_step
@@ -79,6 +83,52 @@ def _derive_evaluation_mode(steps: list[dict[str, Any]]) -> str:
# #endregion ScenarioExecution.RunnerPlan.EvaluationMode
# #region ScenarioExecution.RunnerPlan.BindParams [C:3] [TYPE Function] [SEMANTICS scenario,execution,runnerplan,params,filter,binding]
# @ingroup ScenarioExecution
# @BRIEF Bind launch params.filter_values (UX-6) onto a pinned apply_native_filter step during derivation.
# @PRE params is the typed launch parameter dict supplied by the start boundary (persisted as
# run.parameter_bindings), or None for derivation surfaces that carry no launch params.
# @POST Returns the pinned step unchanged when the param is absent/empty-string (UX-1 current-state
# observe mode); a present valid param stamps param_binding {"filter_values": [str, ...]}.
# @POST A present-but-invalid param raises ValueError("BROWSER_FILTER_VALUES_INVALID") before any
# ScenarioRun/lease/gate exists — the requested filter is never silently dropped.
# @INVARIANT Only browser/apply_native_filter steps are bound; the 038 action_descriptor snapshot and
# the graph's declarative input Ref list stay byte-identical (validate_pinned_runner_plan holds).
# @RELATION DEPENDS_ON -> [ScenarioExecution.BrowserProvider.NativeFilter.Validate]
# @RATIONALE Fail-closed on a present param: silently degrading to current-state observe mode would
# fabricate a PASS for a filter the analyst explicitly requested (ux-flow plan §UX-6, 2026-09-12).
# A bare string is accepted as a single value because the REST/UI RunConfigurationPanel submits
# string-typed param rows; comma-splitting is not attempted — it would invent values the
# analyst never typed. Validation reuses the provider input contract so a derived binding
# always passes admission.
# @REJECTED Stamping the values into the pinned action_descriptor was rejected — 038 descriptors are
# registry-owned and validate_pinned_runner_plan enforces byte-equality with the registry;
# rewriting the graph's declarative input Ref list was rejected for the same provenance reason.
# @REJECTED Fail-closed input (invalid param -> step runs current-state) was rejected: it reproduces the
# false-verification the fail-closed rule forbids — the run would "pass" without testing the
# requested filter.
def _bind_filter_values_param(step: dict[str, Any], params: dict[str, Any] | None) -> dict[str, Any]:
if params and step.get("tool") == "browser" and step.get("action") == "apply_native_filter":
raw = params.get("filter_values")
if raw is not None and raw != "":
value = [raw] if isinstance(raw, str) else raw
code = validate_native_filter_input({"values": value})
if code is not None:
logger.explore(
"Run parameter filter_values rejected at plan derivation",
src="ScenarioExecution.RunnerPlan.BindParams",
claim="POST: filter_values parameter is a bounded string list",
error_code=code,
payload={"logical_step_id": str(step.get("logical_step_id", step.get("id", "")))},
error="present filter_values param must bind or fail closed; silent current-state "
"fallback would verify an unapplied filter",
)
raise ValueError(code)
step["param_binding"] = {"filter_values": list(value)}
return step
# #endregion ScenarioExecution.RunnerPlan.BindParams
# #region ScenarioExecution.RunnerPlan.Derive [C:4] [TYPE Function] [SEMANTICS scenario,execution,runnerplan,derive]
# @ingroup ScenarioExecution
# @BRIEF Derive env targets, topological order and executor mapping from a revision snapshot.
@@ -88,6 +138,8 @@ def _derive_evaluation_mode(steps: list[dict[str, Any]]) -> str:
# @POST A human checkpoint is represented as immutable manual_run_only=true for the 044 start boundary.
# @POST Refuses an un-promoted bootstrap revision — one whose snapshot carries only draft-pack
# provenance (compiled_handle_id/draft_pack_id/draft_pack_digest) and no materialized steps.
# @POST When launch params carry filter_values (UX-6), every pinned browser apply_native_filter step
# carries a validated param_binding; a present-but-invalid param rejects derivation before a row.
# @INVARIANT Every executable step persists an exact version/hash-pinned 038 ActionExecutionDescriptor;
# missing actions, stale registry identity, invalid I/O shape, and unsafe mutation metadata
# reject before a ScenarioRun/lease/adapter can exist.
@@ -96,7 +148,13 @@ def _derive_evaluation_mode(steps: list[dict[str, Any]]) -> str:
# @REJECTED Deriving an empty plan for a provenance-only bootstrap revision was rejected (F3 review
# finding, 2026-09-06): an empty topological_order makes ScenarioExecution.Runner.Walker
# (_advance_run) mark the run "passed" without executing a single step — an automated false PASS.
def derive_runner_plan(db: Session, scenario_id: str, revision_id: str) -> dict[str, Any]:
def derive_runner_plan(
db: Session,
scenario_id: str,
revision_id: str,
*,
params: dict[str, Any] | None = None,
) -> dict[str, Any]:
revision = db.query(ScenarioRevision).filter(
ScenarioRevision.scenario_id == scenario_id,
ScenarioRevision.revision_id == revision_id,
@@ -136,11 +194,14 @@ def derive_runner_plan(db: Session, scenario_id: str, revision_id: str) -> dict[
registry_version=registry_version,
registry_hash=registry_hash,
)
by_id[step_id] = {
**source_step,
"logical_step_id": step_id,
"action_descriptor": descriptor.snapshot(),
}
by_id[step_id] = _bind_filter_values_param(
{
**source_step,
"logical_step_id": step_id,
"action_descriptor": descriptor.snapshot(),
},
params,
)
pinned_steps = [by_id[str(step.get("logical_step_id", step.get("id", index)))] for index, step in enumerate(steps)]
executor_mapping = {
step_id: by_id[step_id]["action_descriptor"]

View File

@@ -12,6 +12,8 @@
# only through gate approval (036), never by silent dispatch.
# @INVARIANT scenario_content_hash mirrors the pinned revision's content hash; request_hash
# carries the idempotency fingerprint (provenance vs replay identity).
# @INVARIANT Launch params flow into plan derivation: a present filter_values param binds into pinned
# apply_native_filter steps (UX-6) or rejects the start typed before any row exists.
# @INVARIANT A revision with a human step is manual_run_only: only the trusted manual
# interactive entry point may create its run. Automation is refused before a
# ScenarioRun, PROD gate, notification, or queue side effect exists.
@@ -242,7 +244,9 @@ def start_run(
entry = db.query(ScenarioRegistryEntry).filter(ScenarioRegistryEntry.scenario_id == scenario_id).first()
if entry is None:
raise ValueError("scenario not found")
plan = derive_runner_plan(db, scenario_id, revision_id)
# UX-6: launch params are part of the derived program — filter_values binds into pinned
# apply_native_filter steps here (or rejects the start typed) before any run row exists.
plan = derive_runner_plan(db, scenario_id, revision_id, params=params or None)
_reject_automated_human_plan(plan, trigger_source)
catalog_snapshot = load_published_catalog(published_catalog)
if catalog_snapshot is None and baseline_set and baseline_set_version:

View File

@@ -181,14 +181,17 @@ def is_load_testing_permission(resource: str, action: str) -> bool:
# #region Services.RbacPermissionCatalog.AUTOMATIONPERMISSIONS [C:1] [TYPE Constant] [SEMANTICS rbac,automation,permission]
# @ingroup Services
# @BRIEF Canonical scenario-automation permission pairs (SCAUTO-FR-010/011/013, 046):
# @BRIEF Canonical scenario-automation permission pairs (SCAUTO-FR-010/011/013/019, 046):
# read (the six operational reads, any authenticated principal type per DG-2),
# manage (schedule/trigger-rule/policy CRUD + enable/disable), trigger (external API run
# trigger), prod (PROD-classified external trigger authorization, 036 approval-gate
# semantics). Mirrors dashboard:loadtest:PROD — a stricter authorization than TRIGGER.
# @DATA_CONTRACT scenario:automation:manage -> ("scenario:automation", "MANAGE");
# @DATA_CONTRACT scenario:automation:read -> ("scenario:automation", "READ");
# scenario:automation:manage -> ("scenario:automation", "MANAGE");
# scenario:automation:trigger -> ("scenario:automation", "TRIGGER");
# scenario:automation:prod -> ("scenario:automation", "PROD")
AUTOMATION_PERMISSIONS: tuple[tuple[str, str], ...] = (
("scenario:automation", "READ"),
("scenario:automation", "MANAGE"),
("scenario:automation", "TRIGGER"),
("scenario:automation", "PROD"),
@@ -396,8 +399,8 @@ def discover_declared_permissions(plugin_loader=None) -> set[tuple[str, str]]:
# and the run surface historically bound it via a module constant. The explicit catalog
# constant guarantees sync/catalog discovery always sees scenario RUN/RUN_PROD.
permissions.update(SCENARIO_RUN_PERMISSIONS)
# Automation REST routes define MANAGE/TRIGGER/PROD. Read/list/metrics routes
# are authenticated-only and intentionally do not create a canonical READ grant.
# Automation REST routes define READ/MANAGE/TRIGGER/PROD. The explicit constant
# pins all four so sync/catalog discovery sees READ even before route re-scan.
permissions.update(AUTOMATION_PERMISSIONS)
# The Superset SQL risk class is bound by the MCP catalog, not a route has_permission
# call; the explicit constant guarantees sync creates the rows for admin assignment.

View File

@@ -94,5 +94,79 @@ class TestGetReady:
assert "admin" not in detail_str
assert "FATAL" not in detail_str
def test_ready_providers_block_schema_stable(self, client):
"""T040/SC-010: the cached provider snapshot passes through verbatim with stable keys."""
snapshot = {
"provider_loop": {"state": "ready"},
"evidence_storage": {"state": "ready"},
"superset": {
"state": "ready",
"provider": "superset",
"version": "038.4.0",
"capabilities": {
"features": ["dataset_field_read", "superset_metric", "superset_query_envelope"],
"registry_fingerprint": "a" * 64,
},
"dependencies": {
"evidence_storage_ready": True,
"registered_bindings": 1,
"unavailable_bindings": 0,
},
},
"screenshot": {
"state": "ready",
"provider": "screenshot",
"version": "038.4.0",
"capabilities": {"features": ["screenshot"], "registry_fingerprint": "a" * 64},
"dependencies": {
"evidence_storage_ready": True,
"provider_loop_running": True,
"registered_bindings": 1,
"unavailable_bindings": 0,
},
},
"browser": {
"state": "degraded",
"reason_code": "BROWSER_EXECUTABLE_MISSING",
"provider": "browser",
"version": "038.4.0",
"capabilities": {"features": ["browser"], "registry_fingerprint": "a" * 64},
"dependencies": {
"chromium_executable_resolved": False,
"provider_loop_running": True,
"registered_bindings": 1,
"unavailable_bindings": 0,
},
},
"bindings": {"registered": 1, "screenshot_unavailable": [], "browser_unavailable": []},
}
class _Root:
last_provider_readiness = snapshot
mock_session = MagicMock()
mock_session.execute.return_value = True
with (
patch("src.api.routes.ready.SessionLocal", return_value=mock_session),
patch("src.dependencies.get_live_execution_composition_root", return_value=_Root()),
):
resp = client.get("/api/ready")
assert resp.status_code == 200
body = resp.json()
assert body["status"] == "ready"
providers = body["providers"]
assert set(providers) == {
"provider_loop", "evidence_storage", "superset", "screenshot", "browser", "bindings",
}
for name in ("superset", "screenshot", "browser"):
entry = providers[name]
assert entry["provider"] == name
assert isinstance(entry["version"], str) and entry["version"]
assert set(entry["capabilities"]) == {"features", "registry_fingerprint"}
assert isinstance(entry["dependencies"], dict)
assert all(isinstance(v, (bool, int)) for v in entry["dependencies"].values())
# #endregion Test.Api.Ready

View File

@@ -9,6 +9,9 @@
# @TEST_EDGE missing_manage_scope -> 403 on schedule mutation
# @TEST_EDGE missing_trigger_scope -> 403 on direct trigger
# @TEST_EDGE missing_prod_scope -> 403 on PROD direct trigger
# @TEST_EDGE missing_read_scope -> 403 on all six operational reads (DG-2, T019)
# @TEST_EDGE foreign_owner_rows -> filtered out of every read collection, no totals leak
# @TEST_EDGE disabled_human_schedule -> identical 409 AUTOMATION_INELIGIBLE_HUMAN_STEP as enabled (DEF-02)
# @TEST_EDGE idempotency_reuse -> 409 on changed request hash
# @TEST_INVARIANT ScenarioExecution.Runner.Start: The 046 API trigger supplies a server-owned
# automation source; a persisted human graph returns the typed manual-only
@@ -33,7 +36,15 @@ from src.app import app
from src.core.database import SessionLocal
from src.dependencies import get_config_manager, get_current_user
from src.models.auth import Permission, Role, User
from src.models.scenario_registry import ScenarioRegistryEntry
from src.models.scenario_automation import (
AutomationPolicy,
ScenarioNotificationEvent,
ScenarioRetentionDeletion,
ScenarioSchedule,
ScenarioTriggerRule,
)
from src.models.scenario_registry import ScenarioRegistryEntry, ScenarioRevision
from src.models.scenario_run import ScenarioRun
from src.services.dashboard_testing.scenario.templates import (
ACTION_REGISTRY_VERSION,
action_registry_fingerprint,
@@ -416,4 +427,375 @@ class TestScenarioAutomationAnonymousReads:
resp = anonymous.get(path)
assert resp.status_code == 401, (path, resp.text)
# #endregion Test.Api.ScenarioAutomation.Sec01
# #region Test.Api.ScenarioAutomation.ReadAcl [C:4] [TYPE Class] [SEMANTICS test,api,scenario,automation,rbac,acl]
# @BRIEF DG-2 (046 T019): six reads require scenario:automation READ plus per-object
# scenario-ownership ACL; foreign rows vanish from collections without leaking totals.
# @TEST_INVARIANT Api.ScenarioAutomation.ReadAcl: missing READ -> 403 on every read;
# owner sees own rows only; admin sees all rows.
# @TEST_INVARIANT Api.ScenarioAutomation.Schedules: disabled schedule bound to a human-step
# revision is rejected with the identical code as enabled, with zero side-effect
# rows (DEF-02) -> VERIFIED_BY: test_disabled_human_schedule_rejected_like_enabled
def _make_acl_user() -> User:
role = Role(id="acl-role-test", name="AclRole")
role.permissions = [Permission(resource="scenario:automation", action="READ")]
user = User(id="acl-viewer-1", username="acl.viewer", email="acl@test.com")
user.roles = [role]
return user
def _make_admin_user() -> User:
role = Role(id="acl-admin-role-test", name="AclAdmin", is_admin=True)
user = User(id="acl-admin-1", username="acl.admin", email="acl.admin@test.com")
user.roles = [role]
return user
_OWN_SCENARIO = "70460000-0000-4000-8000-0000000000a1"
_FOREIGN_SCENARIO = "70460000-0000-4000-8000-0000000000b2"
_HUMAN_SCENARIO = "70460000-0000-4000-8000-0000000000c3"
_HUMAN_REVISION = "70460000-0000-4000-8000-0000000000c4"
_ELIGIBLE_SCENARIO = "70460000-0000-4000-8000-0000000000d5"
_ELIGIBLE_REVISION = "70460000-0000-4000-8000-0000000000d6"
def _seed_acl_rows(db) -> dict[str, str]:
"""Hardcoded ACL fixture: one own + one foreign row per read collection."""
db.add(ScenarioRegistryEntry(
scenario_id=_OWN_SCENARIO, scenario_key="acl-own", name="ACL own", dashboard_id=71,
environment_ids=["env-preprod-01"], owner_id="acl-viewer-1", owner_username="acl.viewer",
lifecycle_status="READY", validation_status="valid",
))
db.add(ScenarioRegistryEntry(
scenario_id=_FOREIGN_SCENARIO, scenario_key="acl-foreign", name="ACL foreign", dashboard_id=72,
environment_ids=["env-preprod-01"], owner_id="foreign-owner-1", owner_username="foreign.owner",
lifecycle_status="READY", validation_status="valid",
))
own_policy = AutomationPolicy(name="acl-policy-own")
foreign_policy = AutomationPolicy(name="acl-policy-foreign")
free_policy = AutomationPolicy(name="acl-policy-free")
db.add_all([own_policy, foreign_policy, free_policy])
db.flush()
own_schedule = ScenarioSchedule(
scenario_id=_OWN_SCENARIO, environment_id="env-preprod-01", cron_expr="0 7 * * *",
policy_id=own_policy.id,
)
foreign_schedule = ScenarioSchedule(
scenario_id=_FOREIGN_SCENARIO, environment_id="env-preprod-01", cron_expr="0 8 * * *",
policy_id=foreign_policy.id,
)
own_rule = ScenarioTriggerRule(scenario_id=_OWN_SCENARIO, environment_id="env-preprod-01", trigger="api")
foreign_rule = ScenarioTriggerRule(scenario_id=_FOREIGN_SCENARIO, environment_id="env-preprod-01", trigger="api")
db.add_all([own_schedule, foreign_schedule, own_rule, foreign_rule])
own_notification = ScenarioNotificationEvent(
event_type="run_finished", scenario_id=_OWN_SCENARIO, severity="info", payload={"k": "own"},
)
foreign_notification = ScenarioNotificationEvent(
event_type="run_finished", scenario_id=_FOREIGN_SCENARIO, severity="info", payload={"k": "foreign"},
)
db.add_all([own_notification, foreign_notification])
db.add(ScenarioRun(
scenario_id=_OWN_SCENARIO, scenario_revision_id="70460000-0000-4000-8000-0000000000e1",
scenario_content_hash="a" * 64, environment_id="env-preprod-01",
idempotency_key="acl-own-run-001", trigger_source="scheduled",
))
db.add(ScenarioRun(
scenario_id=_FOREIGN_SCENARIO, scenario_revision_id="70460000-0000-4000-8000-0000000000e2",
scenario_content_hash="f" * 64, environment_id="env-preprod-01",
idempotency_key="acl-foreign-run-001", trigger_source="scheduled",
))
db.commit()
return {
"own_policy": own_policy.id, "foreign_policy": foreign_policy.id, "free_policy": free_policy.id,
"own_schedule": own_schedule.id, "foreign_schedule": foreign_schedule.id,
"own_rule": own_rule.id, "foreign_rule": foreign_rule.id,
"own_notification": own_notification.id, "foreign_notification": foreign_notification.id,
}
def _drop_acl_rows(db) -> None:
db.query(ScenarioRun).filter(ScenarioRun.idempotency_key.in_(["acl-own-run-001", "acl-foreign-run-001"])).delete()
db.query(ScenarioNotificationEvent).filter(ScenarioNotificationEvent.scenario_id.in_([_OWN_SCENARIO, _FOREIGN_SCENARIO])).delete()
db.query(ScenarioSchedule).filter(ScenarioSchedule.scenario_id.in_([_OWN_SCENARIO, _FOREIGN_SCENARIO])).delete()
db.query(ScenarioTriggerRule).filter(ScenarioTriggerRule.scenario_id.in_([_OWN_SCENARIO, _FOREIGN_SCENARIO])).delete()
db.query(AutomationPolicy).filter(AutomationPolicy.name.in_(["acl-policy-own", "acl-policy-foreign", "acl-policy-free"])).delete()
db.query(ScenarioRegistryEntry).filter(ScenarioRegistryEntry.scenario_id.in_([_OWN_SCENARIO, _FOREIGN_SCENARIO])).delete()
db.commit()
class TestScenarioAutomationReadAcl:
_READ_PATHS = (
"/api/scenario-automation/schedules",
"/api/scenario-automation/trigger-rules",
"/api/scenario-automation/policies",
"/api/scenario-automation/notifications",
"/api/scenario-automation/metrics",
"/api/scenario-automation/retention",
)
def test_reads_require_read_permission(self):
user = _make_user_with_permissions([("scenario:automation", "MANAGE"), ("scenario:automation", "TRIGGER")])
with _client_for(user) as client:
for path in self._READ_PATHS:
resp = client.get(path)
assert resp.status_code == 403, (path, resp.text)
def test_read_acl_filters_foreign_rows_without_totals_leak(self):
session = SessionLocal()
try:
ids = _seed_acl_rows(session)
finally:
session.close()
try:
with _client_for(_make_acl_user()) as client:
schedules = client.get("/api/scenario-automation/schedules")
assert schedules.status_code == 200, schedules.text
schedule_ids = {item["id"] for item in schedules.json()}
assert ids["own_schedule"] in schedule_ids
assert ids["foreign_schedule"] not in schedule_ids
rules = client.get("/api/scenario-automation/trigger-rules")
assert rules.status_code == 200, rules.text
rule_ids = {item["id"] for item in rules.json()}
assert ids["own_rule"] in rule_ids
assert ids["foreign_rule"] not in rule_ids
notifications = client.get("/api/scenario-automation/notifications", params={"limit": 100})
assert notifications.status_code == 200, notifications.text
notification_ids = {item["id"] for item in notifications.json()}
assert ids["own_notification"] in notification_ids
assert ids["foreign_notification"] not in notification_ids
policies = client.get("/api/scenario-automation/policies")
assert policies.status_code == 200, policies.text
policy_ids = {item["id"] for item in policies.json()}
assert ids["own_policy"] in policy_ids
assert ids["free_policy"] in policy_ids
assert ids["foreign_policy"] not in policy_ids
metrics = client.get("/api/scenario-automation/metrics")
assert metrics.status_code == 200, metrics.text
body = metrics.json()
assert body["schedules_total"] == 1
assert body["trigger_rules_total"] == 1
assert body["total_runs"] == 1
retention = client.get("/api/scenario-automation/retention")
assert retention.status_code == 200, retention.text
finally:
cleanup = SessionLocal()
try:
_drop_acl_rows(cleanup)
finally:
cleanup.close()
def test_admin_read_bypasses_acl_filter(self):
session = SessionLocal()
try:
ids = _seed_acl_rows(session)
finally:
session.close()
try:
with _client_for(_make_admin_user()) as client:
schedules = client.get("/api/scenario-automation/schedules")
assert schedules.status_code == 200, schedules.text
schedule_ids = {item["id"] for item in schedules.json()}
assert ids["own_schedule"] in schedule_ids
assert ids["foreign_schedule"] in schedule_ids
policies = client.get("/api/scenario-automation/policies")
policy_ids = {item["id"] for item in policies.json()}
assert ids["foreign_policy"] in policy_ids
finally:
cleanup = SessionLocal()
try:
_drop_acl_rows(cleanup)
finally:
cleanup.close()
def test_disabled_human_schedule_rejected_like_enabled(self):
setup = SessionLocal()
try:
setup.query(ScenarioRegistryEntry).filter(
ScenarioRegistryEntry.scenario_id.in_([_HUMAN_SCENARIO, _ELIGIBLE_SCENARIO])
).delete()
setup.commit()
setup.add(ScenarioRegistryEntry(
scenario_id=_HUMAN_SCENARIO, scenario_key="acl-human", name="ACL human fixture",
dashboard_id=73, environment_ids=["env-preprod-01"],
owner_id="acl-viewer-1", owner_username="acl.viewer",
lifecycle_status="READY", validation_status="valid", current_revision_id=_HUMAN_REVISION,
))
setup.add(ScenarioRevision(
revision_id=_HUMAN_REVISION, scenario_id=_HUMAN_SCENARIO, content_hash="c" * 64,
graph_snapshot={
"action_registry_version": ACTION_REGISTRY_VERSION,
"action_registry_hash": action_registry_fingerprint(),
"steps": [_action_step("human-acl-046", "human", "human_checkpoint")],
"dependencies": [],
},
created_by="acl.viewer", activation_status="current",
))
setup.add(ScenarioRegistryEntry(
scenario_id=_ELIGIBLE_SCENARIO, scenario_key="acl-eligible", name="ACL eligible fixture",
dashboard_id=74, environment_ids=["env-preprod-01"],
owner_id="acl-viewer-1", owner_username="acl.viewer",
lifecycle_status="READY", validation_status="valid", current_revision_id=_ELIGIBLE_REVISION,
))
setup.add(ScenarioRevision(
revision_id=_ELIGIBLE_REVISION, scenario_id=_ELIGIBLE_SCENARIO, content_hash="e" * 64,
graph_snapshot={
"action_registry_version": ACTION_REGISTRY_VERSION,
"action_registry_hash": action_registry_fingerprint(),
"steps": [_action_step("step-acl-046", "assertion", "structural_assert")],
"dependencies": [], "environment_ids": ["env-preprod-01"],
},
created_by="acl.viewer", activation_status="current",
))
setup.commit()
finally:
setup.close()
user = _make_user_with_permissions([("scenario:automation", "MANAGE")])
try:
with _client_for(user) as client:
enabled = client.post(
"/api/scenario-automation/schedules",
json={
"scenario_id": _HUMAN_SCENARIO, "environment_id": "env-preprod-01",
"cron_expr": "0 7 * * *", "revision_policy": "pinned",
"revision_id": _HUMAN_REVISION, "enabled": True,
},
)
assert enabled.status_code == 409, enabled.text
assert enabled.json()["detail"]["code"] == "AUTOMATION_INELIGIBLE_HUMAN_STEP"
disabled = client.post(
"/api/scenario-automation/schedules",
json={
"scenario_id": _HUMAN_SCENARIO, "environment_id": "env-preprod-01",
"cron_expr": "0 7 * * *", "revision_policy": "pinned",
"revision_id": _HUMAN_REVISION, "enabled": False,
},
)
assert disabled.status_code == 409, disabled.text
assert disabled.json()["detail"]["code"] == "AUTOMATION_INELIGIBLE_HUMAN_STEP"
created = client.post(
"/api/scenario-automation/schedules",
json={
"scenario_id": _ELIGIBLE_SCENARIO, "environment_id": "env-preprod-01",
"cron_expr": "0 6 * * *", "revision_policy": "pinned",
"revision_id": _ELIGIBLE_REVISION, "enabled": True,
},
)
assert created.status_code == 201, created.text
schedule_id = created.json()["id"]
patch = client.patch(
f"/api/scenario-automation/schedules/{schedule_id}",
json={
"scenario_id": _HUMAN_SCENARIO, "environment_id": "env-preprod-01",
"cron_expr": "0 6 * * *", "revision_policy": "pinned",
"revision_id": _HUMAN_REVISION, "enabled": False,
},
)
assert patch.status_code == 409, patch.text
assert patch.json()["detail"]["code"] == "AUTOMATION_INELIGIBLE_HUMAN_STEP"
verify = SessionLocal()
try:
assert verify.query(ScenarioSchedule).filter(
ScenarioSchedule.scenario_id == _HUMAN_SCENARIO
).count() == 0
row = verify.query(ScenarioSchedule).filter(ScenarioSchedule.id == schedule_id).first()
assert row is not None
assert row.scenario_id == _ELIGIBLE_SCENARIO
assert row.revision_id == _ELIGIBLE_REVISION
finally:
verify.close()
# Remove the scheduler job through the API so no in-memory APScheduler
# registration survives the fixture teardown.
deleted = client.delete(f"/api/scenario-automation/schedules/{schedule_id}")
assert deleted.status_code == 204
finally:
cleanup = SessionLocal()
try:
cleanup.query(ScenarioSchedule).filter(
ScenarioSchedule.scenario_id.in_([_HUMAN_SCENARIO, _ELIGIBLE_SCENARIO])
).delete()
cleanup.query(ScenarioRevision).filter(
ScenarioRevision.scenario_id.in_([_HUMAN_SCENARIO, _ELIGIBLE_SCENARIO])
).delete()
cleanup.query(ScenarioRegistryEntry).filter(
ScenarioRegistryEntry.scenario_id.in_([_HUMAN_SCENARIO, _ELIGIBLE_SCENARIO])
).delete()
cleanup.commit()
finally:
cleanup.close()
# #endregion Test.Api.ScenarioAutomation.ReadAcl
# #region Test.Api.ScenarioAutomation.RetentionReceipts [C:3] [TYPE Class] [SEMANTICS test,api,scenario,automation,retention,receipts,acl]
# @BRIEF 046 T021: the retention read projects deletion receipts under the same DG-2 ACL as the
# other five operational reads (READ grant + per-object scenario ownership, no totals leak).
# @TEST_INVARIANT Api.ScenarioAutomation.ReadAcl: foreign deletion receipts are filtered from the
# projection without leaking their count. -> VERIFIED_BY:
# test_retention_receipts_projected_with_acl
class TestScenarioAutomationRetentionReceipts:
def test_retention_receipts_projected_with_acl(self):
setup = SessionLocal()
try:
setup.query(ScenarioRetentionDeletion).filter(
ScenarioRetentionDeletion.scenario_id.in_([_OWN_SCENARIO, _FOREIGN_SCENARIO])
).delete()
setup.query(ScenarioRegistryEntry).filter(
ScenarioRegistryEntry.scenario_id.in_([_OWN_SCENARIO, _FOREIGN_SCENARIO])
).delete()
setup.add(ScenarioRegistryEntry(
scenario_id=_OWN_SCENARIO, scenario_key="acl-receipt-own", name="Receipt own",
dashboard_id=81, environment_ids=["env-preprod-01"],
owner_id="acl-viewer-1", owner_username="acl.viewer",
lifecycle_status="READY", validation_status="valid",
))
setup.add(ScenarioRegistryEntry(
scenario_id=_FOREIGN_SCENARIO, scenario_key="acl-receipt-foreign", name="Receipt foreign",
dashboard_id=82, environment_ids=["env-preprod-01"],
owner_id="foreign-owner-1", owner_username="foreign.owner",
lifecycle_status="READY", validation_status="valid",
))
setup.add(ScenarioRetentionDeletion(
target_type="artifact", target_id="receipt-own-001", scenario_id=_OWN_SCENARIO,
state="deletion_pending", holds_snapshot={"reasons": ["active_operation"]},
))
setup.add(ScenarioRetentionDeletion(
target_type="artifact", target_id="receipt-foreign-001", scenario_id=_FOREIGN_SCENARIO,
state="tombstoned", holds_snapshot={"reasons": []},
))
setup.commit()
finally:
setup.close()
try:
with _client_for(_make_acl_user()) as client:
body = client.get("/api/scenario-automation/retention").json()
assert body["tiers"]["raw_vlm"] == 7
receipt_targets = {item["target_id"] for item in body["deletions"]}
assert "receipt-own-001" in receipt_targets
assert "receipt-foreign-001" not in receipt_targets
assert body["deletions_total"] == 1
with _client_for(_make_admin_user()) as client:
admin_body = client.get("/api/scenario-automation/retention").json()
admin_targets = {item["target_id"] for item in admin_body["deletions"]}
assert {"receipt-own-001", "receipt-foreign-001"} <= admin_targets
finally:
cleanup = SessionLocal()
try:
cleanup.query(ScenarioRetentionDeletion).filter(
ScenarioRetentionDeletion.scenario_id.in_([_OWN_SCENARIO, _FOREIGN_SCENARIO])
).delete()
cleanup.query(ScenarioRegistryEntry).filter(
ScenarioRegistryEntry.scenario_id.in_([_OWN_SCENARIO, _FOREIGN_SCENARIO])
).delete()
cleanup.commit()
finally:
cleanup.close()
# #endregion Test.Api.ScenarioAutomation.RetentionReceipts
# #endregion Test.Api.ScenarioAutomation

View File

@@ -3,6 +3,7 @@
# @LAYER Test
# @RELATION VERIFIES -> [Api.ScenarioRun.Center.List]
# @TEST_EDGE: empty->empty envelope; filters->scoped rows; waiting_for_me->only pending-human runs
# @TEST_EDGE: pending_approval->only pending_approval runs (UX-7 badge projection); page_size=1 bounded
from __future__ import annotations
from datetime import UTC, datetime
@@ -137,4 +138,52 @@ def test_waiting_for_me_returns_only_pending_human_runs(run_center_env):
body = run_center_env.client().get("/api/scenario-runs", params={"waiting_for_me": True}).json()
assert body["total"] == 1
assert body["items"][0]["id"] == "run-1"
def _seed_pending_approval_runs(env: RunCenterEnv) -> None:
"""Self-contained seeding so the shared seed() totals used by earlier tests stay intact."""
session = env.session_factory()
try:
now = datetime.now(UTC)
for i, (run_id, idem) in enumerate([("run-pa-1", "idem-pa-1"), ("run-pa-2", "idem-pa-2")]):
session.add(ScenarioRun(
id=run_id, scenario_id="s-1", scenario_revision_id="r-1",
scenario_content_hash=chr(ord("c") + i) * 64, environment_id="ss-prod",
status="pending_approval", phase="preflight", trigger_source="schedule",
idempotency_key=idem, runner_plan={}, parameter_bindings={}, target_snapshot={},
created_at=now,
))
session.commit()
finally:
session.close()
def test_pending_approval_projection_returns_only_pending_approval_runs(run_center_env):
run_center_env.seed()
_seed_pending_approval_runs(run_center_env)
client = run_center_env.client()
body = client.get("/api/scenario-runs", params={"pending_approval": True}).json()
assert body["total"] == 2
assert {item["id"] for item in body["items"]} == {"run-pa-1", "run-pa-2"}
assert all(item["status"] == "pending_approval" for item in body["items"])
assert all(item["pending_checkpoint"] is None for item in body["items"])
# waiting_for_me semantics unchanged: pending-approval runs are not human checkpoints.
waiting = client.get("/api/scenario-runs", params={"waiting_for_me": True}).json()
assert waiting["total"] == 1
assert waiting["items"][0]["id"] == "run-1"
# Same listing visibility: the unfiltered listing still sees every run.
assert client.get("/api/scenario-runs").json()["total"] == 4
def test_pending_approval_projection_bounded_envelope(run_center_env):
run_center_env.seed()
_seed_pending_approval_runs(run_center_env)
client = run_center_env.client()
body = client.get("/api/scenario-runs", params={"pending_approval": True, "page_size": 1}).json()
assert body["total"] == 2
assert len(body["items"]) == 1
# #endregion Test.ScenarioRun.Center

View File

@@ -0,0 +1,347 @@
# #region Test.ScenarioExecution.BrowserLimitsCleanup [C:4] [TYPE Module] [SEMANTICS test,scenario,execution,provider,browser,limits,cleanup,mutation]
# @BRIEF 044 T034 rounds 3-4: context/auth timeout bound (120s), session page bound (3), and
# restore_fixture cleanup execution with typed failure (BROWSER_MUTATION_CLEANUP_FAILED).
# @RELATION BINDS_TO -> [ScenarioExecution.BrowserProvider.Transport]
# @RELATION VERIFIES -> [ScenarioExecution.BrowserProvider.MutationSQL.Cleanup]
# @RELATION VERIFIES -> [ScenarioExecution.BrowserProvider.Factory]
# @TEST_FIXTURE: fake page/context/service seams; hardcoded pre-image rows and hashes (INVARIANT-01).
# @TEST_EDGE cleanup restored_hash == pre_hash -> fixture_restored checkpoint, outcome completed
# @TEST_EDGE cleanup restored_hash != pre_hash -> BrowserTransportCleanupFailed before evidence
# @TEST_EDGE provider maps cleanup failure -> inconclusive BROWSER_MUTATION_CLEANUP_FAILED,
# receipt reconciliation_required/completed (mutation effect is known, env is dirty)
# @TEST_EDGE page overflow -> extras closed oldest-first, driving page survives; lost driving page
# -> typed BROWSER_PAGE_LOST
# @TEST_EDGE context/auth bound -> asyncio.TimeoutError from open_session on a slow login
# @TEST_INVARIANT ScenarioExecution.BrowserProvider: A mutation with cleanup_policy=restore_fixture
# never reports completed while the fixture is un-restored. ->
# VERIFIED_BY: cleanup_success, cleanup_failure, provider_cleanup_mapping
# @TEST_INVARIANT ScenarioExecution.BrowserProvider.Transport: Session resources are bounded
# (120s context/auth, 3 pages) with typed failures, never silent growth. ->
# VERIFIED_BY: page_bound_*, context_auth_timeout
from __future__ import annotations
import asyncio
import uuid
from pathlib import Path
from typing import Any
from unittest.mock import AsyncMock
import pytest
from src.core.database import SessionLocal
from src.models.provider_operation import ProviderOperationReceipt
from src.services.dashboard_testing.execution.live_adapter import LiveAdapterResult
from src.services.dashboard_testing.execution.live_binding import LiveExecutionBinding
from src.services.dashboard_testing.execution.provider_runtime import ProviderEventLoop
from src.services.dashboard_testing.execution.providers import browser as browser_provider_module
from src.services.dashboard_testing.execution.providers import browser_transport as transport_module
from src.services.dashboard_testing.execution.providers.browser import build_browser_provider
from src.services.dashboard_testing.execution.providers.browser_transport import (
BrowserTransportCleanupFailed,
_execute_on_page,
build_playwright_browser_transport,
)
_PNG = b"\x89PNG\r\n\x1a\n" + b"limits-cleanup" * 4
_PRE_HASH = "a" * 64
_MUTATION_INPUT = {
"table": "global_sales",
"key_columns": ["game"],
"assignments": {"sales": 0},
"database_id": 7,
}
_MUTATION_CONTRACT = {
"fixture_lease_id": "fixture-lease-lc",
"target_keys": ["Wii Sports"],
"field_allowlist": ["sales"],
"precondition_hash": _PRE_HASH,
"cleanup_policy": "restore_fixture",
"retry_safe": False,
}
# #region Test.ScenarioExecution.BrowserLimitsCleanup.Fakes [C:1] [TYPE Module]
class _FakePage:
"""Drives _execute_on_page: evaluate switches on the script kind (mutation vs cleanup)."""
def __init__(self, *, mutation_flow: dict, cleanup_result: dict | None = None, pages: int = 1):
self._mutation_flow = mutation_flow
self._cleanup_result = cleanup_result
self.url = "http://superset.test/superset/dashboard/42/"
self._closed = False
self.context = _FakeContext(self, extra_pages=pages - 1)
self.evaluated: list[str] = []
def is_closed(self) -> bool:
return self._closed
async def evaluate(self, script: str) -> Any:
self.evaluated.append(script)
if '"pre_rows"' in script:
return self._cleanup_result
return self._mutation_flow
async def screenshot(self, **_kwargs) -> bytes:
return _PNG
async def reload(self, **_kwargs) -> None:
return None
async def title(self) -> str:
return "Dashboard"
class _FakeExtraPage:
def __init__(self) -> None:
self.closed = False
def is_closed(self) -> bool:
return self.closed
async def close(self) -> None:
self.closed = True
class _FakeContext:
def __init__(self, driving: Any, *, extra_pages: int):
self.pages: list[Any] = [driving] + [_FakeExtraPage() for _ in range(extra_pages)]
self.closed = False
async def close(self) -> None:
self.closed = True
class _FakeService:
pass
# #endregion Test.ScenarioExecution.BrowserLimitsCleanup.Fakes
# #region Test.ScenarioExecution.BrowserLimitsCleanup.Cleanup [C:3] [TYPE Function]
# @BRIEF restore_fixture reverts the mutation and adds the fixture_restored checkpoint on success.
@pytest.mark.asyncio()
async def test_cleanup_success_restores_fixture_and_checkpoints():
page = _FakePage(
mutation_flow={
"precondition_ok": True,
"pre_rows": [{"game": "Wii Sports", "sales": 82.53}],
"post_rows": [{"game": "Wii Sports", "sales": 0}],
"pre_hash": _PRE_HASH,
"post_hash": "b" * 64,
},
cleanup_result={"restored_hash": _PRE_HASH, "rows_restored": 1},
)
outcome = await _execute_on_page(
_FakeService(), page, "row_edit",
action_input={**_MUTATION_INPUT, "mutation_contract": _MUTATION_CONTRACT},
timeout_seconds=30,
)
assert outcome.checkpoints == ("dashboard_open", "row_edited", "fixture_restored")
assert outcome.effect_state == "completed"
assert len(page.evaluated) == 2 # mutation script + cleanup script
assert '"pre_rows"' in page.evaluated[1]
@pytest.mark.asyncio()
async def test_cleanup_failure_raises_typed_before_evidence():
page = _FakePage(
mutation_flow={
"precondition_ok": True,
"pre_rows": [{"game": "Wii Sports", "sales": 82.53}],
"post_rows": [{"game": "Wii Sports", "sales": 0}],
"pre_hash": _PRE_HASH,
"post_hash": "b" * 64,
},
cleanup_result={"restored_hash": "c" * 64, "rows_restored": 1},
)
with pytest.raises(BrowserTransportCleanupFailed):
await _execute_on_page(
_FakeService(), page, "row_edit",
action_input={**_MUTATION_INPUT, "mutation_contract": _MUTATION_CONTRACT},
timeout_seconds=30,
)
@pytest.mark.asyncio()
async def test_retain_policy_skips_cleanup():
page = _FakePage(
mutation_flow={
"precondition_ok": True,
"pre_rows": [{"game": "Wii Sports", "sales": 82.53}],
"post_rows": [{"game": "Wii Sports", "sales": 0}],
"pre_hash": _PRE_HASH,
"post_hash": "b" * 64,
},
)
contract = {**_MUTATION_CONTRACT, "cleanup_policy": "retain"}
outcome = await _execute_on_page(
_FakeService(), page, "row_edit",
action_input={**_MUTATION_INPUT, "mutation_contract": contract},
timeout_seconds=30,
)
assert outcome.checkpoints == ("dashboard_open", "row_edited")
assert len(page.evaluated) == 1
# #endregion Test.ScenarioExecution.BrowserLimitsCleanup.Cleanup
# #region Test.ScenarioExecution.BrowserLimitsCleanup.PageBound [C:3] [TYPE Function]
# @BRIEF The session page bound closes extras oldest-first; a lost driving page is typed.
@pytest.mark.asyncio()
async def test_page_bound_closes_extras_and_keeps_driving_page():
page = _FakePage(
mutation_flow={"precondition_ok": True, "pre_rows": [], "post_rows": [], "pre_hash": _PRE_HASH, "post_hash": "b" * 64},
pages=5,
)
outcome = await _execute_on_page(
_FakeService(), page, "refresh", action_input={}, timeout_seconds=30,
)
assert "refreshed" in outcome.checkpoints
live = [p for p in page.context.pages if not p.is_closed()]
assert len(live) <= transport_module._MAX_PAGES_PER_SESSION
assert page in live
@pytest.mark.asyncio()
async def test_lost_driving_page_is_typed():
page = _FakePage(
mutation_flow={"precondition_ok": True, "pre_rows": [], "post_rows": [], "pre_hash": _PRE_HASH, "post_hash": "b" * 64},
)
page._closed = True
with pytest.raises(ValueError, match="BROWSER_PAGE_LOST"):
await _execute_on_page(_FakeService(), page, "refresh", action_input={}, timeout_seconds=30)
# #endregion Test.ScenarioExecution.BrowserLimitsCleanup.PageBound
# #region Test.ScenarioExecution.BrowserLimitsCleanup.ContextAuthTimeout [C:3] [TYPE Function]
# @BRIEF open_session bounds launch+login at the module constant; a slow login is a typed timeout.
@pytest.mark.asyncio()
async def test_context_auth_timeout_is_bounded(monkeypatch):
monkeypatch.setattr(transport_module, "_CONTEXT_AUTH_TIMEOUT_SECONDS", 0.05)
class _FakePlaywrightCM:
async def __aenter__(self):
return object()
async def __aexit__(self, *args):
return False
import playwright.async_api as pw
monkeypatch.setattr(pw, "async_playwright", lambda: _FakePlaywrightCM())
class _SlowService:
async def _launch_and_login(self, _playwright, _dashboard_id: str):
await asyncio.sleep(1.0)
return (object(), object(), object())
transport = build_playwright_browser_transport(_SlowService())
with pytest.raises((TimeoutError, asyncio.TimeoutError)):
await transport.open_session(42, timeout_seconds=30)
# #endregion Test.ScenarioExecution.BrowserLimitsCleanup.ContextAuthTimeout
# #region Test.ScenarioExecution.BrowserLimitsCleanup.ProviderMapping [C:3] [TYPE Function]
# @BRIEF Cleanup failure maps to inconclusive + reconciliation_required receipt (effect completed).
def _binding() -> LiveExecutionBinding:
return LiveExecutionBinding(
binding_ref=f"bind-lc-{uuid.uuid4().hex[:8]}",
environment_id=f"env-lc-{uuid.uuid4().hex[:8]}",
dashboard_release_id="release-lc",
release_fingerprint="f" * 64,
dashboard_id=42,
query_model_fingerprint="q" * 64,
execution_principal_fingerprint="p" * 64,
rls_security_fingerprint="r" * 64,
browser_safe_checkpoint_ref=None,
browser_action_binding_ref="action-bind-lc",
evidence_owner_type="scenario_run",
evidence_ref_policy="draft_storage_raw_response",
)
@pytest.fixture()
def loop_runtime():
runtime = ProviderEventLoop()
runtime.start()
yield runtime
runtime.stop()
@pytest.fixture(autouse=True)
def _clean_receipts():
yield
with SessionLocal() as db:
db.query(ProviderOperationReceipt).delete()
db.commit()
def test_provider_maps_cleanup_failure_to_reconciliation(loop_runtime, monkeypatch):
binding = _binding()
step = {
"scenario_run_id": f"run-lc-{uuid.uuid4().hex[:8]}",
"live_execution_binding_ref": binding.binding_ref,
"live_execution_binding_snapshot": binding.snapshot(),
"target_snapshot": {
"environment_id": binding.environment_id,
"dashboard_release_id": binding.dashboard_release_id,
"environment_class": "DEV",
},
"execution_principal_fingerprint": binding.execution_principal_fingerprint,
"step_meta": {
"environment_id": binding.environment_id,
"dashboard_id": binding.dashboard_id,
"logical_step_id": "mut-lc",
"attempt": 1,
"action_descriptor": {
"action": "row_edit",
"tool": "browser",
"mutating": True,
"risk": "mutation",
"inputs": dict(_MUTATION_INPUT),
},
"mutation_contract": _MUTATION_CONTRACT,
},
}
from src.services.dashboard_testing.execution.providers.browser_transport import BrowserTransportOutcome
class _CleanupFailTransport:
async def execute(self, dashboard_id, action, *, action_input, timeout_seconds):
raise BrowserTransportCleanupFailed("BROWSER_MUTATION_CLEANUP_FAILED")
monkeypatch.setattr(
browser_provider_module, "is_session_capable_transport", lambda _t: False
)
provider = build_browser_provider(transport=_CleanupFailTransport(), storage=None, loop=loop_runtime)
class _Ctx:
def __init__(self):
self.binding = binding
self.step = step
self.completed: dict = {}
result = provider(_Ctx())
assert result.status == "inconclusive"
assert result.reason_code == "BROWSER_MUTATION_CLEANUP_FAILED"
assert result.details["reconciliation_required"] is True
assert result.details["effect_state"] == "completed"
with SessionLocal() as db:
receipt = (
db.query(ProviderOperationReceipt)
.filter(ProviderOperationReceipt.run_id == step["scenario_run_id"])
.one()
)
assert receipt.status == "reconciliation_required"
assert receipt.effect_state == "completed"
# #endregion Test.ScenarioExecution.BrowserLimitsCleanup.ProviderMapping
# #endregion Test.ScenarioExecution.BrowserLimitsCleanup

View File

@@ -0,0 +1,349 @@
# #region Test.ScenarioExecution.BrowserNativeFilter [C:3] [TYPE Module] [SEMANTICS test,provider,browser,native-filter,selector]
# @BRIEF Verify the read-only apply_native_filter input contract and Playwright transport branch
# without a real browser (fake page/service, monkeypatched playwright seam).
# @RELATION BINDS_TO -> [ScenarioExecution.BrowserProvider.NativeFilter]
# @TEST_EDGE: unknown wait_state -> BROWSER_WAIT_STATE_INVALID before any I/O
# @TEST_EDGE: filter bar absent -> BrowserTransportSelectorNotFound, no evidence, browser/context closed
# @TEST_EDGE: empty typed input -> current-state apply mode (no fabricated value selection)
import pytest
from src.services.dashboard_testing.execution.providers.browser_native_filter import (
BrowserTransportSelectorNotFound,
apply_native_filter_via_ui,
parse_native_filter_input,
resolve_native_filter_input,
validate_native_filter_input,
)
from src.services.dashboard_testing.execution.providers.browser_transport import (
BrowserTransportUnsupported,
build_playwright_browser_transport,
)
PNG_FIXTURE = b"\x89PNG-native-filter-fixture"
# #region Test.ScenarioExecution.BrowserNativeFilter.Fakes [C:2] [TYPE Module] [SEMANTICS test,browser,fake]
# @BRIEF Minimal page/service fakes speaking the Playwright locator protocol used by the transport.
class FakeLocator:
def __init__(
self,
*,
visible: bool = True,
count: int = 1,
text: str = "",
children: dict | None = None,
texts: dict | None = None,
):
self._visible = visible
self._count = count
self._text = text
self._children = children or {}
self._texts = texts or {}
self.clicked = False
async def count(self) -> int:
return self._count
def nth(self, index: int) -> "FakeLocator":
return self
async def is_visible(self) -> bool:
return self._visible
async def click(self, **kwargs) -> None:
self.clicked = True
async def text_content(self) -> str:
return self._text
def locator(self, selector: str) -> "FakeLocator":
return self._children.get(selector, FakeLocator(visible=False, count=0))
def get_by_text(self, text: str, exact: bool = False) -> "FakeLocator":
return self._texts.get(text, FakeLocator(visible=False, count=0))
class FakePage:
def __init__(
self,
*,
selectors: dict | None = None,
texts: dict | None = None,
rendered_charts: int = 3,
):
self._selectors = selectors or {}
self._texts = texts or {}
self._rendered_charts = rendered_charts
self.url = "http://superset.test/superset/dashboard/11/"
self.waited_states: list[str] = []
def locator(self, selector: str) -> FakeLocator:
return self._selectors.get(selector, FakeLocator(visible=False, count=0))
def get_by_text(self, text: str, exact: bool = False) -> FakeLocator:
return self._texts.get(text, FakeLocator(visible=False, count=0))
async def wait_for_load_state(self, state: str, timeout: int | None = None) -> None:
self.waited_states.append(state)
async def screenshot(self, **kwargs) -> bytes:
return PNG_FIXTURE
async def title(self) -> str:
return "Sales Dashboard"
async def evaluate(self, script: str):
return self._rendered_charts
class FakeBrowser:
def __init__(self) -> None:
self.closed = False
async def close(self) -> None:
self.closed = True
class FakeContext:
def __init__(self) -> None:
self.closed = False
async def close(self) -> None:
self.closed = True
class FakeScreenshotService:
"""Stub of ScreenshotService: launch/login, locator fallback and chart settle helpers."""
def __init__(self, page: FakePage):
self.page = page
self.browser = FakeBrowser()
self.context = FakeContext()
self.settled = False
async def _launch_and_login(self, playwright, dashboard_id: str):
return self.browser, self.context, self.page
async def _wait_for_charts_stabilized(self, page, timeout_ms: int = 15000) -> None:
self.settled = True
async def _find_first_visible_locator(self, candidates):
for locator in candidates:
try:
match_count = await locator.count()
for index in range(match_count):
candidate = locator.nth(index)
if await candidate.is_visible():
return candidate
except Exception:
continue
return None
class _FakePlaywrightCM:
async def __aenter__(self):
return object()
async def __aexit__(self, *args) -> bool:
return False
@pytest.fixture()
def playwright_stub(monkeypatch):
monkeypatch.setattr("playwright.async_api.async_playwright", lambda: _FakePlaywrightCM())
return monkeypatch
def _filter_bar_page(**overrides) -> FakePage:
control = FakeLocator(visible=True, children={".ant-select-selection-item": FakeLocator(count=1, text="North")})
bar = FakeLocator(
visible=True,
children={".ant-select-selector": control},
texts={"region": control},
)
selectors = {
'[data-test="dashboard-filters-panel"]': bar,
'button[data-test="filter-bar-apply-button"]': FakeLocator(visible=True),
}
texts = {"South": FakeLocator(visible=True)}
selectors.update(overrides.pop("selectors", {}))
texts.update(overrides.pop("texts", {}))
return FakePage(selectors=selectors, texts=texts, **overrides)
# #endregion Test.ScenarioExecution.BrowserNativeFilter.Fakes
# #region Test.ScenarioExecution.BrowserNativeFilter.Input [C:2] [TYPE Function] [SEMANTICS test,browser,native-filter,input]
# @BRIEF Pure input merge/validation: typed codes, selector_hint authority, fail-closed shapes.
def test_resolve_merges_selector_hint_from_pinned_description():
metadata = {"description": "B01: apply_native_filter | selector_hint: #native-filter"}
merged = resolve_native_filter_input(metadata, {})
assert merged["selector_hint"] == "#native-filter"
def test_resolve_prefers_explicit_descriptor_inputs():
metadata = {"description": "selector_hint: #stale"}
merged = resolve_native_filter_input(metadata, {"selector_hint": "#pinned", "column": "region", "values": ["South"]})
assert merged["selector_hint"] == "#pinned"
assert merged["column"] == "region"
assert merged["values"] == ["South"]
# UX-6: run-parameter binding stamped at RunnerPlan derivation wins over descriptor-pinned values.
def test_resolve_binds_run_param_filter_values_over_descriptor_values():
metadata = {"param_binding": {"filter_values": ["South"]}}
merged = resolve_native_filter_input(metadata, {"column": "region", "values": ["North"]})
assert merged["column"] == "region"
assert merged["values"] == ["South"]
def test_resolve_without_param_binding_keeps_ux1_descriptor_values():
assert resolve_native_filter_input({}, {"values": ["North"]}) == {"values": ["North"]}
assert resolve_native_filter_input({}, {}) == {}
def test_resolve_ignores_malformed_param_binding_shape_but_keeps_other_keys():
assert resolve_native_filter_input({"param_binding": "bogus"}, {"values": ["North"]}) == {"values": ["North"]}
assert "values" not in resolve_native_filter_input({"param_binding": {}}, {})
assert "values" not in resolve_native_filter_input({"param_binding": {"other_param": ["x"]}}, {})
def test_resolve_keeps_selector_hint_authority_with_param_binding():
metadata = {"description": "selector_hint: #pinned", "param_binding": {"filter_values": ["South"]}}
merged = resolve_native_filter_input(metadata, {"selector_hint": "#explicit"})
assert merged["selector_hint"] == "#explicit"
assert merged["values"] == ["South"]
def test_validate_accepts_empty_input_as_current_state_mode():
assert validate_native_filter_input({}) is None
assert validate_native_filter_input({"column": "region", "values": ["South"], "wait_state": "networkidle"}) is None
def test_validate_rejects_unknown_wait_state():
assert validate_native_filter_input({"wait_state": "bogus"}) == "BROWSER_WAIT_STATE_INVALID"
def test_validate_rejects_non_list_values():
assert validate_native_filter_input({"values": "South"}) == "BROWSER_FILTER_VALUES_INVALID"
assert validate_native_filter_input({"values": [1, 2]}) == "BROWSER_FILTER_VALUES_INVALID"
def test_validate_rejects_oversized_values():
assert validate_native_filter_input({"values": ["v"] * 101}) == "BROWSER_FILTER_VALUES_INVALID"
def test_validate_rejects_bad_identity_shape():
assert validate_native_filter_input({"filter_id": 42}) == "BROWSER_FILTER_INPUT_INVALID"
assert validate_native_filter_input({"column": " "}) == "BROWSER_FILTER_INPUT_INVALID"
def test_parse_normalizes_and_raises_typed_codes():
parsed = parse_native_filter_input({"column": "region", "values": ["South"]})
assert parsed == {
"filter_id": None,
"filter_name": None,
"column": "region",
"selector_hint": None,
"values": ["South"],
"wait_state": None,
}
with pytest.raises(ValueError, match="BROWSER_WAIT_STATE_INVALID"):
parse_native_filter_input({"wait_state": "bogus"})
# #endregion Test.ScenarioExecution.BrowserNativeFilter.Input
# #region Test.ScenarioExecution.BrowserNativeFilter.UiFlow [C:3] [TYPE Function] [SEMANTICS test,browser,native-filter,selector]
# @BRIEF apply_native_filter_via_ui: typed selector failure, value selection, current-state mode.
@pytest.mark.asyncio()
async def test_ui_flow_applies_typed_values_and_observes_charts():
page = _filter_bar_page()
service = FakeScreenshotService(page)
details = await apply_native_filter_via_ui(
service, page, parse_native_filter_input({"column": "region", "values": ["South"]}),
timeout_seconds=5,
)
assert details["applied"] is True
assert details["applied_mode"] == "values"
assert details["applied_values"] == ["South"]
assert details["chart_data_observed"] is True
assert service.settled is True
@pytest.mark.asyncio()
async def test_ui_flow_empty_input_applies_current_state():
page = _filter_bar_page()
service = FakeScreenshotService(page)
details = await apply_native_filter_via_ui(
service, page, parse_native_filter_input({}),
timeout_seconds=5,
)
assert details["applied"] is True
assert details["applied_mode"] == "current_state"
assert details["applied_values"] == ["North"]
@pytest.mark.asyncio()
async def test_ui_flow_missing_filter_bar_raises_typed_selector_failure():
page = FakePage()
service = FakeScreenshotService(page)
with pytest.raises(BrowserTransportSelectorNotFound, match="BROWSER_SELECTOR_NOT_FOUND"):
await apply_native_filter_via_ui(service, page, parse_native_filter_input({}), timeout_seconds=5)
# #endregion Test.ScenarioExecution.BrowserNativeFilter.UiFlow
# #region Test.ScenarioExecution.BrowserNativeFilter.Transport [C:3] [TYPE Function] [SEMANTICS test,browser,transport,native-filter]
# @BRIEF Transport branch: supported action -> checkpoints+evidence; selector miss -> typed raise; unknown -> unsupported.
@pytest.mark.asyncio()
async def test_transport_apply_native_filter_passes_with_checkpoints(playwright_stub):
page = _filter_bar_page()
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
outcome = await transport.execute(
11, "apply_native_filter",
action_input={"column": "region", "values": ["South"]},
timeout_seconds=5,
)
assert outcome.checkpoints == ("dashboard_open", "filter_applied", "charts_settled")
assert outcome.evidence_png == PNG_FIXTURE
assert outcome.page_url == "http://superset.test/superset/dashboard/11/"
assert outcome.details["applied"] is True
assert outcome.details["applied_values"] == ["South"]
assert outcome.details["chart_data_observed"] is True
assert service.context.closed is True
assert service.browser.closed is True
@pytest.mark.asyncio()
async def test_transport_selector_miss_raises_typed_and_closes_browser(playwright_stub):
page = FakePage()
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
with pytest.raises(BrowserTransportSelectorNotFound, match="BROWSER_SELECTOR_NOT_FOUND"):
await transport.execute(11, "apply_native_filter", action_input={}, timeout_seconds=5)
assert service.context.closed is True
assert service.browser.closed is True
@pytest.mark.asyncio()
async def test_transport_unknown_action_still_unsupported(playwright_stub):
service = FakeScreenshotService(FakePage())
transport = build_playwright_browser_transport(service)
with pytest.raises(BrowserTransportUnsupported, match="BROWSER_ACTION_NOT_SUPPORTED"):
await transport.execute(11, "navigate_dashboard", action_input={}, timeout_seconds=5)
@pytest.mark.asyncio()
async def test_transport_invalid_wait_state_raises_typed(playwright_stub):
page = _filter_bar_page()
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
with pytest.raises(ValueError, match="BROWSER_WAIT_STATE_INVALID"):
await transport.execute(11, "apply_native_filter", action_input={"wait_state": "bogus"}, timeout_seconds=5)
# #endregion Test.ScenarioExecution.BrowserNativeFilter.Transport
# #endregion Test.ScenarioExecution.BrowserNativeFilter

View File

@@ -0,0 +1,780 @@
# #region Test.ScenarioExecution.BrowserReadOnlyActions [C:3] [TYPE Module] [SEMANTICS test,provider,browser,readonly,navigate,extract,download,table-filter]
# @BRIEF Verify the T034 round-2 read-only action catalog: typed input validation (BROWSER_*_INVALID
# before I/O), selector miss (BROWSER_SELECTOR_NOT_FOUND, fail-closed), happy path
# checkpoints+evidence for each action, bounded extract (10 MiB) and bounded download
# (25 MiB), and session integration (navigate_tab stamps active_tab on the checkpoint).
# @RELATION BINDS_TO -> [ScenarioExecution.BrowserProvider.ReadOnlyActions]
# @RELATION BINDS_TO -> [ScenarioExecution.BrowserProvider.Transport]
# @RELATION BINDS_TO -> [ScenarioExecution.BrowserProvider.Admission]
# @TEST_EDGE: each action invalid input -> BROWSER_*_INVALID before any I/O
# @TEST_EDGE: selector miss per action -> BrowserTransportSelectorNotFound, no evidence
# @TEST_EDGE: download oversize -> BROWSER_DOWNLOAD_TOO_LARGE before storage
# @TEST_EDGE: extract output oversize -> BROWSER_EXTRACT_TOO_LARGE before evidence
# @TEST_EDGE: navigate_tab updates checkpoint.active_tab in the run-scoped session
import tempfile
import uuid
from pathlib import Path
import pytest
from src.core.database import SessionLocal
from src.models.provider_capacity import CapacityLease, CapacityQuota
from src.models.scenario_registry import ScenarioRegistryEntry
from src.models.scenario_run import ScenarioRun
from src.services.dashboard_testing.execution.live_binding import LiveExecutionBinding
from src.services.dashboard_testing.execution.live_composition import LiveProviderContext
from src.services.dashboard_testing.execution.provider_runtime import ProviderEventLoop
from src.services.dashboard_testing.execution.providers.browser import build_browser_provider
from src.services.dashboard_testing.execution.providers.browser_readonly_actions import (
validate_readonly_action_input,
)
from src.services.dashboard_testing.execution.providers.browser_session import (
BrowserSessionManager,
fold_checkpoint_state,
)
from src.services.dashboard_testing.execution.providers.browser_transport import (
BrowserTransportOutcome,
BrowserTransportSelectorNotFound,
BrowserTransportUnsupported,
build_playwright_browser_transport,
)
PNG_FIXTURE = b"\x89PNG-readonly-actions-fixture"
# #region Test.ScenarioExecution.BrowserReadOnlyActions.Fakes [C:2] [TYPE Module] [SEMANTICS test,browser,readonly,fake]
# @BRIEF Minimal page/service/download fakes speaking the Playwright protocol used by the transport.
class FakeLocator:
def __init__(self, *, visible=True, count=1, text="", children=None, texts=None, eval_result=None):
self._visible = visible
self._count = count
self._text = text
self._children = children or {}
self._texts = texts or {}
self._eval_result = eval_result
self.clicked = False
self.filled = None
self.scrolled = False
async def count(self) -> int:
return self._count
def nth(self, index: int) -> "FakeLocator":
return self
async def is_visible(self) -> bool:
return self._visible
async def click(self, **kwargs) -> None:
self.clicked = True
async def fill(self, value, **kwargs) -> None:
self.filled = value
async def scroll_into_view_if_needed(self, **kwargs) -> None:
self.scrolled = True
async def text_content(self) -> str:
return self._text
def locator(self, selector: str) -> "FakeLocator":
return self._children.get(selector, FakeLocator(visible=False, count=0))
def get_by_text(self, text: str, exact: bool = False) -> "FakeLocator":
return self._texts.get(text, FakeLocator(visible=False, count=0))
async def evaluate(self, script: str, arg=None):
return self._eval_result
class _FakeDownload:
def __init__(self, data: bytes, path: Path):
path.write_bytes(data)
self._path = path
async def path(self):
return self._path
class _FakeDownloadInfo:
def __init__(self, data: bytes, tmp_dir: Path):
self._download = _FakeDownload(data, tmp_dir / "download.bin")
@property
def value(self):
async def _get():
return self._download
return _get()
class _FakeDownloadCM:
def __init__(self, data: bytes, tmp_dir: Path):
self._info = _FakeDownloadInfo(data, tmp_dir)
async def __aenter__(self):
return self._info
async def __aexit__(self, *args) -> bool:
return False
class FakePage:
def __init__(self, *, selectors=None, texts=None, rendered_charts=3, eval_result=None, download_data=b""):
self._selectors = selectors or {}
self._texts = texts or {}
self._rendered_charts = rendered_charts
self._eval_result = eval_result
self._download_data = download_data
self._tmp_dir = Path(tempfile.mkdtemp())
self.url = "http://superset.test/superset/dashboard/11/"
self.waited_states: list[str] = []
def locator(self, selector: str) -> FakeLocator:
return self._selectors.get(selector, FakeLocator(visible=False, count=0))
def get_by_text(self, text: str, exact: bool = False) -> FakeLocator:
return self._texts.get(text, FakeLocator(visible=False, count=0))
async def wait_for_load_state(self, state: str, timeout: int | None = None) -> None:
self.waited_states.append(state)
async def screenshot(self, **kwargs) -> bytes:
return PNG_FIXTURE
async def title(self) -> str:
return "Dashboard"
async def evaluate(self, script: str):
return self._eval_result
def expect_download(self, timeout=None):
return _FakeDownloadCM(self._download_data, self._tmp_dir)
class FakeBrowser:
def __init__(self) -> None:
self.closed = False
async def close(self) -> None:
self.closed = True
class FakeContext:
def __init__(self) -> None:
self.closed = False
async def close(self) -> None:
self.closed = True
class FakeScreenshotService:
def __init__(self, page: FakePage):
self.page = page
self.browser = FakeBrowser()
self.context = FakeContext()
async def _launch_and_login(self, playwright, dashboard_id: str):
return self.browser, self.context, self.page
async def _wait_for_charts_stabilized(self, page, timeout_ms: int = 15000) -> None:
pass
async def _find_first_visible_locator(self, candidates):
for locator in candidates:
try:
match_count = await locator.count()
for index in range(match_count):
candidate = locator.nth(index)
if await candidate.is_visible():
return candidate
except Exception:
continue
return None
class _FakePlaywrightCM:
async def __aenter__(self):
return object()
async def __aexit__(self, *args) -> bool:
return False
@pytest.fixture()
def playwright_stub(monkeypatch):
monkeypatch.setattr("playwright.async_api.async_playwright", lambda: _FakePlaywrightCM())
return monkeypatch
# #region Test.ScenarioExecution.BrowserReadOnlyActions.ProviderFakes [C:2] [TYPE Module] [SEMANTICS test,browser,provider,fake]
# @BRIEF Session-capable fake transport for provider-level round-2 tests.
class FakeSessionToken:
def __init__(self, dashboard_id: int) -> None:
self.dashboard_id = dashboard_id
self.closed = False
class FakeSessionTransport:
def __init__(self, *, evidence: bytes | None = PNG_FIXTURE, download_bytes: bytes | None = None):
self.evidence = evidence
self.download_bytes = download_bytes
self.open_count = 0
self.tokens: list[FakeSessionToken] = []
self.executed: list[tuple[FakeSessionToken, str, dict]] = []
async def open_session(self, dashboard_id, *, timeout_seconds):
self.open_count += 1
token = FakeSessionToken(dashboard_id)
self.tokens.append(token)
return token
async def execute_in_session(self, session, action, *, action_input, timeout_seconds):
self.executed.append((session, action, dict(action_input)))
details: dict = {}
checkpoints: tuple[str, ...] = ("dashboard_open",)
if action == "navigate_tab":
details = {"tab": action_input.get("tab"), "tab_navigated": True}
checkpoints = ("dashboard_open", "tab_navigated")
elif action == "apply_table_filter":
details = {"column": action_input.get("column"), "value": action_input.get("value"), "table_filter_applied": True}
checkpoints = ("dashboard_open", "table_filter_applied")
elif action == "inspect_filter_state":
details = {"filter_count": 2, "filter_controls": [{"text": "Region", "selected_values": ["North"]}]}
checkpoints = ("dashboard_open", "filter_state_inspected")
elif action == "download":
details = {"artifact_type": action_input.get("artifact_type") or "xlsx", "byte_size": len(self.download_bytes or b""), "downloaded": True}
checkpoints = ("dashboard_open", "downloaded")
return BrowserTransportOutcome(
checkpoints=checkpoints,
page_url=f"http://superset.test/superset/dashboard/{session.dashboard_id}/",
evidence_png=self.evidence,
details={"title": "Dashboard", **details},
effect_state="completed",
download_bytes=self.download_bytes,
)
async def close_session(self, session):
session.closed = True
class FakeStorage:
def __init__(self) -> None:
self.stored: dict[str, bytes] = {}
def store(self, run_id: str, sha256: str, data: bytes) -> str:
self.stored[f"draft:{run_id}:{sha256}"] = data
return f"draft:{run_id}:{sha256}"
def build_binding() -> LiveExecutionBinding:
return LiveExecutionBinding(
binding_ref=f"bind-{uuid.uuid4().hex[:8]}",
environment_id=f"env-{uuid.uuid4().hex[:8]}",
dashboard_release_id="release-1",
release_fingerprint="f" * 64,
dashboard_id=42,
query_model_fingerprint="q" * 64,
execution_principal_fingerprint="p" * 64,
rls_security_fingerprint="r" * 64,
browser_safe_checkpoint_ref=None,
browser_action_binding_ref="action-bind-1",
evidence_owner_type="scenario_run",
evidence_ref_policy="draft_storage_raw_response",
)
def descriptor(action: str) -> dict:
return {"action": action, "tool": "browser", "mutating": False, "risk": "read_only"}
def build_step(binding: LiveExecutionBinding, run_id: str | None = None, **overrides):
step = {
"scenario_run_id": run_id or str(uuid.uuid4()),
"live_execution_binding_ref": binding.binding_ref,
"live_execution_binding_snapshot": binding.snapshot(),
"target_snapshot": {"environment_id": binding.environment_id, "dashboard_release_id": binding.dashboard_release_id, "environment_class": "DEV"},
"execution_principal_fingerprint": binding.execution_principal_fingerprint,
"step_meta": {
"environment_id": binding.environment_id,
"dashboard_id": binding.dashboard_id,
"logical_step_id": str(uuid.uuid4()),
"action_descriptor": descriptor("open_dashboard"),
},
}
step.update(overrides)
return step
# #endregion Test.ScenarioExecution.BrowserReadOnlyActions.ProviderFakes
# #endregion Test.ScenarioExecution.BrowserReadOnlyActions.Fakes
# #region Test.ScenarioExecution.BrowserReadOnlyActions.Validate [C:2] [TYPE Function] [SEMANTICS test,browser,readonly,validate]
# @BRIEF Pure input validation returns typed codes or None; no I/O.
@pytest.mark.parametrize(
"action, inputs, expected",
[
("navigate_tab", {"tab": "charts"}, None),
("navigate_tab", {}, "BROWSER_NAVIGATE_TAB_INVALID"),
("navigate_tab", {"tab": ""}, "BROWSER_NAVIGATE_TAB_INVALID"),
("navigate_tab", {"tab": "charts", "selector_hint": " "}, "BROWSER_NAVIGATE_TAB_INVALID"),
("apply_table_filter", {"column": "region", "value": "North"}, None),
("apply_table_filter", {"column": ""}, "BROWSER_TABLE_FILTER_INVALID"),
("apply_table_filter", {"column": "region", "value": 42}, "BROWSER_TABLE_FILTER_INVALID"),
("extract_table", {}, None),
("extract_table", {"max_rows": 0}, "BROWSER_EXTRACT_BOUNDS_INVALID"),
("extract_table", {"max_rows": 10001}, "BROWSER_EXTRACT_BOUNDS_INVALID"),
("extract_table", {"max_columns": 101}, "BROWSER_EXTRACT_BOUNDS_INVALID"),
("scroll_to", {"selector": ".chart"}, None),
("scroll_to", {}, "BROWSER_SCROLL_INVALID"),
("scroll_to", {"selector": ".chart", "direction": "diagonal"}, "BROWSER_SCROLL_INVALID"),
("click", {"selector": ".button"}, None),
("click", {}, "BROWSER_CLICK_INVALID"),
("select_rows", {"row_keys": ["row1"]}, None),
("select_rows", {}, "BROWSER_SELECT_ROWS_INVALID"),
("select_rows", {"row_keys": []}, "BROWSER_SELECT_ROWS_INVALID"),
("select_rows", {"row_keys": [1, 2]}, "BROWSER_SELECT_ROWS_INVALID"),
("download", {}, None),
("download", {"artifact_type": "xlsx"}, None),
("download", {"artifact_type": "bogus"}, "BROWSER_DOWNLOAD_INVALID"),
("inspect_filter_state", {}, None),
("inspect_columns", {}, None),
],
)
def test_validate_readonly_action_input_typed_codes(action, inputs, expected):
assert validate_readonly_action_input(action, inputs) == expected
# #endregion Test.ScenarioExecution.BrowserReadOnlyActions.Validate
# #region Test.ScenarioExecution.BrowserReadOnlyActions.CheckpointFold [C:2] [TYPE Function] [SEMANTICS test,browser,session,checkpoint]
# @BRIEF fold_checkpoint_state stamps navigate_tab/apply_table_filter/inspect_filter_state state.
def test_fold_navigate_tab_stamps_active_tab():
base = fold_checkpoint_state(None, dashboard_id=42, action="open_dashboard", action_input={}, details={})
folded = fold_checkpoint_state(base, dashboard_id=42, action="navigate_tab", action_input={"tab": "charts"}, details={"tab": "charts", "tab_navigated": True})
assert folded["active_tab"] == "charts"
assert folded["checkpoint_seq"] == 2
def test_fold_apply_table_filter_upserts_by_column():
base = fold_checkpoint_state(None, dashboard_id=42, action="open_dashboard", action_input={}, details={})
f1 = fold_checkpoint_state(base, dashboard_id=42, action="apply_table_filter", action_input={"column": "region"}, details={"column": "region", "value": "North", "table_filter_applied": True})
f2 = fold_checkpoint_state(f1, dashboard_id=42, action="apply_table_filter", action_input={"column": "region"}, details={"column": "region", "value": "South", "table_filter_applied": True})
assert f2["table_filter_state"] == [{"column": "region", "value": "South"}]
def test_fold_inspect_filter_state_stamps_observed():
base = fold_checkpoint_state(None, dashboard_id=42, action="open_dashboard", action_input={}, details={})
folded = fold_checkpoint_state(base, dashboard_id=42, action="inspect_filter_state", action_input={}, details={"filter_count": 2, "filter_controls": [{"text": "Region", "selected_values": ["North"]}]})
assert folded["filter_state_observed"]["filter_count"] == 2
assert folded["filter_state_observed"]["controls"] == [{"text": "Region", "selected_values": ["North"]}]
def test_fold_unknown_action_only_bumps_sequence():
base = fold_checkpoint_state(None, dashboard_id=42, action="open_dashboard", action_input={}, details={})
folded = fold_checkpoint_state(base, dashboard_id=42, action="bogus_action", action_input={}, details={})
assert folded["checkpoint_seq"] == 2
assert folded["active_tab"] is None
assert folded["table_filter_state"] == []
# #endregion Test.ScenarioExecution.BrowserReadOnlyActions.CheckpointFold
# #region Test.ScenarioExecution.BrowserReadOnlyActions.Transport [C:3] [TYPE Function] [SEMANTICS test,browser,readonly,transport]
# @BRIEF Transport-level happy paths, selector miss and typed invalid input for each round-2 action.
@pytest.mark.asyncio()
@pytest.mark.parametrize(
"action, inputs, selectors_key, expected_checkpoint",
[
("navigate_tab", {"tab": "charts"}, "tab", "tab_navigated"),
("scroll_to", {"selector": ".chart"}, "scroll", "scrolled"),
("click", {"selector": ".btn"}, "click", "clicked"),
],
)
async def test_transport_happy_path_simple_actions(playwright_stub, action, inputs, selectors_key, expected_checkpoint):
tab_locator = FakeLocator(visible=True)
scroll_locator = FakeLocator(visible=True)
click_locator = FakeLocator(visible=True)
selectors = {
'[data-test="tab-charts"]': tab_locator,
'.chart': scroll_locator,
'.btn': click_locator,
}
page = FakePage(selectors=selectors)
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
outcome = await transport.execute(11, action, action_input=inputs, timeout_seconds=5)
assert outcome.checkpoints == ("dashboard_open", expected_checkpoint)
assert outcome.evidence_png == PNG_FIXTURE
assert service.context.closed is True
@pytest.mark.asyncio()
async def test_transport_navigate_tab_miss_raises_typed(playwright_stub):
page = FakePage() # no tab selectors
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
with pytest.raises(BrowserTransportSelectorNotFound, match="BROWSER_SELECTOR_NOT_FOUND"):
await transport.execute(11, "navigate_tab", action_input={"tab": "charts"}, timeout_seconds=5)
assert service.context.closed is True
@pytest.mark.asyncio()
async def test_transport_navigate_tab_invalid_input_raises_typed(playwright_stub):
page = FakePage()
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
with pytest.raises(ValueError, match="BROWSER_NAVIGATE_TAB_INVALID"):
await transport.execute(11, "navigate_tab", action_input={}, timeout_seconds=5)
@pytest.mark.asyncio()
async def test_transport_inspect_filter_state_happy(playwright_stub):
control = FakeLocator(visible=True, text="Region", children={".ant-select-selection-item": FakeLocator(count=1, text="North")})
bar = FakeLocator(visible=True, children={".ant-select-selector": control})
page = FakePage(selectors={'[data-test="dashboard-filters-panel"]': bar})
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
outcome = await transport.execute(11, "inspect_filter_state", action_input={}, timeout_seconds=5)
assert outcome.checkpoints == ("dashboard_open", "filter_state_inspected")
assert outcome.details["filter_count"] == 1
assert outcome.details["filter_controls"][0]["text"] == "Region"
assert outcome.details["filter_controls"][0]["selected_values"] == ["North"]
@pytest.mark.asyncio()
async def test_transport_inspect_filter_state_bar_missing(playwright_stub):
page = FakePage()
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
with pytest.raises(BrowserTransportSelectorNotFound, match="BROWSER_SELECTOR_NOT_FOUND"):
await transport.execute(11, "inspect_filter_state", action_input={}, timeout_seconds=5)
@pytest.mark.asyncio()
async def test_transport_apply_table_filter_happy(playwright_stub):
filter_control = FakeLocator(visible=True)
input_loc = FakeLocator(visible=True)
confirm = FakeLocator(visible=True)
selectors = {
'[data-test="table-filter"]': filter_control,
".ant-table-filter-dropdown input": input_loc,
".ant-table-filter-dropdown .ant-btn-primary": confirm,
}
page = FakePage(selectors=selectors)
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
outcome = await transport.execute(11, "apply_table_filter", action_input={"column": "region", "value": "North"}, timeout_seconds=5)
assert outcome.checkpoints == ("dashboard_open", "table_filter_applied")
assert outcome.details["column"] == "region"
assert outcome.details["value"] == "North"
assert input_loc.filled == "North"
@pytest.mark.asyncio()
async def test_transport_apply_table_filter_invalid_input(playwright_stub):
page = FakePage()
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
with pytest.raises(ValueError, match="BROWSER_TABLE_FILTER_INVALID"):
await transport.execute(11, "apply_table_filter", action_input={"column": ""}, timeout_seconds=5)
@pytest.mark.asyncio()
async def test_transport_extract_table_happy(playwright_stub):
eval_result = {"columns": ["Name", "Sales"], "rows": [["Wii", 82.75], ["Mario", 40.24]]}
table_loc = FakeLocator(visible=True, eval_result=eval_result)
page = FakePage(selectors={".ant-table-wrapper": table_loc})
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
outcome = await transport.execute(11, "extract_table", action_input={}, timeout_seconds=5)
assert outcome.checkpoints == ("dashboard_open", "table_extracted")
assert outcome.details["columns"] == ["Name", "Sales"]
assert outcome.details["rows"] == [["Wii", 82.75], ["Mario", 40.24]]
assert outcome.details["row_count"] == 2
assert outcome.details["column_count"] == 2
@pytest.mark.asyncio()
async def test_transport_extract_table_missing(playwright_stub):
page = FakePage()
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
with pytest.raises(BrowserTransportSelectorNotFound, match="BROWSER_SELECTOR_NOT_FOUND"):
await transport.execute(11, "extract_table", action_input={}, timeout_seconds=5)
@pytest.mark.asyncio()
async def test_transport_extract_table_oversize_raises_typed(playwright_stub, monkeypatch):
import src.services.dashboard_testing.execution.providers.browser_readonly_actions as mod
monkeypatch.setattr(mod, "_MAX_EXTRACT_OUTPUT_BYTES", 8)
eval_result = {"columns": ["A", "B"], "rows": [["1", "2"], ["3", "4"]]}
table_loc = FakeLocator(visible=True, eval_result=eval_result)
page = FakePage(selectors={".ant-table-wrapper": table_loc})
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
with pytest.raises(ValueError, match="BROWSER_EXTRACT_TOO_LARGE"):
await transport.execute(11, "extract_table", action_input={}, timeout_seconds=5)
@pytest.mark.asyncio()
async def test_transport_inspect_columns_happy(playwright_stub):
header = FakeLocator(count=3, text="Name")
table_loc = FakeLocator(visible=True, children={"th .ant-table-column-title": header})
page = FakePage(selectors={".ant-table-wrapper": table_loc})
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
outcome = await transport.execute(11, "inspect_columns", action_input={}, timeout_seconds=5)
assert outcome.checkpoints == ("dashboard_open", "columns_inspected")
assert outcome.details["columns"] == ["Name", "Name", "Name"]
assert outcome.details["column_count"] == 3
@pytest.mark.asyncio()
async def test_transport_select_rows_happy(playwright_stub):
row_loc = FakeLocator(visible=True)
checkbox = FakeLocator(visible=True)
row_loc._children = {"td .ant-checkbox-input": checkbox}
page = FakePage(texts={"Row1": row_loc, "Row2": row_loc})
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
outcome = await transport.execute(11, "select_rows", action_input={"row_keys": ["Row1", "Row2"]}, timeout_seconds=5)
assert outcome.checkpoints == ("dashboard_open", "rows_selected")
assert outcome.details["row_keys"] == ["Row1", "Row2"]
assert outcome.details["selected_count"] == 2
assert checkbox.clicked is True
@pytest.mark.asyncio()
async def test_transport_select_rows_invalid_input(playwright_stub):
page = FakePage()
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
with pytest.raises(ValueError, match="BROWSER_SELECT_ROWS_INVALID"):
await transport.execute(11, "select_rows", action_input={}, timeout_seconds=5)
@pytest.mark.asyncio()
async def test_transport_download_happy_returns_bytes(playwright_stub):
trigger = FakeLocator(visible=True)
page = FakePage(selectors={'[data-test="download-button"]': trigger}, download_data=b"xlsx-bytes")
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
outcome = await transport.execute(11, "download", action_input={"artifact_type": "xlsx"}, timeout_seconds=5)
assert outcome.checkpoints == ("dashboard_open", "downloaded")
assert outcome.download_bytes == b"xlsx-bytes"
assert outcome.details["downloaded"] is True
assert outcome.details["artifact_type"] == "xlsx"
assert outcome.details["byte_size"] == 10
@pytest.mark.asyncio()
async def test_transport_download_oversize_raises_typed(playwright_stub, monkeypatch):
import src.services.dashboard_testing.execution.providers.browser_readonly_actions as mod
monkeypatch.setattr(mod, "_MAX_DOWNLOAD_BYTES", 4)
trigger = FakeLocator(visible=True)
page = FakePage(selectors={'[data-test="download-button"]': trigger}, download_data=b"too-large")
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
with pytest.raises(ValueError, match="BROWSER_DOWNLOAD_TOO_LARGE"):
await transport.execute(11, "download", action_input={}, timeout_seconds=5)
@pytest.mark.asyncio()
async def test_transport_download_trigger_missing(playwright_stub):
page = FakePage()
service = FakeScreenshotService(page)
transport = build_playwright_browser_transport(service)
with pytest.raises(BrowserTransportSelectorNotFound, match="BROWSER_SELECTOR_NOT_FOUND"):
await transport.execute(11, "download", action_input={}, timeout_seconds=5)
# #endregion Test.ScenarioExecution.BrowserReadOnlyActions.Transport
# #region Test.ScenarioExecution.BrowserReadOnlyActions.Provider [C:3] [TYPE Function] [SEMANTICS test,provider,browser,readonly,provider]
# @BRIEF Provider-level round-2 tests: admission validation, download artifact storage, oversize.
@pytest.fixture()
def loop_runtime():
runtime = ProviderEventLoop()
runtime.start()
yield runtime
runtime.stop()
@pytest.fixture()
def clean_env():
binding = build_binding()
yield binding
with SessionLocal() as db:
db.query(CapacityLease).filter(CapacityLease.environment_id == binding.environment_id).delete()
db.query(CapacityQuota).filter(CapacityQuota.environment_id == binding.environment_id).delete()
db.commit()
def test_provider_navigate_tab_passes_with_checkpoint(loop_runtime, clean_env):
binding = clean_env
transport = FakeSessionTransport()
manager = BrowserSessionManager(transport=transport, event_loop=loop_runtime)
provider = build_browser_provider(
transport=transport, storage=FakeStorage(), loop=loop_runtime, session_manager=manager,
)
run_id = str(uuid.uuid4())
step = build_step(binding, run_id=run_id)
step["step_meta"]["action_descriptor"] = {**descriptor("navigate_tab"), "inputs": {"tab": "charts"}}
result = provider(LiveProviderContext(binding=binding, step=step, completed={}))
assert result.status == "passed"
assert result.details["checkpoints"] == ["dashboard_open", "tab_navigated"]
assert result.details["tab_navigated"] is True
assert result.details["browser_checkpoint"]["active_tab"] == "charts"
manager.close_all()
@pytest.mark.parametrize(
"action, inputs, expected_code",
[
("navigate_tab", {}, "BROWSER_NAVIGATE_TAB_INVALID"),
("apply_table_filter", {"column": ""}, "BROWSER_TABLE_FILTER_INVALID"),
("extract_table", {"max_rows": 0}, "BROWSER_EXTRACT_BOUNDS_INVALID"),
("scroll_to", {}, "BROWSER_SCROLL_INVALID"),
("click", {}, "BROWSER_CLICK_INVALID"),
("select_rows", {}, "BROWSER_SELECT_ROWS_INVALID"),
("download", {"artifact_type": "bogus"}, "BROWSER_DOWNLOAD_INVALID"),
],
)
def test_provider_invalid_readonly_input_rejected_before_io(loop_runtime, clean_env, action, inputs, expected_code):
binding = clean_env
transport = FakeSessionTransport()
provider = build_browser_provider(transport=transport, storage=FakeStorage(), loop=loop_runtime)
step = build_step(binding)
step["step_meta"]["action_descriptor"] = {**descriptor(action), "inputs": inputs}
result = provider(LiveProviderContext(binding=binding, step=step, completed={}))
assert result.status == "inconclusive"
assert result.reason_code == expected_code
assert transport.open_count == 0
def test_provider_download_stores_artifact_ref(loop_runtime, clean_env):
binding = clean_env
download_bytes = b"xlsx-content-bytes"
transport = FakeSessionTransport(download_bytes=download_bytes)
manager = BrowserSessionManager(transport=transport, event_loop=loop_runtime)
storage = FakeStorage()
provider = build_browser_provider(
transport=transport, storage=storage, loop=loop_runtime, session_manager=manager,
)
run_id = str(uuid.uuid4())
step = build_step(binding, run_id=run_id)
step["step_meta"]["action_descriptor"] = {**descriptor("download"), "inputs": {"artifact_type": "xlsx"}}
result = provider(LiveProviderContext(binding=binding, step=step, completed={}))
from hashlib import sha256
assert result.status == "passed"
assert result.details["checkpoints"] == ["dashboard_open", "downloaded"]
assert result.details["downloaded"] is True
# Two artifact refs: screenshot evidence + download artifact.
assert len(result.artifact_refs) == 2
screenshot_ref, download_ref = result.artifact_refs
assert screenshot_ref.startswith("draft:")
assert download_ref.startswith("draft:")
assert result.details["download_artifact_ref"] == download_ref
assert download_ref == f"draft:{run_id}:{sha256(download_bytes).hexdigest()}"
# Per-ref byte lengths and content types include both.
assert result.details["artifact_byte_lengths"][download_ref] == len(download_bytes)
assert result.details["artifact_byte_lengths"][screenshot_ref] == len(PNG_FIXTURE)
assert result.details["artifact_content_types"][screenshot_ref] == "image/png"
assert download_ref in result.details["artifact_content_types"]
# Storage holds the download bytes under the ref.
assert storage.stored[download_ref] == download_bytes
manager.close_all()
def test_provider_download_oversize_is_typed_inconclusive(loop_runtime, clean_env):
binding = clean_env
transport = FakeSessionTransport(download_bytes=b"x" * 64)
storage = FakeStorage()
provider = build_browser_provider(
transport=transport, storage=storage, loop=loop_runtime, max_download_bytes=16,
)
step = build_step(binding)
step["step_meta"]["action_descriptor"] = {**descriptor("download"), "inputs": {"artifact_type": "xlsx"}}
result = provider(LiveProviderContext(binding=binding, step=step, completed={}))
assert result.status == "inconclusive"
assert result.reason_code == "BROWSER_DOWNLOAD_TOO_LARGE"
assert not result.artifact_refs
def test_provider_inspect_filter_state_passes_with_details(loop_runtime, clean_env):
binding = clean_env
transport = FakeSessionTransport()
provider = build_browser_provider(transport=transport, storage=FakeStorage(), loop=loop_runtime)
step = build_step(binding)
step["step_meta"]["action_descriptor"] = descriptor("inspect_filter_state")
result = provider(LiveProviderContext(binding=binding, step=step, completed={}))
assert result.status == "passed"
assert result.details["checkpoints"] == ["dashboard_open", "filter_state_inspected"]
assert result.details["filter_count"] == 2
assert result.details["browser_checkpoint"]["filter_state_observed"]["filter_count"] == 2
# #endregion Test.ScenarioExecution.BrowserReadOnlyActions.Provider
# #region Test.ScenarioExecution.BrowserReadOnlyActions.SessionIntegration [C:3] [TYPE Function] [SEMANTICS test,browser,session,integration]
# @BRIEF Session integration: navigate_tab + apply_table_filter accumulate checkpoint state across steps.
def test_session_navigate_tab_and_apply_table_filter_accumulate_checkpoint(loop_runtime, clean_env):
binding = clean_env
transport = FakeSessionTransport()
manager = BrowserSessionManager(transport=transport, event_loop=loop_runtime)
try:
provider = build_browser_provider(
transport=transport, storage=FakeStorage(), loop=loop_runtime, session_manager=manager,
)
run_id = str(uuid.uuid4())
step1 = build_step(binding, run_id=run_id)
step1["step_meta"]["action_descriptor"] = {**descriptor("navigate_tab"), "inputs": {"tab": "charts"}}
r1 = provider(LiveProviderContext(binding=binding, step=step1, completed={}))
assert r1.status == "passed"
assert r1.details["browser_checkpoint"]["active_tab"] == "charts"
assert r1.details["browser_checkpoint"]["table_filter_state"] == []
step2 = build_step(binding, run_id=run_id)
step2["step_meta"]["action_descriptor"] = {**descriptor("apply_table_filter"), "inputs": {"column": "region", "value": "North"}}
r2 = provider(LiveProviderContext(binding=binding, step=step2, completed={}))
assert r2.status == "passed"
assert r2.details["browser_checkpoint"]["active_tab"] == "charts"
assert r2.details["browser_checkpoint"]["table_filter_state"] == [{"column": "region", "value": "North"}]
# Same live context across steps (launch+login once).
assert transport.open_count == 1
finally:
manager.close_all()
# #endregion Test.ScenarioExecution.BrowserReadOnlyActions.SessionIntegration
# #endregion Test.ScenarioExecution.BrowserReadOnlyActions

View File

@@ -0,0 +1,446 @@
# #region Test.ScenarioExecution.BrowserSession [C:4] [TYPE Module] [SEMANTICS test,provider,browser,session,checkpoint,replay,leak-guard]
# @BRIEF Verify the run-scoped browser session (DG-1 round 1): one context per run across steps,
# checkpoint stamping with filter state, replay recovery after process death, typed missing
# checkpoint, leak-guard close paths, and cross-run isolation — fake browser, no Playwright.
# @RELATION BINDS_TO -> [ScenarioExecution.BrowserProvider.Session]
# @TEST_EDGE: two steps of one run share one context (launch+login once)
# @TEST_EDGE: process death + persisted checkpoint -> open + re-apply replay before the step
# @TEST_EDGE: browser history without checkpoint -> BROWSER_CHECKPOINT_MISSING, zero I/O
# @TEST_EDGE: crash/timeout/explicit/idle close paths all close the context (leak-guard)
import uuid
import pytest
from src.core.database import SessionLocal
from src.models.provider_capacity import CapacityLease, CapacityQuota
from src.models.scenario_registry import ScenarioRegistryEntry
from src.models.scenario_run import ScenarioRun, ScenarioStepRun
from src.services.dashboard_testing.execution.live_binding import LiveExecutionBinding
from src.services.dashboard_testing.execution.live_composition import LiveProviderContext
from src.services.dashboard_testing.execution.provider_runtime import ProviderEventLoop
from src.services.dashboard_testing.execution.providers.browser import build_browser_provider
from src.services.dashboard_testing.execution.providers.browser_session import BrowserSessionManager
from src.services.dashboard_testing.execution.providers.browser_transport import BrowserTransportOutcome
PNG_FIXTURE = b"\x89PNG-browser-session-fixture"
# #region Test.ScenarioExecution.BrowserSession.Fakes [C:3] [TYPE Module] [SEMANTICS test,browser,session,fake]
# @BRIEF Session-capable fake transport and hardcoded fixtures mirroring test_provider_browser.py.
class FakeSessionToken:
def __init__(self, dashboard_id: int) -> None:
self.dashboard_id = dashboard_id
self.closed = False
class FakeSessionTransport:
"""Fake open_session/execute_in_session/close_session; open implies launch+login."""
def __init__(self, *, failure: str | None = None) -> None:
self.failure = failure
self.open_count = 0
self.tokens: list[FakeSessionToken] = []
self.executed: list[tuple[FakeSessionToken, str, dict]] = []
async def open_session(self, dashboard_id, *, timeout_seconds):
self.open_count += 1
token = FakeSessionToken(dashboard_id)
self.tokens.append(token)
return token
async def execute_in_session(self, session, action, *, action_input, timeout_seconds):
self.executed.append((session, action, dict(action_input)))
if self.failure == "timeout":
raise TimeoutError("transport deadline")
if self.failure == "crash":
raise RuntimeError("transport exploded")
details: dict = {}
if action == "apply_native_filter":
details = {
"applied": True,
"applied_mode": "values" if action_input.get("values") else "current_state",
"applied_values": list(action_input.get("values") or ["North"]),
"filter_target": action_input.get("filter_id") or action_input.get("filter_name") or action_input.get("column"),
"chart_data_observed": True,
}
return BrowserTransportOutcome(
checkpoints=("dashboard_open",),
page_url=f"http://superset.test/superset/dashboard/{session.dashboard_id}/",
evidence_png=PNG_FIXTURE,
details={"title": "Dashboard", **details},
effect_state="completed",
)
async def close_session(self, session):
session.closed = True
class FakeStorage:
def __init__(self) -> None:
self.stored: dict[str, bytes] = {}
def store(self, run_id: str, sha256: str, data: bytes) -> str:
self.stored[f"draft:{run_id}:{sha256}"] = data
return f"draft:{run_id}:{sha256}"
def build_binding() -> LiveExecutionBinding:
return LiveExecutionBinding(
binding_ref=f"bind-{uuid.uuid4().hex[:8]}",
environment_id=f"env-{uuid.uuid4().hex[:8]}",
dashboard_release_id="release-1",
release_fingerprint="f" * 64,
dashboard_id=42,
query_model_fingerprint="q" * 64,
execution_principal_fingerprint="p" * 64,
rls_security_fingerprint="r" * 64,
browser_safe_checkpoint_ref=None,
browser_action_binding_ref="action-bind-1",
evidence_owner_type="scenario_run",
evidence_ref_policy="draft_storage_raw_response",
)
def descriptor(action: str = "open_dashboard") -> dict:
return {"action": action, "tool": "browser", "mutating": False, "risk": "read_only"}
def build_step(binding: LiveExecutionBinding, run_id: str | None = None, **overrides):
step = {
"scenario_run_id": run_id or str(uuid.uuid4()),
"live_execution_binding_ref": binding.binding_ref,
"live_execution_binding_snapshot": binding.snapshot(),
"target_snapshot": {"environment_id": binding.environment_id, "dashboard_release_id": binding.dashboard_release_id, "environment_class": "DEV"},
"execution_principal_fingerprint": binding.execution_principal_fingerprint,
"step_meta": {
"environment_id": binding.environment_id,
"dashboard_id": binding.dashboard_id,
"logical_step_id": str(uuid.uuid4()),
"action_descriptor": descriptor(),
},
}
step.update(overrides)
return step
def _persist_run(environment_id: str, run_id: str | None = None) -> str:
"""Persist the minimal registry-entry/run parent chain for scenario_step_runs FK."""
scenario_id = str(uuid.uuid4())
run_id = run_id or str(uuid.uuid4())
with SessionLocal() as db:
db.add(ScenarioRegistryEntry(
scenario_id=scenario_id,
scenario_key=f"sc-{uuid.uuid4().hex[:8]}",
name="Browser session fixture",
dashboard_id=42,
owner_id="qa-1",
owner_username="qa.analyst",
))
db.add(ScenarioRun(
id=run_id,
scenario_id=scenario_id,
scenario_revision_id=str(uuid.uuid4()),
scenario_content_hash="c" * 64,
environment_id=environment_id,
status="running",
phase="execute",
idempotency_key=f"idem-{uuid.uuid4().hex}",
))
db.commit()
return run_id
def _persist_browser_step(run_id: str, *, checkpoint: dict | None, status: str = "passed") -> None:
with SessionLocal() as db:
db.add(ScenarioStepRun(
run_id=run_id,
logical_step_id=f"step-{uuid.uuid4().hex[:6]}",
step_position=1,
attempt=1,
status=status,
step_outcome={
"tool": "browser",
"reason_code": "BROWSER_ACTION_EXECUTED" if status == "passed" else "BROWSER_ACTION_FAILED",
**({"browser_checkpoint": checkpoint} if checkpoint is not None else {}),
},
))
db.commit()
# #endregion Test.ScenarioExecution.BrowserSession.Fakes
@pytest.fixture()
def loop_runtime():
runtime = ProviderEventLoop()
runtime.start()
yield runtime
runtime.stop()
@pytest.fixture()
def binding():
live_binding = build_binding()
yield live_binding
with SessionLocal() as db:
db.query(CapacityLease).filter(CapacityLease.environment_id == live_binding.environment_id).delete()
db.query(CapacityQuota).filter(CapacityQuota.environment_id == live_binding.environment_id).delete()
db.commit()
# #region Test.ScenarioExecution.BrowserSession.Shared [C:3] [TYPE Function] [SEMANTICS test,provider,browser,session,checkpoint]
# @BRIEF (а)+(б): two steps of one run share one context; the stamped checkpoint carries filter state.
# @TEST_INVARIANT BrowserProvider.Session: at most one context per (run, lease) across steps.
def test_two_steps_share_one_context_and_checkpoint_accumulates(loop_runtime, binding):
transport = FakeSessionTransport()
manager = BrowserSessionManager(transport=transport, event_loop=loop_runtime)
try:
provider = build_browser_provider(
transport=transport, storage=FakeStorage(), loop=loop_runtime, session_manager=manager,
)
run_id = str(uuid.uuid4())
step1 = build_step(binding, run_id=run_id)
step1["step_meta"]["action_descriptor"] = {
**descriptor("apply_native_filter"),
"inputs": {"column": "region", "values": ["South"]},
}
result1 = provider(LiveProviderContext(binding=binding, step=step1, completed={}))
step2 = build_step(binding, run_id=run_id)
step2["step_meta"]["action_descriptor"] = {
**descriptor("wait_for_state"),
"inputs": {"state": "networkidle"},
}
result2 = provider(LiveProviderContext(binding=binding, step=step2, completed={}))
assert result1.status == "passed"
assert result2.status == "passed"
# Launch+login happened exactly once; both steps executed on the same live context.
assert transport.open_count == 1
assert len(transport.tokens) == 1
token = transport.tokens[0]
assert transport.executed[0][0] is token
assert transport.executed[1][0] is token
assert [call[1] for call in transport.executed] == ["apply_native_filter", "wait_for_state"]
# UX-1 outcome semantics preserved in the step details.
assert result1.details["applied"] is True
assert result1.details["applied_values"] == ["South"]
# (б) The checkpoint is stamped with the reconstructible filter state slice.
checkpoint1 = result1.details["browser_checkpoint"]
assert checkpoint1["dashboard_id"] == 42
assert checkpoint1["checkpoint_seq"] == 1
assert checkpoint1["native_filter_state"] == [{
"filter_target": "region",
"applied_values": ["South"],
"applied_mode": "values",
"replay_input": {"column": "region", "values": ["South"]},
}]
assert checkpoint1["active_tab"] is None
assert checkpoint1["wait_states"] == []
# The second step accumulates on the same checkpoint without losing the filter state.
checkpoint2 = result2.details["browser_checkpoint"]
assert checkpoint2["checkpoint_seq"] == 2
assert checkpoint2["wait_states"] == ["networkidle"]
assert checkpoint2["native_filter_state"][0]["applied_values"] == ["South"]
# Run-scoped: the context survives the per-step lease release.
assert token.closed is False
finally:
manager.close_all()
def test_provider_auto_builds_session_manager_for_capable_transport(loop_runtime, binding):
transport = FakeSessionTransport()
provider = build_browser_provider(transport=transport, storage=FakeStorage(), loop=loop_runtime)
run_id = str(uuid.uuid4())
first = provider(LiveProviderContext(binding=binding, step=build_step(binding, run_id=run_id), completed={}))
second = provider(LiveProviderContext(binding=binding, step=build_step(binding, run_id=run_id), completed={}))
assert first.status == "passed"
assert second.status == "passed"
assert transport.open_count == 1
assert second.details["browser_checkpoint"]["checkpoint_seq"] == 2
# #endregion Test.ScenarioExecution.BrowserSession.Shared
# #region Test.ScenarioExecution.BrowserSession.Replay [C:3] [TYPE Function] [SEMANTICS test,provider,browser,session,replay,recovery]
# @BRIEF (в): process death loses the in-memory context; recovery replays open + re-apply from the
# last persisted checkpoint, then continues the step.
# @TEST_INVARIANT BrowserProvider.Session: a dead context is never revived; recovery is replay-only.
def test_replay_restores_state_after_process_death(loop_runtime, binding):
transport = FakeSessionTransport()
manager1 = BrowserSessionManager(transport=transport, event_loop=loop_runtime)
provider1 = build_browser_provider(
transport=transport, storage=FakeStorage(), loop=loop_runtime, session_manager=manager1,
)
run_id = _persist_run(binding.environment_id)
step1 = build_step(binding, run_id=run_id)
step1["step_meta"]["action_descriptor"] = {
**descriptor("apply_native_filter"),
"inputs": {"column": "region", "values": ["South"]},
}
result1 = provider1(LiveProviderContext(binding=binding, step=step1, completed={}))
assert result1.status == "passed"
# The runner persists the stamped checkpoint into step_outcome (simulated here).
_persist_browser_step(run_id, checkpoint=result1.details["browser_checkpoint"])
# Process death: the registry (and the loop-bound context) is gone — a new manager owns recovery.
manager2 = BrowserSessionManager(transport=transport, event_loop=loop_runtime)
provider2 = build_browser_provider(
transport=transport, storage=FakeStorage(), loop=loop_runtime, session_manager=manager2,
)
step2 = build_step(binding, run_id=run_id)
result2 = provider2(LiveProviderContext(binding=binding, step=step2, completed={}))
assert result2.status == "passed"
# A NEW context was opened (the dead one was never revived), then the filter state re-applied
# from the persisted checkpoint before the step action executed.
assert transport.open_count == 2
replay_token = transport.tokens[1]
assert transport.executed[1][0] is replay_token
assert transport.executed[1][1] == "apply_native_filter"
assert transport.executed[1][2] == {"column": "region", "values": ["South"]}
assert transport.executed[2][0] is replay_token
assert transport.executed[2][1] == "open_dashboard"
# The step's own checkpoint still carries the recovered filter state.
assert result2.details["browser_checkpoint"]["native_filter_state"][0]["applied_values"] == ["South"]
manager2.close_all()
manager1.close_all()
# #endregion Test.ScenarioExecution.BrowserSession.Replay
# #region Test.ScenarioExecution.BrowserSession.Missing [C:3] [TYPE Function] [SEMANTICS test,provider,browser,session,checkpoint,fail-closed]
# @BRIEF (г): browser history without a declared checkpoint is a typed inconclusive, never a silent
# stateless session; zero browser I/O is performed.
# @TEST_INVARIANT BrowserProvider.Session: no declared checkpoint -> BROWSER_CHECKPOINT_MISSING.
def test_missing_checkpoint_is_typed_inconclusive_without_io(loop_runtime, binding):
run_id = _persist_run(binding.environment_id)
_persist_browser_step(run_id, checkpoint=None, status="inconclusive")
transport = FakeSessionTransport()
manager = BrowserSessionManager(transport=transport, event_loop=loop_runtime)
provider = build_browser_provider(
transport=transport, storage=FakeStorage(), loop=loop_runtime, session_manager=manager,
)
result = provider(LiveProviderContext(binding=binding, step=build_step(binding, run_id=run_id), completed={}))
assert result.status == "inconclusive"
assert result.reason_code == "BROWSER_CHECKPOINT_MISSING"
assert result.details["retry_disposition"] == "manual_only"
assert transport.open_count == 0
assert transport.executed == []
# #endregion Test.ScenarioExecution.BrowserSession.Missing
# #region Test.ScenarioExecution.BrowserSession.LeakGuard [C:3] [TYPE Function] [SEMANTICS test,provider,browser,session,close,leak-guard]
# @BRIEF (д): close runs on explicit terminal, dead-context (crash/timeout) and idle TTL paths.
# @TEST_INVARIANT BrowserProvider.Session: close is idempotent and every abnormal path closes.
def test_terminal_close_releases_context(loop_runtime, binding):
transport = FakeSessionTransport()
manager = BrowserSessionManager(transport=transport, event_loop=loop_runtime)
provider = build_browser_provider(
transport=transport, storage=FakeStorage(), loop=loop_runtime, session_manager=manager,
)
run_id = str(uuid.uuid4())
result = provider(LiveProviderContext(binding=binding, step=build_step(binding, run_id=run_id), completed={}))
assert result.status == "passed"
assert manager.active_run_ids() == [run_id]
assert manager.close(run_id, reason="terminal") is True
assert transport.tokens[0].closed is True
assert manager.active_run_ids() == []
# Idempotent: a repeated close is a no-op reflect, not an error.
assert manager.close(run_id, reason="terminal") is False
@pytest.mark.parametrize(
"failure, expected_code",
[("crash", "BROWSER_ACTION_FAILED"), ("timeout", "BROWSER_ACTION_TIMEOUT")],
)
def test_failed_step_closes_dead_context(loop_runtime, binding, failure, expected_code):
transport = FakeSessionTransport(failure=failure)
manager = BrowserSessionManager(transport=transport, event_loop=loop_runtime)
provider = build_browser_provider(
transport=transport, storage=FakeStorage(), loop=loop_runtime, session_manager=manager,
)
run_id = str(uuid.uuid4())
result = provider(LiveProviderContext(binding=binding, step=build_step(binding, run_id=run_id), completed={}))
assert result.status == "inconclusive"
assert result.reason_code == expected_code
# Leak-guard: the possibly-corrupted context is closed and dropped, never reused.
assert transport.tokens[0].closed is True
assert manager.active_run_ids() == []
def test_idle_reaper_closes_stale_context(loop_runtime, binding):
clock = [1000.0]
transport = FakeSessionTransport()
manager = BrowserSessionManager(
transport=transport, event_loop=loop_runtime, idle_timeout_seconds=60.0, clock=lambda: clock[0],
)
provider = build_browser_provider(
transport=transport, storage=FakeStorage(), loop=loop_runtime, session_manager=manager,
)
run_id = str(uuid.uuid4())
result = provider(LiveProviderContext(binding=binding, step=build_step(binding, run_id=run_id), completed={}))
assert result.status == "passed"
assert transport.tokens[0].closed is False
clock[0] += 120.0
assert manager.reap_idle() == 1
assert transport.tokens[0].closed is True
assert manager.active_run_ids() == []
# #endregion Test.ScenarioExecution.BrowserSession.LeakGuard
# #region Test.ScenarioExecution.BrowserSession.Isolation [C:3] [TYPE Function] [SEMANTICS test,provider,browser,session,isolation]
# @BRIEF (е): contexts of different runs never intersect — separate handles, independent close, and
# the persisted-checkpoint lookup is strictly run-scoped.
# @TEST_INVARIANT BrowserProvider.Session: a context is never shared across runs.
def test_contexts_never_shared_across_runs(loop_runtime, binding):
transport = FakeSessionTransport()
manager = BrowserSessionManager(transport=transport, event_loop=loop_runtime)
provider = build_browser_provider(
transport=transport, storage=FakeStorage(), loop=loop_runtime, session_manager=manager,
)
run_a = _persist_run(binding.environment_id)
run_b = str(uuid.uuid4())
step_a = build_step(binding, run_id=run_a)
step_a["step_meta"]["action_descriptor"] = {
**descriptor("apply_native_filter"),
"inputs": {"column": "region", "values": ["South"]},
}
result_a = provider(LiveProviderContext(binding=binding, step=step_a, completed={}))
result_b = provider(LiveProviderContext(binding=binding, step=build_step(binding, run_id=run_b), completed={}))
assert result_a.status == "passed"
assert result_b.status == "passed"
assert transport.open_count == 2
token_a, token_b = transport.tokens
assert token_a is not token_b
assert manager.active_run_ids() == sorted([run_a, run_b])
# Closing run A leaves run B's context alive.
assert manager.close(run_a, reason="terminal") is True
assert token_a.closed is True
assert token_b.closed is False
assert manager.active_run_ids() == [run_b]
# The persisted-checkpoint lookup is run-scoped: run B cannot recover run A's state.
_persist_browser_step(run_a, checkpoint=result_a.details["browser_checkpoint"])
manager2 = BrowserSessionManager(transport=transport, event_loop=loop_runtime)
provider2 = build_browser_provider(
transport=transport, storage=FakeStorage(), loop=loop_runtime, session_manager=manager2,
)
result_b2 = provider2(LiveProviderContext(binding=binding, step=build_step(binding, run_id=run_b), completed={}))
assert result_b2.status == "passed"
# run_b had no browser history: a fresh context opened without any filter replay.
new_token = transport.tokens[2]
assert transport.executed[-1][0] is new_token
assert transport.executed[-1][1] == "open_dashboard"
assert result_b2.details["browser_checkpoint"]["native_filter_state"] == []
manager2.close_all()
manager.close_all()
# #endregion Test.ScenarioExecution.BrowserSession.Isolation
# #endregion Test.ScenarioExecution.BrowserSession

View File

@@ -0,0 +1,373 @@
# #region Test.ScenarioExecution.DispatchCapacityLifecycle [C:4] [TYPE Module] [SEMANTICS test,scenario,execution,capacity,dispatch,heartbeat,session,cancel]
# @BRIEF 044 T032 acceptance: dispatcher-driven lease expiry, provider heartbeat before I/O,
# run-terminal/cancel session finalization (DG-1 B-wave finalizer).
# @RELATION BINDS_TO -> [ScenarioExecution.DispatchRuns]
# @RELATION BINDS_TO -> [ScenarioExecution.CapacityManager.Service]
# @RELATION VERIFIES -> [ScenarioExecution.BrowserProvider.Session.ManagerRegistry]
# @TEST_FIXTURE: hardcoded runs/leases/bindings; fake transports/services; no live environment.
# @TEST_EDGE expired claimed lease -> freed by the dispatch tick (reconcile drives capacity)
# @TEST_EDGE lease lost/expired between claim and submit -> typed capacity refusal, zero provider I/O
# @TEST_EDGE terminal run -> close_run_sessions(run_terminal:*); waiting_human -> no close
# @TEST_EDGE cancelled run -> close_run_sessions(cancelled)
# @TEST_INVARIANT ScenarioExecution.CapacityManager.Service: No provider I/O happens without a live
# capacity lease; an expired lease is freed by the dispatcher tick. ->
# VERIFIED_BY: dispatch_tick_reconciles_expired_leases,
# browser_heartbeat_*, screenshot_heartbeat_*
# @TEST_INVARIANT ScenarioExecution.BrowserProvider.Session: A terminal or cancelled run never
# keeps a run-scoped session alive. -> VERIFIED_BY: dispatch_finalizer_*,
# cancel_run_closes_sessions
from __future__ import annotations
import asyncio
import uuid
from datetime import UTC, datetime, timedelta
import pytest
from src.core.database import SessionLocal
from src.models.provider_operation import ProviderOperationReceipt
from src.models.scenario_run import ScenarioRun
from src.models.provider_capacity import CapacityLease, CapacityQuota
from src.services.dashboard_testing.execution import cancel_lifecycle, dispatch_runs
from src.services.dashboard_testing.execution.cancel_lifecycle import cancel_run
from src.services.dashboard_testing.execution.capacity import claim_capacity
from src.services.dashboard_testing.execution.executor_registry import ScenarioExecutorRegistry
from src.services.dashboard_testing.execution.live_adapter import LiveAdapterResult
from src.services.dashboard_testing.execution.live_binding import LiveExecutionBinding
from src.services.dashboard_testing.execution.provider_runtime import ProviderEventLoop
from src.services.dashboard_testing.execution.providers import browser as browser_provider_module
from src.services.dashboard_testing.execution.providers import screenshot as screenshot_provider_module
from src.services.dashboard_testing.execution.providers.browser import build_browser_provider
from src.services.dashboard_testing.execution.providers.browser_transport import BrowserTransportOutcome
from src.services.dashboard_testing.execution.providers.screenshot import build_screenshot_provider
from src.services.dashboard_testing.execution.runner import dispatch_queued_runs
from src.services.dashboard_testing.scenario.templates import (
ACTION_REGISTRY_VERSION,
action_registry_fingerprint,
resolve_action_descriptor,
)
_PNG = b"\x89PNG\r\n\x1a\n" + b"b2-capacity-lifecycle" * 4
_ENV = "env-dispatch-capacity-b2"
_SCENARIO_ID = "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaa1"
_REVISION_ID = "bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbb1"
# #region Test.ScenarioExecution.DispatchCapacityLifecycle.Fixtures [C:1] [TYPE Module]
@pytest.fixture()
def loop_runtime():
runtime = ProviderEventLoop()
runtime.start()
yield runtime
runtime.stop()
@pytest.fixture()
def clean_b2_state():
yield
with SessionLocal() as db:
for model in (CapacityLease, CapacityQuota):
db.query(model).filter(model.environment_id == _ENV).delete()
db.query(ProviderOperationReceipt).delete()
db.commit()
def _binding() -> LiveExecutionBinding:
return LiveExecutionBinding(
binding_ref=f"bind-b2-{uuid.uuid4().hex[:8]}",
environment_id=_ENV,
dashboard_release_id="release-b2",
release_fingerprint="f" * 64,
dashboard_id=42,
query_model_fingerprint="q" * 64,
execution_principal_fingerprint="p" * 64,
rls_security_fingerprint="r" * 64,
browser_safe_checkpoint_ref=None,
browser_action_binding_ref="action-bind-b2",
evidence_owner_type="scenario_run",
evidence_ref_policy="draft_storage_raw_response",
)
def _browser_step(binding: LiveExecutionBinding, run_id: str) -> dict:
return {
"scenario_run_id": run_id,
"live_execution_binding_ref": binding.binding_ref,
"live_execution_binding_snapshot": binding.snapshot(),
"target_snapshot": {
"environment_id": binding.environment_id,
"dashboard_release_id": binding.dashboard_release_id,
"environment_class": "DEV",
},
"execution_principal_fingerprint": binding.execution_principal_fingerprint,
"step_meta": {
"environment_id": binding.environment_id,
"dashboard_id": binding.dashboard_id,
"logical_step_id": "browser-b2",
"action_descriptor": {
"action": "open_dashboard",
"tool": "browser",
"mutating": False,
"risk": "read_only",
},
},
}
def _screenshot_step(binding: LiveExecutionBinding, run_id: str) -> dict:
step = _browser_step(binding, run_id)
step["logical_step_id"] = "screenshot-b2"
return step
class _FakeTransport:
def __init__(self) -> None:
self.calls: list[tuple[int, str]] = []
async def execute(self, dashboard_id, action, *, action_input, timeout_seconds):
self.calls.append((dashboard_id, action))
return BrowserTransportOutcome(
checkpoints=("dashboard_open",),
page_url="http://superset.test/superset/dashboard/42/",
evidence_png=_PNG,
details={"title": "Dashboard"},
effect_state="none",
)
class _FakeCaptureService:
def __init__(self) -> None:
self.calls: list[int] = []
async def capture_dashboard(self, dashboard_id: int, output_path: str):
self.calls.append(dashboard_id)
with open(output_path, "wb") as handle:
handle.write(_PNG)
return [output_path], []
class _FakeStorage:
def store(self, run_id: str, sha256: str, data: bytes) -> str:
return f"draft:{run_id}:{sha256}"
class _Context:
def __init__(self, binding: LiveExecutionBinding, step: dict) -> None:
self.binding = binding
self.step = step
self.completed: dict = {}
# #endregion Test.ScenarioExecution.DispatchCapacityLifecycle.Fixtures
# #region Test.ScenarioExecution.DispatchCapacityLifecycle.Expiry [C:3] [TYPE Function]
# @BRIEF An expired claimed lease is freed by the dispatcher tick, before any admission decision.
def test_dispatch_tick_reconciles_expired_leases(seeded_execution, clean_b2_state):
lease = claim_capacity(
seeded_execution,
environment_id=_ENV,
environment_class="PROD",
workload_class="browser",
provider_id="browser",
run_id="run-expired-lease-b2",
ttl_seconds=60,
)
seeded_execution.flush()
row = seeded_execution.query(CapacityLease).filter(CapacityLease.id == lease["lease_id"]).one()
row.expires_at = datetime.now(UTC) - timedelta(seconds=5)
seeded_execution.flush()
outcomes = dispatch_queued_runs(seeded_execution, worker_id="b2-expiry")
assert outcomes == []
seeded_execution.refresh(row)
assert row.status == "expired"
quota = (
seeded_execution.query(CapacityQuota)
.filter(CapacityQuota.environment_id == _ENV, CapacityQuota.workload_class == "browser")
.one()
)
assert quota.active_units == 0
# #endregion Test.ScenarioExecution.DispatchCapacityLifecycle.Expiry
# #region Test.ScenarioExecution.DispatchCapacityLifecycle.Heartbeat [C:3] [TYPE Function]
# @BRIEF Providers heartbeat their lease before any loop submission; a lost lease refuses I/O typed.
def test_browser_heartbeat_precedes_transport_and_lost_lease_is_typed(loop_runtime, clean_b2_state, monkeypatch):
calls: list[str] = []
real_heartbeat = browser_provider_module.heartbeat_capacity
def spy_heartbeat(db, lease_id, **kwargs):
calls.append(f"heartbeat:{lease_id}")
return real_heartbeat(db, lease_id, **kwargs)
monkeypatch.setattr(browser_provider_module, "heartbeat_capacity", spy_heartbeat)
transport = _FakeTransport()
binding = _binding()
provider = build_browser_provider(transport=transport, storage=_FakeStorage(), loop=loop_runtime)
result = provider(_Context(binding, _browser_step(binding, f"run-{uuid.uuid4()}")))
assert result.status == "passed"
assert len(calls) == 1
assert transport.calls == [(binding.dashboard_id, "open_dashboard")]
def dead_heartbeat(db, lease_id, **kwargs):
from src.services.dashboard_testing.execution.capacity import CapacityUnavailable
raise CapacityUnavailable("CAPACITY_LEASE_NOT_ACTIVE")
monkeypatch.setattr(browser_provider_module, "heartbeat_capacity", dead_heartbeat)
dead_transport = _FakeTransport()
dead_provider = build_browser_provider(transport=dead_transport, storage=_FakeStorage(), loop=loop_runtime)
refused = dead_provider(_Context(_binding(), _browser_step(_binding(), f"run-{uuid.uuid4()}")))
assert refused.status == "inconclusive"
assert refused.reason_code == "BROWSER_CAPACITY_UNAVAILABLE"
assert dead_transport.calls == []
def test_screenshot_heartbeat_precedes_capture_and_lost_lease_is_typed(loop_runtime, clean_b2_state, monkeypatch):
calls: list[str] = []
real_heartbeat = screenshot_provider_module.heartbeat_capacity
def spy_heartbeat(db, lease_id, **kwargs):
calls.append(f"heartbeat:{lease_id}")
return real_heartbeat(db, lease_id, **kwargs)
monkeypatch.setattr(screenshot_provider_module, "heartbeat_capacity", spy_heartbeat)
service = _FakeCaptureService()
binding = _binding()
provider = build_screenshot_provider(service=service, storage=_FakeStorage(), loop=loop_runtime)
result = provider(_Context(binding, _screenshot_step(binding, f"run-{uuid.uuid4()}")))
assert result.status == "passed"
assert len(calls) == 1
assert service.calls == [binding.dashboard_id]
def dead_heartbeat(db, lease_id, **kwargs):
from src.services.dashboard_testing.execution.capacity import CapacityUnavailable
raise CapacityUnavailable("CAPACITY_LEASE_NOT_ACTIVE")
monkeypatch.setattr(screenshot_provider_module, "heartbeat_capacity", dead_heartbeat)
dead_service = _FakeCaptureService()
dead_provider = build_screenshot_provider(service=dead_service, storage=_FakeStorage(), loop=loop_runtime)
refused = dead_provider(_Context(_binding(), _screenshot_step(_binding(), f"run-{uuid.uuid4()}")))
assert refused.status == "inconclusive"
assert refused.reason_code == "SCREENSHOT_CAPACITY_UNAVAILABLE"
assert dead_service.calls == []
# #endregion Test.ScenarioExecution.DispatchCapacityLifecycle.Heartbeat
# #region Test.ScenarioExecution.DispatchCapacityLifecycle.Finalizer [C:3] [TYPE Function]
# @BRIEF Terminal runs close their run-scoped sessions; a human wait keeps the session (replay-safe).
def _queued_run(run_id: str, plan: dict) -> ScenarioRun:
return ScenarioRun(
id=run_id,
scenario_id=_SCENARIO_ID,
scenario_revision_id=_REVISION_ID,
scenario_content_hash="e" * 64,
environment_id="env-queued-dispatch-044",
status="queued",
phase="preflight",
parameter_bindings={"fixture": "dispatch-capacity-b2"},
target_snapshot={"environment_id": "env-queued-dispatch-044"},
trigger_source="manual",
idempotency_key=run_id,
runner_plan=plan,
execution_principal_fingerprint="f" * 64,
)
def _assertion_plan(step_id: str) -> dict:
return {
"action_registry_version": ACTION_REGISTRY_VERSION,
"action_registry_hash": action_registry_fingerprint(),
"topological_order": [step_id],
"dependencies": [],
"steps": [{
"logical_step_id": step_id,
"tool": "assertion",
"action": "structural_assert",
"action_descriptor": resolve_action_descriptor(
tool="assertion", action="structural_assert",
registry_version=ACTION_REGISTRY_VERSION, registry_hash=action_registry_fingerprint(),
).snapshot(),
}],
}
def test_dispatch_finalizer_closes_sessions_on_terminal(seeded_execution, monkeypatch):
closed: list[tuple[str, str]] = []
monkeypatch.setattr(
dispatch_runs, "close_run_sessions",
lambda run_id, *, reason: closed.append((run_id, reason)) or 1,
)
run = _queued_run(f"run-terminal-{uuid.uuid4().hex[:8]}", _assertion_plan("assert-b2"))
seeded_execution.add(run)
seeded_execution.flush()
registry = ScenarioExecutorRegistry()
registry.register(
"assertion",
lambda step, _completed: {"status": "passed", "step_outcome": {"status": "passed"}, "artifact_refs": []},
action="structural_assert",
)
outcomes = dispatch_queued_runs(seeded_execution, worker_id="b2-finalizer", registry=registry)
assert [result["status"] for result in outcomes] == ["passed"]
assert closed == [(run.id, "run_terminal:passed")]
def test_dispatch_finalizer_skips_waiting_human(seeded_execution, monkeypatch):
closed: list[tuple[str, str]] = []
monkeypatch.setattr(
dispatch_runs, "close_run_sessions",
lambda run_id, *, reason: closed.append((run_id, reason)) or 1,
)
plan = _assertion_plan("human-b2")
plan["steps"] = [{
"logical_step_id": "human-b2",
"tool": "human",
"action": "human_checkpoint",
"action_descriptor": resolve_action_descriptor(
tool="human", action="human_checkpoint",
registry_version=ACTION_REGISTRY_VERSION, registry_hash=action_registry_fingerprint(),
).snapshot(),
}]
run = _queued_run(f"run-human-{uuid.uuid4().hex[:8]}", plan)
seeded_execution.add(run)
seeded_execution.flush()
outcomes = dispatch_queued_runs(seeded_execution, worker_id="b2-finalizer-human")
assert [result["status"] for result in outcomes] == ["waiting_human"]
assert closed == []
# #endregion Test.ScenarioExecution.DispatchCapacityLifecycle.Finalizer
# #region Test.ScenarioExecution.DispatchCapacityLifecycle.Cancel [C:3] [TYPE Function]
# @BRIEF A cancelled run closes its run-scoped sessions in the same terminal transition.
def test_cancel_run_closes_sessions(seeded_execution, monkeypatch):
closed: list[tuple[str, str]] = []
monkeypatch.setattr(
cancel_lifecycle, "close_run_sessions",
lambda run_id, *, reason: closed.append((run_id, reason)) or 1,
)
run = _queued_run(f"run-cancel-{uuid.uuid4().hex[:8]}", _assertion_plan("assert-cancel-b2"))
run.status = "running"
run.phase = "executing"
seeded_execution.add(run)
seeded_execution.flush()
cancelled = cancel_run(seeded_execution, run.id, drain_in_flight=False)
assert cancelled.status == "cancelled"
assert closed == [(run.id, "cancelled")]
# #endregion Test.ScenarioExecution.DispatchCapacityLifecycle.Cancel
# #endregion Test.ScenarioExecution.DispatchCapacityLifecycle

View File

@@ -0,0 +1,353 @@
# #region Test.ScenarioExecution.DueAdmissionAtomicity [C:5] [TYPE Module] [SEMANTICS test,scenario,execution,due,admission,atomicity,crash,replay,concurrent,outbox,approval]
# @defgroup Test Due-event admission atomicity (046 T020): one pinned run/gate/outbox under crash, replay, concurrency.
# @BRIEF Prove crash/replay/concurrent due events converge to exactly one pinned run + one PROD
# gate, that the terminal transition + 047 queue signal + notification outbox commit in one
# transaction, and that a pending_approval run is never dispatched before gate approval.
# @RELATION BINDS_TO -> [Core.Scheduler.ExecuteScheduledScenario]
# @RELATION BINDS_TO -> [Core.Scheduler.ExecuteQueuedScenarioDispatch]
# @RELATION VERIFIES -> [ScenarioExecution.Approval.CreateGate]
# @RELATION VERIFIES -> [ScenarioExecution.Runner.TerminalSignal]
# @RELATION VERIFIES -> [ScenarioExecution.Runner.QueuedDispatch]
# @TEST_CONTRACT: [ConcurrentOrCrashedDueAdmission] -> [SinglePinnedRunGateOutbox]
# @TEST_FIXTURE: due_admission_graphs -> INLINE hardcoded assertion graph snapshots (1==1 pass, 1!=2 fail)
# @TEST_EDGE concurrent_same_second_callbacks -> the unique idempotency key converges both workers to one run+gate
# @TEST_EDGE crash_between_run_flush_and_gate_insert -> the whole admission transaction rolls back; replay persists one pair
# @TEST_EDGE crash_inside_terminal_outbox -> transition/queue/notification roll back together; retry emits exactly one of each
# @TEST_EDGE pending_approval_dispatch_tick -> the dispatcher CAS never claims an unapproved run
# @TEST_INVARIANT Core.Scheduler.ExecuteScheduledScenario: two concurrent same-second due callbacks
# converge on exactly one ScenarioRun and one ActionApprovalGate via the unique
# idempotency key (storage-layer convergence). -> VERIFIED_BY:
# test_concurrent_due_callbacks_converge_to_one_run_and_gate
# @TEST_INVARIANT Core.Scheduler.ExecuteScheduledScenario: a crash between run flush and gate
# insert rolls back the single admission transaction, so no orphan run or gate
# survives and the replayed due creates exactly one pair. -> VERIFIED_BY:
# test_crash_between_run_and_gate_leaves_no_orphan_and_retry_converges
# @TEST_INVARIANT ScenarioExecution.Runner.TerminalSignal: run transition, InvestigationQueueItem
# and ScenarioNotificationEvent are flush-only effects of the caller's single
# transaction; a crash inside the outbox write leaves the run queued with zero
# partial side effects, and the retry emits exactly one of each. -> VERIFIED_BY:
# test_terminal_transition_queue_and_outbox_share_one_transaction
# @TEST_INVARIANT ScenarioExecution.Runner.QueuedDispatch: a pending_approval PROD run is not a
# dispatch candidate; gate approval alone transitions it to queued for the CAS.
# -> VERIFIED_BY: test_pending_approval_run_is_not_dispatched_until_gate_approval
# @RATIONALE T020 acceptance is proven against the real callback composition (SessionLocal monkeypatch
# only): the admission atomicity guarantee is the single caller-owned transaction, not
# replay-side repair.
# @REJECTED Repairing a missing PROD gate during idempotent replay was rejected — the replay
# invariant forbids any side effect, and the single-transaction boundary already makes a
# persisted run-without-gate state unreachable through every admission path.
from __future__ import annotations
from datetime import UTC, datetime
import threading
from unittest.mock import MagicMock
from sqlalchemy.orm import sessionmaker
from src.models.scenario_approval import ActionApprovalGate
from src.models.scenario_automation import ScenarioNotificationEvent
from src.models.scenario_investigation import InvestigationQueueItem
from src.models.scenario_registry import ScenarioRegistryEntry, ScenarioRevision
from src.models.scenario_run import ScenarioRun
from src.services.dashboard_testing.execution.approval import decide_approval_gate
from src.services.dashboard_testing.execution.executor_registry import ScenarioExecutorRegistry
from src.services.dashboard_testing.execution.runner import dispatch_queued_runs, start_run
from src.services.dashboard_testing.scenario.templates import (
ACTION_REGISTRY_VERSION,
action_registry_fingerprint,
resolve_action_descriptor,
)
_SCENARIO_ID = "90460000-0000-4000-8000-0000000000a2"
_SCHEDULE_ID = "91460000-0000-4000-8000-0000000000a2"
_REVISION_ID = "92460000-0000-4000-8000-0000000000a2"
_TERMINAL_RUN_ID = "93460000-0000-4000-8000-0000000000a2"
_FIRED_AT = datetime(2026, 9, 12, 10, 0, 5, tzinfo=UTC)
# #region Test.ScenarioExecution.DueAdmissionAtomicity.Fixtures [C:2] [TYPE Function] [SEMANTICS test,fixture,graph,hardcoded]
# @ingroup Test.ScenarioExecution.DueAdmissionAtomicity
# @BRIEF Hardcoded graphs/plans: pass (1==1) and fail (1!=2) single assertion steps, one registry seed.
def _graph(actual: int, expected: int, step_id: str) -> dict:
return {
"action_registry_version": ACTION_REGISTRY_VERSION,
"action_registry_hash": action_registry_fingerprint(),
"steps": [{
"logical_step_id": step_id,
"tool": "assertion",
"action": "structural_assert",
"action_descriptor": resolve_action_descriptor(
tool="assertion", action="structural_assert",
registry_version=ACTION_REGISTRY_VERSION,
registry_hash=action_registry_fingerprint(),
).snapshot(),
"actual": actual,
"expected": expected,
}],
"dependencies": [],
}
def _plan(actual: int, expected: int, step_id: str) -> dict:
graph = _graph(actual, expected, step_id)
return {
"action_registry_version": ACTION_REGISTRY_VERSION,
"action_registry_hash": action_registry_fingerprint(),
"topological_order": [step_id],
"dependencies": [],
"steps": graph["steps"],
}
def _persist_scenario(db) -> None:
db.add(ScenarioRegistryEntry(
scenario_id=_SCENARIO_ID,
scenario_key="due-admission-atomicity-046",
name="Due admission atomicity 046",
dashboard_id=46,
environment_ids=["prod", "preprod"],
owner_id="analyst-046",
owner_username="analyst.046",
lifecycle_status="READY",
validation_status="valid",
current_revision_id=_REVISION_ID,
))
db.add(ScenarioRevision(
revision_id=_REVISION_ID,
scenario_id=_SCENARIO_ID,
content_hash="a046".replace("-", "").ljust(64, "0"),
graph_snapshot=_graph(1, 1, "assert-due-046"),
execution_template_hash="",
template_version="v1",
schema_version=1,
compatibility_family="default",
change_summary={"reason": "hardcoded due-admission atomicity fixture"},
created_by="analyst-046",
activation_status="current",
))
db.flush()
def _frozen(moment: datetime) -> type[datetime]:
class _FrozenDatetime(datetime):
@classmethod
def now(cls, tz=None):
return moment if tz is None else moment.astimezone(tz)
return _FrozenDatetime
def _fire_due() -> None:
from src.core import scheduler as scheduler_module
scheduler_module.execute_scheduled_scenario(
schedule_id=_SCHEDULE_ID,
scenario_id=_SCENARIO_ID,
revision_policy="current",
revision_id=None,
environment_id="prod",
cron_expr="* * * * *",
timezone="UTC",
policy_id=None,
)
# #endregion Test.ScenarioExecution.DueAdmissionAtomicity.Fixtures
# #region Test.ScenarioExecution.DueAdmissionAtomicity.Concurrent [C:5] [TYPE Function] [SEMANTICS test,scenario,execution,concurrent,due,unique-convergence]
# @ingroup Test.ScenarioExecution.DueAdmissionAtomicity
# @BRIEF Two barrier-aligned due callbacks racing the same frozen second persist exactly one run+gate.
def test_concurrent_due_callbacks_converge_to_one_run_and_gate(registry_engine, monkeypatch):
from src.core import scheduler as scheduler_module
factory = sessionmaker(bind=registry_engine)
seed = factory()
_persist_scenario(seed)
seed.commit()
seed.close()
monkeypatch.setattr(scheduler_module, "datetime", _frozen(_FIRED_AT))
monkeypatch.setattr(scheduler_module, "SessionLocal", factory)
barrier = threading.Barrier(2)
escaped: list[BaseException] = []
def _due() -> None:
barrier.wait(timeout=10)
try:
_fire_due()
except BaseException as exc: # the callback contract: every failure is contained
escaped.append(exc)
threads = [threading.Thread(target=_due, name=f"due-callback-{index}") for index in range(2)]
for thread in threads:
thread.start()
for thread in threads:
thread.join(timeout=30)
assert not escaped, f"due callback failure escaped: {escaped}"
verify = factory()
try:
runs = verify.query(ScenarioRun).filter(ScenarioRun.scenario_id == _SCENARIO_ID).all()
gates = verify.query(ActionApprovalGate).all()
assert len(runs) == 1
assert len(gates) == 1
assert runs[0].idempotency_key == f"scheduled-{_SCHEDULE_ID}-2026-09-12T10:00:05+00:00"
assert runs[0].status == "pending_approval"
assert gates[0].owner_id == runs[0].id
assert gates[0].request_hash == runs[0].request_hash
assert gates[0].status == "pending"
finally:
verify.close()
# #endregion Test.ScenarioExecution.DueAdmissionAtomicity.Concurrent
# #region Test.ScenarioExecution.DueAdmissionAtomicity.CrashRunGate [C:5] [TYPE Function] [SEMANTICS test,scenario,execution,crash,rollback,replay]
# @ingroup Test.ScenarioExecution.DueAdmissionAtomicity
# @BRIEF A crash after the run flush but before the gate insert rolls back both; the replayed due persists one pair.
def test_crash_between_run_and_gate_leaves_no_orphan_and_retry_converges(registry_session, monkeypatch):
from src.core import scheduler as scheduler_module
from src.services.dashboard_testing.execution import start_run as start_run_module
_persist_scenario(registry_session)
registry_session.commit()
monkeypatch.setattr(scheduler_module, "datetime", _frozen(_FIRED_AT))
monkeypatch.setattr(scheduler_module, "SessionLocal", lambda: registry_session)
explore = MagicMock()
monkeypatch.setattr(scheduler_module.logger, "explore", explore)
real_create_gate = start_run_module.create_prod_gate
def _crashing_gate(*args, **kwargs):
raise RuntimeError("simulated crash after run flush before gate insert")
monkeypatch.setattr(start_run_module, "create_prod_gate", _crashing_gate)
_fire_due()
assert registry_session.query(ScenarioRun).filter(ScenarioRun.scenario_id == _SCENARIO_ID).count() == 0
assert registry_session.query(ActionApprovalGate).count() == 0
crash_logs = [
call for call in explore.call_args_list
if "simulated crash after run flush before gate insert" in str(call.kwargs.get("error"))
]
assert crash_logs, f"crash was not contained by the callback: {explore.call_args_list}"
monkeypatch.setattr(start_run_module, "create_prod_gate", real_create_gate)
_fire_due()
run = registry_session.query(ScenarioRun).filter(ScenarioRun.scenario_id == _SCENARIO_ID).one()
gate = registry_session.query(ActionApprovalGate).filter(ActionApprovalGate.owner_id == run.id).one()
assert run.status == "pending_approval"
assert run.idempotency_key == f"scheduled-{_SCHEDULE_ID}-2026-09-12T10:00:05+00:00"
assert gate.status == "pending"
assert gate.request_hash == run.request_hash
# #endregion Test.ScenarioExecution.DueAdmissionAtomicity.CrashRunGate
# #region Test.ScenarioExecution.DueAdmissionAtomicity.TerminalOutbox [C:5] [TYPE Function] [SEMANTICS test,scenario,execution,terminal,outbox,transaction,crash]
# @ingroup Test.ScenarioExecution.DueAdmissionAtomicity
# @BRIEF Transition record, 047 queue signal and notification outbox share the tick transaction:
# a crash inside the outbox write rolls back all three; the retry emits exactly one of each.
def test_terminal_transition_queue_and_outbox_share_one_transaction(registry_session, monkeypatch):
from src.core import scheduler as scheduler_module
from src.services.dashboard_testing.execution import terminal_effects
_persist_scenario(registry_session)
registry_session.add(ScenarioRun(
id=_TERMINAL_RUN_ID,
scenario_id=_SCENARIO_ID,
scenario_revision_id=_REVISION_ID,
scenario_content_hash="f" * 64,
environment_id="preprod",
status="queued",
phase="preflight",
parameter_bindings={},
target_snapshot={"environment_id": "preprod", "environment_class": "PREPROD"},
trigger_source="scheduled",
idempotency_key="due-terminal-atomicity-046",
runner_plan=_plan(1, 2, "assert-due-terminal-046"),
execution_principal_fingerprint="a" * 64,
))
registry_session.commit()
monkeypatch.setattr(scheduler_module, "SessionLocal", lambda: registry_session)
real_persist_notification = terminal_effects.persist_notification
def _crashing_outbox(*args, **kwargs):
raise RuntimeError("simulated crash inside notification outbox write")
monkeypatch.setattr(terminal_effects, "persist_notification", _crashing_outbox)
scheduler_module.execute_scheduled_queued_scenario_dispatch()
persisted = registry_session.query(ScenarioRun).filter_by(id=_TERMINAL_RUN_ID).one()
assert persisted.status == "queued"
assert registry_session.query(InvestigationQueueItem).filter_by(run_id=_TERMINAL_RUN_ID).count() == 0
assert registry_session.query(ScenarioNotificationEvent).filter_by(run_id=_TERMINAL_RUN_ID).count() == 0
monkeypatch.setattr(terminal_effects, "persist_notification", real_persist_notification)
scheduler_module.execute_scheduled_queued_scenario_dispatch()
terminal = registry_session.query(ScenarioRun).filter_by(id=_TERMINAL_RUN_ID).one()
assert terminal.status == "failed"
assert terminal.phase == "completed"
assert registry_session.query(InvestigationQueueItem).filter_by(run_id=_TERMINAL_RUN_ID).count() == 1
notifications = registry_session.query(ScenarioNotificationEvent).filter_by(run_id=_TERMINAL_RUN_ID).all()
assert [event.event_type for event in notifications] == ["failed"]
scheduler_module.execute_scheduled_queued_scenario_dispatch()
assert registry_session.query(InvestigationQueueItem).filter_by(run_id=_TERMINAL_RUN_ID).count() == 1
assert registry_session.query(ScenarioNotificationEvent).filter_by(run_id=_TERMINAL_RUN_ID).count() == 1
# #endregion Test.ScenarioExecution.DueAdmissionAtomicity.TerminalOutbox
# #region Test.ScenarioExecution.DueAdmissionAtomicity.ApprovalBeforeCas [C:5] [TYPE Function] [SEMANTICS test,scenario,execution,approval,gate,dispatch,cas]
# @ingroup Test.ScenarioExecution.DueAdmissionAtomicity
# @BRIEF A due PROD run stays pending_approval across dispatch ticks until its gate is approved;
# only then does the queued->running CAS claim it, exactly once.
def test_pending_approval_run_is_not_dispatched_until_gate_approval(registry_session):
_persist_scenario(registry_session)
registry_session.commit()
run = start_run(
registry_session,
scenario_id=_SCENARIO_ID,
revision_id=_REVISION_ID,
params={},
environment_id="prod",
actor="ops.oncall-046",
idempotency_key="due-admission-approval-046",
trigger_source="scheduled",
)
registry_session.commit()
gate = registry_session.query(ActionApprovalGate).filter_by(owner_id=run.id).one()
run_id, gate_id = run.id, gate.id
assert run.status == "pending_approval"
assert gate.status == "pending"
calls: list[str] = []
registry = ScenarioExecutorRegistry()
def recording_adapter(step, _completed):
calls.append(step["logical_step_id"])
return {"status": "passed", "step_outcome": {"status": "passed"}, "artifact_refs": []}
registry.register("assertion", recording_adapter, action="structural_assert")
before = dispatch_queued_runs(registry_session, worker_id="due-worker-before-approval", registry=registry)
assert before == []
assert registry_session.query(ScenarioRun).filter_by(id=run_id).one().status == "pending_approval"
assert calls == []
decide_approval_gate(registry_session, gate_id, decision="approve", actor_id="ops.approver-046")
registry_session.commit()
assert registry_session.query(ScenarioRun).filter_by(id=run_id).one().status == "queued"
assert registry_session.query(ActionApprovalGate).filter_by(id=gate_id).one().status == "approved"
after = dispatch_queued_runs(registry_session, worker_id="due-worker-after-approval", registry=registry)
assert [outcome["status"] for outcome in after] == ["passed"]
assert calls == ["assert-due-046"]
assert registry_session.query(ScenarioRun).filter_by(id=run_id).one().status == "passed"
assert dispatch_queued_runs(registry_session, worker_id="due-worker-replay", registry=registry) == []
notifications = registry_session.query(ScenarioNotificationEvent).filter_by(run_id=run_id).all()
assert [event.event_type for event in notifications] == ["completed"]
# #endregion Test.ScenarioExecution.DueAdmissionAtomicity.ApprovalBeforeCas
# #endregion Test.ScenarioExecution.DueAdmissionAtomicity

View File

@@ -38,9 +38,13 @@ from hashlib import sha256
from types import SimpleNamespace
from unittest.mock import AsyncMock
import pytest
from src.core.config_models import ScenarioLiveExecutionBindingConfig
from src.core.database import SessionLocal
from src.core.superset_client._chart_data import ChartDataResponse
from src.core.utils.network import SupersetAPIError
from src.models.provider_operation import ProviderOperationReceipt
from src.models.scenario_artifact import ScenarioArtifact
from src.models.scenario_run import ScenarioStepRun
from src.schemas.dashboard_testing import (
@@ -157,6 +161,17 @@ class _Resolver:
# #endregion Test.ScenarioExecution.LiveBinding.Fakes
# #region Test.ScenarioExecution.LiveBinding.ReceiptCleanup [C:1] [TYPE Function]
@pytest.fixture(autouse=True)
def _clean_provider_operation_receipts():
"""Superset/sql_evidence receipts live in the global test DB; keep each test isolated."""
yield
with SessionLocal() as db:
db.query(ProviderOperationReceipt).delete()
db.commit()
# #endregion Test.ScenarioExecution.LiveBinding.ReceiptCleanup
# #region Test.ScenarioExecution.LiveBinding.QueryModel [C:1] [TYPE Function]
def _query_model() -> DashboardQueryModel:
return DashboardQueryModel(
@@ -566,4 +581,90 @@ def test_lifespan_bootstrap_mismatch_and_unavailable_are_no_call(monkeypatch):
client.execute_chart_data_raw.assert_not_awaited()
# #endregion Test.ScenarioExecution.LiveBinding.LifespanFailClosed
# #region Test.ScenarioExecution.LiveBinding.Receipts [C:2] [TYPE Function]
# @BRIEF SCEX-FR-020 coverage: superset adapter opens a running receipt before 037 I/O and
# finalizes it with a terminal status on every outcome.
# @TEST_INVARIANT ScenarioExecution.ProviderOperations.Service: A receipt is opened before external
# I/O and reaches exactly one terminal status per attempt. -> VERIFIED_BY:
# superset_receipt_running_during_io_then_completed, superset_receipt_failed_on_error
def test_superset_receipt_running_during_io_then_completed():
seen: dict[str, list[str]] = {}
def _probe(*_args, **_kwargs):
with SessionLocal() as db:
seen["statuses"] = [
row.status
for row in db.query(ProviderOperationReceipt).filter(
ProviderOperationReceipt.run_id == "run-044",
).all()
]
return ChartDataResponse(
parsed={"result": [{"data": {"revenue": 4400}}], "query_id": "q-044"},
raw_bytes=_RAW_RESPONSE_044,
source_response_hash=_RAW_RESPONSE_SHA_044,
)
client = AsyncMock()
client.execute_chart_data_raw.side_effect = _probe
binding = LiveExecutionBinding.from_snapshot(_BINDING_044)
resolver = _Resolver(_resolved(binding, client, _EvidenceStore()))
outcome = superset_adapter_from(resolver)(_STEP_044, {})
assert outcome.status == "passed"
# Receipt was opened in running state BEFORE the 037 client call.
assert seen["statuses"] == ["running"]
with SessionLocal() as db:
rows = db.query(ProviderOperationReceipt).filter(
ProviderOperationReceipt.run_id == "run-044",
).all()
assert len(rows) == 1
row = rows[0]
assert row.status == "completed"
assert row.effect_state == "none"
assert row.provider_id == "superset"
assert row.action == "execute_query"
assert row.logical_step_id == "superset-metric-044"
assert row.attempt == 1
assert row.terminal_at is not None
def test_superset_receipt_failed_on_query_error():
client = AsyncMock()
client.execute_chart_data_raw.side_effect = SupersetAPIError("preprod denied")
binding = LiveExecutionBinding.from_snapshot(_BINDING_044)
resolver = _Resolver(_resolved(binding, client, _EvidenceStore()))
outcome = superset_adapter_from(resolver)(_STEP_044, {})
assert outcome.status == "failed"
with SessionLocal() as db:
rows = db.query(ProviderOperationReceipt).filter(
ProviderOperationReceipt.run_id == "run-044",
).all()
assert len(rows) == 1
assert rows[0].status == "failed"
assert rows[0].effect_state == "none"
def test_superset_receipt_failed_unknown_on_exception():
client = AsyncMock()
client.execute_chart_data_raw.side_effect = RuntimeError("transport exploded")
binding = LiveExecutionBinding.from_snapshot(_BINDING_044)
resolver = _Resolver(_resolved(binding, client, _EvidenceStore()))
outcome = superset_adapter_from(resolver)(_STEP_044, {})
assert outcome.status == "inconclusive"
assert outcome.reason_code == "SUPERSET_QUERY_EXECUTION_ERROR"
with SessionLocal() as db:
rows = db.query(ProviderOperationReceipt).filter(
ProviderOperationReceipt.run_id == "run-044",
).all()
assert len(rows) == 1
assert rows[0].status == "failed"
assert rows[0].effect_state == "unknown"
# #endregion Test.ScenarioExecution.LiveBinding.Receipts
# #endregion Test.ScenarioExecution.LiveBinding

View File

@@ -17,6 +17,7 @@ from src.services.dashboard_testing.execution.live_composition import LiveProvid
from src.services.dashboard_testing.execution.provider_runtime import ProviderEventLoop
from src.services.dashboard_testing.execution.providers.browser import (
BrowserTransportOutcome,
BrowserTransportSelectorNotFound,
BrowserTransportUnsupported,
build_browser_provider,
)
@@ -38,13 +39,19 @@ class FakeTransport:
self.effect_state = effect_state
self.failure = failure
self.calls: list[tuple[int, str]] = []
self.inputs: list[dict] = []
async def execute(self, dashboard_id, action, *, action_input, timeout_seconds):
self.calls.append((dashboard_id, action))
self.inputs.append(action_input)
if self.failure == "timeout":
raise TimeoutError("transport deadline")
if self.failure == "unsupported":
raise BrowserTransportUnsupported("BROWSER_ACTION_NOT_SUPPORTED")
if self.failure == "selector_not_found":
raise BrowserTransportSelectorNotFound("BROWSER_SELECTOR_NOT_FOUND")
if self.failure == "wait_state_invalid":
raise ValueError("BROWSER_WAIT_STATE_INVALID")
if self.failure == "crash":
raise RuntimeError("transport exploded")
return BrowserTransportOutcome(
@@ -448,4 +455,152 @@ def test_browser_unrecognized_evidence_falls_back_to_png(loop_runtime, clean_env
ref = result.artifact_refs[0]
assert result.details["artifact_content_types"][ref] == "image/png"
# #endregion Test.ScenarioExecution.BrowserProvider.MimeSniff
# #region Test.ScenarioExecution.BrowserProvider.NativeFilter [C:3] [TYPE Function] [SEMANTICS test,provider,browser,native-filter,selector]
# @BRIEF apply_native_filter: admitted read-only action with typed input merge, checkpoints and
# durable evidence; selector failure and invalid input are typed inconclusive before/without I/O.
# @TEST_INVARIANT The mutation catalog is not extended: apply_native_filter with mutating=true is
# BROWSER_ACTION_NOT_SUPPORTED, never a mutation path.
def test_apply_native_filter_passes_with_checkpoints_and_evidence(loop_runtime, clean_env):
binding = clean_env
transport = FakeTransport(checkpoints=("dashboard_open", "filter_applied", "charts_settled"))
storage = FakeStorage()
provider = build_browser_provider(transport=transport, storage=storage, loop=loop_runtime)
step = build_step(binding)
step["step_meta"]["action_descriptor"] = {
**descriptor("apply_native_filter"),
"inputs": {"column": "region", "values": ["South"]},
}
result = provider(LiveProviderContext(binding=binding, step=step, completed={}))
from hashlib import sha256
digest = sha256(PNG_FIXTURE).hexdigest()
assert result.status == "passed"
assert result.reason_code == "BROWSER_ACTION_EXECUTED"
assert result.details["checkpoints"] == ["dashboard_open", "filter_applied", "charts_settled"]
assert result.details["sha256"] == digest
assert result.details["effect_state"] == "none"
assert "operation_id" not in result.details
assert result.artifact_refs == [f"draft:{step['scenario_run_id']}:{digest}"]
assert transport.calls == [(binding.dashboard_id, "apply_native_filter")]
assert transport.inputs == [{"column": "region", "values": ["South"]}]
def test_apply_native_filter_selector_failure_is_typed_inconclusive(loop_runtime, clean_env):
binding = clean_env
transport = FakeTransport(failure="selector_not_found")
provider = build_browser_provider(transport=transport, storage=FakeStorage(), loop=loop_runtime)
step = build_step(binding)
step["step_meta"]["action_descriptor"] = descriptor("apply_native_filter")
result = provider(LiveProviderContext(binding=binding, step=step, completed={}))
assert result.status == "inconclusive"
assert result.reason_code == "BROWSER_SELECTOR_NOT_FOUND"
assert not result.artifact_refs
with SessionLocal() as db:
lease = db.query(CapacityLease).filter(CapacityLease.run_id == step["scenario_run_id"]).one()
assert lease.status == "released"
# UX-6: a param_binding stamped by RunnerPlan derivation overrides descriptor-pinned values
# and reaches the transport as the typed values input.
def test_apply_native_filter_run_param_binding_reaches_transport(loop_runtime, clean_env):
binding = clean_env
transport = FakeTransport(checkpoints=("dashboard_open", "filter_applied", "charts_settled"))
provider = build_browser_provider(transport=transport, storage=FakeStorage(), loop=loop_runtime)
step = build_step(binding)
step["step_meta"]["action_descriptor"] = {
**descriptor("apply_native_filter"),
"inputs": {"column": "region", "values": ["North"]},
}
step["step_meta"]["param_binding"] = {"filter_values": ["South"]}
result = provider(LiveProviderContext(binding=binding, step=step, completed={}))
assert result.status == "passed"
assert transport.inputs == [{"column": "region", "values": ["South"]}]
def test_apply_native_filter_mutating_descriptor_rejected_before_io(loop_runtime, clean_env):
binding = clean_env
transport = FakeTransport()
provider = build_browser_provider(transport=transport, storage=FakeStorage(), loop=loop_runtime)
step = build_step(binding)
step["step_meta"]["action_descriptor"] = {**descriptor("apply_native_filter", mutating=True), "inputs": dict(MUTATION_INPUTS)}
step["step_meta"]["mutation_contract"] = dict(VALID_CONTRACT)
result = provider(LiveProviderContext(binding=binding, step=step, completed={}))
assert result.status == "inconclusive"
assert result.reason_code == "BROWSER_ACTION_NOT_SUPPORTED"
assert transport.calls == []
def test_unknown_read_only_action_still_not_supported(loop_runtime, clean_env):
binding = clean_env
transport = FakeTransport()
provider = build_browser_provider(transport=transport, storage=FakeStorage(), loop=loop_runtime)
step = build_step(binding)
step["step_meta"]["action_descriptor"] = descriptor("navigate_dashboard")
result = provider(LiveProviderContext(binding=binding, step=step, completed={}))
assert result.status == "inconclusive"
assert result.reason_code == "BROWSER_ACTION_NOT_SUPPORTED"
assert transport.calls == []
@pytest.mark.parametrize(
"inputs, expected_code",
[
({"wait_state": "bogus"}, "BROWSER_WAIT_STATE_INVALID"),
({"values": "South"}, "BROWSER_FILTER_VALUES_INVALID"),
({"values": ["v"] * 101}, "BROWSER_FILTER_VALUES_INVALID"),
({"filter_id": 42}, "BROWSER_FILTER_INPUT_INVALID"),
],
)
def test_apply_native_filter_invalid_input_rejected_before_io(loop_runtime, clean_env, inputs, expected_code):
binding = clean_env
transport = FakeTransport()
provider = build_browser_provider(transport=transport, storage=FakeStorage(), loop=loop_runtime)
step = build_step(binding)
step["step_meta"]["action_descriptor"] = {**descriptor("apply_native_filter"), "inputs": inputs}
result = provider(LiveProviderContext(binding=binding, step=step, completed={}))
assert result.status == "inconclusive"
assert result.reason_code == expected_code
assert transport.calls == []
def test_apply_native_filter_forwards_pinned_selector_hint(loop_runtime, clean_env):
binding = clean_env
transport = FakeTransport()
provider = build_browser_provider(transport=transport, storage=FakeStorage(), loop=loop_runtime)
step = build_step(binding)
step["step_meta"]["action_descriptor"] = {**descriptor("apply_native_filter"), "inputs": {"column": "region"}}
step["step_meta"]["description"] = "B01: apply_native_filter | selector_hint: #native-filter"
result = provider(LiveProviderContext(binding=binding, step=step, completed={}))
assert result.status == "passed"
assert transport.inputs == [{"column": "region", "selector_hint": "#native-filter"}]
def test_transport_wait_state_invalid_maps_to_typed_inconclusive(loop_runtime, clean_env):
binding = clean_env
transport = FakeTransport(failure="wait_state_invalid")
provider = build_browser_provider(transport=transport, storage=FakeStorage(), loop=loop_runtime)
step = build_step(binding)
step["step_meta"]["action_descriptor"] = descriptor("wait_for_state")
result = provider(LiveProviderContext(binding=binding, step=step, completed={}))
assert result.status == "inconclusive"
assert result.reason_code == "BROWSER_WAIT_STATE_INVALID"
# #endregion Test.ScenarioExecution.BrowserProvider.NativeFilter
# #endregion Test.ScenarioExecution.BrowserProvider

View File

@@ -243,7 +243,7 @@ def test_readiness_payload_never_contains_secrets():
serialized = json.dumps(readiness).lower()
for forbidden in ("password", "secret", "token", "cookie", "credential"):
assert forbidden not in serialized
assert set(readiness) == {"provider_loop", "evidence_storage", "screenshot", "browser", "bindings"}
assert set(readiness) == {"provider_loop", "evidence_storage", "superset", "screenshot", "browser", "bindings"}
digest = sha256(json.dumps(readiness, sort_keys=True).encode()).hexdigest()
assert len(digest) == 64
# #endregion Test.ScenarioExecution.ProviderContract.Redaction

View File

@@ -1,19 +1,26 @@
# #region Test.ScenarioExecution.ProviderOperations [C:3] [TYPE Module] [SEMANTICS test,provider,operation,receipt,reconciliation]
# #region Test.ScenarioExecution.ProviderOperations [C:3] [TYPE Module] [SEMANTICS test,provider,operation,receipt,reconciliation,cancel]
# @BRIEF Verify durable receipts: pre-I/O open, write-once CAS terminals, reconciliation, late history.
# @RELATION BINDS_TO -> [ScenarioExecution.ProviderOperations.Service]
# @TEST_EDGE: duplicate_attempt -> PROVIDER_OPERATION_DUPLICATE
# @TEST_EDGE: late_overwrite -> PROVIDER_OPERATION_TERMINAL, status unchanged
# @TEST_EDGE: reconcile_wrong_source -> PROVIDER_OPERATION_TERMINAL
# @TEST_EDGE: cancel_terminal -> no-op, receipt unchanged, acknowledged completed
# @TEST_EDGE: cancel_unknown -> reconciliation_required until a reconciler verdict
# @TEST_EDGE: cancel_repeat -> idempotent, single cancel_request history entry
import uuid
from datetime import UTC, datetime, timedelta
import pytest
from src.models.provider_operation import ProviderOperationReceipt
from src.services.dashboard_testing.execution.provider_operations import (
cancel_provider_operation,
complete_provider_operation,
open_provider_operation,
reconcile_provider_operation,
reconcile_stale_provider_operations,
record_late_response,
register_provider_reconciler,
)
@@ -126,4 +133,158 @@ def test_late_response_appends_history_without_status_change(seeded_execution):
finally:
_cleanup(seeded_execution, [run_id])
# #endregion Test.ScenarioExecution.ProviderOperations.LateResponse
# #region Test.ScenarioExecution.ProviderOperations.Cancel [C:2] [TYPE Function] [SEMANTICS test,provider,operation,cancel,reconciliation]
# @BRIEF SCEX-FR-020: cancel is operation-aware — stopped/completed/unknown acknowledge semantics.
# @TEST_INVARIANT ScenarioExecution.ProviderOperations.Service.Cancel: Terminal receipts are unchanged
# on late cancel; running/reconciliation_required receipts mutate with a durable
# cancel_request history entry; unknown leaves reconciliation_required for the
# ReconcileWorker. -> VERIFIED_BY: stopped, completed_noop, unknown_reconcile,
# idempotent, escalation_unknown_to_stopped, invalid_inputs
def test_cancel_running_receipt_acknowledge_stopped(seeded_execution):
run_id = str(uuid.uuid4())
try:
receipt = open_provider_operation(seeded_execution, attempt=1, **_identity(run_id=run_id))
cancelled = cancel_provider_operation(
seeded_execution, receipt["operation_id"], reason="user abort", acknowledge="stopped",
)
assert cancelled["status"] == "cancelled"
assert cancelled["effect_state"] == "not_started"
assert cancelled["cancellation"] == "stopped"
row = seeded_execution.query(ProviderOperationReceipt).filter(
ProviderOperationReceipt.operation_id == receipt["operation_id"],
).one()
assert row.cancellation_requested is True
assert row.cancellation_deadline_at is not None
assert row.terminal_at is not None
assert row.history[-1]["kind"] == "cancel_request"
assert row.history[-1]["acknowledge"] == "stopped"
assert row.history[-1]["reason"] == "user abort"
# CAS: cancelled receipt cannot be completed afterwards
with pytest.raises(ValueError, match="PROVIDER_OPERATION_TERMINAL"):
complete_provider_operation(seeded_execution, receipt["operation_id"], status="completed", effect_state="completed")
finally:
_cleanup(seeded_execution, [run_id])
def test_cancel_completed_receipt_is_noop(seeded_execution):
run_id = str(uuid.uuid4())
try:
receipt = open_provider_operation(seeded_execution, attempt=1, **_identity(run_id=run_id))
complete_provider_operation(seeded_execution, receipt["operation_id"], status="completed", effect_state="completed")
result = cancel_provider_operation(
seeded_execution, receipt["operation_id"], reason="late abort", acknowledge="stopped",
)
assert result["status"] == "completed"
assert result["cancellation"] == "completed"
row = seeded_execution.query(ProviderOperationReceipt).filter(
ProviderOperationReceipt.operation_id == receipt["operation_id"],
).one()
assert row.cancellation_requested is False
assert row.history == []
finally:
_cleanup(seeded_execution, [run_id])
def test_cancel_failed_receipt_returns_completed_acknowledge(seeded_execution):
run_id = str(uuid.uuid4())
try:
receipt = open_provider_operation(seeded_execution, attempt=1, **_identity(run_id=run_id))
complete_provider_operation(seeded_execution, receipt["operation_id"], status="failed", effect_state="not_started")
result = cancel_provider_operation(
seeded_execution, receipt["operation_id"], reason="late", acknowledge="stopped",
)
assert result["status"] == "failed"
assert result["cancellation"] == "completed"
finally:
_cleanup(seeded_execution, [run_id])
def test_cancel_unknown_requires_reconciliation_and_worker_resolves(seeded_execution):
run_id = str(uuid.uuid4())
try:
receipt = open_provider_operation(seeded_execution, attempt=1, **_identity(run_id=run_id))
result = cancel_provider_operation(
seeded_execution, receipt["operation_id"], reason="operator stop", acknowledge="unknown",
)
assert result["status"] == "reconciliation_required"
assert result["effect_state"] == "unknown"
assert result["cancellation"] == "unknown"
row = seeded_execution.query(ProviderOperationReceipt).filter(
ProviderOperationReceipt.operation_id == receipt["operation_id"],
).one()
row.updated_at = datetime.now(UTC).replace(tzinfo=None) - timedelta(seconds=60)
seeded_execution.flush()
register_provider_reconciler(
"browser",
lambda projected: {"resolution": "failed", "effect_state": "not_started", "note": "target confirmed not started"},
)
summary = reconcile_stale_provider_operations(seeded_execution, older_than_seconds=30, limit=10)
assert summary["resolved"] >= 1
seeded_execution.refresh(row)
assert row.status == "failed"
assert any(h["kind"] == "cancel_request" and h["acknowledge"] == "unknown" for h in row.history)
assert any(h["kind"] == "reconciliation" for h in row.history)
finally:
_cleanup(seeded_execution, [run_id])
def test_cancel_is_idempotent_for_same_acknowledge(seeded_execution):
run_id = str(uuid.uuid4())
try:
receipt = open_provider_operation(seeded_execution, attempt=1, **_identity(run_id=run_id))
first = cancel_provider_operation(
seeded_execution, receipt["operation_id"], reason="abort", acknowledge="unknown",
)
second = cancel_provider_operation(
seeded_execution, receipt["operation_id"], reason="abort", acknowledge="unknown",
)
assert first["operation_id"] == second["operation_id"]
assert first["status"] == second["status"] == "reconciliation_required"
row = seeded_execution.query(ProviderOperationReceipt).filter(
ProviderOperationReceipt.operation_id == receipt["operation_id"],
).one()
cancel_entries = [h for h in row.history if h["kind"] == "cancel_request"]
assert len(cancel_entries) == 1
finally:
_cleanup(seeded_execution, [run_id])
def test_cancel_esc_unknown_then_stopped_resolves_to_cancelled(seeded_execution):
run_id = str(uuid.uuid4())
try:
receipt = open_provider_operation(seeded_execution, attempt=1, **_identity(run_id=run_id))
cancel_provider_operation(
seeded_execution, receipt["operation_id"], reason="operator stop", acknowledge="unknown",
)
result = cancel_provider_operation(
seeded_execution, receipt["operation_id"], reason="target confirmed stop", acknowledge="stopped",
)
assert result["status"] == "cancelled"
assert result["cancellation"] == "stopped"
row = seeded_execution.query(ProviderOperationReceipt).filter(
ProviderOperationReceipt.operation_id == receipt["operation_id"],
).one()
cancel_entries = [h for h in row.history if h["kind"] == "cancel_request"]
assert [h["acknowledge"] for h in cancel_entries] == ["unknown", "stopped"]
finally:
_cleanup(seeded_execution, [run_id])
@pytest.mark.parametrize("acknowledge,reason", [("maybe", "x"), ("", "x"), ("stopped", "")])
def test_cancel_rejects_invalid_inputs(seeded_execution, acknowledge, reason):
run_id = str(uuid.uuid4())
try:
receipt = open_provider_operation(seeded_execution, attempt=1, **_identity(run_id=run_id))
with pytest.raises(ValueError, match="PROVIDER_OPERATION_CANCEL_INVALID"):
cancel_provider_operation(seeded_execution, receipt["operation_id"], reason=reason, acknowledge=acknowledge)
finally:
_cleanup(seeded_execution, [run_id])
def test_cancel_rejects_unknown_operation_id(seeded_execution):
with pytest.raises(ValueError, match="PROVIDER_OPERATION_UNKNOWN"):
cancel_provider_operation(seeded_execution, "00000000-0000-0000-0000-000000000000", reason="x", acknowledge="stopped")
# #endregion Test.ScenarioExecution.ProviderOperations.Cancel
# #endregion Test.ScenarioExecution.ProviderOperations

View File

@@ -1,10 +1,14 @@
# #region Test.ScenarioExecution.ProviderPreflight [C:3] [TYPE Module] [SEMANTICS test,provider,health,readiness]
# @BRIEF Verify the readiness snapshot: registered/degraded/unregistered states and crash safety.
# @BRIEF Verify the readiness snapshot: registered/degraded/unregistered states, crash safety,
# pinned provider identity (version/capability fingerprint), SCEX-FR-022 redaction, and
# fail-closed admission of unready providers (SC-010).
# @RELATION BINDS_TO -> [ScenarioExecution.ProviderPreflight]
# @TEST_EDGE: check_crash -> degraded PROVIDER_CHECK_CRASHED, never raised
# @TEST_EDGE: unregistered_capability -> unregistered with stable reason
# @TEST_EDGE: browser_probe_failure -> degraded with typed reason, no startup blocker
# @TEST_EDGE: unready_provider_dispatch -> typed BINDING_UNAVAILABLE, zero provider invocations
import asyncio
import json
import uuid
from src.services.dashboard_testing.execution.live_binding import LiveExecutionBinding
@@ -15,6 +19,10 @@ from src.services.dashboard_testing.execution.providers.preflight import (
check_browser_executable,
check_evidence_storage,
)
from src.services.dashboard_testing.scenario.templates import (
ACTION_REGISTRY_VERSION,
action_registry_fingerprint,
)
def build_binding() -> LiveExecutionBinding:
@@ -57,8 +65,9 @@ def test_ready_snapshot_reports_registered_capabilities():
storage_check=lambda: None,
)
assert readiness["screenshot"] == {"state": "ready"}
assert readiness["browser"] == {"state": "ready"}
assert readiness["screenshot"]["state"] == "ready"
assert readiness["browser"]["state"] == "ready"
assert readiness["superset"]["state"] == "ready"
assert readiness["evidence_storage"] == {"state": "ready"}
assert readiness["provider_loop"]["state"] in {"ready", "not_started"}
assert readiness["bindings"]["screenshot_unavailable"] == []
@@ -96,7 +105,8 @@ def test_check_crash_becomes_degraded_reason():
storage_check=crashing_check,
)
assert readiness["browser"] == {"state": "degraded", "reason_code": "PROVIDER_CHECK_CRASHED"}
assert readiness["browser"]["state"] == "degraded"
assert readiness["browser"]["reason_code"] == "PROVIDER_CHECK_CRASHED"
assert readiness["evidence_storage"] == {"state": "degraded", "reason_code": "PROVIDER_CHECK_CRASHED"}
@@ -112,11 +122,160 @@ def test_unregistered_capability_reports_stable_reason():
storage_check=lambda: None,
)
assert readiness["screenshot"] == {"state": "unregistered", "reason_code": "SCREENSHOT_PROVIDER_UNREGISTERED"}
assert readiness["browser"] == {"state": "unregistered", "reason_code": "BROWSER_PROVIDER_UNREGISTERED"}
assert readiness["screenshot"]["state"] == "unregistered"
assert readiness["screenshot"]["reason_code"] == "SCREENSHOT_PROVIDER_UNREGISTERED"
assert readiness["browser"]["state"] == "unregistered"
assert readiness["browser"]["reason_code"] == "BROWSER_PROVIDER_UNREGISTERED"
assert readiness["superset"]["state"] == "unregistered"
assert readiness["superset"]["reason_code"] == "SUPERSET_PROVIDER_UNREGISTERED"
# #endregion Test.ScenarioExecution.ProviderPreflight.Degraded
# #region Test.ScenarioExecution.ProviderPreflight.Identity [C:2] [TYPE Function] [SEMANTICS test,provider,readiness,version,fingerprint]
# @BRIEF SC-010/SCEX-FR-022: every provider entry carries provider/version/capability
# fingerprint/dependency diagnostics sourced from the pinned 038 ActionRegistry.
# @TEST_INVARIANT ScenarioExecution.ProviderPreflight.Snapshot: version ==
# ACTION_REGISTRY_VERSION and capabilities.registry_fingerprint ==
# action_registry_fingerprint() for superset/screenshot/browser entries.
def test_provider_entries_carry_identity_version_and_capabilities():
root = registered_root()
readiness = build_provider_readiness(
root,
run_async=asyncio.run,
browser_check=lambda run_async=None: None,
storage_check=lambda: None,
)
for name in ("superset", "screenshot", "browser"):
entry = readiness[name]
assert entry["provider"] == name
assert entry["version"] == ACTION_REGISTRY_VERSION
assert entry["capabilities"]["registry_fingerprint"] == action_registry_fingerprint()
assert entry["capabilities"]["features"], name
assert isinstance(entry["dependencies"], dict) and entry["dependencies"], name
assert readiness["browser"]["capabilities"]["features"] == sorted({
"browser", "table_filter", "pagination", "cross_dashboard",
"row_edit", "bulk_edit", "xlsx_export", "text_filter",
})
assert readiness["screenshot"]["capabilities"]["features"] == ["screenshot"]
assert readiness["superset"]["capabilities"]["features"] == sorted({
"superset_metric", "superset_query_envelope", "dataset_field_read",
})
# Redacted dependency diagnostics: booleans and binding counts only.
assert readiness["browser"]["dependencies"] == {
"chromium_executable_resolved": True,
"provider_loop_running": readiness["provider_loop"]["state"] == "ready",
"registered_bindings": 1,
"unavailable_bindings": 0,
}
assert readiness["screenshot"]["dependencies"]["evidence_storage_ready"] is True
assert readiness["superset"]["dependencies"] == {
"evidence_storage_ready": True,
"registered_bindings": 0,
"unavailable_bindings": 1,
}
# #endregion Test.ScenarioExecution.ProviderPreflight.Identity
# #region Test.ScenarioExecution.ProviderPreflight.Redaction [C:2] [TYPE Function] [SEMANTICS test,provider,readiness,redaction]
# @BRIEF SCEX-FR-022: the serialized payload carries no credentials, cookies, SQL, or host paths.
def test_readiness_payload_is_redacted():
root = registered_root()
readiness = build_provider_readiness(
root,
run_async=asyncio.run,
browser_check=lambda run_async=None: None,
storage_check=lambda: None,
)
payload = json.dumps(readiness).lower()
for marker in ("password", "secret", "token", "cookie", "credential", "api_key", "authorization"):
assert marker not in payload
def strings(node):
if isinstance(node, dict):
for value in node.values():
yield from strings(value)
elif isinstance(node, list):
for value in node:
yield from strings(value)
elif isinstance(node, str):
yield node
for value in strings(readiness):
assert not value.startswith("/"), value
assert "\\" not in value, value
# #endregion Test.ScenarioExecution.ProviderPreflight.Redaction
# #region Test.ScenarioExecution.ProviderPreflight.FailClosedAdmission [C:3] [TYPE Function] [SEMANTICS test,provider,readiness,fail-closed,admission]
# @BRIEF SC-010: an unready provider admits zero new operations — dispatch against an
# unavailable/unregistered binding refuses I/O with a typed reason and never invokes any
# registered provider callable.
# @TEST_INVARIANT ScenarioExecution.LiveCompositionRoot.Runtime: browser/screenshot adapters return
# typed BINDING_UNAVAILABLE before any provider I/O when the binding is not ready.
def _dispatch_step(binding: LiveExecutionBinding) -> dict:
return {
"scenario_run_id": str(uuid.uuid4()),
"live_execution_binding_ref": binding.binding_ref,
"live_execution_binding_snapshot": binding.snapshot(),
"target_snapshot": {
"environment_id": binding.environment_id,
"dashboard_release_id": binding.dashboard_release_id,
"environment_class": "DEV",
},
"execution_principal_fingerprint": binding.execution_principal_fingerprint,
"step_meta": {"environment_id": binding.environment_id, "dashboard_id": binding.dashboard_id},
}
def test_unready_provider_admits_zero_new_operations():
from src.services.dashboard_testing.execution.live_composition import LiveExecutionCompositionRoot
root = LiveExecutionCompositionRoot()
unavailable = build_binding()
root.mark_browser_unavailable(unavailable.binding_ref)
root.mark_screenshot_unavailable(unavailable.binding_ref)
calls: list[str] = []
other = build_binding()
def spy(context: LiveProviderContext) -> LiveAdapterResult:
calls.append(context.binding.binding_ref)
return LiveAdapterResult(status="passed", reason_code="SHOULD_NOT_RUN")
root.register_browser(other, spy)
root.register_screenshot(other, spy)
step = _dispatch_step(unavailable)
browser_result = root.browser_adapter()(step, {})
screenshot_result = root.screenshot_adapter()(step, {})
assert browser_result.status == "inconclusive"
assert browser_result.reason_code == "BROWSER_BINDING_UNAVAILABLE"
assert screenshot_result.status == "inconclusive"
assert screenshot_result.reason_code == "SCREENSHOT_BINDING_UNAVAILABLE"
assert calls == []
def test_degraded_readiness_blocks_browser_capability_admission(monkeypatch):
from src.services.dashboard_testing.scenario import capability_authority
class _Root:
last_provider_readiness = {
"browser": {"state": "degraded", "reason_code": "BROWSER_EXECUTABLE_MISSING"},
}
monkeypatch.setattr("src.dependencies.get_live_execution_composition_root", lambda: _Root())
assert capability_authority.resolve_browser_availability() is False
# #endregion Test.ScenarioExecution.ProviderPreflight.FailClosedAdmission
# #region Test.ScenarioExecution.ProviderPreflight.RealChecks [C:2] [TYPE Function] [SEMANTICS test,provider,health,smoke]
# @BRIEF Real checks never crash and return None or a stable BROWSER_*/storage reason code.
def test_real_checks_return_stable_outcomes():

View File

@@ -4,6 +4,7 @@
# @TEST_EDGE: binding_mismatch -> typed inconclusive before I/O
# @TEST_EDGE: capacity_exhausted -> SCREENSHOT_CAPACITY_UNAVAILABLE, no capture
# @TEST_EDGE: oversize_capture -> SCREENSHOT_TOO_LARGE, no evidence stored
# @TEST_EDGE: capture_timeout -> receipt finalized failed/unknown, no PASS
import uuid
from datetime import UTC, datetime
@@ -11,6 +12,7 @@ import pytest
from src.core.database import SessionLocal
from src.models.provider_capacity import CapacityLease, CapacityQuota
from src.models.provider_operation import ProviderOperationReceipt
from src.services.dashboard_testing.execution.capacity import claim_capacity
from src.services.dashboard_testing.execution.live_binding import LiveExecutionBinding
from src.services.dashboard_testing.execution.live_composition import LiveProviderContext
@@ -305,4 +307,99 @@ def test_capture_unrecognized_bytes_fall_back_to_jpeg(loop_runtime, clean_env):
assert result.details["artifact_content_types"][ref] == "image/jpeg"
# #endregion Test.ScenarioExecution.ScreenshotProvider.MimeSniff
# #endregion Test.ScenarioExecution.ScreenshotProvider.Evidence
# #region Test.ScenarioExecution.ScreenshotProvider.Receipts [C:2] [TYPE Function] [SEMANTICS test,provider,screenshot,receipt]
# @BRIEF SCEX-FR-020 coverage: open receipt before I/O, complete with terminal status after.
# @TEST_INVARIANT ScenarioExecution.ProviderOperations.Service: A receipt is opened before external
# I/O and reaches exactly one terminal status per attempt. -> VERIFIED_BY:
# capture_receipt_running_during_io_then_completed, timeout_receipt_failed_unknown,
# empty_capture_receipt_failed
def _receipts_for(run_id: str):
with SessionLocal() as db:
return (
db.query(ProviderOperationReceipt)
.filter(ProviderOperationReceipt.run_id == run_id)
.all()
)
def _clean_receipts(run_id: str):
with SessionLocal() as db:
db.query(ProviderOperationReceipt).filter(ProviderOperationReceipt.run_id == run_id).delete()
db.commit()
def test_capture_receipt_running_during_io_then_completed(loop_runtime, clean_env):
binding = clean_env
step = build_step(binding)
run_id = step["scenario_run_id"]
seen: dict[str, list[str]] = {}
class ReceiptProbeService(FakeCaptureService):
async def capture_dashboard(self, dashboard_id: int, output_path: str):
seen["statuses"] = [row.status for row in _receipts_for(run_id)]
return await super().capture_dashboard(dashboard_id, output_path)
provider = build_screenshot_provider(service=ReceiptProbeService(), storage=FakeStorage(), loop=loop_runtime)
try:
result = provider(LiveProviderContext(binding=binding, step=step, completed={}))
assert result.status == "passed"
assert result.details["operation_id"]
# Receipt was opened in running state BEFORE the capture I/O ran.
assert seen["statuses"] == ["running"]
rows = _receipts_for(run_id)
assert len(rows) == 1
row = rows[0]
assert row.operation_id == result.details["operation_id"]
assert row.status == "completed"
assert row.effect_state == "none"
assert row.provider_id == "screenshot"
assert row.action == "capture_dashboard"
assert row.capacity_lease_id is not None
assert row.terminal_at is not None
finally:
_clean_receipts(run_id)
def test_capture_timeout_receipt_failed_unknown(loop_runtime, clean_env):
binding = clean_env
step = build_step(binding)
run_id = step["scenario_run_id"]
class TimeoutLoop:
is_running = True
def submit(self, factory, timeout=None):
raise TimeoutError()
provider = build_screenshot_provider(service=FakeCaptureService(), storage=FakeStorage(), loop=TimeoutLoop())
try:
result = provider(LiveProviderContext(binding=binding, step=step, completed={}))
assert result.status == "inconclusive"
assert result.reason_code == "SCREENSHOT_CAPTURE_TIMEOUT"
rows = _receipts_for(run_id)
assert len(rows) == 1
assert rows[0].status == "failed"
assert rows[0].effect_state == "unknown"
finally:
_clean_receipts(run_id)
def test_empty_capture_receipt_failed(loop_runtime, clean_env):
binding = clean_env
step = build_step(binding)
run_id = step["scenario_run_id"]
provider = build_screenshot_provider(service=FakeCaptureService(paths=[]), storage=FakeStorage(), loop=loop_runtime)
try:
result = provider(LiveProviderContext(binding=binding, step=step, completed={}))
assert result.status == "inconclusive"
assert result.reason_code == "SCREENSHOT_CAPTURE_EMPTY"
rows = _receipts_for(run_id)
assert len(rows) == 1
assert rows[0].status == "failed"
assert rows[0].effect_state == "none"
finally:
_clean_receipts(run_id)
# #endregion Test.ScenarioExecution.ScreenshotProvider.Receipts
# #endregion Test.ScenarioExecution.ScreenshotProvider

View File

@@ -0,0 +1,382 @@
# #region Test.ScenarioExecution.SupersetProviderContract [C:3] [TYPE Module] [SEMANTICS test,scenario,execution,live-binding,superset,contract]
# @BRIEF 044 T042 offline contract suite for the Superset/sql_evidence provider boundary.
# @RELATION BINDS_TO -> [ScenarioExecution.LiveBinding]
# @RELATION VERIFIES -> [ScenarioExecution.LiveBinding.Execute]
# @RELATION VERIFIES -> [ScenarioExecution.LiveBinding.Payload]
# @RELATION VERIFIES -> [ScenarioExecution.LiveBinding.Evidence]
# @RELATION VERIFIES -> [ScenarioExecution.ProviderOperations.Service]
# @TEST_CONTRACT: exact persisted binding + resolver -> typed outcome; any identity defect or
# external failure is typed non-pass with a durable receipt and no silent retry.
# @TEST_FIXTURE: _BINDING/_STEP/_RAW hardcoded identity (INVARIANT-01: no live environment).
# @TEST_EDGE binding_missing/invalid/mismatch/unavailable -> typed inconclusive, spy client never called
# @TEST_EDGE external_api_error -> failed SUPERSET_QUERY_FAILED + receipt failed/none
# @TEST_EDGE external_exception -> inconclusive SUPERSET_QUERY_EXECUTION_ERROR + receipt failed/unknown
# @TEST_EDGE timeout -> TimeoutError re-raised at adapter level, receipt finalized failed/unknown first
# @TEST_EDGE evidence_digest_mismatch / evidence_ref_mismatch -> inconclusive, nothing stored
# @TEST_EDGE duplicate receipt open (same run/step/attempt) -> never masks the outcome, one row
# @TEST_EDGE sql_evidence adapter shares the exact same bound execution path
# @TEST_INVARIANT ScenarioExecution.LiveBinding: No client is constructed from run metadata,
# environment labels, or principal fingerprints; missing/mismatched identity never
# reaches live I/O. -> VERIFIED_BY: missing/invalid/mismatch/unavailable tests
# @TEST_INVARIANT ScenarioExecution.LiveBinding.Evidence: Passed requires verified SHA-256 of the
# exact raw bytes and the owned opaque ref. -> VERIFIED_BY: evidence tests
# @TEST_INVARIANT ScenarioExecution.ProviderOperations.Service: Receipt opens before I/O and
# reaches exactly one terminal status; a raced duplicate open never masks the
# adapter outcome. -> VERIFIED_BY: receipt_ordering/duplicate_open tests
from __future__ import annotations
import asyncio
from hashlib import sha256
from unittest.mock import AsyncMock
import pytest
from src.core.database import SessionLocal
from src.core.superset_client._chart_data import ChartDataResponse
from src.core.utils.network import SupersetAPIError
from src.models.provider_operation import ProviderOperationReceipt
from src.schemas.dashboard_testing import (
ChartQueryModel,
DashboardQueryModel,
MetricDescriptor,
VizType,
)
from src.services.dashboard_testing.execution.live_binding import (
LiveExecutionBinding,
ResolvedLiveExecutionBinding,
sql_evidence_adapter_from,
superset_adapter_from,
)
_RAW = b'{"result":[{"data":{"revenue":7700}}],"query_id":"q-superset-contract"}'
_RAW_SHA = sha256(_RAW).hexdigest()
_BINDING = {
"binding_ref": "live-binding-superset-contract",
"environment_id": "env-preprod-superset-contract",
"dashboard_release_id": "release-superset-contract",
"release_fingerprint": "a" * 64,
"dashboard_id": 77,
"query_model_fingerprint": "sha256:model-superset-contract",
"execution_principal_fingerprint": "b" * 64,
"rls_security_fingerprint": "c" * 64,
"browser_safe_checkpoint_ref": None,
"browser_action_binding_ref": None,
"evidence_owner_type": "scenario_run",
"evidence_ref_policy": "draft_storage_raw_response",
}
_STEP = {
"logical_step_id": "superset-metric-contract",
"scenario_run_id": "run-superset-contract",
"execution_principal_fingerprint": "b" * 64,
"target_snapshot": {
"environment_id": "env-preprod-superset-contract",
"dashboard_release_id": "release-superset-contract",
},
"live_execution_binding_ref": "live-binding-superset-contract",
"live_execution_binding_snapshot": _BINDING,
"step_meta": {
"environment_id": "env-preprod-superset-contract",
"dashboard_id": 77,
"chart_id": 770,
"result_key": "revenue",
"normalized_filters": {"schema_version": 1, "filters": [], "filters_hash": "sha256:filters-contract"},
},
}
# #region Test.ScenarioExecution.SupersetProviderContract.Fakes [C:1] [TYPE Class]
class _EvidenceStore:
def __init__(self, *, broken_ref: bool = False) -> None:
self.stored: list[tuple[str, str, bytes]] = []
self.broken_ref = broken_ref
def store(self, run_id: str, digest: str, data: bytes) -> str:
self.stored.append((run_id, digest, data))
if self.broken_ref:
return f"draft:{run_id}:tampered-ref"
return f"draft:{run_id}:{digest}"
class _Resolver:
def __init__(self, resolved: ResolvedLiveExecutionBinding | None) -> None:
self.resolved = resolved
self.references: list[str] = []
def resolve(self, binding_ref: str) -> ResolvedLiveExecutionBinding | None:
self.references.append(binding_ref)
return self.resolved
# #endregion Test.ScenarioExecution.SupersetProviderContract.Fakes
# #region Test.ScenarioExecution.SupersetProviderContract.ReceiptCleanup [C:1] [TYPE Function]
@pytest.fixture(autouse=True)
def _clean_provider_operation_receipts():
"""Superset receipts live in the global test DB; keep each test isolated."""
yield
with SessionLocal() as db:
db.query(ProviderOperationReceipt).filter(
ProviderOperationReceipt.run_id == "run-superset-contract"
).delete()
db.commit()
# #endregion Test.ScenarioExecution.SupersetProviderContract.ReceiptCleanup
def _query_model() -> DashboardQueryModel:
return DashboardQueryModel(
environment_id="env-preprod-superset-contract",
dashboard_id=77,
title="Revenue contract",
query_model_fingerprint="sha256:model-superset-contract",
charts=[
ChartQueryModel(
chart_id=770,
slice_name="Revenue",
viz_type=VizType.TABLE,
dataset_id=771,
dataset_name="sales",
metrics=[MetricDescriptor(metric_name="revenue", label="Revenue", expression_type="SIMPLE")],
)
],
)
def _resolved(binding: LiveExecutionBinding, client: AsyncMock, evidence: _EvidenceStore) -> ResolvedLiveExecutionBinding:
return ResolvedLiveExecutionBinding(
binding=binding,
superset_client=client,
query_model=_query_model(),
evidence_storage=evidence,
run_async=asyncio.run,
)
def _receipts() -> list[ProviderOperationReceipt]:
with SessionLocal() as db:
return (
db.query(ProviderOperationReceipt)
.filter(ProviderOperationReceipt.run_id == "run-superset-contract")
.all()
)
def _ok_client() -> AsyncMock:
client = AsyncMock()
client.execute_chart_data_raw.return_value = ChartDataResponse(
parsed={"result": [{"data": {"revenue": 7700}}], "query_id": "q-superset-contract"},
raw_bytes=_RAW,
source_response_hash=_RAW_SHA,
)
return client
# #region Test.ScenarioExecution.SupersetProviderContract.Success [C:2] [TYPE Function]
# @BRIEF Typed success: passed outcome, receipt completed, fingerprints on the receipt identity,
# exact sha256 + owned evidence ref in the result.
def test_success_typed_outcome_receipt_and_owned_evidence():
client = _ok_client()
evidence = _EvidenceStore()
resolver = _Resolver(_resolved(LiveExecutionBinding.from_snapshot(_BINDING), client, evidence))
outcome = superset_adapter_from(resolver)(_STEP, {})
assert outcome.status == "passed"
assert outcome.reason_code == "SUPERSET_QUERY_EXECUTED"
assert outcome.details["sha256"] == _RAW_SHA
assert outcome.details["source_response_hash"] == _RAW_SHA
expected_ref = f"draft:run-superset-contract:{_RAW_SHA}"
assert outcome.artifact_refs == [expected_ref]
assert outcome.artifact_digests == {expected_ref: _RAW_SHA}
assert evidence.stored == [("run-superset-contract", _RAW_SHA, _RAW)]
receipts = _receipts()
assert len(receipts) == 1
receipt = receipts[0]
assert receipt.status == "completed"
assert receipt.effect_state == "none"
assert receipt.provider_id == "superset"
assert receipt.action == "execute_query"
assert receipt.binding_ref == "live-binding-superset-contract"
assert receipt.execution_principal_fingerprint == "b" * 64
assert receipt.terminal_at is not None
client.execute_chart_data_raw.assert_awaited_once()
# #endregion Test.ScenarioExecution.SupersetProviderContract.Success
# #region Test.ScenarioExecution.SupersetProviderContract.IdentityFailClosed [C:3] [TYPE Function]
# @BRIEF Every identity defect is typed inconclusive before any live I/O; spy client never called.
@pytest.mark.parametrize(
("step_override", "resolved", "expected_code"),
[
({"live_execution_binding_ref": None, "live_execution_binding_snapshot": None}, None, "SUPERSET_BINDING_MISSING"),
({"live_execution_binding_snapshot": {"binding_ref": "broken"}}, None, "SUPERSET_BINDING_INVALID"),
(
{"execution_principal_fingerprint": "d" * 64},
"resolved",
"SUPERSET_BINDING_MISMATCH",
),
(
{"target_snapshot": {"environment_id": "env-other", "dashboard_release_id": "release-superset-contract"}},
"resolved",
"SUPERSET_BINDING_MISMATCH",
),
({}, None, "SUPERSET_BINDING_UNAVAILABLE"),
],
)
def test_identity_defects_fail_closed_without_io(step_override, resolved, expected_code):
client = _ok_client()
binding = LiveExecutionBinding.from_snapshot(_BINDING)
resolver = _Resolver(
_resolved(binding, client, _EvidenceStore()) if resolved == "resolved" else None
)
step = {**_STEP, **step_override}
outcome = superset_adapter_from(resolver)(step, {})
assert outcome.status == "inconclusive"
assert outcome.reason_code == expected_code
client.execute_chart_data_raw.assert_not_awaited()
assert _receipts() == []
# #endregion Test.ScenarioExecution.SupersetProviderContract.IdentityFailClosed
# #region Test.ScenarioExecution.SupersetProviderContract.ExternalFailures [C:3] [TYPE Function]
# @BRIEF 037 failure taxonomy is preserved end to end and finalizes the receipt; no silent retry.
def test_external_api_error_is_typed_failed_with_failed_receipt():
client = _ok_client()
client.execute_chart_data_raw.side_effect = SupersetAPIError("preprod 403")
resolver = _Resolver(_resolved(LiveExecutionBinding.from_snapshot(_BINDING), client, _EvidenceStore()))
outcome = superset_adapter_from(resolver)(_STEP, {})
assert outcome.status == "failed"
assert outcome.reason_code == "SUPERSET_QUERY_FAILED"
receipts = _receipts()
assert len(receipts) == 1
assert receipts[0].status == "failed"
assert receipts[0].effect_state == "none"
def test_external_exception_is_typed_inconclusive_with_unknown_effect_receipt():
client = _ok_client()
client.execute_chart_data_raw.side_effect = RuntimeError("transport exploded")
resolver = _Resolver(_resolved(LiveExecutionBinding.from_snapshot(_BINDING), client, _EvidenceStore()))
outcome = superset_adapter_from(resolver)(_STEP, {})
assert outcome.status == "inconclusive"
assert outcome.reason_code == "SUPERSET_QUERY_EXECUTION_ERROR"
receipts = _receipts()
assert len(receipts) == 1
assert receipts[0].status == "failed"
assert receipts[0].effect_state == "unknown"
def test_external_timeout_reraises_and_receipt_is_finalized_failed_unknown():
client = _ok_client()
client.execute_chart_data_raw.side_effect = TimeoutError("chart-data timed out")
resolver = _Resolver(_resolved(LiveExecutionBinding.from_snapshot(_BINDING), client, _EvidenceStore()))
with pytest.raises(TimeoutError):
superset_adapter_from(resolver)(_STEP, {})
receipts = _receipts()
assert len(receipts) == 1
assert receipts[0].status == "failed"
assert receipts[0].effect_state == "unknown"
# #endregion Test.ScenarioExecution.SupersetProviderContract.ExternalFailures
# #region Test.ScenarioExecution.SupersetProviderContract.Evidence [C:3] [TYPE Function]
# @BRIEF SCEX-FR-018: passed requires the verified digest and the owned opaque ref; nothing else passes.
def test_evidence_digest_mismatch_is_inconclusive_and_nothing_stored():
client = _ok_client()
client.execute_chart_data_raw.return_value = ChartDataResponse(
parsed={"result": [{"data": {"revenue": 7700}}], "query_id": "q-superset-contract"},
raw_bytes=_RAW,
source_response_hash="0" * 64,
)
evidence = _EvidenceStore()
resolver = _Resolver(_resolved(LiveExecutionBinding.from_snapshot(_BINDING), client, evidence))
outcome = superset_adapter_from(resolver)(_STEP, {})
assert outcome.status == "inconclusive"
assert outcome.reason_code == "SUPERSET_EVIDENCE_INVALID"
assert evidence.stored == []
receipts = _receipts()
assert len(receipts) == 1
assert receipts[0].status == "failed"
def test_evidence_ref_mismatch_is_inconclusive():
client = _ok_client()
evidence = _EvidenceStore(broken_ref=True)
resolver = _Resolver(_resolved(LiveExecutionBinding.from_snapshot(_BINDING), client, evidence))
outcome = superset_adapter_from(resolver)(_STEP, {})
assert outcome.status == "inconclusive"
assert outcome.reason_code == "SUPERSET_EVIDENCE_REF_INVALID"
receipts = _receipts()
assert len(receipts) == 1
assert receipts[0].status == "failed"
# #endregion Test.ScenarioExecution.SupersetProviderContract.Evidence
# #region Test.ScenarioExecution.SupersetProviderContract.ReceiptOrdering [C:3] [TYPE Function]
# @BRIEF The receipt is opened in running state before the 037 client call and terminally closed after.
def test_receipt_opens_running_before_io():
seen: dict[str, list[str]] = {}
def _probe(*_args, **_kwargs):
seen["statuses"] = [row.status for row in _receipts()]
return ChartDataResponse(
parsed={"result": [{"data": {"revenue": 7700}}], "query_id": "q-superset-contract"},
raw_bytes=_RAW,
source_response_hash=_RAW_SHA,
)
client = _ok_client()
client.execute_chart_data_raw.side_effect = _probe
resolver = _Resolver(_resolved(LiveExecutionBinding.from_snapshot(_BINDING), client, _EvidenceStore()))
outcome = superset_adapter_from(resolver)(_STEP, {})
assert outcome.status == "passed"
assert seen["statuses"] == ["running"]
assert _receipts()[0].status == "completed"
def test_duplicate_receipt_open_never_masks_outcome():
"""A raced duplicate open (same run/step/attempt) is logged and the adapter still completes."""
resolver = _Resolver(_resolved(LiveExecutionBinding.from_snapshot(_BINDING), _ok_client(), _EvidenceStore()))
adapter = superset_adapter_from(resolver)
first = adapter(_STEP, {})
second = adapter(_STEP, {})
assert first.status == "passed"
assert second.status == "passed"
assert len(_receipts()) == 1
# #endregion Test.ScenarioExecution.SupersetProviderContract.ReceiptOrdering
# #region Test.ScenarioExecution.SupersetProviderContract.SqlEvidence [C:2] [TYPE Function]
# @BRIEF The read-only sql_evidence adapter executes through the exact same bound boundary.
def test_sql_evidence_adapter_shares_bound_path():
client = _ok_client()
evidence = _EvidenceStore()
resolver = _Resolver(_resolved(LiveExecutionBinding.from_snapshot(_BINDING), client, evidence))
outcome = sql_evidence_adapter_from(resolver)(_STEP, {})
assert outcome.status == "passed"
assert outcome.reason_code == "SUPERSET_QUERY_EXECUTED"
client.execute_chart_data_raw.assert_awaited_once()
missing = sql_evidence_adapter_from(_Resolver(None))(
{**_STEP, "live_execution_binding_ref": None, "live_execution_binding_snapshot": None}, {}
)
assert missing.status == "inconclusive"
assert missing.reason_code == "SUPERSET_BINDING_MISSING"
# #endregion Test.ScenarioExecution.SupersetProviderContract.SqlEvidence
# #endregion Test.ScenarioExecution.SupersetProviderContract

View File

@@ -14,12 +14,31 @@
# @TEST_EDGE: capacity_blocked -> active run in env rejects candidate via max_concurrent_per_env
# @TEST_EDGE: stale_rule_environment_class -> rule metadata cannot classify PROD
# @TEST_EDGE: dedup_blocked -> recent duplicate fingerprint rejects candidate
# @TEST_EDGE: dispatch_deploy_to_preprod -> real 044 start persists a pinned queued run
# @TEST_EDGE: dispatch_etl_completed -> real 044 start persists a run with trigger_source=etl_completed
# @TEST_EDGE: dispatch_disabled_or_mismatched_rule -> zero durable run rows
# @TEST_EDGE: dispatch_capacity_shared_across_rules -> one event creates at most max_concurrent runs per env
# @TEST_EDGE: dispatch_prod_gate -> PROD event run persists pending_approval with one durable gate
# @TEST_EDGE: dispatch_dedup_recent_completed -> a completed run inside the window blocks the duplicate
# @TEST_EDGE: dispatch_event_redelivery -> the same event twice yields exactly one run
# @TEST_INVARIANT ScenarioAutomation.Trigger: A real event source is passed to 044 before
# ScenarioRun creation; it is never post-create provenance mutation.
# @TEST_INVARIANT ScenarioExecution.EnvironmentPolicy: Event/rule data carries target identity,
# never PROD authority; 044 resolves the configured class at its common durable
# start boundary. -> VERIFIED_BY: test_trigger_rule_environment_class_does_not_decide_prod_authority,
# test_dispatch_trigger_event_calls_run_boundary_with_pinned_current_revision
# test_dispatch_trigger_event_calls_run_boundary_with_pinned_current_revision,
# test_dispatch_prod_event_creates_pending_approval_run_with_durable_gate
# @TEST_INVARIANT ScenarioAutomation.Policy: the event-path dedup window is honored across active
# AND recent completed runs. -> VERIFIED_BY:
# test_dispatch_dedup_window_blocks_recent_completed_duplicate
# @TEST_INVARIANT ScenarioAutomation.Policy: capacity consumed by a run created earlier in the
# same dispatch is visible to later rules of the same event. -> VERIFIED_BY:
# test_dispatch_capacity_gate_is_shared_across_rules_of_one_event
# @TEST_INVARIANT ScenarioAutomation.Trigger.Dispatch: redelivering one event never duplicates a
# run (dedup/idempotency). -> VERIFIED_BY: test_dispatch_redelivered_event_creates_no_duplicate_run
# @NOTE manual_run_only typed reject (no durable rows) is already proven by
# test_scenario_manual_run_only.py::test_trigger_dispatch_rejects_human_revision_before_run_creation
# and is referenced here instead of duplicated.
from src.services.dashboard_testing.automation.trigger import dispatch_trigger_event, handle_trigger_event
@@ -185,4 +204,296 @@ def test_dispatch_trigger_event_calls_run_boundary_with_pinned_current_revision(
assert calls[0]["trigger_source"] == "release_created"
assert calls[0]["config_manager"].get_environment("preprod").stage == "PREPROD"
# #endregion Test.ScenarioAutomation.Trigger.Policy
# #region Test.ScenarioAutomation.Trigger.DispatchGates [C:4] [TYPE Function] [SEMANTICS test,scenario,automation,trigger,dispatch,gates]
# @ingroup Test.ScenarioAutomation.Trigger
# @BRIEF Drive dispatch_trigger_event with the real 044 start_run over hardcoded fixtures:
# deploy/release/ETL events persist pinned runs, disabled/mismatched rules stay inert,
# and capacity/PROD/dedup gates hold on the event path.
# @TEST_FIXTURE: dispatch_gate_graphs -> INLINE hardcoded assertion graph (automatable)
# @TEST_FIXTURE: dispatch_gate_envs -> conftest INLINE_CONFIG_MANAGER (preprod/prod)
from datetime import UTC, datetime
from src.models.scenario_approval import ActionApprovalGate
from src.models.scenario_automation import AutomationPolicy, ScenarioTriggerRule
from src.models.scenario_registry import ScenarioRegistryEntry, ScenarioRevision
from src.models.scenario_run import ScenarioRun
from src.services.dashboard_testing.scenario.templates import (
ACTION_REGISTRY_VERSION,
action_registry_fingerprint,
resolve_action_descriptor,
)
_DISPATCH_SCENARIOS = {
"deploy": "80460000-0000-4000-8000-0000000000d1",
"release": "80460000-0000-4000-8000-0000000000d2",
"etl": "80460000-0000-4000-8000-0000000000d3",
"capacity_a": "80460000-0000-4000-8000-0000000000d4",
"capacity_b": "80460000-0000-4000-8000-0000000000d5",
"prod": "80460000-0000-4000-8000-0000000000d6",
"dedup": "80460000-0000-4000-8000-0000000000d7",
"redelivery": "80460000-0000-4000-8000-0000000000d8",
"inert": "80460000-0000-4000-8000-0000000000d9",
"blocked_capacity": "80460000-0000-4000-8000-0000000000da",
}
_DISPATCH_REVISION = "81460000-0000-4000-8000-0000000000e1"
_DISPATCH_POLICY = "82460000-0000-4000-8000-0000000000f1"
def _dispatch_graph() -> dict:
return {
"action_registry_version": ACTION_REGISTRY_VERSION,
"action_registry_hash": action_registry_fingerprint(),
"steps": [{
"logical_step_id": "assert-dispatch-046",
"tool": "assertion",
"action": "structural_assert",
"action_descriptor": resolve_action_descriptor(
tool="assertion", action="structural_assert",
registry_version=ACTION_REGISTRY_VERSION,
registry_hash=action_registry_fingerprint(),
).snapshot(),
"actual": 1,
"expected": 1,
}],
"dependencies": [],
}
def _persist_dispatch_scenario(db, scenario_id: str, *, environment_ids: list[str]) -> str:
revision_id = f"81460000-0000-4000-8000-{scenario_id[-12:]}"
db.add(ScenarioRegistryEntry(
scenario_id=scenario_id,
scenario_key=f"dispatch-gate-046-{scenario_id[-4:]}",
name="Dispatch gate 046",
dashboard_id=46,
environment_ids=environment_ids,
owner_id="analyst-046",
owner_username="analyst.046",
lifecycle_status="READY",
validation_status="valid",
current_revision_id=revision_id,
))
db.add(ScenarioRevision(
revision_id=revision_id,
scenario_id=scenario_id,
content_hash="6" * 64,
graph_snapshot=_dispatch_graph(),
execution_template_hash="",
template_version="v1",
schema_version=1,
compatibility_family="default",
change_summary={"reason": "hardcoded dispatch-gate fixture"},
created_by="analyst-046",
activation_status="current",
))
db.flush()
return revision_id
def _add_rule(db, rule_id: str, trigger: str, scenario_id: str, environment_id: str,
*, enabled: bool = True, revision_policy: str = "pinned",
revision_id: str | None = _DISPATCH_REVISION, policy_id: str | None = None) -> None:
db.add(ScenarioTriggerRule(
id=rule_id,
trigger=trigger,
scenario_id=scenario_id,
revision_policy=revision_policy,
revision_id=revision_id,
environment_id=environment_id,
policy_id=policy_id,
enabled=enabled,
))
db.flush()
def _run_count(db, scenario_id: str) -> int:
return db.query(ScenarioRun).filter(ScenarioRun.scenario_id == scenario_id).count()
def test_dispatch_deploy_to_preprod_creates_pinned_queued_run(registry_session):
revision_id = _persist_dispatch_scenario(registry_session, _DISPATCH_SCENARIOS["deploy"], environment_ids=["preprod"])
_add_rule(registry_session, "rule-deploy-046", "deploy_to_preprod", _DISPATCH_SCENARIOS["deploy"], "preprod", revision_id=revision_id)
created = dispatch_trigger_event(
registry_session, {"type": "deploy_to_preprod", "fingerprint": "deploy-046-1"}
)
assert len(created) == 1
run = registry_session.query(ScenarioRun).filter(ScenarioRun.scenario_id == _DISPATCH_SCENARIOS["deploy"]).one()
assert run.id == created[0]
assert run.status == "queued"
assert run.trigger_source == "deploy_to_preprod"
assert run.scenario_revision_id == revision_id
assert run.environment_id == "preprod"
assert (run.parameter_bindings or {}).get("automation_fingerprint") == "deploy-046-1"
assert run.idempotency_key == "trigger-rule-deploy-046-deploy-046-1"
assert registry_session.query(ActionApprovalGate).count() == 0
def test_dispatch_release_created_resolves_current_revision_run(registry_session):
revision_id = _persist_dispatch_scenario(registry_session, _DISPATCH_SCENARIOS["release"], environment_ids=["preprod"])
_add_rule(
registry_session, "rule-release-046", "release_created", _DISPATCH_SCENARIOS["release"],
"preprod", revision_policy="current", revision_id=None,
)
created = dispatch_trigger_event(
registry_session, {"type": "release_created", "fingerprint": "rel-046-1"}
)
assert len(created) == 1
run = registry_session.query(ScenarioRun).filter(ScenarioRun.scenario_id == _DISPATCH_SCENARIOS["release"]).one()
assert run.trigger_source == "release_created"
assert run.scenario_revision_id == revision_id
assert run.status == "queued"
def test_dispatch_etl_completed_creates_run(registry_session):
revision_id = _persist_dispatch_scenario(registry_session, _DISPATCH_SCENARIOS["etl"], environment_ids=["preprod"])
_add_rule(registry_session, "rule-etl-046", "etl_completed", _DISPATCH_SCENARIOS["etl"], "preprod", revision_id=revision_id)
created = dispatch_trigger_event(
registry_session, {"type": "etl_completed", "fingerprint": "etl-046-1"}
)
assert len(created) == 1
run = registry_session.query(ScenarioRun).filter(ScenarioRun.scenario_id == _DISPATCH_SCENARIOS["etl"]).one()
assert run.trigger_source == "etl_completed"
assert run.environment_id == "preprod"
def test_dispatch_skips_disabled_rule_and_event_mismatch_without_durable_rows(registry_session):
revision_id = _persist_dispatch_scenario(registry_session, _DISPATCH_SCENARIOS["inert"], environment_ids=["preprod"])
_add_rule(registry_session, "rule-disabled-046", "deploy_to_preprod", _DISPATCH_SCENARIOS["inert"], "preprod", enabled=False, revision_id=revision_id)
_add_rule(registry_session, "rule-mismatch-046", "etl_completed", _DISPATCH_SCENARIOS["inert"], "preprod", revision_id=revision_id)
created = dispatch_trigger_event(
registry_session, {"type": "deploy_to_preprod", "fingerprint": "deploy-046-inert"}
)
assert created == []
assert _run_count(registry_session, _DISPATCH_SCENARIOS["inert"]) == 0
assert registry_session.query(ActionApprovalGate).count() == 0
def test_dispatch_capacity_gate_blocks_candidate_with_active_env_run(registry_session):
revision_id = _persist_dispatch_scenario(registry_session, _DISPATCH_SCENARIOS["blocked_capacity"], environment_ids=["preprod"])
registry_session.add(AutomationPolicy(
id=_DISPATCH_POLICY, name="dispatch-cap-046", max_concurrent_per_env=1,
dedup_window_seconds=0, overlap_rule="block",
))
registry_session.add(ScenarioRun(
id="existing-queued-046",
scenario_id=_DISPATCH_SCENARIOS["blocked_capacity"],
scenario_revision_id=revision_id,
scenario_content_hash="6" * 64,
environment_id="preprod",
status="queued",
phase="walker",
parameter_bindings={"automation_fingerprint": "other-fingerprint"},
trigger_source="deploy_to_preprod",
idempotency_key="trigger-existing-queued-046",
))
_add_rule(
registry_session, "rule-blocked-cap-046", "deploy_to_preprod",
_DISPATCH_SCENARIOS["blocked_capacity"], "preprod", revision_id=revision_id, policy_id=_DISPATCH_POLICY,
)
registry_session.flush()
created = dispatch_trigger_event(
registry_session, {"type": "deploy_to_preprod", "fingerprint": "deploy-046-cap"}
)
assert created == []
assert registry_session.query(ScenarioRun).filter(
ScenarioRun.environment_id == "preprod"
).count() == 1
def test_dispatch_capacity_gate_is_shared_across_rules_of_one_event(registry_session):
revision_a = _persist_dispatch_scenario(registry_session, _DISPATCH_SCENARIOS["capacity_a"], environment_ids=["preprod"])
revision_b = _persist_dispatch_scenario(registry_session, _DISPATCH_SCENARIOS["capacity_b"], environment_ids=["preprod"])
registry_session.add(AutomationPolicy(
id=_DISPATCH_POLICY, name="dispatch-shared-cap-046", max_concurrent_per_env=1,
dedup_window_seconds=0, overlap_rule="block",
))
_add_rule(registry_session, "rule-cap-a-046", "deploy_to_preprod", _DISPATCH_SCENARIOS["capacity_a"], "preprod", revision_id=revision_a, policy_id=_DISPATCH_POLICY)
_add_rule(registry_session, "rule-cap-b-046", "deploy_to_preprod", _DISPATCH_SCENARIOS["capacity_b"], "preprod", revision_id=revision_b, policy_id=_DISPATCH_POLICY)
registry_session.flush()
created = dispatch_trigger_event(
registry_session, {"type": "deploy_to_preprod", "fingerprint": "deploy-046-shared"}
)
assert len(created) == 1
assert registry_session.query(ScenarioRun).filter(
ScenarioRun.environment_id == "preprod"
).count() == 1
def test_dispatch_prod_event_creates_pending_approval_run_with_durable_gate(registry_session):
revision_id = _persist_dispatch_scenario(registry_session, _DISPATCH_SCENARIOS["prod"], environment_ids=["prod"])
_add_rule(registry_session, "rule-prod-046", "release_created", _DISPATCH_SCENARIOS["prod"], "prod", revision_id=revision_id)
created = dispatch_trigger_event(
registry_session, {"type": "release_created", "fingerprint": "rel-046-prod"}
)
assert len(created) == 1
run = registry_session.query(ScenarioRun).filter(ScenarioRun.scenario_id == _DISPATCH_SCENARIOS["prod"]).one()
assert run.status == "pending_approval"
assert run.trigger_source == "release_created"
assert (run.target_snapshot or {}).get("environment_class") == "PROD"
gate = registry_session.query(ActionApprovalGate).filter(ActionApprovalGate.owner_id == run.id).one()
assert gate.owner_type == "scenario_run"
assert gate.status == "pending"
def test_dispatch_dedup_window_blocks_recent_completed_duplicate(registry_session):
revision_id = _persist_dispatch_scenario(registry_session, _DISPATCH_SCENARIOS["dedup"], environment_ids=["preprod"])
registry_session.add(AutomationPolicy(
id=_DISPATCH_POLICY, name="dispatch-dedup-046", max_concurrent_per_env=5,
dedup_window_seconds=3600, overlap_rule="block",
))
registry_session.add(ScenarioRun(
id="completed-duplicate-046",
scenario_id=_DISPATCH_SCENARIOS["dedup"],
scenario_revision_id=revision_id,
scenario_content_hash="6" * 64,
environment_id="preprod",
status="passed",
phase="done",
parameter_bindings={"automation_fingerprint": "rel-046-dup"},
trigger_source="release_created",
idempotency_key="trigger-completed-duplicate-046",
created_at=datetime.now(UTC),
))
_add_rule(
registry_session, "rule-dedup-046", "release_created",
_DISPATCH_SCENARIOS["dedup"], "preprod", revision_id=revision_id, policy_id=_DISPATCH_POLICY,
)
registry_session.flush()
created = dispatch_trigger_event(
registry_session, {"type": "release_created", "fingerprint": "rel-046-dup"}
)
assert created == []
assert _run_count(registry_session, _DISPATCH_SCENARIOS["dedup"]) == 1
def test_dispatch_redelivered_event_creates_no_duplicate_run(registry_session):
revision_id = _persist_dispatch_scenario(registry_session, _DISPATCH_SCENARIOS["redelivery"], environment_ids=["preprod"])
_add_rule(registry_session, "rule-redelivery-046", "etl_completed", _DISPATCH_SCENARIOS["redelivery"], "preprod", revision_id=revision_id)
event = {"type": "etl_completed", "fingerprint": "etl-046-redelivery"}
first = dispatch_trigger_event(registry_session, event)
second = dispatch_trigger_event(registry_session, event)
assert len(first) == 1
assert second == []
assert _run_count(registry_session, _DISPATCH_SCENARIOS["redelivery"]) == 1
# #endregion Test.ScenarioAutomation.Trigger.DispatchGates
# #endregion Test.ScenarioAutomation.Trigger

View File

@@ -0,0 +1,200 @@
# #region Test.ScenarioExecution.CompareBaselineTypedFailure [C:4] [TYPE Module] [SEMANTICS test,scenario,execution,compare,baseline,dispatch,typed]
# @defgroup Test.ScenarioExecution.CompareBaselineTypedFailure UX-2 residual: a compare_to_baseline execution
# failure is a typed step outcome, never a queued-dispatch-loop crash.
# @RELATION BINDS_TO -> [ScenarioExecution.OfflineExecutors.Assertion]
# @RELATION BINDS_TO -> [ScenarioExecution.Dispatch.Step]
# @RELATION VERIFIES -> [ScenarioExecution.OfflineExecutors.Assertion]
# @RELATION VERIFIES -> [ScenarioExecution.Dispatch.Step]
# @TEST_FIXTURE: ux2_compare_baseline_044 -> INLINE hardcoded compiled baseline_ref expectation
# @TEST_EDGE compiled_baseline_ref -> the graph compiler emits Expected(kind=baseline_ref); feeding it to
# NormalizedValue raised pydantic ValidationError and closed the whole live run as
# QUEUED_DISPATCH_ERROR (ss-prod B01 run 6de8d0d9, 2026-09-12).
# @TEST_INVARIANT ScenarioExecution.OfflineExecutors.Assertion: a compiled baseline_ref expectation is
# typed inconclusive BASELINE_EVIDENCE_UNAVAILABLE, never a ValidationError.
# -> VERIFIED_BY: test_assertion_baseline_ref_expectation_is_typed_inconclusive
# @TEST_INVARIANT ScenarioExecution.Dispatch.Step: an executor exception is contained at the step
# boundary as typed inconclusive EXECUTOR_STEP_ERROR; the run terminalizes honestly
# and QUEUED_DISPATCH_ERROR stays reserved for infrastructure closure.
# -> VERIFIED_BY: test_dispatch_executor_exception_is_typed_not_queued_dispatch_error
# @TEST_INVARIANT ScenarioExecution.OfflineExecutors.Assertion: literal-value assertions keep the
# existing 037 compare_values behavior (pass/fail untouched).
# -> VERIFIED_BY: test_assertion_literal_comparison_behavior_unchanged,
# test_dispatch_compare_to_baseline_run_terminalizes_inconclusive
from __future__ import annotations
from src.models.scenario_run import ScenarioRun, ScenarioStepRun
from src.services.dashboard_testing.execution.executor_registry import ScenarioExecutorRegistry
from src.services.dashboard_testing.execution.executors import assertion
from src.services.dashboard_testing.execution.offline_executors import assertion as offline_assertion
from src.services.dashboard_testing.execution.registry_builder import _build_default_registry
from src.services.dashboard_testing.execution.runner import dispatch_queued_runs
from src.services.dashboard_testing.scenario.templates import (
ACTION_REGISTRY_VERSION,
action_registry_fingerprint,
resolve_action_descriptor,
)
_SCENARIO_ID = "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaa1"
_REVISION_ID = "bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbb1"
# Exact compiled expectation shape emitted by ScenarioGraph.Compiler.BuildStep for compare_to_baseline.
_COMPILED_BASELINE_REF = {
"kind": "baseline_ref",
"ref": "baseline.default",
"predicate": "approx_eq",
"description": "compare to approved baseline",
}
def _step(step_id: str, tool: str, action: str, **extra: object) -> dict:
return {
"logical_step_id": step_id,
"tool": tool,
"action": action,
"action_descriptor": resolve_action_descriptor(
tool=tool,
action=action,
registry_version=ACTION_REGISTRY_VERSION,
registry_hash=action_registry_fingerprint(),
).snapshot(),
**extra,
}
def _plan(step_id: str, tool: str, action: str, **extra: object) -> dict:
return {
"action_registry_version": ACTION_REGISTRY_VERSION,
"action_registry_hash": action_registry_fingerprint(),
"topological_order": [step_id],
"dependencies": [],
"steps": [_step(step_id, tool, action, **extra)],
}
def _run(run_id: str, plan: dict, *, status: str = "queued") -> ScenarioRun:
return ScenarioRun(
id=run_id,
scenario_id=_SCENARIO_ID,
scenario_revision_id=_REVISION_ID,
scenario_content_hash="e" * 64,
environment_id="env-preprod-02",
status=status,
phase="preflight",
parameter_bindings={"fixture": "ux2-compare-baseline-044"},
target_snapshot={"environment_id": "env-preprod-02"},
trigger_source="manual",
idempotency_key=run_id,
runner_plan=plan,
execution_principal_fingerprint="f" * 64,
)
# #region Test.ScenarioExecution.CompareBaselineTypedFailure.AssertionUnit [C:3] [TYPE Function]
# @ingroup Test.ScenarioExecution.CompareBaselineTypedFailure
# @BRIEF The assertion executor fails closed typed on a compiled baseline_ref expectation.
def test_assertion_baseline_ref_expectation_is_typed_inconclusive():
step = _step("cmp-ux2-044", "assertion", "compare_to_baseline", expected=_COMPILED_BASELINE_REF)
outcome = offline_assertion(step, {})
assert outcome["status"] == "inconclusive"
assert outcome["error_code"] == "BASELINE_EVIDENCE_UNAVAILABLE"
assert outcome["step_outcome"]["reason_code"] == "BASELINE_EVIDENCE_UNAVAILABLE"
assert outcome["step_outcome"]["baseline_ref"] == "baseline.default"
assert outcome["artifact_refs"] == []
# #endregion Test.ScenarioExecution.CompareBaselineTypedFailure.AssertionUnit
# #region Test.ScenarioExecution.CompareBaselineTypedFailure.LiteralUnchanged [C:2] [TYPE Function]
# @ingroup Test.ScenarioExecution.CompareBaselineTypedFailure
# @BRIEF Literal-value assertions keep the existing 037 compare_values pass/fail behavior.
def test_assertion_literal_comparison_behavior_unchanged():
passing = assertion({
"logical_step_id": "lit-pass-ux2-044",
"expected": {"kind": "integer", "canonical_value": "1"},
"actual": {"kind": "integer", "canonical_value": "1"},
"policy": {"type": "exact"},
}, {})
assert passing["status"] == "passed"
assert passing["error_code"] is None
failing = assertion({
"logical_step_id": "lit-fail-ux2-044",
"expected": {"kind": "integer", "canonical_value": "2"},
"actual": {"kind": "integer", "canonical_value": "1"},
"policy": {"type": "exact"},
}, {})
assert failing["status"] == "failed"
assert failing["error_code"] == "FAIL"
# #endregion Test.ScenarioExecution.CompareBaselineTypedFailure.LiteralUnchanged
# #region Test.ScenarioExecution.CompareBaselineTypedFailure.DispatchTerminal [C:4] [TYPE Function]
# @ingroup Test.ScenarioExecution.CompareBaselineTypedFailure
# @BRIEF A compare_to_baseline dispatch terminalizes the run honestly instead of crashing the loop.
def test_dispatch_compare_to_baseline_run_terminalizes_inconclusive(seeded_execution):
run = _run(
"80460000-0000-4000-8000-0000000000c1",
_plan("phase-4-B01-compare_to_baseline", "assertion", "compare_to_baseline", expected=_COMPILED_BASELINE_REF),
)
seeded_execution.add(run)
seeded_execution.flush()
outcome = dispatch_queued_runs(seeded_execution, worker_id="ux2-worker", registry=_build_default_registry())
step = seeded_execution.query(ScenarioStepRun).filter_by(
run_id=run.id, logical_step_id="phase-4-B01-compare_to_baseline"
).one()
assert outcome[0]["status"] == "inconclusive"
assert run.status == "inconclusive"
assert run.phase == "completed"
assert run.error_code is None
assert run.error_code != "QUEUED_DISPATCH_ERROR"
assert step.status == "inconclusive"
assert step.error_code == "BASELINE_EVIDENCE_UNAVAILABLE"
assert step.step_outcome["step_outcome"]["reason_code"] == "BASELINE_EVIDENCE_UNAVAILABLE"
assert step.step_outcome["decision_policy_id"] == "baseline-semantic"
assert step.step_outcome["reason_codes"] == ["COMPARISON_INCONCLUSIVE"]
# A repeated dispatch tick is a durable no-op: the run is already terminal.
replay = dispatch_queued_runs(seeded_execution, worker_id="ux2-worker", registry=_build_default_registry())
assert replay == []
assert seeded_execution.query(ScenarioStepRun).filter_by(run_id=run.id).count() == 1
# #endregion Test.ScenarioExecution.CompareBaselineTypedFailure.DispatchTerminal
# #region Test.ScenarioExecution.CompareBaselineTypedFailure.DispatchBoundary [C:4] [TYPE Function]
# @ingroup Test.ScenarioExecution.CompareBaselineTypedFailure
# @BRIEF Any executor exception is contained at the step boundary; QUEUED_DISPATCH_ERROR never fires.
def test_dispatch_executor_exception_is_typed_not_queued_dispatch_error(seeded_execution):
run = _run(
"80460000-0000-4000-8000-0000000000c2",
_plan("adapter-crash-ux2-044", "assertion", "structural_assert"),
)
seeded_execution.add(run)
seeded_execution.flush()
def exploding_executor(_step, _completed):
raise RuntimeError("executor exploded")
# Live-faithful composition: the default registry, with one executor replaced by an exploding one.
registry = _build_default_registry()
registry.register("assertion", exploding_executor, action="structural_assert")
outcome = dispatch_queued_runs(seeded_execution, worker_id="ux2-boundary-worker", registry=registry)
step = seeded_execution.query(ScenarioStepRun).filter_by(
run_id=run.id, logical_step_id="adapter-crash-ux2-044"
).one()
assert outcome[0]["status"] == "inconclusive"
assert run.status == "inconclusive"
assert run.error_code is None
assert run.error_code != "QUEUED_DISPATCH_ERROR"
assert step.status == "inconclusive"
assert step.error_code == "EXECUTOR_STEP_ERROR"
assert step.step_outcome["step_outcome"]["reason_code"] == "EXECUTOR_STEP_ERROR"
assert dispatch_queued_runs(seeded_execution, worker_id="ux2-boundary-worker", registry=registry) == []
# #endregion Test.ScenarioExecution.CompareBaselineTypedFailure.DispatchBoundary
# #endregion Test.ScenarioExecution.CompareBaselineTypedFailure

View File

@@ -10,6 +10,10 @@
# not dispatch; only the CAS-winning dispatcher advances a queued manual human
# graph to its waiting_human lifecycle checkpoint. -> VERIFIED_BY:
# test_dispatch_skips_ineligible_rows_and_waits_for_human
# @TEST_EDGE: executor_exception -> typed inconclusive EXECUTOR_STEP_ERROR at the dispatch step
# boundary; the run terminalizes honestly and QUEUED_DISPATCH_ERROR stays
# infrastructure-only. -> VERIFIED_BY:
# test_dispatch_executor_exception_is_typed_inconclusive_not_dispatch_error
# @TEST_INVARIANT ScenarioExecution.Runner.RejectMalformedPlan: a legacy queued plan lacking an
# ActionExecutionDescriptor blocks and removes active leases before any adapter call.
# -> VERIFIED_BY: test_malformed_legacy_plan_blocks_and_cleans_lease
@@ -18,8 +22,6 @@
# -> VERIFIED_BY: test_prod_browser_mutation_never_calls_provider
from __future__ import annotations
from datetime import UTC, datetime
from src.models.scenario_run import ScenarioRun, ScenarioStepRun
from src.models.scenario_worker import ScenarioStepLease
from src.services.dashboard_testing.execution.executor_registry import ScenarioExecutorRegistry
@@ -161,8 +163,11 @@ def test_malformed_legacy_plan_blocks_and_cleans_lease(seeded_execution):
# #region Test.ScenarioExecution.QueuedDispatch.ErrorCleanup [C:4] [TYPE Function]
# @BRIEF A post-claim executor exception terminalizes the run without active step or lease state.
def test_dispatch_exception_closes_claimed_step_and_lease(seeded_execution):
# @ingroup Test.ScenarioExecution.QueuedDispatch
# @BRIEF A post-claim executor exception is contained at the dispatch step boundary as a typed
# inconclusive EXECUTOR_STEP_ERROR step outcome; the run terminalizes honestly and the
# infrastructure closure (QUEUED_DISPATCH_ERROR) stays untouched.
def test_dispatch_executor_exception_is_typed_inconclusive_not_dispatch_error(seeded_execution):
run = _run(
"80440000-0000-4000-8000-000000000119",
_plan("adapter-error-044", "assertion", "structural_assert"),
@@ -182,10 +187,16 @@ def test_dispatch_exception_closes_claimed_step_and_lease(seeded_execution):
).one()
assert outcome[0]["status"] == "inconclusive"
assert run.status == "inconclusive"
assert run.error_code == "QUEUED_DISPATCH_ERROR"
assert run.phase == "completed"
assert run.error_code is None
assert run.error_code != "QUEUED_DISPATCH_ERROR"
assert step.status == "inconclusive"
leases = seeded_execution.query(ScenarioStepLease).filter_by(run_id=run.id).all()
assert leases and all(lease.expires_at <= datetime.now(UTC).replace(tzinfo=None) for lease in leases)
assert step.error_code == "EXECUTOR_STEP_ERROR"
assert step.step_outcome["step_outcome"]["reason_code"] == "EXECUTOR_STEP_ERROR"
assert step.finished_at is not None
# A repeated tick is a durable no-op: the run is already terminal.
assert dispatch_queued_runs(seeded_execution, worker_id="error-worker", registry=registry) == []
# #endregion Test.ScenarioExecution.QueuedDispatch.ErrorCleanup

View File

@@ -0,0 +1,264 @@
# #region Test.ScenarioAutomation.RetentionHolds [C:4] [TYPE Module] [SEMANTICS test,scenario,automation,retention,holds,deletion,receipts]
# @BRIEF 046 T021 (offline): retention holds (approved-baseline, active operations, analytics
# minimum window) and the durable mark->eligible->delete-bytes->verify-absent->tombstone
# deletion pipeline with idempotent retries.
# @RELATION BINDS_TO -> [ScenarioAutomation.Retention.ApplyHolds]
# @RELATION BINDS_TO -> [ScenarioAutomation.Retention.DeletionPipeline]
# @RELATION VERIFIES -> [Models.ScenarioAutomation.RetentionDeletion]
# @TEST_EDGE approved_baseline_pin -> item held, never deletable while pin alive
# @TEST_EDGE active_operation_run -> artifact held until the run leaves queued/running
# @TEST_EDGE analytics_min_window -> metadata pruning never violates the 047 minimum history window
# @TEST_EDGE storage_delete_failure -> receipt stays deletion_pending, never deleted while bytes survive
# @TEST_EDGE bytes_survived_after_delete -> verify-absent fails closed, state stays deletion_pending
# @TEST_EDGE retry_same_deletion_id -> no-op on tombstoned, re-attempt on deletion_pending
# @TEST_INVARIANT ScenarioAutomation.ProductionOperations (046 line 20): deletion is
# mark->eligible-after-all-holds->delete bytes->verify absent->tombstone/audit;
# failure remains deletion_pending; deletion of baseline-held artifacts is a
# @REJECTED path. -> VERIFIED_BY: test_advance_storage_failure_stays_deletion_pending
# / test_advance_bytes_survived_never_reports_deleted /
# test_apply_holds_approved_baseline_holds_item
from __future__ import annotations
from datetime import UTC, datetime, timedelta
import pytest
from sqlalchemy import create_engine
from sqlalchemy.orm import sessionmaker
from src.models.mapping import Base
from src.models.scenario_automation import ScenarioRetentionDeletion
from src.models.scenario_run import ScenarioRun
from src.services.dashboard_testing.automation.deletions import (
advance_retention_deletions,
mark_deletion,
retry_deletion,
)
from src.services.dashboard_testing.automation.retention import (
apply_holds,
run_retention_days,
tier_limits,
)
# ── Hardcoded [EXT] byte store: delete/exists with injectable failure modes ──
class FakeByteStore:
def __init__(self, files=None, fail_delete=frozenset(), phantom=frozenset()):
self.files = dict(files or {})
self.fail_delete = set(fail_delete)
self.phantom = set(phantom)
def delete(self, ref):
if ref in self.fail_delete:
raise RuntimeError("storage backend unavailable")
self.files.pop(ref, None)
return True
def exists(self, ref):
return ref in self.phantom or ref in self.files
@pytest.fixture
def db():
engine = create_engine(
"sqlite:///file::memory:?cache=shared&uri=true",
connect_args={"check_same_thread": False},
)
Base.metadata.create_all(bind=engine)
session = sessionmaker(bind=engine)()
try:
yield session
finally:
session.rollback()
session.close()
engine.dispose()
def _days_ago(days: int) -> str:
return (datetime.now(UTC) - timedelta(days=days)).isoformat()
# #region Test.ScenarioAutomation.RetentionHolds.Pure [C:2] [TYPE Block] [SEMANTICS test,scenario,automation,retention,holds]
# @BRIEF Pure hold evaluation over existing dict structures.
def test_apply_holds_approved_baseline_holds_item():
item = {"id": "a1", "baseline_pin": {"baseline_set_id": "bs-1", "baseline_set_version": "v3"}}
# No live catalog collector wired (None) -> fail-closed: any resolved pin holds.
deletable, held = apply_holds([item], {})
assert [entry["id"] for entry in deletable] == []
assert held == [{"id": "a1", "reasons": ["approved_baseline"]}]
# A live collector reporting the pin alive also holds it.
_, held = apply_holds([item], {"approved_baselines": {"bs-1@v3"}})
assert held[0]["reasons"] == ["approved_baseline"]
# A live collector reporting the pin retired from the approved catalog releases the hold.
deletable, held = apply_holds([item], {"approved_baselines": set()})
assert [entry["id"] for entry in deletable] == ["a1"]
assert held == []
def test_apply_holds_active_operation_holds_item():
items = [
{"id": "art-1", "retention_class": "artifacts", "run_id": "run-9"},
{"id": "art-2", "retention_class": "artifacts", "run_id": "run-done"},
{"id": "run-9", "retention_class": "run_metadata"},
]
deletable, held = apply_holds(items, {"active_operations": {"run-9"}})
deletable_ids = {entry["id"] for entry in deletable}
held_ids = {entry["id"]: entry["reasons"] for entry in held}
assert "art-1" not in deletable_ids
assert held_ids["art-1"] == ["active_operation"]
# run_metadata items are identified by their own id.
assert held_ids["run-9"] == ["active_operation"]
assert "art-2" in deletable_ids
def test_apply_holds_analytics_window_holds_recent_metadata():
items = [
{"id": "meta-200d", "retention_class": "run_metadata", "created_at": _days_ago(200)},
{"id": "meta-400d", "retention_class": "run_metadata", "created_at": _days_ago(400)},
]
deletable, held = apply_holds(items, {"analytics_min_window_days": 365})
held_ids = {entry["id"] for entry in held}
assert held_ids == {"meta-200d"}
assert [entry["id"] for entry in deletable] == ["meta-400d"]
assert held[0]["reasons"] == ["analytics_window"]
def test_run_retention_days_keeps_held_items():
items = [
{"id": "art-held", "retention_class": "artifacts", "created_at": _days_ago(60), "run_id": "run-9"},
{"id": "art-free", "retention_class": "artifacts", "created_at": _days_ago(60)},
{"id": "meta-window", "retention_class": "run_metadata", "created_at": _days_ago(200)},
]
holds = {"active_operations": {"run-9"}, "analytics_min_window_days": 365}
kept = run_retention_days(items, tier_limits(), holds=holds)
kept_ids = {item["id"] for item in kept}
# Both hold classes survive pruning; only the unheld expired artifact is pruned.
assert kept_ids == {"art-held", "meta-window"}
# #endregion Test.ScenarioAutomation.RetentionHolds.Pure
# #region Test.ScenarioAutomation.RetentionHolds.Pipeline [C:3] [TYPE Block] [SEMANTICS test,scenario,automation,retention,deletion,receipts]
# @BRIEF Durable deletion receipts: idempotent mark, sweep state machine, retry idempotency.
def test_mark_deletion_is_idempotent(db):
first = mark_deletion(
db, target_type="artifact", target_id="art-1", scenario_id="sc-1",
content_ref="draft:run-1:" + "a" * 64, retention_class="screenshots",
)
second = mark_deletion(db, target_type="artifact", target_id="art-1", scenario_id="sc-1")
db.commit()
assert first.id == second.id
assert first.state == "deletion_pending"
assert db.query(ScenarioRetentionDeletion).count() == 1
def test_advance_tombstones_after_bytes_deleted_and_verified(db):
store = FakeByteStore(files={"draft:run-1:" + "a" * 64: b"bytes"})
receipt = mark_deletion(
db, target_type="artifact", target_id="art-1", scenario_id="sc-1",
content_ref="draft:run-1:" + "a" * 64,
)
db.commit()
summary = advance_retention_deletions(db, byte_store=store)
db.commit()
assert summary["tombstoned"] == 1
db.refresh(receipt)
assert receipt.state == "tombstoned"
assert receipt.verified_absent_at is not None
assert receipt.tombstoned_at is not None
assert receipt.error_code is None
assert store.exists(receipt.content_ref) is False
def test_advance_storage_failure_stays_deletion_pending(db):
ref = "draft:run-1:" + "b" * 64
store = FakeByteStore(files={ref: b"bytes"}, fail_delete={ref})
receipt = mark_deletion(db, target_type="artifact", target_id="art-2", content_ref=ref)
db.commit()
summary = advance_retention_deletions(db, byte_store=store)
db.commit()
db.refresh(receipt)
# Normative: failure remains deletion_pending; never report deleted while bytes survive.
assert receipt.state == "deletion_pending"
assert receipt.error_code == "RETENTION_DELETE_FAILED"
assert receipt.attempts == 1
assert store.exists(ref) is True
assert summary["failed"] == 1
def test_advance_bytes_survived_never_reports_deleted(db):
ref = "draft:run-1:" + "c" * 64
store = FakeByteStore(files={ref: b"bytes"}, phantom={ref})
receipt = mark_deletion(db, target_type="artifact", target_id="art-3", content_ref=ref)
db.commit()
advance_retention_deletions(db, byte_store=store)
db.commit()
db.refresh(receipt)
assert receipt.state == "deletion_pending"
assert receipt.error_code == "RETENTION_BYTES_SURVIVED"
assert receipt.verified_absent_at is None
def test_advance_holds_block_eligibility_until_release(db):
run = ScenarioRun(
scenario_id="70460000-0000-4000-8000-0000000000f1",
scenario_revision_id="70460000-0000-4000-8000-0000000000f2",
scenario_content_hash="e" * 64,
environment_id="env-preprod-01",
status="queued",
idempotency_key="retention-hold-run-001",
)
db.add(run)
db.flush()
receipt = mark_deletion(
db, target_type="artifact", target_id="art-4", scenario_id=run.scenario_id,
content_ref="draft:" + run.id + ":" + "d" * 64, run_id=run.id,
)
db.commit()
store = FakeByteStore(files={receipt.content_ref: b"bytes"})
advance_retention_deletions(db, byte_store=store)
db.commit()
db.refresh(receipt)
assert receipt.state == "deletion_pending"
assert receipt.holds_snapshot["reasons"] == ["active_operation"]
# Active operation finishes -> next sweep releases the hold and completes deletion.
run.status = "passed"
db.commit()
summary = advance_retention_deletions(db, byte_store=store)
db.commit()
db.refresh(receipt)
assert receipt.state == "tombstoned"
assert summary["tombstoned"] == 1
def test_retry_same_deletion_id_noop_on_tombstoned(db):
receipt = mark_deletion(db, target_type="artifact", target_id="art-5", content_ref=None)
db.commit()
advance_retention_deletions(db, byte_store=FakeByteStore())
db.commit()
db.refresh(receipt)
assert receipt.state == "tombstoned"
attempts = receipt.attempts
retried = retry_deletion(db, receipt.id, byte_store=FakeByteStore())
db.commit()
assert retried.id == receipt.id
assert retried.state == "tombstoned"
assert retried.attempts == attempts
def test_retry_same_deletion_id_reattempts_pending(db):
ref = "draft:run-1:" + "f" * 64
store = FakeByteStore(files={ref: b"bytes"}, fail_delete={ref})
receipt = mark_deletion(db, target_type="artifact", target_id="art-6", content_ref=ref)
db.commit()
advance_retention_deletions(db, byte_store=store)
db.commit()
db.refresh(receipt)
assert receipt.state == "deletion_pending"
healthy = FakeByteStore(files={ref: b"bytes"})
retried = retry_deletion(db, receipt.id, byte_store=healthy)
db.commit()
db.refresh(retried)
assert retried.state == "tombstoned"
assert retried.attempts == 2
assert healthy.exists(ref) is False
# #endregion Test.ScenarioAutomation.RetentionHolds.Pipeline
# #endregion Test.ScenarioAutomation.RetentionHolds

View File

@@ -121,3 +121,88 @@ def test_malformed_pinned_policy_is_rejected():
validate_pinned_runner_plan(malformed)
validate_pinned_runner_plan({"topological_order": [], "decision_policy": None})
# #endregion Test.ScenarioExecution.RunnerPlan.Policy
# #region Test.ScenarioExecution.RunnerPlan.Params [C:3] [TYPE Module] [SEMANTICS test,scenario,execution,runnerplan,params,filter]
# @ingroup Test
# @BRIEF UX-6: launch params.filter_values bind into pinned apply_native_filter steps at derivation.
# @RELATION VERIFIES -> [ScenarioExecution.RunnerPlan.BindParams]
# @TEST_EDGE: params.filter_values=["South"] -> pinned step carries param_binding {"filter_values": ["South"]}
# @TEST_EDGE: params absent/other actions -> no param_binding; provider stays in UX-1 current-state mode
# @TEST_EDGE: present-but-invalid param -> typed BROWSER_FILTER_VALUES_INVALID before any run row exists
_APPLY_FILTER_GRAPH = {
"action_registry_version": "038.4.0",
"action_registry_hash": action_registry_fingerprint(),
"environment_ids": ["env-dev"],
"dependencies": [],
"steps": [
{
"logical_step_id": "filter-step",
"tool": "browser",
"action": "apply_native_filter",
"description": "B01: apply_native_filter | selector_hint: #native-filter",
"inputs": [{"name": "param.filter_values", "kind": "parameter", "value_type": "string_list"}],
},
{
"logical_step_id": "open-step",
"tool": "browser",
"action": "open_dashboard",
},
],
}
_SCENARIO = "11111111-1111-4111-8111-111111111111"
_REVISION = "22222222-2222-4222-8222-222222222222"
def _seed_apply_filter_graph(session) -> None:
from src.models.scenario_registry import ScenarioRevision
revision = session.query(ScenarioRevision).filter(
ScenarioRevision.revision_id == _REVISION
).first()
revision.graph_snapshot = dict(_APPLY_FILTER_GRAPH)
def test_filter_values_param_binds_into_pinned_apply_native_filter_step(seeded_registry):
_seed_apply_filter_graph(seeded_registry)
plan = derive_runner_plan(seeded_registry, _SCENARIO, _REVISION, params={"filter_values": ["South"]})
filter_step = next(s for s in plan["steps"] if s["logical_step_id"] == "filter-step")
assert filter_step["param_binding"] == {"filter_values": ["South"]}
open_step = next(s for s in plan["steps"] if s["logical_step_id"] == "open-step")
assert "param_binding" not in open_step
def test_filter_values_binding_is_deterministic_and_part_of_plan_hash(seeded_registry):
_seed_apply_filter_graph(seeded_registry)
first = derive_runner_plan(seeded_registry, _SCENARIO, _REVISION, params={"filter_values": ["South"]})
second = derive_runner_plan(seeded_registry, _SCENARIO, _REVISION, params={"filter_values": ["South"]})
assert first == second
other = derive_runner_plan(seeded_registry, _SCENARIO, _REVISION, params={"filter_values": ["North"]})
assert other["plan_hash"] != first["plan_hash"]
assert other["steps"][0]["param_binding"] == {"filter_values": ["North"]}
def test_absent_param_leaves_steps_unbound_for_current_state_mode(seeded_registry):
_seed_apply_filter_graph(seeded_registry)
plan = derive_runner_plan(seeded_registry, _SCENARIO, _REVISION)
assert "param_binding" not in plan["steps"][0]
no_key = derive_runner_plan(seeded_registry, _SCENARIO, _REVISION, params={})
assert "param_binding" not in no_key["steps"][0]
explicit_none = derive_runner_plan(seeded_registry, _SCENARIO, _REVISION, params={"filter_values": None})
assert "param_binding" not in explicit_none["steps"][0]
def test_empty_and_string_filter_values_bind_current_state_and_single_value(seeded_registry):
_seed_apply_filter_graph(seeded_registry)
empty = derive_runner_plan(seeded_registry, _SCENARIO, _REVISION, params={"filter_values": []})
assert empty["steps"][0]["param_binding"] == {"filter_values": []}
single = derive_runner_plan(seeded_registry, _SCENARIO, _REVISION, params={"filter_values": "South"})
assert single["steps"][0]["param_binding"] == {"filter_values": ["South"]}
@pytest.mark.parametrize("bad_value", [42, {"a": 1}, [1, 2], ["South", 42], [" "], ["v"] * 101])
def test_invalid_filter_values_param_rejected_at_derivation(seeded_registry, bad_value):
_seed_apply_filter_graph(seeded_registry)
with pytest.raises(ValueError, match="BROWSER_FILTER_VALUES_INVALID"):
derive_runner_plan(seeded_registry, _SCENARIO, _REVISION, params={"filter_values": bad_value})
# #endregion Test.ScenarioExecution.RunnerPlan.Params

View File

@@ -6,7 +6,9 @@
# @RELATION BINDS_TO -> [Core.Scheduler.ExecuteRevisionMaterialization]
# @TEST_CONTRACT: [PersistedQueuedOrCancellingScenarioRun] -> [DurableScheduledDispatchOrFinalization]
# @TEST_FIXTURE: scheduler_callback_runs -> INLINE persisted SQLite ScenarioRun/ScenarioStepRun rows
# @TEST_EDGE: worker_exception -> typed QUEUED_DISPATCH_ERROR rather than uncaught callback failure
# @TEST_EDGE: worker_exception -> typed persisted terminal result, never an uncaught callback failure
# (malformed plan -> blocked ACTION_DESCRIPTOR_REQUIRED; executor exception ->
# inconclusive EXECUTOR_STEP_ERROR at the dispatch step boundary, UX-2 2026-09-12)
# @TEST_EDGE: database_exception -> rollback, structured error log, and close without persisted mutation
# @TEST_INVARIANT Core.Scheduler.Start: Registration installs the exact durable-dispatch and
# cancel-drain callback IDs, five-second intervals, singleton overlap limits, and
@@ -17,9 +19,9 @@
# @TEST_INVARIANT Core.Scheduler.ExecuteScenarioCancelFinalizer: An expired persisted cancellation
# finalizes without an HTTP handler or a permanently running scheduler. -> VERIFIED_BY:
# test_due_callbacks_dispatch_once_and_finalize_cancel
# @TEST_INVARIANT Core.Scheduler.ExecuteQueuedScenarioDispatch: A per-run executor failure persists
# QUEUED_DISPATCH_ERROR as a typed inconclusive result. -> VERIFIED_BY:
# test_dispatch_callback_persists_typed_error_for_executor_exception
# @TEST_INVARIANT Core.Scheduler.ExecuteQueuedScenarioDispatch: A per-run plan failure persists a
# typed blocked result (ACTION_DESCRIPTOR_REQUIRED) without escaping the callback.
# -> VERIFIED_BY: test_dispatch_callback_persists_typed_error_for_executor_exception
# @TEST_INVARIANT Core.Scheduler.ExecuteQueuedScenarioDispatch: A callback database-edge failure is
# rolled back, logged, and closed rather than escaping the scheduler. -> VERIFIED_BY:
# test_dispatch_callback_rolls_back_logs_and_closes_on_database_edge

View File

@@ -24,6 +24,18 @@
# (schedule_id, fired_at-second); a same-second double-fire creates exactly one
# run and one gate. -> VERIFIED_BY: test_scheduled_callback_double_fire_same_second_creates_single_run,
# test_scheduled_idempotency_key_is_second_deterministic
# @TEST_INVARIANT ScenarioExecution.Runner.Start: due admission resolves the BaselineSelectionPin
# before the idempotency lookup and stamps it atomically into target_snapshot and
# runner_plan; the PROD gate binds the same canonical request hash. -> VERIFIED_BY:
# test_scheduled_callback_stamps_baseline_pin_in_due_identity
# @TEST_INVARIANT ScenarioExecution.Runner.Start: a replayed due under the same idempotency key
# with a changed resolved baseline conflicts typed (IDEMPOTENCY_KEY_REUSED ->
# SCHEDULED_SCENARIO_START_REJECTED) and never silently reuses the prior run.
# -> VERIFIED_BY: test_scheduled_replay_with_changed_baseline_rejects_typed
# @TEST_INVARIANT Core.Scheduler.ScheduledIdempotencyKey: a genuinely new due (next second) with a
# changed baseline creates a new run pinned to the new canonical request and never
# mutates the earlier run. -> VERIFIED_BY:
# test_scheduled_next_due_with_changed_baseline_creates_new_run
from __future__ import annotations
import hashlib
@@ -256,6 +268,137 @@ def test_scheduled_callback_double_fire_same_second_creates_single_run(registry_
# #endregion Test.ScenarioExecution.ScheduledStart.DoubleFire
# #region Test.ScenarioExecution.ScheduledStart.BaselineStamp [C:4] [TYPE Function] [SEMANTICS test,scheduled,baseline,pin,due-identity]
# @ingroup Test.ScenarioExecution.ScheduledStart
# @BRIEF A due admission stamps the resolved BaselineSelectionPin into the run target/plan atomically with the gate hash.
def test_scheduled_callback_stamps_baseline_pin_in_due_identity(
registry_session, monkeypatch, published_baseline_pin_for_registry_starts,
):
pin = published_baseline_pin_for_registry_starts
_persist_scenario(
registry_session,
current_revision_id=_CURRENT_REVISION_ID,
graph_by_revision={_CURRENT_REVISION_ID: _assertion_graph()},
)
_fire(registry_session, monkeypatch)
run = registry_session.query(ScenarioRun).filter(ScenarioRun.scenario_id == _SCENARIO_ID).one()
target = run.target_snapshot or {}
assert target.get("baseline_pin") == pin
# The scheduled admission passes no baseline selector (schedules pin revision policy only);
# the stamped pin itself is the due baseline identity inside the canonical request hash.
assert target.get("baseline_set") is None
assert target.get("baseline_set_version") is None
assert target["baseline_pin"]["baseline_set_id"] == pin["baseline_set_id"]
assert target["baseline_pin"]["baseline_set_version"] == pin["baseline_set_version"]
assert (run.runner_plan or {}).get("baseline_pin") == pin
gate = registry_session.query(ActionApprovalGate).filter(ActionApprovalGate.owner_id == run.id).one()
assert gate.request_hash == run.request_hash
# #endregion Test.ScenarioExecution.ScheduledStart.BaselineStamp
# #region Test.ScenarioExecution.ScheduledStart.BaselineConflict [C:4] [TYPE Function] [SEMANTICS test,scheduled,baseline,pin,idempotency,conflict]
# @ingroup Test.ScenarioExecution.ScheduledStart
# @BRIEF Same-second due replay with a changed resolved baseline rejects typed; the pinned run is never silently reused.
def test_scheduled_replay_with_changed_baseline_rejects_typed(
registry_session, monkeypatch, published_baseline_pin_for_registry_starts,
):
from src.core import scheduler as scheduler_module
original_pin = published_baseline_pin_for_registry_starts
_persist_scenario(
registry_session,
current_revision_id=_CURRENT_REVISION_ID,
graph_by_revision={_CURRENT_REVISION_ID: _assertion_graph()},
)
class _FrozenDatetime(datetime):
@classmethod
def now(cls, tz=None):
return datetime(2026, 9, 12, 12, 0, 5, tzinfo=tz or UTC)
monkeypatch.setattr(scheduler_module, "datetime", _FrozenDatetime)
_fire(registry_session, monkeypatch)
assert registry_session.query(ScenarioRun).filter(ScenarioRun.scenario_id == _SCENARIO_ID).count() == 1
mutated_pin = {**original_pin, "catalog_revision_id": "cr-046-mutated-baseline-generation"}
monkeypatch.setattr(
"src.services.dashboard_testing.execution.start_run.resolve_baseline_pin",
lambda **_kwargs: mutated_pin,
)
explore = MagicMock()
monkeypatch.setattr(scheduler_module.logger, "explore", explore)
_fire(registry_session, monkeypatch)
runs = registry_session.query(ScenarioRun).filter(ScenarioRun.scenario_id == _SCENARIO_ID).all()
assert len(runs) == 1
assert (runs[0].target_snapshot or {}).get("baseline_pin") == original_pin
assert registry_session.query(ActionApprovalGate).count() == 1
rejections = [
call for call in explore.call_args_list
if call.kwargs.get("error_code") == "SCHEDULED_SCENARIO_START_REJECTED"
]
assert rejections, f"no typed rejection logged: {explore.call_args_list}"
assert rejections[0].kwargs.get("error") == "IDEMPOTENCY_KEY_REUSED"
# #endregion Test.ScenarioExecution.ScheduledStart.BaselineConflict
# #region Test.ScenarioExecution.ScheduledStart.BaselineNextDue [C:4] [TYPE Function] [SEMANTICS test,scheduled,baseline,pin,new-due,new-run]
# @ingroup Test.ScenarioExecution.ScheduledStart
# @BRIEF A new due second with a changed baseline mints a new run pinned to the new canonical request; the earlier run is untouched.
def test_scheduled_next_due_with_changed_baseline_creates_new_run(
registry_session, monkeypatch, published_baseline_pin_for_registry_starts,
):
from src.core import scheduler as scheduler_module
original_pin = published_baseline_pin_for_registry_starts
_persist_scenario(
registry_session,
current_revision_id=_CURRENT_REVISION_ID,
graph_by_revision={_CURRENT_REVISION_ID: _assertion_graph()},
)
class _FrozenDatetime(datetime):
@classmethod
def now(cls, tz=None):
return datetime(2026, 9, 12, 12, 0, 5, tzinfo=tz or UTC)
monkeypatch.setattr(scheduler_module, "datetime", _FrozenDatetime)
_fire(registry_session, monkeypatch)
mutated_pin = {**original_pin, "catalog_revision_id": "cr-046-next-baseline-generation"}
monkeypatch.setattr(
"src.services.dashboard_testing.execution.start_run.resolve_baseline_pin",
lambda **_kwargs: mutated_pin,
)
class _NextSecondDatetime(datetime):
@classmethod
def now(cls, tz=None):
return datetime(2026, 9, 12, 12, 0, 6, tzinfo=tz or UTC)
monkeypatch.setattr(scheduler_module, "datetime", _NextSecondDatetime)
_fire(registry_session, monkeypatch)
runs = (
registry_session.query(ScenarioRun)
.filter(ScenarioRun.scenario_id == _SCENARIO_ID)
.order_by(ScenarioRun.idempotency_key.asc())
.all()
)
assert len(runs) == 2
assert registry_session.query(ActionApprovalGate).count() == 2
first, second = runs
assert first.idempotency_key.endswith("2026-09-12T12:00:05+00:00")
assert second.idempotency_key.endswith("2026-09-12T12:00:06+00:00")
assert (first.target_snapshot or {}).get("baseline_pin") == original_pin
assert (second.target_snapshot or {}).get("baseline_pin") == mutated_pin
assert first.request_hash != second.request_hash
# #endregion Test.ScenarioExecution.ScheduledStart.BaselineNextDue
# #region Test.ScenarioExecution.ScheduledStart.KeyContract [C:2] [TYPE Function] [SEMANTICS test,idempotency,key,second-granularity]
# @ingroup Test.ScenarioExecution.ScheduledStart
# @BRIEF The key helper maps a whole wall-clock second to one key and splits across seconds.

View File

@@ -70,6 +70,26 @@ def test_baseline_refuses_incomplete_existing_schema(tmp_path: Path, monkeypatch
_run_alembic_upgrade()
# #region Test.AlembicMigrations.RevisionIdLength [C:2] [TYPE Function]
# @BRIEF Every Alembic revision id must fit alembic_version.version_num VARCHAR(32).
# @TEST_INVARIANT: alembic_revision_id_fits_varchar32 -> VERIFIED_BY: [Test.AlembicMigrations.RevisionIdLength]
# @REJECTED Catching this at `alembic upgrade` time was rejected — the stamp failure happens after the
# DDL and, on a dialect without transactional DDL, would leave a half-migrated schema
# (incident 2026-09-14: "0024_scenario_retention_deletions" is 33 chars > VARCHAR(32)).
def test_revision_ids_fit_varchar32() -> None:
"""Static guard: no migration revision id may exceed the version_num VARCHAR(32) column."""
import re
versions_dir = Path(__file__).parent.parent / "alembic" / "versions"
offenders = []
for path in sorted(versions_dir.glob("*.py")):
match = re.search(r'^revision\s*=\s*["\']([^"\']+)["\']', path.read_text(), re.M)
if match and len(match.group(1)) > 32:
offenders.append((path.name, len(match.group(1))))
assert not offenders, f"revision ids exceeding VARCHAR(32): {offenders}"
# #endregion Test.AlembicMigrations.RevisionIdLength
# #region Test.AlembicMigrations.RunAlembicUpgrade [C:1] [TYPE Function]
def _run_alembic_upgrade(revision: str = "head") -> None:
"""Run a named Alembic upgrade programmatically against DATABASE_URL."""

View File

@@ -0,0 +1,335 @@
# #region Test.McpAutomationParity [C:4] [TYPE Module] [SEMANTICS test,mcp,automation,rbac,acl,parity]
# @BRIEF 046 T019 / DG-2: REST<->MCP admission parity for the four automation reads plus the
# disabled==enabled HumanStep rejection parity (DEF-02) on the MCP surface.
# @RELATION VERIFIES -> [McpServer.ToolsAutomation.Register]
# @RELATION VERIFIES -> [McpServer.RbacServer]
# @RELATION VERIFIES -> [ScenarioAutomation.Eligibility.Assert]
# @TEST_INVARIANT Reads admit any authenticated principal type with scenario:automation READ
# (human) or the mcp:read scope (service); anonymous -> unauthenticated (401),
# lacking grant -> permission_denied (403), foreign scenario -> not_found (404).
# @TEST_INVARIANT A disabled schedule bound to a human-step revision is rejected with
# AUTOMATION_INELIGIBLE_HUMAN_STEP and zero side-effect rows (DEF-02, MCP side).
from __future__ import annotations
from uuid import uuid4
import pytest
from mcp.server.fastmcp import FastMCP
from sqlalchemy import create_engine, event
from sqlalchemy.orm import sessionmaker
import src.mcp_server.rbac_server as rbac_module
import src.mcp_server.tools_automation as automation_module
from src.mcp_server import server as mcp_server
from src.mcp_server.auth import _access_token_context
from src.mcp_server.tools_automation import register_automation_tools
from src.models.auth import Permission, Role, User
from src.models.mapping import Base
from src.models.scenario_automation import AutomationPolicy, ScenarioSchedule
from src.models.scenario_registry import ScenarioRegistryEntry, ScenarioRevision
from src.services.dashboard_testing.scenario.templates import (
ACTION_REGISTRY_VERSION,
action_registry_fingerprint,
resolve_action_descriptor,
)
# #region Test.McpAutomationParity.Fixtures [C:3] [TYPE Block] [SEMANTICS test,mcp,sqlite,fixture]
# @ingroup Test.McpAutomationParity
# @BRIEF Isolated shared SQLite wired into the automation/rbac SessionLocal seams; hardcoded
# owner/foreign fixture rows for ACL assertions.
@pytest.fixture
def isolated_db(tmp_path, monkeypatch):
engine = create_engine(f"sqlite:///{tmp_path / 'mcp_automation.db'}", connect_args={"check_same_thread": False})
event.listen(engine, "connect", lambda connection, _: connection.execute("PRAGMA foreign_keys=ON"))
Base.metadata.create_all(engine)
factory = sessionmaker(bind=engine)
monkeypatch.setattr(automation_module, "SessionLocal", factory)
monkeypatch.setattr(rbac_module, "SessionLocal", factory)
try:
yield factory
finally:
engine.dispose()
def _seed_user(factory, username: str, permissions: list[tuple[str, str]], *, admin: bool = False) -> None:
role = Role(name=f"role-{username}", is_admin=admin)
role.permissions = [Permission(resource=resource, action=action) for resource, action in permissions]
with factory() as db:
db.add(User(username=username, password_hash="x", is_active=True, roles=[role]))
db.commit()
def _seed_acl_rows(factory) -> dict[str, str]:
"""Hardcoded fixture: scenario owned by 'acl.reader' plus one foreign scenario, one
schedule and one referenced policy each."""
with factory() as db:
own_policy = AutomationPolicy(name="mcp-acl-policy-own")
foreign_policy = AutomationPolicy(name="mcp-acl-policy-foreign")
db.add_all([own_policy, foreign_policy])
db.flush()
own_scenario = str(uuid4())
foreign_scenario = str(uuid4())
db.add(ScenarioRegistryEntry(
scenario_id=own_scenario, scenario_key=f"own-{uuid4().hex[:8]}", name="own",
dashboard_id=81, environment_ids=["env-dev"],
owner_id="acl.reader", owner_username="acl.reader",
lifecycle_status="READY", validation_status="valid",
))
db.add(ScenarioRegistryEntry(
scenario_id=foreign_scenario, scenario_key=f"foreign-{uuid4().hex[:8]}", name="foreign",
dashboard_id=82, environment_ids=["env-dev"],
owner_id="foreign.owner", owner_username="foreign.owner",
lifecycle_status="READY", validation_status="valid",
))
own_schedule = ScenarioSchedule(
scenario_id=own_scenario, environment_id="env-dev", cron_expr="0 7 * * *",
policy_id=own_policy.id,
)
foreign_schedule = ScenarioSchedule(
scenario_id=foreign_scenario, environment_id="env-dev", cron_expr="0 8 * * *",
policy_id=foreign_policy.id,
)
db.add_all([own_schedule, foreign_schedule])
db.commit()
return {
"own_scenario": own_scenario, "foreign_scenario": foreign_scenario,
"own_schedule": own_schedule.id, "foreign_schedule": foreign_schedule.id,
"own_policy": own_policy.id, "foreign_policy": foreign_policy.id,
}
def _seed_human_scenario(factory, owner: str) -> tuple[str, str]:
"""Human-step current revision fixture for the DEF-02 disabled==enabled rejection."""
scenario_id = str(uuid4())
revision_id = str(uuid4())
graph = {
"action_registry_version": ACTION_REGISTRY_VERSION,
"action_registry_hash": action_registry_fingerprint(),
"steps": [{
"logical_step_id": "human",
"tool": "human",
"action": "human_checkpoint",
"action_descriptor": resolve_action_descriptor(
tool="human", action="human_checkpoint",
registry_version=ACTION_REGISTRY_VERSION,
registry_hash=action_registry_fingerprint(),
).snapshot(),
}],
"dependencies": [],
"environment_ids": ["env-dev"],
}
with factory() as db:
db.add(ScenarioRegistryEntry(
scenario_id=scenario_id, scenario_key=f"human-{uuid4().hex[:8]}", name="human fixture",
dashboard_id=83, environment_ids=["env-dev"], owner_id=owner, owner_username=owner,
lifecycle_status="READY", validation_status="valid", current_revision_id=revision_id,
))
db.add(ScenarioRevision(
revision_id=revision_id, scenario_id=scenario_id, content_hash="h" * 64,
graph_snapshot=graph, execution_template_hash="", template_version="v1",
schema_version=1, compatibility_family="default", change_summary={},
created_by=owner, activation_status="current",
))
db.commit()
return scenario_id, revision_id
def _token(subject: str, *, scopes: list[str], principal_type: str = "user"):
return _access_token_context.set(mcp_server.AccessToken(
token=f"token-{uuid4().hex[:8]}", client_id=subject, scopes=scopes,
subject=subject, claims={"principal_type": principal_type},
))
def _unwrap(value):
if isinstance(value, tuple):
return _unwrap(value[1])
if isinstance(value, dict) and set(value) == {"result"}:
return _unwrap(value["result"])
return value
def _plain_server():
"""FastMCP without the RbacFastMCP catalog layer — proves tool-body semantic codes."""
server = FastMCP("automation-body-probe")
register_automation_tools(server)
return server
def _rbac_server():
server = mcp_server._build_probe_server()
server._record = lambda **_: None
return server
# #endregion Test.McpAutomationParity.Fixtures
# #region Test.McpAutomationParity.BodyCodes [C:4] [TYPE Function] [SEMANTICS test,mcp,automation,semantic-codes]
# @ingroup Test.McpAutomationParity
# @BRIEF Tool-body admission codes match the REST semantics: 401 unauthenticated,
# 403 permission_denied, 404 not_found (foreign == missing).
@pytest.mark.asyncio
async def test_read_body_semantic_codes(isolated_db) -> None:
ids = _seed_acl_rows(isolated_db)
_seed_user(isolated_db, "acl.reader", [("scenario:automation", "READ")])
server = _plain_server()
# Anonymous (401 semantic): no token context at all.
with pytest.raises(Exception, match="unauthenticated"):
await server.call_tool("list_scenario_schedules", {})
# Service principal without the mcp:read scope (403 semantic).
token = _token("svc-noscope", scopes=["mcp"], principal_type="service")
try:
with pytest.raises(Exception, match="permission_denied"):
await server.call_tool("list_scenario_schedules", {})
finally:
_access_token_context.reset(token)
# Foreign scenario addressed by a READ holder resolves exactly like a missing one (404).
token = _token("acl.reader", scopes=["mcp"])
try:
foreign_policy = _unwrap(await server.call_tool(
"get_scenario_automation_policy", {"scenario_id": ids["foreign_scenario"]}
))
missing_policy = _unwrap(await server.call_tool(
"get_scenario_automation_policy", {"scenario_id": f"missing-{uuid4()}"}
))
assert foreign_policy == {"status": "not_found", "scenario_id": ids["foreign_scenario"]}
# Foreign and missing projections are indistinguishable apart from the echoed id.
assert {**foreign_policy, "scenario_id": None} == {**missing_policy, "scenario_id": None}
foreign_metrics = _unwrap(await server.call_tool(
"get_scenario_automation_metrics", {"scenario_id": ids["foreign_scenario"]}
))
assert foreign_metrics == {"status": "not_found", "scenario_id": ids["foreign_scenario"]}
finally:
_access_token_context.reset(token)
# #endregion Test.McpAutomationParity.BodyCodes
# #region Test.McpAutomationParity.Acl [C:4] [TYPE Function] [SEMANTICS test,mcp,automation,acl]
# @ingroup Test.McpAutomationParity
# @BRIEF Per-object ACL on reads: owners see own rows only; service principals with scope are
# admitted and see only rows they own; admins see all rows.
@pytest.mark.asyncio
async def test_read_acl_filters_foreign_rows(isolated_db) -> None:
ids = _seed_acl_rows(isolated_db)
_seed_user(isolated_db, "acl.reader", [("scenario:automation", "READ")])
_seed_user(isolated_db, "acl.admin", [], admin=True)
server = _plain_server()
token = _token("acl.reader", scopes=["mcp"])
try:
listed = _unwrap(await server.call_tool("list_scenario_schedules", {}))
schedule_ids = {item["id"] for item in listed}
assert ids["own_schedule"] in schedule_ids
assert ids["foreign_schedule"] not in schedule_ids
# Addressing the foreign scenario directly yields an empty projection (no leak).
foreign_only = _unwrap(await server.call_tool(
"list_scenario_schedules", {"scenario_id": ids["foreign_scenario"]}
))
assert foreign_only == []
own_metrics = _unwrap(await server.call_tool(
"get_scenario_automation_metrics", {"scenario_id": ids["own_scenario"]}
))
assert own_metrics["schedules_total"] == 1
own_policy = _unwrap(await server.call_tool(
"get_scenario_automation_policy", {"scenario_id": ids["own_scenario"]}
))
assert own_policy["id"] == ids["own_policy"]
finally:
_access_token_context.reset(token)
token = _token("svc-automation", scopes=["mcp", "mcp:read"], principal_type="service")
try:
listed = _unwrap(await server.call_tool("list_scenario_schedules", {}))
assert listed == []
finally:
_access_token_context.reset(token)
token = _token("acl.admin", scopes=["mcp"])
try:
listed = _unwrap(await server.call_tool("list_scenario_schedules", {}))
schedule_ids = {item["id"] for item in listed}
assert ids["own_schedule"] in schedule_ids
assert ids["foreign_schedule"] in schedule_ids
finally:
_access_token_context.reset(token)
# #endregion Test.McpAutomationParity.Acl
# #region Test.McpAutomationParity.RbacCatalog [C:4] [TYPE Function] [SEMANTICS test,mcp,automation,catalog,rbac]
# @ingroup Test.McpAutomationParity
# @BRIEF RbacFastMCP catalog parity: scoped service principals pass reads and are denied
# human-only mutations; humans without READ are denied reads.
@pytest.mark.asyncio
async def test_catalog_admission_parity(isolated_db) -> None:
_seed_user(isolated_db, "acl.reader", [("scenario:automation", "READ")])
_seed_user(isolated_db, "acl.manager", [("scenario:automation", "MANAGE")])
server = _rbac_server()
token = _token("svc-automation", scopes=["mcp", "mcp:read"], principal_type="service")
try:
listed = _unwrap(await server.call_tool("list_scenario_schedules", {}))
assert listed == []
with pytest.raises(Exception, match="permission_denied"):
await server.call_tool("upsert_scenario_schedule", {"request": {
"scenario_id": "sc", "idempotency_key": f"k-{uuid4()}", "environment_id": "env-dev",
"revision_id": str(uuid4()), "cron_expr": "0 7 * * *",
}})
finally:
_access_token_context.reset(token)
token = _token("acl.manager", scopes=["mcp"])
try:
with pytest.raises(Exception, match="permission_denied"):
await server.call_tool("list_scenario_schedules", {})
finally:
_access_token_context.reset(token)
# Human-only mutation surface unchanged: a service principal is refused by owner() even
# when invoked below the catalog layer.
plain = _plain_server()
token = _token("svc-automation", scopes=["mcp", "mcp:read"], principal_type="service")
try:
with pytest.raises(Exception, match="human_principal_required"):
await plain.call_tool("upsert_scenario_schedule", {"request": {
"scenario_id": "sc", "idempotency_key": f"k-{uuid4()}", "environment_id": "env-dev",
"revision_id": str(uuid4()), "cron_expr": "0 7 * * *",
}})
finally:
_access_token_context.reset(token)
# #endregion Test.McpAutomationParity.RbacCatalog
# #region Test.McpAutomationParity.DisabledReject [C:4] [TYPE Function] [SEMANTICS test,mcp,automation,eligibility,def02]
# @ingroup Test.McpAutomationParity
# @BRIEF DEF-02 MCP side: a disabled schedule bound to a human-step revision is rejected with
# the identical code as enabled and leaves zero side-effect rows.
# @TEST_INVARIANT ScenarioAutomation.Eligibility.Assert: enabled=False does not bypass the
# revision-bound guard -> VERIFIED_BY: test_disabled_human_schedule_rejected
@pytest.mark.asyncio
async def test_disabled_human_schedule_rejected(isolated_db) -> None:
_seed_user(isolated_db, "acl.manager", [
("scenario:automation", "READ"), ("scenario:automation", "MANAGE"),
])
scenario_id, revision_id = _seed_human_scenario(isolated_db, owner="acl.manager")
server = _rbac_server()
token = _token("acl.manager", scopes=["mcp"])
try:
with isolated_db() as db:
before = db.query(ScenarioSchedule).count()
for enabled in (True, False):
with pytest.raises(Exception, match="AUTOMATION_INELIGIBLE_HUMAN_STEP"):
await server.call_tool("upsert_scenario_schedule", {"request": {
"scenario_id": scenario_id, "idempotency_key": f"disabled-{enabled}-{uuid4()}",
"environment_id": "env-dev", "revision_id": revision_id,
"revision_policy": "pinned", "cron_expr": "0 7 * * *", "enabled": enabled,
}})
with isolated_db() as db:
assert db.query(ScenarioSchedule).count() == before
finally:
_access_token_context.reset(token)
# #endregion Test.McpAutomationParity.DisabledReject
# #endregion Test.McpAutomationParity

View File

@@ -106,6 +106,7 @@ def _published_catalog_snapshot() -> dict:
def _seed_principal(factory: sessionmaker, user_id: str) -> None:
role = Role(name=f"role-{user_id}", is_admin=False, permissions=[
Permission(resource="scenario", action="RUN"), Permission(resource="scenario", action="RUN_PROD"),
Permission(resource="scenario:automation", action="READ"), # DG-2 (046 T019): read grant
Permission(resource="scenario:automation", action="MANAGE"),
Permission(resource="dashboard:testing", action="WRITE"),
Permission(resource="dashboard:testing", action="EXECUTE"), # create_agent_run (T029i)

View File

@@ -395,8 +395,10 @@ def test_mcp_catalog_is_explicit_and_unique() -> None:
"get_scenario_automation_policy",
"get_scenario_automation_metrics",
):
assert _MCP_CATALOG_BY_NAME[name].permission is None
assert _MCP_CATALOG_BY_NAME[name].service_allowed is False
# DG-2 (046 T019): reads require scenario:automation READ and admit scoped service
# principals; per-object ACL is enforced inside the tool bodies.
assert _MCP_CATALOG_BY_NAME[name].permission == ("scenario:automation", "READ")
assert _MCP_CATALOG_BY_NAME[name].service_allowed is True
for name in (
"upsert_scenario_schedule",
"delete_scenario_schedule",

226
docs/mcp-client-setup.md Normal file
View File

@@ -0,0 +1,226 @@
#region Doc.McpClientSetup [C:2] [TYPE ADR] [SEMANTICS mcp,onboarding,runbook,oauth,rbac,scenario,gitea]
@BRIEF Onboarding-гайд для BI-аналитика: как подключить внешнего MCP-агента (kilocode / opencode /
openwebui / другой MCP-клиент) к ss-tools, какие scopes нужны по фазам работы, типовые
цепочки инструментов, PROD-гейт, typed errors и два канала Gitea-конфигурации.
@RELATION DEPENDS_ON -> [McpServer.RbacLayer]
@RELATION DEPENDS_ON -> [ScenarioExecution.PublishedCatalogSource]
@RELATION DEPENDS_ON -> [Api.ScenarioLiveBindings]
@RATIONALE UX-3 (docs/reports/ux-flow-improvement-plan-2026-09-11.md): единственный путь создания
сценария тестирования дашборда — внешний MCP-агент, но документа по подключению не было;
сюда же влез остаток UX-2 (runbook двух каналов Gitea).
# Подключение MCP-агента к ss-tools (гайд для BI-аналитика)
Документ описывает, как подключить внешнего MCP-клиента (агента) к ss-tools и довести работу до
первого smoke-check. Путь «создать сценарий тестирования дашборда» существует **только** через
внешнего агента — в продуктовом UI агентских контролов нет (frontend boundary 2026-09-08).
## 1. Подключение клиента
**Endpoint:** `{ss-tools}/mcp` (например, `http://127.0.0.1:8000/mcp` на локальном стенде).
**Транспорт:** Streamable HTTP.
**OAuth-флоу** (проверено live 2026-09-11, см. `docs/2026-09-11-sales-prod-mcp-replay.md`
§Execution facts): REST login + OAuth discovery → **Dynamic Client Registration** (public client)
→ **PKCE** consent в браузере → access token. Клиент регистрируется сам при первом подключении;
заранее создавать client_id не нужно. Логинитесь своей учётной записью ss-tools — права агента
равны правам вашего пользователя (см. §2).
### Конфиги клиентов (типовые; точные поля зависят от версии клиента)
**kilocode** — в настройках MCP (`mcp_settings.json`):
```json
{
"mcpServers": {
"ss-tools": {
"url": "http://127.0.0.1:8000/mcp",
"transport": "streamable-http"
}
}
}
```
**opencode** — в `opencode.json`:
```json
{
"mcp": {
"ss-tools": {
"type": "remote",
"url": "http://127.0.0.1:8000/mcp",
"oauth": true
}
}
}
```
**openwebui** — Admin Panel → Settings → External Tools: добавить MCP-сервер с URL
`http://127.0.0.1:8000/mcp`, тип подключения Streamable HTTP / OAuth. Версии openwebui отличаются
по названиям полей; ищите «MCP» / «Streamable HTTP» в настройках инструментов.
При первом вызове инструмента клиент откроет браузер для PKCE-consent: подтвердите доступ под
своим пользователем.
### Smoke-check
| Шаг | Ожидание |
|---|---|
| `tools/list` | до 62 tools — каталог фильтруется по вашим RBAC-правам, поэтому реальное число зависит от роли (исторический ориентир: live-replay 2026-09-11 фиксировал 61 tool; каталог растёт только аддитивно) |
| `search_dashboards(environment_id="ss-prod", query="sales")` | находит dashboard ID 11 («Sales Dashboard») — так на live-стенде 2026-09-11 |
Если `tools/list` пуст — у пользователя нет ни одного нужного permission (см. §2).
## 2. Scopes по фазам работы
Права проверяются по каталогу `backend/src/mcp_server/rbac_server.py` (permission объявлен рядом с
каждым tool). Нужные grants выдаёт администратор в RBAC ss-tools.
| Фаза | Permission | Инструменты |
|---|---|---|
| Inspect / search | без спец-scopes (достаточно аутентификации) | `search_dashboards`, `inspect_dashboard_context`, `inspect_scenario`, `validate_scenario`, `generate_draft_pack`, `bootstrap_authoring_scenario` |
| Authoring | `dashboard:testing EXECUTE` + `dashboard:testing WRITE` | `create_agent_run` (EXECUTE), `register_draft_pack` (WRITE) |
| Run (не-PROD) | `scenario RUN` | `start_scenario_run`, `list_checkpoints`, `decide_checkpoint` |
| Approve (PROD) | `scenario RUN_PROD` | `start_scenario_run` на PROD-среде, `list_pending_approvals`, `decide_approval`, `publish_baseline_catalog` |
| Automation | `scenario:automation MANAGE` | `upsert_scenario_schedule`, `delete_scenario_schedule`, `upsert_scenario_trigger_rule`, `upsert_scenario_automation_policy` (list/get — без спец-scopes) |
Для `start_scenario_run` сервер сам выбирает `RUN` vs `RUN_PROD` по environment policy: неизвестная
среда — fail-closed отказ, PROD-среда — требуется `RUN_PROD`.
**Human-only правило.** 47 из 62 инструментов каталога помечены `service_allowed=False` — service
principal (`SERVICE_JWT`-интеграции) не подойдёт: вся сценарная работа, approvals, checkpoints и
publish доступны только живому пользователю с OAuth-токеном. Не пытайтесь подключить агента под
service account.
## 3. Типовые фразы агенту и что он сделает
Цепочки подтверждены live-прогоном `docs/2026-09-11-sales-prod-mcp-replay.md` (E2E-EXT-002,
стенд, dashboard 11 на ss-prod).
| Фраза | Цепочка инструментов под капотом | Результат |
|---|---|---|
| «Создай сценарий для sales на ss-prod» | `search_dashboards` → `inspect_dashboard_context` → `create_agent_run` → `inspect_scenario` → `validate_scenario` → `generate_draft_pack` → `register_draft_pack` → `bootstrap_authoring_scenario` | Сценарий + current revision в реестре; `context_authority=verified` |
| «Запусти» (на ss-prod) | `start_scenario_run` | `pending_approval` — создан durable PROD-гейт; повторный вызов идемпотентен (тот же гейт, не новый ран) |
| «Запусти» (на DEV/PREPROD) | `start_scenario_run` | ран уходит в исполнение сразу, без гейта |
| «Есть ли раны, ждущие моего решения?» | `list_pending_approvals` | список pending PROD-гейтов |
| «Подтверди ран» | `decide_approval` | гейт approved → scheduler исполняет ран |
| «Поставь на cron» | `upsert_scenario_schedule` | расписание; PROD — гейт на каждый due-ран (см. §4) |
Compile/validate детерминированы (без LLM внутри): агент передаёт в `inspect_scenario` capabilities,
возвращённые `inspect_dashboard_context` (`derived_capabilities`), плюс caller-заявки вроде
`screenshot: true`; сервер не даёт caller-заявке понизить верифицируемый факт.
## 4. PROD-гейт и human checkpoint
- **Approve — только человек.** Решение принимается в UI (панель рана в мониторе) или через MCP
`list_pending_approvals` / `decide_approval` (permission `scenario RUN_PROD`, human-only).
Автоматического approve не существует by design.
- **Scheduled PROD = гейт на КАЖДЫЙ due-ран** (by design): cron создаёт ран и durable гейт; без
вашего решения ран не исполняется. Это подтверждено live-цепочкой cron → PROD gate → approval →
execution (`specs/044-dashboard-scenario-execution/prototype/live_scheduled_run.py`).
- **Автономные расписания — только не-PROD** (см. §8).
- **HumanCheckpoint** (шаг с `tool=human`) переводит ран в `waiting_human`; disposition — через UI
или MCP `list_checkpoints` / `decide_checkpoint`. Иммутабельный маппинг исходов (044):
| Кнопка / disposition | Персистентный исход |
|---|---|
| «Подтвердить соответствие» (`confirm`) | `passed` — проверка пройдена, **не** подтверждение дефекта |
| «Проблема не подтверждена» (`false_positive`) | `inconclusive` |
| «Недостаточно данных» (`inconclusive`) | `inconclusive` |
- Сценарий с human-шагом **не может** стоять на расписании: scheduler завершит такой due-ран typed
`AUTOMATION_INELIGIBLE_HUMAN_STEP`.
## 5. Параметры фильтров
`native_filters_key` из URL Superset **эфемерен и не импортируется** (by design): ссылка на
дашборд с применённым фильтром не переносится в сценарий. Фильтры задаются typed-параметром графа
`filter_values` (объявляется в `inspect_scenario` → `parameters`, образец — `_author_b01` в
`specs/044-dashboard-scenario-execution/prototype/live_scheduled_run.py`:
```json
"parameters": {
"filter_values": {"type": "string_list", "required": true, "default": []}
}
```
Просите агента явно: «добавь параметр `filter_values` и проверь с region=South».
С 2026-09-12 (plan UX-1/UX-6) значение параметра доезжает до исполнения: `apply_native_filter`
— read-only browser-действие (filter bar, чекпоинты `filter_applied`/`charts_settled`), и
`filter_values` из параметров запуска биндится в его input при derivation RunnerPlan
(`Ref(name="param.filter_values")` из графа резолвится сервером). Семантика fail-closed:
- параметр задан → фильтр применяется к заданным значениям;
- параметр не задан / пустая строка → шаг работает в режиме наблюдения текущего состояния
(`current_state`);
- пустой список `[]` → тоже режим наблюдения; невалидный тип параметра → typed reject рана
до создания строк (никакой «тихой» подмены фильтра).
UI (RunConfigurationPanel) сегодня передаёт string-строку параметра = одно значение; список
значений — через MCP `start_scenario_run` c `params.filter_values` списком.
## 6. Troubleshooting: typed errors
Fail-closed: неподдержанное/недоступное — всегда typed non-pass, PASS не синтезируется.
| Код | Трактовка | Что делать |
|---|---|---|
| `AUTOMATION_INELIGIBLE_HUMAN_STEP` | У текущей revision есть human-шаг — автоматический (scheduled/triggered) запуск запрещён by design | Запускайте вручную (`trigger=manual`) или пересоберите граф без HumanCheckpoint |
| `*_BINDING_MISSING` (`SUPERSET_/BROWSER_/SCREENSHOT_BINDING_MISSING`) | Для dashboard/environment нет live-binding — ран стартовал без авторизованного слепка | Зарегистрируйте binding через admin API (§8) |
| `*_BINDING_UNAVAILABLE` | Binding есть, но `enabled=false` | Включите: `PATCH /api/scenario-live-bindings/{ref}` (`{"enabled": true}`) |
| `*_BINDING_INVALID` | Слепок binding не прошёл валидацию формы | Перерегистрируйте binding корректным snapshot |
| `*_BINDING_MISMATCH` | Идентичность шага/релиза не совпала с запиненной в binding (dashboard/release/principal drift) | Перерегистрируйте binding под актуальный релиз и actor |
| `DRAFT_PACK_ACCESS_DENIED` | `register_draft_pack` под чужим/несуществующим AgentRun — владение проверяется по principal | Создайте AgentRun сами (`create_agent_run`) и используйте его id; side effects при отказе — ноль |
| `PUBLISH_HEAD_MOVED` | HEAD целевой ветки Gitea сдвинулся с момента approve (CAS publish) | Запросите новый approve и повторите publish |
| `PUBLISH_SOURCE_UNCONFIGURED` | Не заданы `PUBLISHED_CATALOG_GITEA_*` env (§7) | Настройте process env и перезапустите backend |
| `BROWSER_ACTION_NOT_SUPPORTED` | Изолированный browser-транспорт допускает только read-only каталог (`open_dashboard`, `wait_for_state`, `refresh`, `apply_native_filter`) + контрактные SQL-Lab мутации; запрос не из каталога | Действие не входит в versioned ActionRegistry — проверьте `inspect_scenario`/дескриптор шага; терминал рана честный `inconclusive` |
| `BROWSER_SELECTOR_NOT_FOUND` | Filter-bar контрол не найден в текущей версии Superset UI | Fail-closed по дизайну (план UX-1): без retry; уточните `selector_hint` шага или обновите дескриптор |
## 7. Runbook: два канала Gitea-конфигурации
Gitea используется в двух независимых местах — настройка одного **не** включает другой:
| Канал | Где настраивается | Для чего |
|---|---|---|
| Settings → Git (UI, ConfigManager) | Веб-интерфейс ss-tools | Git-плагин: ветки, коммиты, deploy дашбордов |
| `PUBLISHED_CATALOG_GITEA_URL` / `PUBLISHED_CATALOG_GITEA_TOKEN` / `PUBLISHED_CATALOG_REPO` / `PUBLISHED_CATALOG_REF` (+ опционально `PUBLISHED_CATALOG_PATH_TEMPLATE`) | **Только process env** backend-процесса (`backend/src/services/dashboard_testing/execution/published_catalog_source.py:38-43`) | Baseline catalog: чтение published-слепков при старте рана и publish worker |
`@INVARIANT`: токен читается только из process environment, никогда не из БД и не логируется.
На чистом стенде без `PUBLISHED_CATALOG_*` publish завершится typed `PUBLISH_SOURCE_UNCONFIGURED`
(HTTP 503), а baseline-backed старт — fail-closed `BASELINE_NOT_PUBLISHED`.
**Как проверить:** `GET /api/catalog-publications` — список публикаций; smoke publish на DEV-ветку
через `publish_baseline_catalog` (требует `scenario RUN_PROD` + approval). Отсутствие env сразу
даёт typed `PUBLISH_SOURCE_UNCONFIGURED` — это ожидаемый сигнал, а не сбой.
## 8. DEV/PREPROD: автономные расписания («поставил и забыл»)
Без человеческого гейта расписание работает только вне PROD:
1. Администратор регистрирует live-binding на DEV/PREPROD-среду через admin API
`PUT /api/scenario-live-bindings/{binding_ref}`
(`backend/src/api/routes/dashboard_testing/scenario_live_bindings.py`, permission
`admin:settings WRITE`). Snapshot содержит только identity/fingerprints — никаких секретов;
`execution_principal_fingerprint` = `sha256(actor)` — имя пользователя, чьим OAuth-токеном будут
стартовать раны (пример: `sha256("admin")`).
2. Аналитик просит агента: «поставь сценарий X на cron …» → `upsert_scenario_schedule`
(permission `scenario:automation MANAGE`).
3. Каждый due-ран на DEV/PREPROD исполняется до терминала **без** approve. Условия: revision без
human-шагов (иначе `AUTOMATION_INELIGIBLE_HUMAN_STEP`) и валидный enabled binding (иначе
typed `*_BINDING_*`).
На PROD та же конфигурация даёт гейт на каждый due-ран — автономных PROD-расписаний не бывает
by design.
## Источники
- Live-доказательства: `docs/2026-09-11-sales-prod-mcp-replay.md` (E2E-EXT-002).
- План: `docs/reports/ux-flow-improvement-plan-2026-09-11.md` (UX-2, UX-3, UX-6, UX-7, UX-8).
- Каталог tools/permissions: `backend/src/mcp_server/rbac_server.py`.
- Binding admin API: `backend/src/api/routes/dashboard_testing/scenario_live_bindings.py`.
- Publish source env: `backend/src/services/dashboard_testing/execution/published_catalog_source.py`.
- Typed `filter_values`: `specs/044-dashboard-scenario-execution/prototype/live_scheduled_run.py`
(`_author_b01`).
- Disposition labels: `frontend/src/lib/i18n/locales/ru/dashboard-testing.json`,
`specs/044-dashboard-scenario-execution/spec.md` (Field-run Amendment).
#endregion Doc.McpClientSetup

View File

@@ -0,0 +1,325 @@
# План доработок UX flow создания и запусков сценариев тестирования дашборда — 2026-09-11
## Контекст и источники
- Live-доказательства: `docs/2026-09-11-sales-prod-mcp-replay.md` (E2E-EXT-002),
`docs/reports/agentic-runtime-scheduled-happy-path-2026-09-11.md` (046 T018),
`docs/2026-09-07-sales-prod-mcp-run.md` (полевой прогон).
- Спеки: 038 (модель/IR), 042 (реестр), 043 (редактор), 044 (исполнение), 045 (монитор),
046 (автоматизация), 050 (MCP), 036 (гейты).
- UX-аудит 2026-09-11 (сессия): ортогональная оценка flow для BI-аналитика с внешним
MCP-агентом (kilocode / opencode / openwebui).
## Жёсткие ограничения (не нарушать)
1. **Frontend boundary 2026-09-08**: весь агент — только внешний MCP. В продуктовом UI
запрещены agent chat/prompt/proposal/workspace/handoff контролы. Разрешены: ручной
CRUD/редактор, human approval/review, мониторинг, read-only evidence, обычная навигация.
2. **Fail-closed**: никакого синтеза PASS; неподдержанное/недоступное — typed non-pass.
3. **Human-only гейты**: PROD approve и HumanCheckpoint disposition — только human principal.
4. Детерминизм: compile/validate — без LLM внутри; LLM только в bounded AgentEvaluationSpec.
5. Минимальные изменения; каждая доработка закрывается тестами + обновлением спек/traceability.
## Сводная таблица работ
| # | Работа | Приоритет | Трек | Закрывает |
|---|---|---|---|---|
| UX-1 | Read-only browser action `apply_native_filter` | **P0** | backend provider | главный inconclusive live |
| UX-2 | ~~Baseline pin~~ → регрессионная live-приёмка + runbook env-конфига | ✅ почти закрыто | ops | T045/T046-pin CLOSED 2026-09-11 |
| UX-3 | MCP onboarding guide для аналитика | **P0** | docs | discoverability gap |
| UX-4 | Человекочитаемый терминал рана в мониторе | **P0** | frontend | трактовка typed non-pass |
| UX-5 | Навигационный вход «Сценарии» из карточки дашборда | **P0** | frontend | entry point gap |
| UX-6 | Параметризация native-фильтров (значения по умолчанию) | P1 | backend+frontend | фильтры из URL |
| UX-7 | Видимость pending PROD-гейтов (scheduled поток) | P1 | frontend+docs | scheduled-PROD полуавтомат |
| UX-8 | DEV/PREPROD автономные расписания | P1 | ops+docs | «поставил и забыл» |
| UX-9 | Trigger-rules (deploy/release/ETL) | P2 | backend | T003 CLOSED 2026-09-12, T016 event-path closed; T017 open |
| UX-10 | Production-ops гейты 046 T019–T021 + 044 provider backlog | P2 | backend | production GO — детальный план: `docs/reports/ux10-production-ops-plan-2026-09-12.md` |
| UX-11 | ✅ done | Удаление agent-дрейфа из DashboardHeader + все найденные product-входы в `/agent`: header `agentHref`/`scenarioHref` + кнопки «MCP»/«Create scenario», TopNavbar assistant-кнопка (`ROUTES.agent`), dataset-detail «AI»-ссылка, `DashboardDetailModel.scenarioHref`; i18n-orphans удалены (RU+EN). Negative acceptance: 6/6 `DashboardHeader.ux.test.ts` (нет `a[href*="/agent"]`, нет prompt/textarea/proposal-семантики, ноль fetch на `/agent`|`/api/agent*`|`/api/assistant*`), полный suite 214/214, lint 0 errors. Route-файлы `/agent` не тронуты (050-наследие) |
---
## P0
### UX-1. Read-only browser action `apply_native_filter`
**Проблема**: каждый реальный фильтр-чек (B01) завершается typed
`BROWSER_ACTION_NOT_SUPPORTED` — изолированный транспорт допускает только
`open_dashboard`/`wait_for_state`/`refresh` (+ SQL-Lab мутации `row_edit`/`bulk_edit`).
Это единственный дефект, превращающий ценностный ран в `inconclusive`
(`backend/src/services/dashboard_testing/execution/providers/browser_transport.py:31`,
`browser.py:128`).
**Объём**:
1. Дескриптор действия в versioned 038 ActionRegistry: `browser.apply_native_filter`,
класс read-only (не мутация данных — UI-состояние сессии), typed input
`{filter_id|column, values[], wait_state}`, typed output
`{applied: bool, chart_data_observed: bool, evidence_ref}`.
2. Реализация в `browser_transport.py`: после `open_dashboard` применить фильтр через UI
страницы (filter bar), дождаться `wait_for_chart_data` (bounded deadline), checkpoints
`filter_applied`/`charts_settled`; evidence — post-filter capture через существующий
ScreenshotProvider. Без SQL, без записи.
3. Compiler: `inspect_scenario` уже эмитит `apply_native_filter` при
`native_filters=true` — изменений не требует; убедиться, что descriptor fingerprint
новой версии реестра проходит pin валидацию.
4. PROD-допуск: действие остаётся read-only по 044 mutation contract — отдельный
`risk_level` в дескрипторе, без fixture-scope требований.
**Приёмка**:
- `backend/tests/.../providers/test_browser*.py`: contract-suite (unsupported → typed reject
до I/O; supported → checkpoints + receipt + sha256 evidence).
- Live canary B01 (`live_canary_v2.py` / replay): `apply_native_filter` **passed**,
терминал рана `passed` (не синтетический — с durable evidence).
- Обновить: `specs/044-dashboard-scenario-execution/spec.md` (BrowserProvider readiness),
`specs/038-dashboard-scenario-model` (каталог действий), traceability 044/038.
### UX-2. Baseline pin — фактически закрыт; остаток — регрессия + runbook
**Факт (2026-09-11, позже replay-доки)**: git-сервер настроен (Settings → Git), и обе
блокировки сняты в тот же день:
- **050 T045 CLOSED**: `publish_baseline_catalog` live — MCP gate → `decide_approval` →
publication worker → Gitea contents commit с head-CAS и receipts
(`docs/reports/agentic-runtime-live-mcp-publish-t045-2026-09-11.md`).
- **050 T046 CLOSED**: baseline pin — MCP `--baseline` цепочка несёт полный
`runner_plan.baseline_pin` (set/version/digest из published envelope), walker штампует его
в `AgentEvaluation.baseline_pin` (`docs/reports/agentic-runtime-live-canary-v4-baseline-pin-2026-09-11.md`).
**Остаточный gap (операционный, не код)**: publish-source конфигурируется process env
`PUBLISHED_CATALOG_GITEA_URL/TOKEN/REPO/REF` (`published_catalog_source.py:38-43`,
`@INVARIANT`: токен только из env, никогда не из БД) — это **отдельный канал** от
git-настроек в Settings UI. На чистом стенде/деплое без этих переменных получится typed
`PUBLISH_SOURCE_UNCONFIGURED`.
**Объём**:
1. Runbook-абзац в `docs/mcp-client-setup.md` (UX-3): два канала Gitea-конфигурации
(Settings UI для git-плагина; `PUBLISHED_CATALOG_*` env для baseline catalog), где
задавать, как проверить (`GET /api/catalog-publications` + smoke publish на DEV-ветку).
2. Регрессионная live-приёмка: `live_mcp_replay.py --baseline` до терминала на текущем
стенде — зафиксировать evidence-ссылку в 050 traceability (T046 уже `[x]`, это
страховка от дрейфа).
**Приёмка**: runbook опубликован; регрессионный прогон baseline-pinned рана — терминал
с pin, отличным от null.
### UX-3. MCP onboarding guide
**Проблема**: единственный путь создания сценария — внешний агент, но нет документа,
как его подключить (есть только аудиты/полевые отчёты).
**Объём** — новый `docs/mcp-client-setup.md`:
1. Endpoint `{ss-tools}/mcp`, транспорт Streamable HTTP.
2. OAuth: DCR (public client) + PKCE consent; готовые конфиги для kilocode, opencode,
openwebui (copy-paste блоки).
3. Требуемые scopes по фазам: authoring (`dashboard:testing EXECUTE/WRITE`),
run (`scenario RUN`), approve (`scenario RUN_PROD`), automation (`scenario:automation MANAGE`);
human-only правило (service principal бесполезен почти для всего каталога).
4. Smoke-check: `tools/list` → 61 tool; `search_dashboards("ss-prod","sales")`.
5. Типовые фразы агенту («создай сценарий для sales», «запусти», «поставь на cron») и
что агент сделает под капотом (цепочка из replay doc — одна таблица).
6. Troubleshooting: typed errors → человекочитаемая трактовка
(`AUTOMATION_INELIGIBLE_HUMAN_STEP`, `*_BINDING_MISSING`, `DRAFT_PACK_ACCESS_DENIED`).
7. Ссылка из `README.md`.
**Приёмка**: аналитик без контекста репозитория подключает клиент и проходит smoke-check
по документу; review документа вторым человеком.
### UX-4. Человекочитаемый терминал рана
**Проблема**: терминал `inconclusive` + `error_code=BROWSER_ACTION_NOT_SUPPORTED`
машиночитаем, но аналитик без агента не поймёт причину и что делать.
**Объём** (`frontend/src/routes/dashboard-testing/scenarios/[id]/runs/[runId]/+page.svelte`
+ i18n): терминальный баннер маппит typed reason codes на RU/EN объяснения и next step
(«действие фильтра пока не поддержано провайдером — см. план UX-1», «нужен approve»,
«ожидает вашего решения»). Только отображение существующих полей — никакой новой логики.
**Приёмка**: vitest на маппинг кодов; ручная проверка на записанном inconclusive ране;
045 ux_reference дополнен состоянием баннера.
### UX-5. Навигационный вход из карточки дашборда
**Проблема**: из инвентаря дашбордов нет пути в сценарное рабочее место (entry point
только через знание URL). Разрешено boundary: обычная read-only навигация.
**Объём**: ссылка «Сценарии тестирования» в header `dashboards/[id]` →
`/dashboard-testing/scenarios?dashboard_id={id}` (index уже поддерживает фильтр
`dashboard_id` — `scenarios/+page.svelte`). Никаких agent-контролов.
**Приёмка**: route-test на ссылку и применённый фильтр; invariant index-страницы
(«никогда не стартует ран») сохранён.
---
## P1
### UX-6. Параметризация native-фильтров
**Проблема**: `native_filters_key` из URL эфемерен и не импортируется (by design); но
аналитик хочет «проверь с фильтром region=South». Typed `filter_values` параметр в графе
уже существует (см. `_author_b01` в `live_scheduled_run.py`).
**Объём**:
1. Убедиться/доработать, что `start_scenario_run` (MCP) и REST start принимают
`params.filter_values` и runner биндит их в `apply_native_filter` input (зависит от UX-1).
2. RunConfigurationPanel: поле параметров запуска (если отсутствует) — только typed params
графа, без свободного JSON.
3. Docs (в UX-3): явный абзац «почему ссылка с native_filters_key не импортируется и как
задать фильтры параметром».
**Приёмка**: run с `filter_values=["South"]` исполняет apply_native_filter с этими
значениями (provider test + live); 044 spec — ParameterBinding гарантия зафиксирована.
### UX-7. Видимость pending PROD-гейтов
**Проблема**: scheduled-PROD создаёт гейт на каждый due-ран; аналитик не узнает, что ран
ждёт его approve, если не смотрит монитор.
**Объём**:
1. UI: счётчик/бейдж pending approvals в run-center (`scenario_run_center.py` API уже
агрегирует раны; добавить проекцию `pending_approval` в бейдж Stores, как для active tasks).
2. Docs (в UX-3): MCP-петля «спроси агента: есть ли ожидающие approve раны» →
`list_pending_approvals`/`decide_approval`.
3. (Опционально, если дешево) 046 T017: NotificationEvent на gate-created — только если
не расширяет scope; иначе отдельной задачей.
**Приёмка**: scheduled ран → бейдж виден в UI без открытия монитора; vitest; 045/046 docs.
### UX-8. DEV/PREPROD автономные расписания
**Проблема**: «поставил и забыл» возможен только вне PROD; live-биндинги сейчас только на
ss-prod. **Объём (ops+docs)**: зарегистрировать DEV/PREPROD binding через admin API
`PUT /api/scenario-live-bindings/...` (маршрут и тесты существуют:
`backend/src/api/routes/dashboard_testing/scenario_live_bindings.py`,
`test_scenario_live_bindings_api.py`), задокументировать в UX-3 гайд «автономные
расписания без гейта — только не-PROD». **Приёмка**: cron на DEV-среду исполняется до
терминала без approve; запись в runbook.
### UX-9. Trigger-rules (deploy/release/ETL) — 046 остаток
**Объём**: закрыть остаток T003 (event->run, release_create->run, ETL->run,
disabled/mismatch skip, capacity/PROD/dedup gates) в
`backend/tests/services/dashboard_testing/registry/test_scenario_automation_trigger.py`;
T016 (wiring через policy/dedup в 044 start) и T017 (persisted lifecycle notifications).
**Приёмка**: статусы T003/T016/T017 в `specs/046-dashboard-scenario-automation/tasks.md`
по фактам; SCAUTO-FR-001/003/012 traceability закрыты executable evidence.
### UX-10. Production-ops гейты
**Объём**: 046 T019–T021 (parity/ACL reads, crash/replay due-детерминизм, retention
holds + canary SLO) и 044 provider contract backlog (T028–T034, T040–T042b, PREPROD
canary) — вести по существующим спекам; этот план не детализирует, только фиксирует
зависимость production GO от них.
---
## Порядок внедрения и зависимости
```mermaid
graph LR
UX3[UX-3 guide] --> UX8[UX-8 DEV schedules]
UX2[UX-2 runbook+регрессия] --> UX3
UX1[UX-1 filter action] --> UX6[UX-6 filter params]
UX1 --> CANARY[B01 canary passed]
UX4[UX-4 terminal UX] --> UX7[UX-7 pending gates]
UX5[UX-5 nav entry] --> UX3
UX9 --> UX10
CANARY --> UX10
```
1. **Волна 1 (независимые, параллельно)**: UX-3, UX-4, UX-5 — docs/frontend, без backend-риска.
2. **Волна 2**: UX-1 (provider) → live canary; UX-2 (runbook + регрессия) — параллельно, без блокеров.
3. **Волна 3**: UX-6, UX-7, UX-8 — после волн 1–2.
4. **Волна 4**: UX-9, UX-10 — production closure.
## Проверки на каждую волну (по AGENTS.md)
```bash
cd backend && source .venv/bin/activate
python -m pytest -q tests/services/dashboard_testing/registry/test_scenario_*.py \
tests/api/test_scenario_runs_api.py tests/api/test_scenario_automation_api.py
python -m ruff check src/services/dashboard_testing src/api/routes/dashboard_testing
python specs/044-dashboard-scenario-execution/prototype/validate_static.py
cd frontend && npm run test -- --run && npm run lint && npm run build
```
Live-приёмка волн 2–3 — только на поднятом стенде (`./run.sh --skip-install`), с
fail-closed клиентами из `specs/044-dashboard-scenario-execution/prototype/`.
## Риски
- **UX-1 — UI-автоматизация фильтров хрупка**: локаторы filter bar Superset могут
отличаться по версии; митигация — checkpoints + typed failure (`SELECTOR_NOT_FOUND` →
inconclusive, не failed), без ретрая мутирующих предположений.
- **UX-5 — scope creep в agent-контролы**: только ссылка; любое расширение — отдельный review.
- **UX-7 — дублирование с 046 T017**: не строить второй notification-контур; при конфликте
T017 выносится отдельно.
- **UX-2 — конфигурационный дрейф**: `PUBLISHED_CATALOG_*` env вне Settings UI — риск
«на стенде работает, на новом деплое `PUBLISH_SOURCE_UNCONFIGURED`»; митигация — runbook +
smoke publish в регрессии.
---
## Session state — 2026-09-12 (implementation log)
### Завершено
| Работа | Статус | Доказательства |
|---|---|---|
| UX-3 + UX-2(docs) | ✅ done | `docs/mcp-client-setup.md` (213 строк, контракт `Doc.McpClientSetup` в индексе, все факты трассированы, README.md:179 ссылка) |
| UX-4 | ✅ done | `terminal-reasons.ts` + `TerminalReasonBanner.svelte` + wiring; 19/19 vitest, lint 0 errors; amendment §6 в `specs/045-…/ux_reference.md` |
| UX-1 (реализация) | ✅ done, live-pending | `browser_native_filter.py` (read-only `apply_native_filter`: filter-bar UI, `selector_hint` pin, typed `BROWSER_SELECTOR_NOT_FOUND`/`BROWSER_WAIT_STATE_INVALID`/`BROWSER_FILTER_*`, checkpoints `filter_applied`/`charts_settled`); 51+110+427 pytest, ruff, compileall, validate_static; 044/038 спеки+traceability обновлены, registry version НЕ бампается |
| UX-5 | ✅ done | `ROUTES.dashboardTesting.scenarios(id?)` + deep-link init `?dashboard_id=` + ссылка в DashboardHeader; 73/73 vitest, lint = pristine |
### В работе
- **UX-6** (воркер): биндинг `params.filter_values` → apply_native_filter step input через RunnerPlan; обе поверхности старта уже принимают params (проверено: `scenario_inputs.py:66`, `RunConfigurationPanel.svelte:102`).
- **UX-7** (воркер): pending_approval проекция в run-center (новый query param, семантика `waiting_for_me` НЕ меняется — контракт 045) + сумма в бейдже WaitingStore.
### Live-приёмка 2026-09-12 (Verify worker, evidence `/tmp/kilo/live_ux_*.json`)
- **UX-1 LIVE PASSED**: B01 run `e642d3a8…` терминал `passed` — `apply_native_filter` passed
(durable sha256 `c064bd00…`, checkpoints `filter_applied/charts_settled`), `capture_screenshot`
passed (8 артефактов). Guard-фальсифицируемость доказана в обе стороны: без `--expect-passed`
passed-терминал отвергнут (`UNEXPECTED_SYNTHETIC_PASS`), с флагом принят только по durable
evidence. Селекторный тюнинг (1 итерация): prod Superset strip-ит `data-test` — добавлены
стабильные antd-кандидаты + bounded mount-wait дропдауна (синхронизация, не retry).
- **UX-6 LIVE**: `param_binding={filter_values:[…]}` подтверждён в persisted runner_plan (БД);
`--filter-value USA` → терминал `passed`, `applied_mode=values`, `applied_values=["USA"]`.
- **UX-2**: pin-часть приёмки выполнена — `baseline_pin` non-null из published envelope
({ss-prod-visual, v1, digest}). Residual: шаг `compare_to_baseline` → typed
`inconclusive QUEUED_DISPATCH_ERROR` (dispatch-луп упал исключением на сравнении) —
**взято в отдельный fix-воркер** (typed step failure вместо креша диспетчера).
- **UX-8**: DEV-binding `ux8-dev-live-001` зарегистрирован (PUT 200, GET виден, ss-dev/dashboard 11,
principal sha256("admin")); composition активируется при следующем старте backend.
- Replay-клиент: новые флаги `--expect-passed` (fail-closed default сохранён) и `--filter-value`
(UX-6); фикс проекции `step_outcome` (вложенный ключ).
- Информация: в USA-ране `chart_data_observed=false` при passed-чекпоинтах — зафиксировано
в evidence как есть (счётчик rendered-чартов на момент оценки 0).
### UX-9 — 2026-09-12 (trigger-rules, T003 CLOSED)
- 18 dispatch-level тестов над реальным 044 `start_run` boundary: deploy_to_preprod/release_created/
etl_completed → run (pinned/trigger_source/fingerprint), disabled+mismatch skip без durable rows,
capacity gate (шарится между правилами одного события), PROD event → pending_approval+durable gate,
dedup window по completed-раннам, идемпотентная redelivery.
- Два реальных дефекта `trigger.py` починены (оба пойманы failing-first тестами): (A) dedup supply
не включал completed-ранны внутри окна; (B) ран, созданный ранее в том же dispatch, не потреблял
capacity для следующих правил.
- T003 → `[x]` (items в tasks.md), T016 event-path wiring закрыт тестами (live APScheduler-интеграция
и subscriber — остаток, честно), T017 остаётся open (notify()/persist_notification() без callers —
cross-cutting, не форсирован).
- Регрессия: registry 451 passed, automation API 11 passed, ruff чист, verify_after_edit 0/0/0.
### Открытые остатки
- **UX-1 live canary**: ✅ закрыт 2026-09-12 (passed с durable evidence, см. Live-приёмка).
- **UX-2(ops)**: ✅ закрыт 2026-09-12 — root cause: скомпилированный `expected={kind:baseline_ref}` валидировался `NormalizedValue` → pydantic ValidationError → generic except dispatch-лупа → close-all `QUEUED_DISPATCH_ERROR`. Fix: typed `BASELINE_EVIDENCE_UNAVAILABLE` (D11-таксономия) + граница `EXECUTOR_STEP_ERROR` на `dispatch_step` (executor exception больше не валит диспетчер; `_close_queued_dispatch_error` — только инфраструктура). Legacy-тест переписан; 044 Field-run Amendment + traceability. 266+116+15 passed.
- **UX-8(ops)**: binding зарегистрирован; исполнение на DEV — при следующем старте backend.
- **UX-11**: ✅ закрыт 2026-09-12 (см. сводную таблицу выше). Найдены и удалены ВСЕ product-входы
в `/agent` (DashboardHeader, TopNavbar assistant, datasets «AI», мёртвый `DashboardDetailModel.scenarioHref`)
с negative DOM/network acceptance; 044/036 спеки отмечены. Route-файлы `/agent`, `ROUTES.agent`,
`isAgentPage`, stale chat i18n — 050-наследие, отдельная задача.
### Инциденты и решения
- **Git stash в общем дереве запрещён**: параллельный `git stash pop` отклонён git'ом, стэш-снепшот содержал промежуточные версии файлов воркеров. Все 15 файлов проверены на диске (новее стэша, покрыты зелёными прогонами), stash@{0} дропнут. Правило: воркеры не выполняют деструктивных git-операций.
- **Потеря/восстановление browser.py между turn'ами UX-1**: частичные правки потерялись, восстановлены воркером и покрыты зелёным прогоном (51 passed).
- **Pre-existing дефект `Ui.Input`**: `_inputSeq` в instance-script → дублирующиеся DOM id (`id="input-1"` у каждого инстанса) ломают `getByLabelText`. Отдельная задача (вне UX-5); в тестах обходится placeholder-запросом.
- **Число MCP-инструментов**: зафиксировано «до 62» (catalog фильтруется RBAC'ом по правам пользователя; replay-дока зафиксировала 61, T045 — 62).

View File

@@ -0,0 +1,150 @@
# UX-10 / 044 / 046 — промежуточный отчёт для передачи (2026-09-14)
**Назначение**: передать агенту-разработчику контекст незавершённой волны. Отчёт самодостаточен:
что сделано, где лежит, как проверить, что осталось и какие есть долги/оговорки.
**Репозиторий**: `/home/busya/dev/ss-tools`, ветка `master`, HEAD `b4148cfe`
(`fix(automation): scheduled runs execute as schedule-owner principal; MCP upsert signature parity`).
Изменения **не закоммичены** (рабочее дерево грязное) — 72 tracked-файла изменено, 24 untracked.
---
## 1. Что проверено лично (воспроизводимо)
| Проверка | Команда | Результат |
|---|---|---|
| Backend, полный suite | `cd backend && source .venv/bin/activate && python -m pytest -q` | **11796 passed, 264 skipped, 1 xpassed** (575.73s) |
| Frontend, полный suite | `cd frontend && npm run test -- --run` | **3585 passed / 214 файлов** (58.21s) |
| Alembic, реальный PostgreSQL | `alembic upgrade head` | 0023 → `0024_retention_deletions`, 16 колонок, PK+3 индекса+unique, повторный прогон no-op |
| Alembic heads | `alembic heads` | единственная голова `0024_retention_deletions` |
| Lint | `cd backend && python -m ruff check .` | All checks passed |
| MCP parity тест | `pytest -q tests/test_mcp_automation_parity.py` | 4 passed |
**Live-факты проверены в PostgreSQL** (durable), а не в отчётах воркеров:
`SELECT id,status,trigger_source,error_code FROM scenario_runs` → ран `e642d3a8` = `passed`,
шаг `phase-2-B01-apply_native_filter` passed с `sha256=c064bd0008f7a0e978d3ca6841d7dba0ce186755de71cce46c53dae30d92d1cf`,
шаг `capture_screenshot` — 8 артефактов; ран `6de8d0d9` = `inconclusive` c `QUEUED_DISPATCH_ERROR`
(реальный дефект, найденный и впоследствии исправленный).
---
## 2. Инвентарь изменений
### Новые исходники (untracked)
| Файл | Строк | Назначение |
|---|---|---|
| `backend/src/services/dashboard_testing/automation/deletions.py` | 331 | durable deletion pipeline 046 |
| `backend/src/services/dashboard_testing/execution/providers/browser_session.py` | 501 | run-scoped browser session (acquire/reuse/replay/close) |
| `backend/src/services/dashboard_testing/execution/providers/browser_readonly_actions.py` | 534 | 9 read-only действий каталога |
| `backend/src/services/dashboard_testing/execution/providers/browser_native_filter.py` | 275 | `apply_native_filter` + safe checkpoint |
| `backend/src/services/dashboard_testing/execution/providers/browser_admission.py` | 208 | admission split |
| `backend/alembic/versions/0024_retention_deletions.py` | — | миграция retention receipts |
| `frontend/src/lib/components/scenario-run/terminal-reasons.ts` | 84 | typed terminal reasons |
| `frontend/src/lib/components/scenario-run/TerminalReasonBanner.svelte` | 32 | баннер терминальных причин |
| `docs/mcp-client-setup.md` | 226 | UX-4 гайд |
Новые тесты (untracked): `test_browser_limits_cleanup.py`, `test_browser_native_filter.py`,
`test_browser_readonly_actions.py`, `test_browser_session.py`, `test_dispatch_capacity_lifecycle.py`,
`test_due_admission_atomicity.py`, `test_provider_superset.py`,
`test_scenario_compare_to_baseline_typed_failure.py`, `test_scenario_retention_holds.py`,
`test_mcp_automation_parity.py`.
### Изменённые спеки
`specs/036-agent-test-stabilization/spec.md`, `specs/038-.../contracts/browser-actions.md`,
`specs/044-.../{spec,tasks,traceability}.md`, `specs/045-.../{openapi.yaml,data-model,ux_reference}`,
`specs/046-.../{spec,tasks,traceability,contracts/production-operations.md}`.
### Планы
`docs/reports/ux-flow-improvement-plan-2026-09-11.md`,
`docs/reports/ux10-production-ops-plan-2026-09-12.md` (содержит session-state и лог инцидентов).
---
## 3. Статус задач в спеках
- **044**: 36 задач `[x]`. Открытые `[~]`/`[ ]`: T007, T008, T010, T011, T014e/f/h, T017, T018, T019,
T022, T024, T025, **T042** `[~]`, **T042b** `[ ]` (PREPROD canaries), T043.
- **046**: 9 задач `[x]`. Открытые: T004, T005, T006, T007, T008, T009, T010, T011, T013, T013d,
T013e, T014, T015, T016, **T017** `[ ]` (lifecycle notifications), **T021** `[~]` (retention done,
5/15/50-tab canaries open).
---
## 4. Что осталось (по приоритету)
1. **D1 — 044 T042b: PREPROD canaries** (блокирует production GO).
Требуется: живой стенд + binding класса PREPROD/DEV. Готовый инструмент:
`specs/044-dashboard-scenario-execution/prototype/browser_readonly_canary.py`
(режимы `SS_CANARY_MODE`: `actions` (default), `timeout`, `soak`, `reconcile`, `recovery`, `mutation`).
Env: `SS_STAND_URL/USER/PASSWORD`, `SS_STAND_STAGE` (mutation требует `PREPROD`), `SS_STAND_DASHBOARD_ID`.
Evidence писать в `specs/044-.../evidence/browser-provider/` (**не в /tmp** — см. §6).
Креды стенда: в typed AppConfig (шифрованные; читаются `ConfigManager()._load_config()` при
заданном `ENCRYPTION_KEY`). Среды: `ss-dev` (https://ss-dev.bebesh.ru), `ss-prod` (https://ss-prod.bebesh.ru).
2. **D2 — 046 T021: 5/15/50-tab canaries** — cost/load/timeout/cancel vs ops-v1; пороги versioned по
первому измерению. Зависит от D1.
3. **046 T017**: lifecycle notifications (`notify()`/`persist_notification()` пока без callers).
4. **Superset `RESULT_TOO_LARGE` bound** — query-ответы не имеют явного size-bound на границе 044
(browser-действия имеют). Зафиксировано в статусе T042.
5. **Cancel mid-I/O для browser** — cancel закрывает сессию на терминале, но активная операция
прерывается drain-deadline (30s+5s) + reconcile sweep, а не мгновенным provider-cancel.
Осознанный компромисс drain-first модели (задокументирован).
---
## 5. Ключевые точки входа в код
- Провайдер: `backend/src/services/dashboard_testing/execution/providers/browser.py`
- Транспорт Playwright: `.../providers/browser_transport.py` (сессия, лимиты, cleanup)
- Сессия: `.../providers/browser_session.py`
- Admission: `.../providers/browser_admission.py`
- Capacity/lease: `.../execution/capacity.py`, `provider_operations.py`
- Dispatch: `.../execution/dispatch.py`, `dispatch_runs.py`, `cancel_lifecycle.py`
- Retention: `.../automation/retention.py`, `.../automation/deletions.py`
- Миграция: `backend/alembic/versions/0024_retention_deletions.py`
---
## 6. Известные долги и оговорки (важно для честности)
1. **Evidence live-прогонов лежал в `/tmp/kilo/live_ux_*.json` и не сохранился** (/tmp очищен).
Живые раны персистятся в PostgreSQL (`scenario_runs` / `scenario_step_runs`) — это авторитетный
источник, он проверен. НО: для новых прогонов evidence писать в `specs/044/.../evidence/`, не в /tmp.
2. **Изменения не закоммичены.** Перед продолжением стоит зафиксировать (или согласовать с владельцем).
3. **Ранние волны (UX-1..UX-11) делегировались субагентам**; их конверты изначально принимались как
«verified» без независимого прогона. Сейчас подтверждено собственными прогонами (§1), но при
доработке этих участков перепроверяйте, а не доверяйте старым записям.
4. **Миграционный инцидент 2026-09-14**: revision id `0024_scenario_retention_deletions` (33 симв.)
превышал `alembic_version.version_num VARCHAR(32)` → падение `alembic upgrade` на PostgreSQL на
шаге stamp (transactional DDL откатил чисто). Исправлено: id → `0024_retention_deletions`.
Добавлен guard: `backend/tests/test_alembic_migrations.py::test_revision_ids_fit_varchar32`.
**Правило: revision id ≤ 32 символов; миграционную цепочку проверять на PostgreSQL, не на SQLite.**
5. **Стенд**: `./run.sh --skip-install` (backend :8000, frontend :5173). При старте composition
подхватывает live-bindings из БД. `source backend/.env` в шелле опасен — файл содержит
multiline PEM-ключи; извлекать `DATABASE_URL`/`ENCRYPTION_KEY` через `grep` + `cut`.
6. Инструменты субагентов (`deepseek-v4.1-flash`) были недоступны (403/rate-limit) — часть волны
доработана в основном потоке.
---
## 7. Быстрый старт для принимающего агента
```bash
# 1) Проверить состояние
cd /home/busya/dev/ss-tools && git status --short && git log --oneline -3
# 2) Backend (полный) — ~10 мин
cd backend && source .venv/bin/activate && python -m pytest -q
# 3) Frontend — ~1 мин
cd frontend && npm run test -- --run
# 4) Миграции на реальном PostgreSQL
cd backend && export DATABASE_URL="$(grep -m1 '^DATABASE_URL=' .env | cut -d= -f2- | tr -d '"')"
alembic current && alembic heads
# 5) Стенд
cd /home/busya/dev/ss-tools && ./run.sh --skip-install
```
Порядок продолжения: **сначала §6.2 (закоммитить или согласовать)**, затем D1 (044 T042b) → D2
(046 T021) → 046 T017 / RESULT_TOO_LARGE.

View File

@@ -0,0 +1,291 @@
# План UX-10 — Production-ops гейты (046 T019–T021 + 044 provider backlog) — 2026-09-12
Родительский план: `docs/reports/ux-flow-improvement-plan-2026-09-11.md` (UX-10 строка).
Граф: `docs/api/nav/root.map` (gen-1789200620594, 11166 contracts); модульные карты указаны по имени узла.
Норматив: `specs/046-dashboard-scenario-automation/contracts/production-operations.md` (SCAUTO-FR-019,
status: implemented=false, DEF-02/SEC-01 open), `specs/044-dashboard-scenario-execution/tasks.md` Phase 9.
## Current state (верифицировано по графу и коду 2026-09-12)
| Объект | Граф/код | Статус |
|---|---|---|
| ProviderOperations.Service | `ScenarioExecution.ProviderOperations.Service.map`: Open/Complete/LateResponse/Reconcile (write-once, CAS) + ReconcileWorker + scheduler sweep | receipts есть; **cancel-контракт (stopped/completed/unknown) отсутствует** |
| CapacityManager | `ScenarioExecution.CapacityManager.Service.map` (9c/7f); `claim_capacity` вызывается ТОЛЬКО browser-провайдером (`providers/browser.py:40`); `dispatch_runs.py` capacity не клеймит | admission есть на провайдере; **нет dispatcher-level интеграции для всех провайдеров** |
| BrowserProvider | `ScenarioExecution.BrowserProvider.map` (23c/17f, +NativeFilter 8c/7f от UX-1); транспорт — контекст per-execution (`browser_transport.py:14`) | 4 read-only + 2 mutation действия; **каталог T034 неполный; контекст-модель ≠ T034** |
| Automation reads | REST: `_USER` auth (anon→401), eligibility-guard на create/update (141/175/224 — DEF-02 частично); MCP: `owner()` human-only на всём | **нет READ-scope, нет object ACL, foreign→rows (не 404), паритет REST/MCP reads не доказан** |
| Due-event admission | `_scheduled_idempotency_key` (second-deterministic), start_run idempotency unique | **baseline selection pin не входит в due identity; run+gate+outbox — не одна транзакция** |
| Retention | `ScenarioAutomation.Retention.map`: tier_limits/run_retention/run_retention_days (pure) | **нет holds (baseline/case), нет deletion receipts (bytes-verify), нет canary vs ops-v1 SLO** |
| Readiness payload | `ProviderPreflight.Snapshot` — provider_loop/evidence_storage/browser/screenshot ready | **нет provider/version + capability fingerprint + redacted diagnostics** |
| Superset contract | live health browser/screenshot есть (canary v1/v2); Superset composition есть | **нет Superset-specific contract test + live query** |
## Decision gates (статус)
- **DG-1 (T034) — RESOLVED 2026-09-12 (решение пользователя).** Цель verbatim: «native filters
должны сохраняться — в этом вся суть проверки дашборда — зафиксировать эталонное состояние».
Принято: **run-scoped session** — контекст на (run, lease) живёт через все browser-шаги рана;
полный native-filter-state входит в browser-safe checkpoint после каждого шага; crash → replay
checkpoint (мёртвый контекст не оживляется; без checkpoint → inconclusive). Lease — per-run.
Спека: 044 `## Decision Amendment — BrowserProvider run-scoped session & filter-state continuity
(2026-09-12, DG-1)`.
- **DG-2 (T019) — RESOLVED 2026-09-12 (архитекторский дефолт, revisable).** Reads: аутентифицированный
principal (любой тип) + `scenario:automation READ` + per-object ACL; mutations — human-only
(как сейчас в MCP). @RATIONALE зафиксировать в `scenario_automation.py`/`tools_automation.py`
при реализации A1.
## Волны и работы
### Wave A — 046 correctness (backend, без live-зависимостей) — параллельно 2 воркера
**A1. T019 — automation reads ACL + parity** (046)
- Scope: 6 reads (`schedules, trigger-rules, policies, notifications, metrics, retention`):
REST — добавить `has_permission("scenario:automation","READ")` + per-object ACL
(scenario owner / dashboard ACL / environment visibility; foreign→404, pagination без утечки totals);
MCP — тот же VIEW/READ + ACL (заменить `owner()` на reads, сохранить human-only на mutations);
anonymous→401, lacking→403, foreign→404 — на обеих поверхностях идентично.
- Disabled==enabled rejection parity: guard уже зовётся (REST 141/175, MCP `_revision`) —
acceptance-тестами hardcoded: disabled HumanStep schedule reject идентично REST/MCP
(semantic code + side-effect counts: 0 rows).
- Files: `backend/src/api/routes/dashboard_testing/scenario_automation.py`,
`backend/src/mcp_server/tools_automation.py`, `automation/eligibility.py`,
тесты `backend/tests/api/test_scenario_automation_api.py` (Rbac section расширить) +
`backend/tests/test_mcp_automation*` (найти/создать parity-тест).
- Acceptance: SEC-01/DEF-02 закрыты; production-operations.md status note; 046 traceability строка.
**A2. T020 — atomic due-event admission** (046)
- Scope: (1) baseline selection pin входит в due identity — при due admission резолвить
BaselineSelectionPin и штамповать в run target/idempotency-проекцию (сейчас только revision);
(2) crash/replay/concurrent due events → ровно один pinned run + один gate + один outbox:
unique-convergence тесты (два конкурентных callback'а, crash между run и gate, replay
same-second — UX: уже есть second-key; добавить baseline pin и outbox); (3) transition record +
queue + notification outbox — одна транзакция (проверить `_record_terminal_side_effects` границы).
- Files: `backend/src/core/scheduler.py` (ExecuteScheduledScenario),
`execution/start_run.py`, `execution/runner.py`, `execution/terminal_effects.py`,
тесты `test_scheduled_scenario_start.py` (расширить) + новый crash/replay модуль.
- Acceptance: T020 закрыт; approval сохраняется до dispatcher CAS (уже так — доказать тестом).
### Wave B — provider operations + capacity — 1 воркер, 2–3 раунда (общие файлы)
**B1. 044 T030 — operation-aware cancellation + receipts coverage**
- Scope: `cancel(operation_id)` контракт на ProviderOperations.Service: durable cancel request +
provider acknowledge `stopped|completed|unknown`; `unknown` → reconciliation_required, блок
retry/PASS (SCEX-FR-020, уже есть ReconcileWorker — подключить); покрытие receipts всеми
провайдерами (browser — есть; screenshot/superset/sql_evidence — проверить/добавить Open/Complete).
- Files: `provider_operations.py`, `providers/{browser,screenshot,superset}.py`,
`test_provider_contract.py` + per-provider тесты.
**B2. 044 T032 — dispatcher admission через shared CapacityManager**
- Scope: dispatcher (`dispatch_runs.py`) admission через `ExecutionCapacityManager` для всех
провайдеров (не только browser): claim перед первым I/O шага, heartbeat на длинных шагах,
expiry/release/reconcile на всех терминалах; pinned defaults (2 DEV/PREPROD, 1 PROD — уже
в ProviderRuntime ADR); `capacity_blocked` → queued (не I/O). Lease lifecycle durable+idempotent.
- Files: `execution/capacity.py`, `execution/dispatch_runs.py`, `execution/provider_runtime.py`,
`execution/walker.py`; тесты capacity/starvation (`test_provider_contract.py`, новый starvation-case:
«каждый eligible workload прогрессирует в пределах 2 completed older leases при равном приоритете»).
### Wave C — BrowserProvider production contract (long pole) — 1 long-lived воркер, 4–6 раундов
**C1. 044 T034 — per-run контекст + полный каталог (после DG-1)**
- Раунд 1: per-run контекст на shared loop: контекст/page привязаны к (run, lease), живут между
шагами рана; browser-safe checkpoint declaration + replay для recovery; crash → контекст не
оживляется (recover через checkpoint; без него — inconclusive, текущая семантика).
- Раунд 2: read-only каталог до полного: `navigate_tab, inspect_filter_state, apply_table_filter,
extract_table, scroll_to, inspect_columns, click, select_rows, download` (типизированные
input/output по 038 дескрипторам, checkpoints, evidence, fail-closed typed misses как UX-1).
- Раунд 3: limits enforcement: 120s context/auth, 30s action, 3 pages, 25 MiB downloads,
10 MiB screenshots (typed `*_LIMIT_EXCEEDED`, никогда truncate→PASS).
- Раунд 4: mutation contract: `row_edit/bulk_edit` — fixture lease + cleanup policy,
precondition hash (уже частично в MutationSQL), PROD reject.
- Files: `providers/browser.py`, `providers/browser_transport.py`, `providers/browser_native_filter.py`,
`providers/browser_mutation.py`, новый `providers/browser_session.py`; тесты `test_provider_browser.py`.
- Acceptance: каталог T034 полный; contract score пересчитан в 044 spec (цель 90/100 при runtime).
**C2. 044 T040 — readiness payload** (параллельно C1, отдельный воркер)
- `/api/ready`: provider/version + capability fingerprint + redacted dependency diagnostics
(без credentials/cookies/SQL/capture bytes — SCEX-FR-022). Files: `providers/preflight.py`,
`app.py` lifespan, `test_provider_preflight.py`.
**C3. 044 T042 — Superset contract** (после C2, live на стенде)
- Superset-specific contract test (offline) + один live Superset query через binding
(ss-prod-d11-live-001): typed receipt + principal/RLS fingerprint в evidence.
- Files: `test_provider_superset.py` (новый), live-прогон (Verify-воркер на стенде).
### Wave D — canary evidence (нужна живая среда) — Verify-воркер на стенде
**D1. 044 T042b — BrowserProvider PREPROD canaries**
- PREPROD/DEV: `ss-dev` binding `ux8-dev-live-001` уже зарегистрирован (UX-8; composition
активируется при рестарте backend — стенд рестартовать через `./run.sh --skip-install`).
- Canaries: read-only action canary (по новому каталогу C1), forced timeout/cleanup canary,
safe-checkpoint reconstruction trace, readiness payload capture; owned receipt →
`specs/044-dashboard-scenario-execution/evidence/browser-provider/`.
- GO-criteria: все acceptance vectors + 1 owned evidence receipt.
**D2. 046 T021 — retention holds + receipts + ops-v1 canaries**
- Retention: holds evaluation (approved-baseline refs, active operations/cases) +
deletion pipeline mark→eligible→delete bytes→verify absent→tombstone/audit с receipt
(bytes removed proof); retry same deletion ID идемпотентно; analytics min window гарантирован.
- Files: `automation/retention.py` (расширить), `models/scenario_automation.py` (deletion rows),
API `GET /retention` (проекция receipts).
- Canary suite vs ops-v1 (production-operations.md §15–16): 5/15/50 tabs, warm/cold browser,
401/403/429/5xx, disk failure, crash/restart, duplicate trigger, timeout/cancel/unknown
reconciliation; latency/bytes/cost recording (pricing version, null if unreported);
пороги ops-v1 versioned (dispatch p95<100ms/100 steps и т.д.).
- Отдельный подпункт: cost cap + estimator для evaluation/dispatch (ops-v1 §16: missing estimator
блокирует cost-limited dispatch) — оценить объём по `execution/agent_evaluation.py`.
### Wave E — closure (архитектор)
- production-operations.md status → per-row evidence; 044 tasks T030/T032/T034/T040/T042/T042b
статусы; 046 tasks T019–T021 статусы; полный suite; `make docs-nav` + axiom rebuild; closure report.
## Граф зависимостей
```mermaid
graph TD
DG1[DG-1 контекст-модель] --> C1
DG2[DG-2 reads политика] --> A1
A1 --> E[Wave E closure]
A2 --> E
B1 --> B2 --> E
C1 --> D1
C2 --> C3 --> E
C1 --> D2
D1 --> E
D2 --> E
```
A1 ∥ A2 ∥ B1 ∥ C1 ∥ C2 — пять непересекающихся веток; D — после C1/C2; E — последней.
## Риски
- **T034 контекст-модель**: per-run контекст увеличивает blast radius crash (мёртвый контекст на
середине рана) — митигация: browser-safe checkpoint после каждого шага + fail-closed recover
(уже специфицировано SCEX-FR-007); отдельный тест crash-mid-run.
- **PREPROD/DEV среда**: ss-dev binding зарегистрирован, но composition не проверена исполнением —
D-волна может застрять на ops (миграция: рестарт стенда + smoke `open_dashboard` на ss-dev до canary).
- **ops-v1 thresholds — proposed**: цифры не утверждены (implemented=false); канарейки измеряют
факт, а не «проход/провал» — пороги фиксируются versioned в первом измерении, не выдумываются.
- **Cost estimator**: если `agent_evaluation.py` не имеет pricing hooks, D2 расширяется — scope
зафиксировать после чтения модуля (оценка в раунде 1 D2).
- **Reads ACL regression**: сужение reads может сломать существующие UI-панели (AutomationPanel
читает list endpoints) — frontend smoke `automation_page.ux.test.ts` в A1 acceptance.
## Проверки на каждую волну
```bash
cd backend && source .venv/bin/activate
python -m pytest -q <wave-scoped paths> -v
python -m pytest -q tests/services/dashboard_testing/ tests/api/ # волновая регрессия
python -m ruff check src
cd ../frontend && npm run test -- --run && npm run lint # при затронутом UI
```
Closure: полный `python -m pytest -q` (база 11652) + `make docs-nav` + axiom rebuild full.
---
## Session state — 2026-09-12 (implementation log)
### Завершено
| Работа | Статус | Доказательства |
|---|---|---|
| DG-1/DG-2 | ✅ RESOLVED | 044 spec Decision Amendment (run-scoped session, filter-state continuity — цель пользователя verbatim); план §Decision gates |
| A1 (T019) | ✅ done | READ grant + per-object ACL на 6 reads REST/MCP, foreign==missing, catalog 2.3.0; 15+7+68 passed; SEC-01/DEF-02 CLOSED в production-operations.md |
| A2 (T020) | ✅ done | Structural verdict + 8 falsifiable тестов (concurrent/crash/replay/outbox-atomicity); T020→[x], traceability DONE; production-код не тронут |
| C2 (T040) | ✅ done | readiness payload: provider/version (038 registry) + capabilities fingerprint + redacted deps; T040→[x], SC-010 closed; 736+ passed |
| C1 r1 (T034) | ✅ round 1 | `browser_session.py` (run-scoped session, checkpoint, replay, leak-guards), `browser_admission.py`, `browser_receipt.py`; `BROWSER_CHECKPOINT_MISSING` fail-closed; 65+558 passed; T034→[~] |
| D2-офлайн (T021) | ✅ offline | `apply_holds` (baseline fail-closed, active ops, analytics window только run_metadata — баг найден и починен), `deletions.py` pipeline + модель + миграция 0024, sweep job `scenario_retention_sweep` (03:45 UTC), ACL-проекция в GET /retention; 11+16 passed; T021→[~] |
### Инциденты
- **Биллинг субагентов**: C1-r1 и D2 воркеры упали с «Payment Required» после применения кода.
Доработка вручную в основном потоке: фикс импорта `seed_trace_id` (cot_logger), баг
analytics-window hold (применялся ко всем items вместо run_metadata — убивал 30d тиры),
миграция 0024, wiring receipts в GET /retention (D2↔A1 шов), регистрация sweep job.
- **Правило**: при недоступности делегирования архитектор дорабатывает сам, минимальным диффом.
### Открытые остатки (по приоритету)
1. **C1 r2–4** (T034): полный каталог действий (navigate_tab, inspect_filter_state, apply_table_filter,
extract_table, scroll_to, inspect_columns, click, select_rows, download), limits enforcement,
mutation fixture lease+cleanup, dispatch finalizer `close(run_id)` на терминале (B-wave hook).
2. **B1/B2** (T030/T032): operation-aware cancel (stopped/completed/unknown) + dispatcher-level
admission через shared CapacityManager для всех провайдеров.
3. **C3** (T042): Superset-specific contract test + live query через binding.
4. **D1 + D2-canaries**: PREPROD canaries (ss-dev binding `ux8-dev-live-001`, рестарт backend) +
5/15/50-tab ops-v1 измерения (пороги versioned по первому измерению, не выдумываются).
5. **T017** (046): lifecycle notifications (`notify()` без callers) — cross-cutting.
### Closure gate
- Backend: **11693 passed / 0 failed** (база 11652 → +41), ruff 0 errors, alembic single head 0024.
- Frontend: 3585 passed (последнее фронтовое изменение — UX-11, проверено его полным прогоном).
- Semantic index: rebuild full после этой записи.
## Session state — 2026-09-13 (waves B/C closure)
### Завершено в этой волне
| Работа | Статус | Доказательства |
|---|---|---|
| C1 r1 (T034) | ✅ | run-scoped session + checkpoint + replay; 65+558 passed |
| C1 r2 (T034) | ✅ | 9 read-only действий (`browser_readonly_actions.py`), download 25 MiB; 59+1480 passed |
| C1 r3/r4 (T034) | ✅ | limits: 120s context/auth (asyncio.wait_for), 3 pages bound (BROWSER_PAGE_LOST); mutation cleanup: closed vocabulary {restore_fixture, retain}, revert из pre-image, `fixture_restored` checkpoint, `BROWSER_MUTATION_CLEANUP_FAILED` → receipt reconciliation_required; 7 новых тестов |
| B1 (T030) | ✅ | `cancel_provider_operation` (stopped/completed/unknown, idempotent, terminal immutable); receipts на screenshot/superset; T030→[x], SC-004→[x] |
| B2 (T032) | ✅ | dispatcher tick → reconcile_expired_leases; heartbeat перед submit (оба провайдера, fail-closed); terminal/cancel финализатор `close_run_sessions` (weak manager registry); run-level lease отклонён (@REJECTED в capacity.py); 6 новых тестов; SC-008→closed |
| C3 (T042 offline) | ✅ | `test_provider_superset.py` (14): typed success + receipts + identity fail-closed matrix + external taxonomy + evidence ownership + duplicate-open + sql_evidence parity |
### Инциденты
- Делегирование сломано инфраструктурно (модель субагентов deepseek-v4.1-flash деактивирована на
аккаунте, 403 + rate-limit) — волна B/C доработана в основном потоке по правилу «минимальный
дифф архитектором». Качество не пострадало: найден и починен реальный баг (analytics-window hold),
TDD-красная фаза зафиксирована в трёх местах (FK-фикстуры, reload-фейк, descriptor inputs).
### Closure gate 2026-09-13
- Backend: **11795 passed / 0 failed** (база 11652 → +143 за UX-10 волны A/B/C), ruff 0 errors,
alembic single head 0024.
- T034 [x], T030 [x], T032 [x], T040 [x]; 046 T019 [x], T020 [x], T021 [~].
### Открытые остатки (по приоритету)
1. **D1** (044 T042b): PREPROD canaries — ss-dev binding `ux8-dev-live-001` зарегистрирован;
требуется рестарт backend (composition на старте) + прогон (read-only canary, forced
timeout/cleanup, safe-checkpoint reconstruction trace, readiness payload) + owned receipt в
`specs/044/evidence/browser-provider/`.
2. **D2-canaries** (046 T021): 5/15/50-tab cost/load/timeout/cancel измерения vs ops-v1
(пороги versioned по первому измерению); cost estimator для agent_evaluation — оценка в раунде 1.
3. **Superset RESULT_TOO_LARGE bound**: query responses не имеют явного size-bound на 044 границе
(browser actions имеют); зафиксировано в T042 status — отдельная задача.
4. **T017** (046): lifecycle notifications (`notify()`/`persist_notification()` без callers).
5. **Browser cancel mid-I/O**: cancel закрывает сессию на терминале, но активная операция
прерывается через drain deadline (30s+5s) + reconcile sweep, не через мгновенный
`cancel_provider_operation` из provider I/O — осознанный компромисс drain-first модели.
## Incident — 2026-09-14: migration revision id превысил VARCHAR(32)
**Симптом**: `./run.sh` упал на database preflight — `alembic upgrade head` на PostgreSQL:
`psycopg2.errors.StringDataRightTruncation: value too long for type character varying(32)` при
`UPDATE alembic_version SET version_num='0024_scenario_retention_deletions'` (33 символа).
**Root cause**: длина revision id никем не проверялась; `alembic_version.version_num` — VARCHAR(32).
SQLite-тесты этот класс бага не ловят (в тестах таблицы создаются через `Base.metadata.create_all`,
alembic-цепочка не прогоняется на PG).
**Последствия**: PostgreSQL transactional DDL откатил всё чисто — `alembic_version` осталась на
0023, таблица НЕ создана, частичного состояния нет (проверено запросом к information_schema).
**Fix**:
1. Revision переименован → `0024_retention_deletions` (25 символов); файл
`alembic/versions/0024_retention_deletions.py`; ссылки в `specs/046-…/tasks.md` обновлены.
2. Статический guard: `tests/test_alembic_migrations.py::test_revision_ids_fit_varchar32` —
проверяет ВСЕ revision id ≤ 32 без PostgreSQL (ловит класс бага до деплоя).
3. Live-верификация на PostgreSQL: `alembic upgrade head` → 0024 head; таблица 16 колонок,
PK + 3 индекса + unique (target_type, target_id); повторный upgrade — no-op (идемпотентно);
`alembic heads` — единственная голова.
4. Тесты: 32 passed (alembic + smoke chain + retention holds + automation API), ruff clean.
**Правило (зафиксировано)**: новые alembic-миграции — revision id ≤ 32 символов; guard-тест это
принуждает. Реальная проверка миграционной цепочки только на PostgreSQL (AGENTS.md: «SQLite не
заменяет проверку production migration chain») — инцидент это подтвердил.

View File

@@ -1,10 +1,10 @@
<!-- #region Layout.TopNavbar [C:4] [TYPE Component] [SEMANTICS navbar, search, activity, user-menu, environment] -->
<!-- @ingroup Layout -->
<!-- @BRIEF Unified top navigation bar with Logo, global search, environment selector, activity indicator, user menu, and assistant toggle. -->
<!-- @BRIEF Unified top navigation bar with Logo, global search, environment selector, activity indicator, and user menu. -->
<!-- @LAYER UI -->
<!-- @PRE Auth token initialized via root +layout.svelte. Environment context store loaded. -->
<!-- @POST Search state delegated to TopNavbar.Model. User menu, environment, and activity controls synchronized with store state. -->
<!-- @SIDE_EFFECT Navigates to /agent on assistant click. Triggers window.location.href for logout. Opens task drawer. Calls setSelectedEnvironment. -->
<!-- @SIDE_EFFECT Triggers window.location.href for logout. Opens task drawer. Calls setSelectedEnvironment. -->
<!-- @DATA_CONTRACT SearchQuery → TopNavbar.Model.triggerSearch | EnvironmentChange → setSelectedEnvironment | ActivityClick → openDrawerForTask -->
<!-- @RELATION BINDS_TO -> [EXT:frontend:activityStore] -->
<!-- @RELATION BINDS_TO -> [EXT:frontend:authStore] -->
@@ -12,7 +12,6 @@
<!-- @RELATION BINDS_TO -> [EXT:frontend:sidebarStore] -->
<!-- @RELATION BINDS_TO -> [EXT:frontend:taskDrawerStore] -->
<!-- @RELATION BINDS_TO -> [TopNavbar.Model] -->
<!-- @RELATION DISPATCHES -> [EXT:frontend:agentHandoffRoute] -->
<!-- @RELATION DEPENDS_ON -> [Ui.Icon] -->
<!-- @RELATION DEPENDS_ON -> [Ui.EnvironmentStatsWidget] -->
<!-- @UX_STATE Idle -> Navbar showing current state with all controls. -->
@@ -23,12 +22,14 @@
<!-- @UX_RECOVERY Click outside closes dropdowns. -->
<!-- @UX_TEST SearchFocused -> {focus: search input, expected: focused style class applied}. -->
<!-- @UX_TEST ActivityClick -> {click: activity button, expected: task drawer opens}. -->
<!-- @RATIONALE Search logic extracted into TopNavbar.Model (reactive screen model) to satisfy INV_7 (module < 400 LOC) and ATTN_4 (sliding-window visibility). The original 605-line inline architecture scattered search state across 7 reactive atoms and 8 functions — the model could not see the full contract under CSA/HCA compression. Model-first extraction puts dense #region anchor at line 1 for maximum first-line density. -->
<!-- @RATIONALE UX-11 (2026-09-12): the assistant/MCP handoff button (handleAssistantClick →
ROUTES.agent) was removed — it was a global agent workspace/start entry on every page,
prohibited by the frontend boundary 2026-09-08 (044/036/046 refresh); agent interaction
is external MCP only. -->
<!-- @REJECTED Search logic extracted into TopNavbar.Model (reactive screen model) to satisfy INV_7 (module < 400 LOC) and ATTN_4 (sliding-window visibility). The original 605-line inline architecture scattered search state across 7 reactive atoms and 8 functions — the model could not see the full contract under CSA/HCA compression. Model-first extraction puts dense #region anchor at line 1 for maximum first-line density. -->
<!-- @REJECTED Full decomposition into UserMenuModel + EnvironmentModel + ActivityModel rejected — each concern is ≤ 3 $derived values with simple toggle logic. Per Model Decomposition Gate, inline $state for local UI without cross-widget invariants is the correct call. Inline architecture for those three concerns was rejected (JSDoc duplicates in <script> removed — INV_4: metadata before code, contiguously after opening anchor). -->
<script lang="ts">
import { onMount } from "svelte";
import { goto } from "$app/navigation";
import { page } from "$app/stores";
import { ROUTES } from "$lib/routes";
import { activityStore } from "$lib/stores/activity.svelte.js";
import {
@@ -97,7 +98,7 @@
window.location.href = ROUTES.login();
}
// ── Activity / Assistant actions ───────────────────────────────
// ── Activity actions ───────────────────────────────────────────
function handleActivityClick() {
const runningTask = recentTasks.find((t: any) => t.status === "RUNNING");
if (runningTask) {
@@ -109,14 +110,6 @@
}
}
function handleAssistantClick() {
const query = [
`route=${encodeURIComponent($page.url.pathname)}`,
globalSelectedEnvId ? `envId=${encodeURIComponent(globalSelectedEnvId)}` : "",
].filter(Boolean).join("&");
goto(ROUTES.agent(query));
}
// ── Document click (delegates to model for search, keeps user menu local) ──
function handleDocumentClick(event: MouseEvent) {
const { closeSearch } = model.handleDocumentClick(event);
@@ -265,17 +258,6 @@
</div>
{/if}
<!-- Assistant (MCP handoff entry) -->
<button
class="flex items-center gap-1.5 rounded-lg border border-border-strong bg-surface-card px-3 py-1.5 text-xs font-medium text-text transition hover:bg-primary hover:text-white hover:border-primary"
onclick={handleAssistantClick}
aria-label={$t.assistant?.mcp_entry_tooltip || $t.assistant?.open}
title={$t.assistant?.mcp_entry_tooltip || $t.assistant?.title}
>
<Icon name="aiAssistant" size={16} />
<span>{$t.assistant?.mcp_entry_label || $t.assistant.assistant}</span>
</button>
<!-- Activity Indicator -->
<div
class="relative cursor-pointer rounded-lg p-2 text-text-muted transition-colors hover:bg-surface-muted"

View File

@@ -0,0 +1,32 @@
<!-- #region ScenarioRunMonitor.Component.TerminalBanner [C:3] [TYPE Component] [SEMANTICS scenario,run,monitor,terminal,ux] -->
<!-- @ingroup ScenarioRunMonitor -->
<!-- @BRIEF Human-readable terminal banner: maps typed reason/error codes to RU/EN explanation + next step (UX-4). -->
<!-- @RELATION DEPENDS_ON -> [ScenarioRunMonitor.TerminalReasons] -->
<!-- @RELATION CALLED_BY -> [ScenarioRun.Route.Detail] -->
<!-- @UX_STATE info -> primary tokens; warning -> warning tokens; never destructive; role="status". -->
<script lang="ts">
import { t } from "$lib/i18n/index.svelte.js";
import type { TerminalReason } from "./terminal-reasons";
let { reason }: { reason: TerminalReason } = $props();
const dt = $derived($t.dashboard_testing ?? {});
</script>
<section
role="status"
class="rounded-lg border p-4 {reason.severity === 'info' ? 'border-primary/40 bg-primary-light' : 'border-warning/40 bg-warning-light'}"
aria-label={dt[reason.titleKey]}
>
<h2 class="font-semibold text-text">{dt[reason.titleKey]}</h2>
<p class="mt-1 text-sm text-text-muted">
{dt[reason.bodyKey]}
{#if reason.kind === "unknown"}
<code class="ml-1 rounded bg-surface-card px-1 py-0.5 text-xs">{reason.code}</code>
{/if}
</p>
{#if reason.nextKey}
<p class="mt-1 text-sm text-text">{dt[reason.nextKey]}</p>
{/if}
</section>
<!-- #endregion ScenarioRunMonitor.Component.TerminalBanner -->

View File

@@ -0,0 +1,83 @@
// #region Test.ScenarioRunMonitor.TerminalReasons [C:2] [TYPE Module] [SEMANTICS test,scenario,run,terminal,ux]
// @BRIEF Unit tests for the terminal reason code mapper (UX-4): exact codes, binding family, fallback, selector.
// @RELATION BINDS_TO -> [ScenarioRunMonitor.TerminalReasons]
// @TEST_EDGE: each known code; *_BINDING_* family by suffix; unknown code -> fallback verbatim; null/empty -> null; selector precedence.
import { describe, expect, it } from "vitest";
import { resolveTerminalReason, terminalCodeFor } from "../terminal-reasons";
import type { ScenarioExecutionResult, ScenarioRun } from "$lib/types/scenario-run";
describe("resolveTerminalReason", () => {
it("maps BROWSER_ACTION_NOT_SUPPORTED to an info banner with the UX-1 next step", () => {
const reason = resolveTerminalReason("BROWSER_ACTION_NOT_SUPPORTED");
expect(reason).not.toBeNull();
expect(reason?.kind).toBe("browser_action_unsupported");
expect(reason?.severity).toBe("info");
expect(reason?.code).toBe("BROWSER_ACTION_NOT_SUPPORTED");
expect(reason?.titleKey).toBe("terminal_browser_title");
expect(reason?.nextKey).toBe("terminal_browser_next");
});
it("maps AUTOMATION_INELIGIBLE_HUMAN_STEP to the manual-only info banner", () => {
const reason = resolveTerminalReason("AUTOMATION_INELIGIBLE_HUMAN_STEP");
expect(reason?.kind).toBe("human_step");
expect(reason?.severity).toBe("info");
expect(reason?.titleKey).toBe("terminal_human_step_title");
expect(reason?.nextKey).toBe("terminal_human_step_next");
});
it.each([
"SUPERSET_BINDING_MISSING",
"SUPERSET_BINDING_UNAVAILABLE",
"BROWSER_BINDING_INVALID",
"LIVE_COMPOSITION_BINDING_MISMATCH",
])("maps the binding family code %s to the live-binding warning banner", (code) => {
const reason = resolveTerminalReason(code);
expect(reason?.kind).toBe("live_binding");
expect(reason?.severity).toBe("warning");
expect(reason?.code).toBe(code);
expect(reason?.titleKey).toBe("terminal_binding_title");
expect(reason?.nextKey).toBe("terminal_binding_next");
});
it("falls back to the unknown banner with the code verbatim", () => {
const reason = resolveTerminalReason("SOME_UNLISTED_CODE");
expect(reason?.kind).toBe("unknown");
expect(reason?.severity).toBe("warning");
expect(reason?.code).toBe("SOME_UNLISTED_CODE");
expect(reason?.titleKey).toBe("terminal_unknown_title");
expect(reason?.nextKey).toBeUndefined();
});
it("does not treat a non-binding suffix as the binding family", () => {
expect(resolveTerminalReason("SUPERSET_QUERY_EXECUTION_ERROR")?.kind).toBe("unknown");
});
it.each([null, undefined, ""])("returns null for %s so the banner stays hidden", (code) => {
expect(resolveTerminalReason(code)).toBeNull();
});
});
describe("terminalCodeFor", () => {
const run = { id: "run-1", error_code: "BROWSER_ACTION_NOT_SUPPORTED" } as ScenarioRun;
const result = {
run_id: "run-1",
failures: [
{ logical_step_id: "s1", status: "inconclusive", error_code: null, attempt: 1 },
{ logical_step_id: "s2", status: "failed", error_code: "SUPERSET_BINDING_MISSING", attempt: 1 },
],
} as unknown as ScenarioExecutionResult;
it("prefers the run-level error_code", () => {
expect(terminalCodeFor(run, result)).toBe("BROWSER_ACTION_NOT_SUPPORTED");
});
it("falls back to the first failed-step error_code", () => {
expect(terminalCodeFor({ ...run, error_code: null }, result)).toBe("SUPERSET_BINDING_MISSING");
});
it("returns null without a run or without any code", () => {
expect(terminalCodeFor(null, result)).toBeNull();
expect(terminalCodeFor({ ...run, error_code: null }, null)).toBeNull();
});
});
// #endregion Test.ScenarioRunMonitor.TerminalReasons

View File

@@ -0,0 +1,84 @@
// #region ScenarioRunMonitor.TerminalReasons [C:2] [TYPE Module] [SEMANTICS scenario,run,monitor,terminal,ux]
// @defgroup ScenarioRunMonitor Run monitor screen model and components.
// @BRIEF Maps typed terminal reason/error codes (044 execution) to RU/EN explanation + next step for the run banner.
// @RELATION CALLED_BY -> [ScenarioRun.Route.Detail]
// @INVARIANT Display-only: resolves from existing run/result fields; never triggers requests or new logic.
import type { ScenarioExecutionResult, ScenarioRun } from "$lib/types/scenario-run";
export type TerminalReasonKind = "browser_action_unsupported" | "live_binding" | "human_step" | "unknown";
export interface TerminalReason {
kind: TerminalReasonKind;
severity: "info" | "warning";
/** Raw typed code as emitted by the backend (shown verbatim for the fallback kind). */
code: string;
/** i18n keys inside the dashboard_testing namespace. */
titleKey: string;
bodyKey: string;
/** Absent for the fallback kind (no actionable next step beyond the raw code). */
nextKey?: string;
}
// #region ScenarioRunMonitor.TerminalReasons.Table [C:1] [TYPE Block] [SEMANTICS scenario,run,terminal,codes]
// @ingroup ScenarioRunMonitor
// @BRIEF Extensible constant table: exact typed code -> banner descriptor i18n keys.
const TERMINAL_REASON_TABLE: Record<string, Omit<TerminalReason, "code">> = {
BROWSER_ACTION_NOT_SUPPORTED: {
kind: "browser_action_unsupported",
severity: "info",
titleKey: "terminal_browser_title",
bodyKey: "terminal_browser_body",
nextKey: "terminal_browser_next",
},
AUTOMATION_INELIGIBLE_HUMAN_STEP: {
kind: "human_step",
severity: "info",
titleKey: "terminal_human_step_title",
bodyKey: "terminal_human_step_body",
nextKey: "terminal_human_step_next",
},
};
/** Live-binding family (live_binding.py / live_composition.py): <TOOL>_BINDING_(MISSING|UNAVAILABLE|INVALID|MISMATCH). */
const BINDING_FAMILY = /^[A-Z0-9_]+_BINDING_(MISSING|UNAVAILABLE|INVALID|MISMATCH)$/;
// #endregion ScenarioRunMonitor.TerminalReasons.Table
// #region ScenarioRunMonitor.TerminalReasons.Resolve [C:2] [TYPE Function] [SEMANTICS scenario,run,terminal,mapping]
// @ingroup ScenarioRunMonitor
// @BRIEF Resolve a typed terminal code to a banner descriptor; null input hides the banner.
// @TEST_EDGE: exact known codes; *_BINDING_* family by suffix; unknown code -> fallback with code verbatim; null/empty -> null.
export function resolveTerminalReason(code: string | null | undefined): TerminalReason | null {
if (!code) return null;
const exact = TERMINAL_REASON_TABLE[code];
if (exact) return { ...exact, code };
if (BINDING_FAMILY.test(code)) {
return {
kind: "live_binding",
severity: "warning",
code,
titleKey: "terminal_binding_title",
bodyKey: "terminal_binding_body",
nextKey: "terminal_binding_next",
};
}
return {
kind: "unknown",
severity: "warning",
code,
titleKey: "terminal_unknown_title",
bodyKey: "terminal_unknown_body",
};
}
// #endregion ScenarioRunMonitor.TerminalReasons.Resolve
// #region ScenarioRunMonitor.TerminalReasons.Select [C:2] [TYPE Function] [SEMANTICS scenario,run,terminal,selector]
// @ingroup ScenarioRunMonitor
// @BRIEF Pick the display code for the terminal banner from already-loaded run/result fields.
// @TEST_EDGE: run.error_code wins; else first failed-step error_code from result.failures; else null.
export function terminalCodeFor(run: ScenarioRun | null, result: ScenarioExecutionResult | null): string | null {
if (!run) return null;
if (run.error_code) return run.error_code;
return result?.failures?.find((failure) => failure.error_code)?.error_code ?? null;
}
// #endregion ScenarioRunMonitor.TerminalReasons.Select
// #endregion ScenarioRunMonitor.TerminalReasons

View File

@@ -1,6 +1,4 @@
{
"title": "AI Assistant",
"open": "Open assistant",
"close": "Close assistant",
"send": "Send",
"input_placeholder": "Type a command...",
@@ -17,7 +15,6 @@
"sample_command_migration": "run migration from dev to prod for dashboard 42",
"sample_command_status": "check task status task-123",
"you": "You",
"assistant": "Assistant",
"task_id": "task_id",
"open_task_drawer": "Open Task Drawer",
"thinking": "Thinking",
@@ -173,8 +170,6 @@
"delete_dialog_title": "Delete conversation",
"diff_for_dashboard": "Diff for dashboard {id}:\n\n{diff}",
"diff_empty_for_dashboard": "Diff for dashboard {id} is empty.",
"ask_ai_dashboard": "Ask AI about the dashboard",
"ask_ai_dataset": "Ask AI about the dataset",
"llm_auto_retry_fallback": "Auto-retry in {seconds}s…",
"attach_file": "Attach file",
"stop_generation": "Stop",
@@ -196,8 +191,5 @@
"handoff_step1": "Install an MCP client — for example Claude Desktop or Cursor.",
"handoff_step2": "Add a server using the URL above (transport: streamable HTTP).",
"handoff_step3": "Use the registered OAuth client to complete its PKCE authorization request while signed in.",
"handoff_docs_hint": "Details: docs/INSTALL.md → the «MCP client» section.",
"mcp_entry_label": "MCP assistant",
"mcp_entry_short": "MCP",
"mcp_entry_tooltip": "Instructions to connect the AI assistant (MCP)"
"handoff_docs_hint": "Details: docs/INSTALL.md → the «MCP client» section."
}

View File

@@ -1,5 +1,5 @@
{
"create_scenario": "Create test scenario",
"view_scenarios": "Test scenarios",
"scenario_mode_title": "Dashboard Test Scenario Agent",
"stages": {
"context": "Context",
@@ -178,6 +178,17 @@
"monitor_cancel": "Cancel run",
"monitor_state_title": "State",
"monitor_state_hint": "Elapsed — events update the timeline.",
"terminal_browser_title": "Browser action is not supported by the provider yet",
"terminal_browser_body": "The browser provider does not support this action yet, so no result was synthesized.",
"terminal_browser_next": "Known gap — see plan UX-1 (read-only apply_native_filter).",
"terminal_binding_title": "Live provider is not configured for this environment",
"terminal_binding_body": "The live-provider binding is missing, unavailable, invalid, or mismatched for the run environment.",
"terminal_binding_next": "Contact the administrator (scenario-live-bindings).",
"terminal_human_step_title": "Scenario contains a human step",
"terminal_human_step_body": "Automated launch is not eligible: the scenario contains a human step.",
"terminal_human_step_next": "Launch the scenario manually and complete the analyst checkpoint.",
"terminal_unknown_title": "Terminal state without a synthesized result",
"terminal_unknown_body": "Typed non-pass — no result synthesized. Code:",
"automation_aria": "Scenario automation",
"automation_title": "Scenario automation",
"automation_subtitle": "Schedule/trigger management for scenario runs (not the 009 cron scheduler, not 037 verification runs).",

View File

@@ -1,6 +1,4 @@
{
"title": "AI Ассистент",
"open": "Открыть ассистента",
"close": "Закрыть ассистента",
"send": "Отправить",
"input_placeholder": "Введите команду...",
@@ -17,7 +15,6 @@
"sample_command_migration": "запусти миграцию с dev на prod для дашборда 42",
"sample_command_status": "проверь статус задачи task-123",
"you": "Вы",
"assistant": "Ассистент",
"task_id": "task_id",
"open_task_drawer": "Открыть Task Drawer",
"thinking": "Думаю",
@@ -173,8 +170,6 @@
"delete_dialog_title": "Удалить диалог",
"diff_for_dashboard": "Diff для дашборда {id}:\n\n{diff}",
"diff_empty_for_dashboard": "Diff для дашборда {id} пуст.",
"ask_ai_dashboard": "Спросить AI о дашборде",
"ask_ai_dataset": "Спросить AI о датасете",
"llm_auto_retry_fallback": "Auto-retry через {seconds}с…",
"attach_file": "Прикрепить файл",
"stop_generation": "Остановить",
@@ -196,8 +191,5 @@
"handoff_step1": "Установите MCP-клиент — например, Claude Desktop или Cursor.",
"handoff_step2": "Добавьте сервер по URL выше (транспорт: streamable HTTP).",
"handoff_step3": "Войдите в Superset Tools и завершите PKCE-запрос авторизации, который откроет зарегистрированный OAuth-клиент.",
"handoff_docs_hint": "Подробности: docs/INSTALL.md → раздел «MCP клиент».",
"mcp_entry_label": "MCP-ассистент",
"mcp_entry_short": "MCP",
"mcp_entry_tooltip": "Инструкция по подключению AI-ассистента (MCP)"
"handoff_docs_hint": "Подробности: docs/INSTALL.md → раздел «MCP клиент»."
}

View File

@@ -1,5 +1,5 @@
{
"create_scenario": "Создать сценарий тестирования",
"view_scenarios": "Сценарии тестирования",
"scenario_mode_title": "Сценарий тестирования дашборда",
"stages": {
"context": "Контекст",
@@ -178,6 +178,17 @@
"monitor_cancel": "Отменить запуск",
"monitor_state_title": "Состояние",
"monitor_state_hint": "Elapsed — события обновляют таймлайн.",
"terminal_browser_title": "Действие браузера пока не поддержано провайдером",
"terminal_browser_body": "Провайдер браузера пока не поддерживает это действие, поэтому результат не синтезирован.",
"terminal_browser_next": "Известный gap — см. план UX-1 (read-only apply_native_filter).",
"terminal_binding_title": "Live-провайдер не настроен для этого окружения",
"terminal_binding_body": "Привязка live-провайдера отсутствует, недоступна, некорректна или не совпала для окружения запуска.",
"terminal_binding_next": "Обратитесь к администратору (scenario-live-bindings).",
"terminal_human_step_title": "Сценарий содержит ручной шаг",
"terminal_human_step_body": "Автоматический запуск недопустим: сценарий содержит ручной шаг.",
"terminal_human_step_next": "Запустите сценарий вручную и пройдите чекпоинт аналитика.",
"terminal_unknown_title": "Терминальное состояние без синтеза результата",
"terminal_unknown_body": "Typed non-pass — результат не синтезирован. Код:",
"automation_aria": "Автоматизация сценариев",
"automation_title": "Автоматизация сценариев",
"automation_subtitle": "Schedule/trigger management для scenario runs (не cron-планировщик 009 и не verification run 037).",

View File

@@ -19,7 +19,7 @@
// @RELATION DEPENDS_ON -> [GitStatusModel]
// @RELATION CALLS -> [Frontend.CotLogger]
// @RATIONALE Dashboards.DetailModel centralizes all dashboard-detail state (metadata, thumbnail, task history, validation history, git) into one screen-level model following the model-first pattern. It uses GitStatusModel as a sub-model for git operations — keeping git state management reusable across detail and hub pages. Thumbnail blob URL lifecycle (createObjectURL / revokeObjectURL) is managed within the model to prevent memory leaks.
// @REJECTED Separate models per tab (TaskHistoryModel, ThumbnailModel, GitStatusModel standalone) rejected — detail data is tightly coupled (loadDashboardPage() loads all in parallel, and tab switches are instant since data is preloaded). Inline $state in +page.svelte rejected — model-first allows verifying invariants (thumbnail release, env context fallback, parallel data loading) in vitest without a DOM.
// @REJECTED Separate models per tab (TaskHistoryModel, ThumbnailModel, GitStatusModel standalone) rejected — detail data is tightly coupled (loadDashboardPage() loads all in parallel, and tab switches are instant since data is preloaded). Inline $state in +page.svelte rejected — model-first allows verifying invariants (thumbnail release, env context fallback, parallel data loading) in vitest without a DOM. UX-11 (2026-09-12): the scenarioHref derived route to /agent (contextVersion=2 + build_dashboard_test_scenario intent, 039 T007) was removed with its last consumer (DashboardHeader MCP/Create-scenario links) — agent workspace/start entry from product surfaces is prohibited by the frontend boundary 2026-09-08; reviving it requires a fresh boundary contract update.
import { goto } from '$app/navigation';
import { ROUTES } from '$lib/routes';
@@ -123,21 +123,6 @@ export class DashboardDetailModel {
gitDashboardRef = $derived(this.dashboard?.slug ? String(this.dashboard.slug).trim() : '');
resolvedDashboardId = $derived(this.dashboard?.id ?? (/^\d+$/.test(String(this.dashboardRef || '')) ? Number(this.dashboardRef) : null));
/** 039 T007: scenario entry href carrying contextVersion=2 + scenario intent (AGUI-FR-002). */
scenarioHref = $derived.by(() => {
const objectId = String(this.resolvedDashboardId ?? this.dashboardRef ?? "");
const params = new URLSearchParams({
objectType: "dashboard",
objectId,
objectName: this.dashboard?.title ?? "",
envId: this.envId || "",
route: `/dashboards/${objectId}`,
contextVersion: "2",
intent: "build_dashboard_test_scenario",
});
return `/agent?${params.toString()}`;
});
// ── Actions ───────────────────────────────────────────────────
async loadDashboardPage(): Promise<void> {

View File

@@ -106,8 +106,13 @@ export const ROUTES = {
// ── Dashboard Testing (042 registry / 044 runs / 045 center / 046 automation / 047 analytics) ──
dashboardTesting: {
/** 042 scenario registry hub (list/detail entry point). */
scenarios: () => '/dashboard-testing/scenarios',
/** 042 scenario registry hub (list/detail entry point). Pass dashboardId to deep-link
* the registry pre-filtered to one dashboard (UX-5 read-only nav entry from the
* dashboard card); omit it for the unfiltered hub. */
scenarios: (dashboardId?: string | number) =>
dashboardId !== undefined && dashboardId !== null && String(dashboardId) !== ''
? `/dashboard-testing/scenarios?dashboard_id=${encodeURIComponent(String(dashboardId))}`
: '/dashboard-testing/scenarios',
/** Scenario detail surface: launch configuration, revisions, registry facts (D2). */
scenarioDetail: (id: string) =>
`/dashboard-testing/scenarios/${encodeURIComponent(id)}`,

View File

@@ -1,9 +1,12 @@
// #region Test.ScenarioRuns [C:3] [TYPE Module] [SEMANTICS test,scenario,run,store,waiting,badge]
// @BRIEF Unit tests for the waiting-for-me scenario-run count store — initial state, refresh,
// dedup, fail-closed auth disabling, error retry, and subscriber contracts.
// #region Test.ScenarioRuns [C:3] [TYPE Module] [SEMANTICS test,scenario,run,store,waiting,pending,badge]
// @BRIEF Unit tests for the waiting-for-me + pending-approval scenario-run count store — initial
// state, combined refresh sum, dedup, fail-closed auth disabling, per-projection failure
// leave-last-value, error retry, and subscriber contracts (plan UX-7).
// @RELATION DEPENDS_ON -> [Stores.ScenarioRuns.WaitingStore]
// @TEST_EDGE: external_fail -> fetchApi rejects (network); store keeps last count and stays enabled
// @TEST_EDGE: invalid_type -> fetchApi resolves without a numeric total; count falls back to 0
// @TEST_EDGE: partial_fail -> one projection rejects non-auth; the other updates, failed keeps last value
// @TEST_EDGE: partial_auth_fail -> 401/403 on either projection disables both polls (fail-closed)
import { describe, it, expect, beforeEach, vi } from 'vitest';
vi.mock('$lib/api', () => ({
@@ -13,10 +16,27 @@ vi.mock('$lib/api', () => ({
}));
const WAITING_URL = '/scenario-runs?waiting_for_me=true&page_size=1';
const PENDING_URL = '/scenario-runs?pending_approval=true&page_size=1';
type MockResult = { total: number } | { status: number; message: string } | Error | string;
/** Queue URL-aware responses: `waiting` answers waiting_for_me, `pending` answers pending_approval. */
async function mockFetchTotals(waiting: MockResult, pending: MockResult = waiting): Promise<void> {
const { api } = await import('$lib/api');
const impl = (url: string): Promise<unknown> => {
const result = url.includes('pending_approval=true') ? pending : waiting;
if (result instanceof Error || typeof result === 'string' || 'status' in result) {
return Promise.reject(result);
}
return Promise.resolve(result);
};
vi.mocked(api.fetchApi).mockImplementation(impl as never);
}
describe('scenarioRuns waiting-count store', () => {
beforeEach(() => {
beforeEach(async () => {
vi.resetModules();
await mockFetchTotals({ total: 0 });
});
it('has correct initial state', async () => {
@@ -26,85 +46,108 @@ describe('scenarioRuns waiting-count store', () => {
expect(scenarioWaitingCount.disabled).toBe(false);
});
it('refresh() fetches the waiting total and updates the count', async () => {
it('refresh() fetches both totals and updates the combined count', async () => {
const { api } = await import('$lib/api');
vi.mocked(api.fetchApi).mockClear();
vi.mocked(api.fetchApi).mockResolvedValue({ total: 4, items: [{}] });
await mockFetchTotals({ total: 4 }, { total: 2 });
const { scenarioWaitingCount } = await import('../scenarioRuns.svelte.js');
const result = await scenarioWaitingCount.refresh();
expect(api.fetchApi).toHaveBeenCalledWith(WAITING_URL);
expect(result).toBe(4);
expect(scenarioWaitingCount.current).toBe(4);
const urls = vi.mocked(api.fetchApi).mock.calls.map((c) => c[0]);
expect(urls).toContain(WAITING_URL);
expect(urls).toContain(PENDING_URL);
expect(result).toBe(6);
expect(scenarioWaitingCount.current).toBe(6);
});
it('refresh() falls back to 0 when total is missing or non-numeric', async () => {
const { api } = await import('$lib/api');
vi.mocked(api.fetchApi).mockClear();
vi.mocked(api.fetchApi).mockResolvedValue({});
it('refresh() falls back to 0 when a total is missing or non-numeric', async () => {
const { scenarioWaitingCount } = await import('../scenarioRuns.svelte.js');
await scenarioWaitingCount.refresh();
expect(scenarioWaitingCount.current).toBe(0);
});
it('refresh() dedups concurrent calls into one request', async () => {
it('refresh() dedups concurrent calls into one combined request pair', async () => {
const { api } = await import('$lib/api');
vi.mocked(api.fetchApi).mockClear();
let resolvePromise!: (v: { total: number }) => void;
vi.mocked(api.fetchApi).mockReturnValue(new Promise((r) => { resolvePromise = r; }) as never);
const gate = new Promise<{ total: number }>((r) => { resolvePromise = r; });
vi.mocked(api.fetchApi).mockImplementation(() => gate as never);
const { scenarioWaitingCount } = await import('../scenarioRuns.svelte.js');
const p1 = scenarioWaitingCount.refresh();
const p2 = scenarioWaitingCount.refresh();
expect(api.fetchApi).toHaveBeenCalledTimes(1);
expect(api.fetchApi).toHaveBeenCalledTimes(2); // one per bounded projection
resolvePromise({ total: 2 });
expect(await p1).toBe(2);
expect(await p2).toBe(2);
expect(await p1).toBe(4); // 2 + 2
expect(await p2).toBe(4);
});
it('401 disables further polling', async () => {
const { api } = await import('$lib/api');
vi.mocked(api.fetchApi).mockClear();
vi.mocked(api.fetchApi).mockRejectedValue({ status: 401, message: 'Unauthorized' });
await mockFetchTotals({ status: 401, message: 'Unauthorized' }, { status: 401, message: 'Unauthorized' });
const { scenarioWaitingCount } = await import('../scenarioRuns.svelte.js');
await scenarioWaitingCount.refresh();
expect(scenarioWaitingCount.disabled).toBe(true);
await scenarioWaitingCount.refresh();
expect(api.fetchApi).toHaveBeenCalledTimes(1);
});
it('403 disables further polling', async () => {
const { api } = await import('$lib/api');
vi.mocked(api.fetchApi).mockClear();
vi.mocked(api.fetchApi).mockRejectedValue({ status: 403, message: 'Forbidden' });
const { scenarioWaitingCount } = await import('../scenarioRuns.svelte.js');
await scenarioWaitingCount.refresh();
expect(scenarioWaitingCount.disabled).toBe(true);
});
it('generic error keeps the store enabled (retriable)', async () => {
const { api } = await import('$lib/api');
vi.mocked(api.fetchApi).mockClear();
vi.mocked(api.fetchApi).mockRejectedValue(new Error('Network error'));
const { scenarioWaitingCount } = await import('../scenarioRuns.svelte.js');
await scenarioWaitingCount.refresh();
expect(scenarioWaitingCount.disabled).toBe(false);
await scenarioWaitingCount.refresh();
expect(api.fetchApi).toHaveBeenCalledTimes(2);
});
it('error without .message string is handled', async () => {
it('401 on only one projection still disables both polls (fail-closed)', async () => {
const { api } = await import('$lib/api');
vi.mocked(api.fetchApi).mockClear();
vi.mocked(api.fetchApi).mockRejectedValue('boom');
await mockFetchTotals({ total: 4 }, { status: 401, message: 'Unauthorized' });
const { scenarioWaitingCount } = await import('../scenarioRuns.svelte.js');
const result = await scenarioWaitingCount.refresh();
expect(scenarioWaitingCount.disabled).toBe(true);
// successful projection keeps its authoritative value; failed one keeps last (0)
expect(result).toBe(4);
await scenarioWaitingCount.refresh();
expect(api.fetchApi).toHaveBeenCalledTimes(2);
});
it('403 disables further polling', async () => {
await mockFetchTotals({ status: 403, message: 'Forbidden' });
const { scenarioWaitingCount } = await import('../scenarioRuns.svelte.js');
await scenarioWaitingCount.refresh();
expect(scenarioWaitingCount.disabled).toBe(true);
});
it('generic error keeps the store enabled (retriable) and never fabricates counts', async () => {
const { api } = await import('$lib/api');
vi.mocked(api.fetchApi).mockClear();
await mockFetchTotals(new Error('Network error'));
const { scenarioWaitingCount } = await import('../scenarioRuns.svelte.js');
const result = await scenarioWaitingCount.refresh();
expect(result).toBe(0);
expect(scenarioWaitingCount.disabled).toBe(false);
await scenarioWaitingCount.refresh();
expect(api.fetchApi).toHaveBeenCalledTimes(4);
});
it('non-auth failure leaves the last value of the failed projection only', async () => {
await mockFetchTotals({ total: 3 }, { total: 1 });
const { scenarioWaitingCount } = await import('../scenarioRuns.svelte.js');
await scenarioWaitingCount.refresh();
expect(scenarioWaitingCount.current).toBe(4);
await mockFetchTotals({ total: 5 }, new Error('Network error'));
const result = await scenarioWaitingCount.refresh();
expect(result).toBe(6); // waiting updated to 5, pending keeps last value 1
});
it('error without .message string is handled', async () => {
vi.mocked((await import('$lib/api')).api.fetchApi).mockClear();
await mockFetchTotals('boom');
const { scenarioWaitingCount } = await import('../scenarioRuns.svelte.js');
const result = await scenarioWaitingCount.refresh();
@@ -113,9 +156,7 @@ describe('scenarioRuns waiting-count store', () => {
});
it('subscribe fires immediately and after refresh; unsubscribe stops it', async () => {
const { api } = await import('$lib/api');
vi.mocked(api.fetchApi).mockClear();
vi.mocked(api.fetchApi).mockResolvedValue({ total: 5 });
await mockFetchTotals({ total: 3 }, { total: 2 });
const { scenarioWaitingCount } = await import('../scenarioRuns.svelte.js');
const spy = vi.fn();
@@ -128,7 +169,7 @@ describe('scenarioRuns waiting-count store', () => {
spy.mockClear();
unsubscribe();
vi.mocked(api.fetchApi).mockResolvedValue({ total: 6 });
await mockFetchTotals({ total: 4 }, { total: 3 });
await scenarioWaitingCount.refresh();
expect(spy).not.toHaveBeenCalled();
});
@@ -136,11 +177,12 @@ describe('scenarioRuns waiting-count store', () => {
it('refreshScenarioWaiting is the named poll alias', async () => {
const { api } = await import('$lib/api');
vi.mocked(api.fetchApi).mockClear();
vi.mocked(api.fetchApi).mockResolvedValue({ total: 1 });
const { refreshScenarioWaiting } = await import('../scenarioRuns.svelte.js');
await refreshScenarioWaiting();
expect(api.fetchApi).toHaveBeenCalledWith(WAITING_URL);
const urls = vi.mocked(api.fetchApi).mock.calls.map((c) => c[0]);
expect(urls).toContain(WAITING_URL);
expect(urls).toContain(PENDING_URL);
});
});
// #endregion Test.ScenarioRuns

View File

@@ -1,21 +1,30 @@
// #region Stores.ScenarioRuns.WaitingStore [C:3] [TYPE Store] [SEMANTICS scenario,run,store,waiting,badge,sidebar]
// #region Stores.ScenarioRuns.WaitingStore [C:3] [TYPE Store] [SEMANTICS scenario,run,store,waiting,pending,badge,sidebar]
// @ingroup Stores
// @BRIEF Waiting-for-me scenario-run count for the sidebar approval badge (045 RUNMON-FR-009):
// makes the human checkpoint loop discoverable instead of a page the analyst must guess.
// @BRIEF Waiting-for-me + pending-approval scenario-run counts for the sidebar decision badge
// (045 RUNMON-FR-009, plan UX-7): makes both human checkpoints and PROD-approval gates
// discoverable instead of pages the analyst must guess.
// @LAYER UI
// @RELATION DEPENDS_ON -> [Api.ApiModule.FetchApi]
// @UX_STATE Idle -> waiting count 0, badge hidden.
// @UX_STATE Ready -> waiting count N>0, sidebar "Тестирование дашбордов" badge shows N.
// @UX_STATE Disabled -> run center not accessible (401/403) -> polling stops, count stays 0.
// @INVARIANT refresh() reads only the authoritative server total (waiting_for_me=true); it never
// derives the count client-side, mirroring GlobalRunCenterModel's server-narrowing rule.
// @INVARIANT A non-2xx authority answer never fabricates a count — it leaves the last value and,
// for 401/403, disables further polling (bounded, fail-closed).
// @RATIONALE Mirrors Stores.Health.HealthStore: module-level $state atoms + getters + subscriber set
// keeps Svelte 5 reactivity for the Sidebar $derived badge while allowing an interval poll.
// The count request uses page_size=1 because only `total` is consumed (bounded payload).
// @REJECTED Polling the full run list to count client-side rejected — the run center already returns
// an exact server total for waiting_for_me, so a heavy list fetch would be wasteful.
// @UX_STATE Idle -> combined decision count 0, badge hidden.
// @UX_STATE Ready -> waiting + pending-approval totals > 0, sidebar "Тестирование дашбордов"
// badge shows the sum ("Ожидают моего решения" — both are decisions the analyst owes).
// @UX_STATE Disabled -> run center not accessible (401/403) -> polling stops, counts keep last values.
// @INVARIANT Each poll reads only its authoritative server total (waiting_for_me=true /
// pending_approval=true, page_size=1); the badge is their sum and is never derived
// client-side, mirroring GlobalRunCenterModel's server-narrowing rule.
// @INVARIANT A non-2xx authority answer never fabricates a count — it leaves the last value of
// the failed projection and, for 401/403 from either poll, disables further polling
// (bounded, fail-closed).
// @RATIONALE pending_approval uses the same listing visibility (scenario:RUN) as waiting_for_me:
// the projection is a read-only aggregate; the approve decision itself stays behind
// scenario:RUN_PROD (mcp list_pending_approvals/decide_approval, service_allowed=False),
// so the badge grants nothing while keeping the gate visible. Mirrors
// Stores.Health.HealthStore: module-level $state atoms + getters + subscriber set
// keeps Svelte 5 reactivity for the Sidebar $derived badge with an interval poll.
// @REJECTED Deriving the pending count client-side from the full run list rejected — same reason
// as the waiting count: the server already returns an exact total (page_size=1, bounded).
// @REJECTED Gating the badge behind RUN_PROD rejected — RUN-holders must learn a gate is pending
// to route it; the count itself has no mutation capability.
import { api } from '../api';
import { SvelteSet } from 'svelte/reactivity';
import { log } from '$lib/cot-logger';
@@ -24,41 +33,77 @@ interface RunCenterTotal {
total: number;
}
const WAITING_URL = '/scenario-runs?waiting_for_me=true&page_size=1';
const PENDING_APPROVAL_URL = '/scenario-runs?pending_approval=true&page_size=1';
let _waitingCount = $state(0);
let _pendingApprovalCount = $state(0);
let _loading = $state(false);
let _isDisabled = false;
const _subs = new SvelteSet<(_v: number) => void>();
let _inflight: Promise<number> | null = null;
function _notify(): void {
_subs.forEach((fn) => fn(_waitingCount));
function _decisionCount(): number {
return _waitingCount + _pendingApprovalCount;
}
function _notify(): void {
_subs.forEach((fn) => fn(_decisionCount()));
}
interface TotalOutcome {
ok: boolean;
total: number;
status?: number;
error?: unknown;
}
// #region Stores.ScenarioRuns.WaitingStore.FetchTotal [C:2] [TYPE Function]
// @BRIEF Fetch one bounded projection total; a failure is returned as data, never thrown.
// @POST Resolves {ok:true,total} from the authoritative server envelope, or {ok:false,status} on error.
async function _fetchTotal(url: string): Promise<TotalOutcome> {
try {
const resp = await api.fetchApi<RunCenterTotal>(url);
const total = typeof resp?.total === 'number' && resp.total >= 0 ? resp.total : 0;
return { ok: true, total };
} catch (error: unknown) {
const apiError = error as { status?: number; message?: string };
return { ok: false, total: 0, status: apiError?.status, error };
}
}
// #endregion Stores.ScenarioRuns.WaitingStore.FetchTotal
// #region Stores.ScenarioRuns.WaitingStore.Refresh [C:3] [TYPE Function]
// @BRIEF Fetch the waiting-for-me run total; dedup concurrent calls; fail closed on auth denial.
// @PRE Caller may invoke from an interval; repeated in-flight calls share one request.
// @POST Resolves the current waiting count; on 401/403 disables further polling.
// @BRIEF Fetch both decision totals (waiting + pending approval); dedup concurrent calls; fail closed on auth denial.
// @PRE Caller may invoke from an interval; repeated in-flight calls share one combined request pair.
// @POST Resolves the summed decision count; a failed projection keeps its last value; on 401/403
// from either poll disables further polling.
async function refresh(): Promise<number> {
if (_isDisabled) return _waitingCount;
if (_isDisabled) return _decisionCount();
if (_inflight) return _inflight;
_loading = true;
_inflight = (async () => {
try {
const resp = await api.fetchApi<RunCenterTotal>('/scenario-runs?waiting_for_me=true&page_size=1');
_waitingCount = typeof resp?.total === 'number' && resp.total >= 0 ? resp.total : 0;
_notify();
return _waitingCount;
} catch (error: unknown) {
const apiError = error as { status?: number; message?: string };
if (apiError?.status === 401 || apiError?.status === 403) {
log('ScenarioRunsStore', 'EXPLORE', 'Run center not accessible — stop waiting-count polling', {}, String(apiError.status));
const [waiting, pending] = await Promise.all([
_fetchTotal(WAITING_URL),
_fetchTotal(PENDING_APPROVAL_URL),
]);
if (waiting.ok) _waitingCount = waiting.total;
if (pending.ok) _pendingApprovalCount = pending.total;
const authDenied = [waiting, pending].some(
(r) => !r.ok && (r.status === 401 || r.status === 403),
);
if (authDenied) {
log('ScenarioRunsStore', 'EXPLORE', 'Run center not accessible — stop decision-count polling', {}, '401/403');
_isDisabled = true;
} else {
log('ScenarioRunsStore', 'EXPLORE', 'Waiting-count refresh failed', {}, error instanceof Error ? error.message : String(error));
} else if (!waiting.ok || !pending.ok) {
const failed = !waiting.ok ? waiting : pending;
log('ScenarioRunsStore', 'EXPLORE', 'Decision-count refresh failed', {}, failed.error instanceof Error ? failed.error.message : String(failed.error));
}
return _waitingCount;
_notify();
return _decisionCount();
} finally {
_loading = false;
_inflight = null;
@@ -69,11 +114,11 @@ async function refresh(): Promise<number> {
// #endregion Stores.ScenarioRuns.WaitingStore.Refresh
export const scenarioWaitingCount = {
get current() { return _waitingCount; },
get current() { return _decisionCount(); },
get loading() { return _loading; },
get disabled() { return _isDisabled; },
subscribe(fn: (_v: number) => void) {
fn(_waitingCount);
fn(_decisionCount());
_subs.add(fn);
return () => { _subs.delete(fn); };
},

View File

@@ -2,17 +2,27 @@
<!-- @ingroup ScenarioRegistry -->
<!-- @BRIEF Scenario registry hub: searchable/filterable, paginated list of 042 scenarios; the
navigation landing for the dashboard-testing workspace and the target of the "← Сценарии"
back-links. Filters (q/status/owner/tag/dashboard_id) and pagination are server-applied. -->
back-links. Filters (q/status/owner/tag/dashboard_id) and pagination are server-applied.
The ?dashboard_id= deep-link (UX-5 entry from the dashboard card) pre-fills the dashboard
filter before the first load. -->
<!-- @RELATION DEPENDS_ON -> [ScenarioRegistry.Model] -->
<!-- @RELATION DEPENDS_ON -> [Routes.RoutesRegistry] -->
<!-- @RELATION DEPENDS_ON -> [Ui.PageHeader] -->
<!-- @UX_STATE loading -> progress placeholder; error -> typed recovery; empty -> empty state; ready -> list. -->
<!-- @UX_STATE deep-linked -> dashboard filter initialized from ?dashboard_id= at component
creation, i.e. before the first fetch and without a filter-change event. -->
<!-- @UX_FEEDBACK Filters and page controls reload the list server-side; rows link to edit/analytics/last run. -->
<!-- @UX_RECOVERY Retry button on load failure; filters can be cleared by emptying them. -->
<!-- @INVARIANT The index only reads registry entries (dashboard:testing READ); it never starts a run,
mutates a revision, or launches an agent (no side effects beyond navigation). -->
<!-- @INVARIANT Changing any filter resets pagination to page 1. -->
<!-- @INVARIANT Changing any filter resets pagination to page 1. Query-param initialization is
NOT a filter change: the first load fires once, already scoped, at page 1. -->
<!-- @REJECTED Initializing the filter inside the mount $effect (or reacting to page.url changes
continuously) was rejected — it couples the deep-link to the load effect and turns the
init into an observable filter-change event; the $state initializer runs once at component
creation, strictly before the first loadList, with no re-fire on later URL edits. -->
<script lang="ts">
import { page } from "$app/state";
import { t } from "$lib/i18n/index.svelte.js";
import { ROUTES } from "$lib/routes";
import { PageHeader, Input, Select, Badge, Button } from "$lib/ui";
@@ -25,7 +35,10 @@
let status = $state("");
let owner = $state("");
let tag = $state("");
let dashboardId = $state("");
// UX-5: ?dashboard_id= deep-link initializes the filter at creation time, before the
// first loadList fires in $effect — the inaugural fetch is already dashboard-scoped.
// Non-numeric values are dropped later by numericId() when the query is built.
let dashboardId = $state(page.url.searchParams.get("dashboard_id") ?? "");
let currentPage = $state(1);
let started = $state(false);

View File

@@ -3,7 +3,8 @@
<!-- @BRIEF Live scenario run monitor: timeline, human checkpoint actions, final result + provenance. -->
<!-- @RELATION DEPENDS_ON -> [ScenarioRunMonitor.Model] -->
<!-- @RELATION BINDS_TO -> [EXT:SvelteKit:page.params] -->
<!-- @UX_STATE loading -> progress; live -> timeline; waiting_human -> checkpoint panel; pending_approval -> PROD approval decision panel (D4); terminal -> result; error -> recovery. -->
<!-- @UX_STATE loading -> progress; live -> timeline; waiting_human -> checkpoint panel; pending_approval -> PROD approval decision panel (D4); terminal -> typed reason banner + result; error -> recovery. -->
<!-- @UX_FEEDBACK terminal non-pass with a typed error/reason code -> human-readable banner (RU/EN explanation + next step, UX-4); unknown code -> fallback banner with the code verbatim. -->
<!-- @INVARIANT The page recovers by run id after disconnect (RUNMON-FR-003) and never parses prose. -->
<script lang="ts">
import { page } from "$app/state";
@@ -16,6 +17,8 @@
import HumanCheckpointPanel from "$lib/components/scenario-run/HumanCheckpointPanel.svelte";
import ApprovalDecisionPanel from "$lib/components/scenario-run/ApprovalDecisionPanel.svelte";
import ScenarioResultView from "$lib/components/scenario-run/ScenarioResultView.svelte";
import TerminalReasonBanner from "$lib/components/scenario-run/TerminalReasonBanner.svelte";
import { resolveTerminalReason, terminalCodeFor } from "$lib/components/scenario-run/terminal-reasons";
import type { ScenarioStepRun } from "$lib/types/scenario-run";
const model = new RunMonitorModel();
@@ -29,6 +32,7 @@
: ((model.result?.snapshot?.steps ?? []) as unknown as ScenarioStepRun[]),
);
const waitingStep = $derived(steps.find((step) => step.status === "waiting_human"));
const terminalReason = $derived(resolveTerminalReason(terminalCodeFor(model.run, model.result)));
let resultRequested = $state("");
$effect(() => {
@@ -83,6 +87,9 @@
<div class="grid gap-6 lg:grid-cols-[1fr_340px]">
<section class="space-y-4">
<RunTimeline {steps} />
{#if ["failed", "cancelled", "blocked", "inconclusive"].includes(model.run.status) && terminalReason}
<TerminalReasonBanner reason={terminalReason} />
{/if}
{#if ["passed", "failed", "cancelled", "blocked", "inconclusive"].includes(model.run.status) && model.result}
<ScenarioResultView result={model.result} />
{/if}

View File

@@ -3,6 +3,7 @@
// @RELATION BINDS_TO -> [ScenarioRun.Route.Detail]
// @RELATION BINDS_TO -> [ScenarioRunMonitor.Model]
// @TEST_EDGE: waiting_human -> checkpoint panel + dispose resumes; terminal -> result + provenance render
// @TEST_EDGE: terminal non-pass with typed code -> reason banner (UX-4); unlisted code -> fallback banner with code verbatim
import { fireEvent, render, screen, waitFor } from "@testing-library/svelte";
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
import { page } from "$app/state";
@@ -72,5 +73,36 @@ describe("RunMonitor route (run.ux)", () => {
expect(screen.getByText(/scenario_revision_id/)).toBeTruthy();
expect(screen.getByText(/044\.1\.0/)).toBeTruthy();
});
it("explains a typed terminal code with a human-readable banner (UX-4)", async () => {
const inconclusiveRun = { ...terminalRun, status: "inconclusive", error_code: "BROWSER_ACTION_NOT_SUPPORTED" };
const inconclusiveResult = { ...result, status: "inconclusive", passed: 0, failed: 0, inconclusive: 1 };
vi.spyOn(api, "fetchApi").mockResolvedValueOnce(inconclusiveRun).mockResolvedValueOnce(inconclusiveResult);
render(DetailPage);
const banner = await screen.findByRole("status");
expect(banner.textContent).toContain("Действие браузера пока не поддержано провайдером");
expect(banner.textContent).toContain("план UX-1");
});
it("shows the fallback banner with the code verbatim for an unlisted typed code", async () => {
const unknownRun = { ...terminalRun, status: "blocked", error_code: "SUPERSET_BINDING_MISSING" };
vi.spyOn(api, "fetchApi").mockResolvedValueOnce(unknownRun).mockResolvedValueOnce({ ...result, status: "blocked" });
render(DetailPage);
const banner = await screen.findByRole("status");
expect(banner.textContent).toContain("Live-провайдер не настроен");
expect(banner.textContent).toContain("scenario-live-bindings");
});
it("renders an unlisted typed code as-is in the fallback banner", async () => {
const unknownRun = { ...terminalRun, status: "failed", error_code: "SOME_UNLISTED_CODE" };
vi.spyOn(api, "fetchApi").mockResolvedValueOnce(unknownRun).mockResolvedValueOnce(result);
render(DetailPage);
const banner = await screen.findByRole("status");
expect(banner.textContent).toContain("Typed non-pass");
expect(banner.textContent).toContain("SOME_UNLISTED_CODE");
});
});
// #endregion Test.ScenarioRun.UX

View File

@@ -0,0 +1,65 @@
// #region Test.ScenarioRegistry.Route.IndexDeepLink [C:3] [TYPE Module] [SEMANTICS test,scenario,registry,route,index,deeplink,ux]
// @RELATION VERIFIES -> [ScenarioRegistry.Route.Index]
// @TEST_EDGE: ?dashboard_id=42 -> dashboard filter pre-filled at creation; the FIRST and only
// initial fetch is already dashboard-scoped at page 1 (init is not a filter change).
// @TEST_INVARIANT: index-never-starts-a-run -> the deep-linked page still issues exactly one
// read-only list request; no run/mutation endpoint is called.
import { render, screen } from "@testing-library/svelte";
import { beforeEach, describe, expect, it, vi } from "vitest";
vi.mock("$app/state", () => ({
page: {
url: new URL("http://localhost/dashboard-testing/scenarios?dashboard_id=42"),
params: {},
},
}));
import { api } from "$lib/api";
import IndexPage from "../+page.svelte";
const entry = {
scenario_id: "s1",
scenario_key: "sales-smoke",
name: "Sales Smoke",
description: null,
dashboard_id: 42,
environment_ids: ["env-prod"],
owner_id: "u1",
owner_username: "alice",
tags: [],
metadata_version: "1",
current_revision_id: "rev-1",
lifecycle_status: "READY",
validation_status: "valid",
last_run_id: "run-1",
last_successful_run_id: "run-1",
health: "pass",
baseline_compatibility: null,
last_modified_at: null,
created_at: null,
};
beforeEach(() => {
vi.restoreAllMocks();
});
describe("Scenario registry index ?dashboard_id= deep link (UX-5)", () => {
it("initializes the dashboard filter from the URL before the first load", async () => {
const fetchMock = vi.spyOn(api, "fetchApi").mockResolvedValue({ items: [entry], total: 1 });
render(IndexPage);
expect(await screen.findByText("Sales Smoke")).toBeTruthy();
// exactly one inaugural fetch, already scoped by the deep-linked dashboard id
expect(fetchMock).toHaveBeenCalledTimes(1);
expect(fetchMock).toHaveBeenCalledWith(
"/dashboard-testing/scenarios?dashboard_id=42&page=1&page_size=20",
);
// the filter input reflects the deep-linked value
// (queried by placeholder: Ui.Input generates duplicate ids — pre-existing a11y defect,
// getByLabelText would resolve to the first input sharing the id)
const filterInput = screen.getByPlaceholderText("id дашборда") as HTMLInputElement;
expect(filterInput.value).toBe("42");
});
});
// #endregion Test.ScenarioRegistry.Route.IndexDeepLink

View File

@@ -1,10 +1,28 @@
<!-- #region Id.DashboardHeader [C:2] [TYPE Component] [SEMANTICS sveltekit, dashboard, header, git, branch] -->
<!-- @ingroup Routes -->
<!-- @BRIEF Top title area, breadcrumb, Git branch selector, and action buttons for dashboard detail. -->
<!-- @BRIEF Top title area, breadcrumb, Git branch selector, action buttons, and the read-only
"Test scenarios" registry nav link (UX-5) for dashboard detail. -->
<!-- @LAYER UI -->
<!-- @RELATION DEPENDS_ON -> [EXT:frontend:BranchSelector] -->
<!-- @RELATION DEPENDS_ON -> [Ui.Icon] -->
<!-- @UX_STATE Idle -> Header with back button, title, Git status badge, and action buttons. -->
<!-- @RELATION DEPENDS_ON -> [Routes.RoutesRegistry] -->
<!-- @UX_STATE Idle -> Header with back button, title, Git status badge, action buttons, and a
read-only link to the scenario registry pre-filtered by this dashboard id. -->
<!-- @INVARIANT The "Test scenarios" entry is pure read-only navigation to
/dashboard-testing/scenarios?dashboard_id={id} — it never starts a run and adds no
agent/prompt/handoff control (frontend boundary 2026-09-08). -->
<!-- @INVARIANT The header renders no a[href*="/agent"], no prompt/textarea/proposal semantics,
and initiates no agent-endpoint requests (UX-11 negative acceptance, 044 refresh
2026-09-08; enforced by Test.DashboardTesting.DashboardHeaderEntry). -->
<!-- @RATIONALE UX-11 (2026-09-12): agentHref/scenarioHref derived routes to /agent
(contextVersion=2 + build_dashboard_test_scenario intent, 039 T007/T009) and their
MCP/Create-scenario buttons were removed as agent workspace/start handoff controls —
runtime drift against the 044/036/046 frontend boundary; scenario authoring belongs to
the external MCP surface. -->
<!-- @REJECTED Keeping the /agent links behind a feature flag or permission gate rejected —
the boundary prohibits the control class in the product frontend entirely, not its
visibility; a hidden route affordance would keep the drift alive. Reverting to the
039 agent-entry design requires a fresh boundary contract update, not a local flag. -->
<script lang="ts">
import { t } from "$lib/i18n/index.svelte.js";
import Icon from "$lib/ui/Icon.svelte";
@@ -29,32 +47,11 @@
loadDashboardPage
} = $props();
let agentHref = $derived.by(() => {
const objectId = String(dashboardRef || resolvedDashboardId || "");
const params = new URLSearchParams({
objectType: "dashboard",
objectId,
objectName: dashboard?.title || "",
envId: envId || "",
route: `/dashboards/${objectId}`,
});
return ROUTES.agent(params.toString());
});
// 039 T007/T009: scenario mode opens /agent with contextVersion=2 + scenario intent (AGUI-FR-002).
let scenarioHref = $derived.by(() => {
const objectId = String(dashboardRef || resolvedDashboardId || "");
const params = new URLSearchParams({
objectType: "dashboard",
objectId,
objectName: dashboard?.title || "",
envId: envId || "",
route: `/dashboards/${objectId}`,
contextVersion: "2",
intent: "build_dashboard_test_scenario",
});
return ROUTES.agent(params.toString());
});
// UX-5: read-only nav entry to the scenario registry pre-filtered by this dashboard.
// Null/non-numeric refs degrade to the unfiltered registry hub (no param).
let scenariosHref = $derived(
ROUTES.dashboardTesting.scenarios(resolvedDashboardId ?? undefined),
);
</script>
<div class="flex flex-col gap-4 xl:flex-row xl:items-start xl:justify-between">
@@ -111,18 +108,10 @@
{isStartingBackup ? $t.common?.loading : $t.dashboard?.run_backup}
</button>
<a
href={agentHref}
class="inline-flex items-center justify-center gap-1.5 rounded-lg border border-border-strong bg-surface-card px-4 py-2 text-sm font-medium text-text transition-colors hover:bg-primary hover:text-white"
title={$t.assistant?.mcp_entry_tooltip || $t.assistant?.ask_ai_dashboard || "Connect the AI assistant (MCP)"}
>
<Icon name="aiAssistant" size={16} />
<span>{$t.assistant?.mcp_entry_short || "MCP"}</span>
</a>
<a
href={scenarioHref}
href={scenariosHref}
class="inline-flex items-center justify-center rounded-lg border border-border-strong bg-surface-card px-4 py-2 text-sm font-medium text-text transition-colors hover:bg-primary hover:text-white"
>
{$t.dashboard_testing?.create_scenario || "Create test scenario"}
{$t.dashboard_testing?.view_scenarios || "Test scenarios"}
</a>
<a
href={ROUTES.loadTesting.detail(resolvedDashboardId)}

View File

@@ -2,49 +2,135 @@
* @vitest-environment jsdom
*/
// #region Test.DashboardTesting.DashboardHeaderEntry [C:3] [TYPE Module] [SEMANTICS test,dashboard-testing,entry,ux]
// @BRIEF 039 T006: DashboardHeader scenario entry — three ids, env missing, AI link preserved.
// @RELATION BINDS_TO -> [DashboardDetailModel]
// @BRIEF UX-5 positive: read-only scenario-registry link; UX-11 negative: agent-drift removal
// acceptance (frontend boundary 2026-09-08, 044 Production contract refresh).
// @RELATION VERIFIES -> [Id.DashboardHeader]
// @TEST_EDGE: ux5_registry_link -> href carries the resolved numeric dashboard id; null id
// degrades to the unfiltered registry hub; no run-start control is added by the link.
// @TEST_EDGE: agent_link_absent -> no a[href*="/agent"] and no prompt/textarea/proposal
// semantics anywhere in the rendered header (agent workspace/start handoff is prohibited).
// @TEST_EDGE: agent_network_absent -> rendering and interacting with the header fires zero
// fetches to /agent or /api/agent*|/api/assistant* endpoints (route-independent check;
// live git traffic through BranchSelector proves the recorder is not vacuous).
// @TEST_INVARIANT: Id.DashboardHeader read-only/no-agent-control invariant -> VERIFIED_BY:
// [ux5_registry_link, agent_link_absent, agent_network_absent]
import { describe, it, expect, vi } from "vitest";
import { DashboardDetailModel } from "$lib/models/DashboardDetailModel.svelte.ts";
import { render, screen } from "@testing-library/svelte";
import DashboardHeader from "../DashboardHeader.svelte";
function withModel(ref: string, envId: string, id?: number) {
const m = new DashboardDetailModel();
(m as { dashboardRef: string }).dashboardRef = ref;
(m as { envId: string }).envId = envId;
if (id !== undefined) (m as { dashboard: { id: number; title: string } | null }).dashboard = { id, title: `Dash ${id}` };
return m;
function headerProps(over: Record<string, unknown> = {}) {
return {
dashboard: { id: 42, title: "Dash 42", slug: "dash-42" },
resolvedDashboardId: 42,
dashboardRef: "42",
envId: "env-prod",
gitDashboardRef: "",
hasGitRepo: false,
currentBranch: "prod",
gitMeta: { pillClass: "", dotClass: "", label: "Git" },
gitStatus: null,
isStartingBackup: false,
showGitManager: false,
goBack: () => {},
handleBranchChange: () => {},
runBackupTask: () => {},
loadDashboardPage: () => {},
...over,
};
}
describe("DashboardHeader scenario entry (T006)", () => {
it("scenarioHref carries contextVersion=2 + intent + correct objectId", () => {
const m = withModel("42", "env-prod", 42);
const params = new URLSearchParams(m.scenarioHref.split("?")[1]);
expect(params.get("contextVersion")).toBe("2");
expect(params.get("intent")).toBe("build_dashboard_test_scenario");
expect(params.get("objectId")).toBe("42");
expect(params.get("envId")).toBe("env-prod");
function scenariosLink(): HTMLAnchorElement {
return screen.getByRole("link", { name: "Сценарии тестирования" }) as HTMLAnchorElement;
}
/** URLs of recorded fetches that hit an agent workspace/run surface. */
function agentCalls(calls: string[]): string[] {
return calls.filter((url) => /\/agent|\/api\/assistant/.test(url));
}
describe("DashboardHeader scenarios registry entry (UX-5)", () => {
it("renders a read-only registry link with ?dashboard_id for this dashboard", () => {
render(DashboardHeader, { props: headerProps() });
expect(scenariosLink().getAttribute("href")).toBe(
"/dashboard-testing/scenarios?dashboard_id=42",
);
});
it("env missing does not drop context (envId empty, context still valid)", () => {
const m = withModel("7", "", 7);
const params = new URLSearchParams(m.scenarioHref.split("?")[1]);
expect(params.get("envId")).toBe("");
expect(params.get("intent")).toBe("build_dashboard_test_scenario");
it("never reuses a stale dashboard id across instances", () => {
const first = render(DashboardHeader, { props: headerProps() });
expect(scenariosLink().getAttribute("href")).toContain("dashboard_id=42");
first.unmount();
render(DashboardHeader, {
props: headerProps({
dashboard: { id: 7, title: "Dash 7", slug: "dash-7" },
resolvedDashboardId: 7,
dashboardRef: "7",
}),
});
expect(scenariosLink().getAttribute("href")).toBe(
"/dashboard-testing/scenarios?dashboard_id=7",
);
});
it("distinct dashboards never reuse stale id", () => {
const a = withModel("1", "env-prod", 1);
const b = withModel("2", "env-prod", 2);
expect(a.scenarioHref).not.toBe(b.scenarioHref);
expect(a.scenarioHref).toContain("objectId=1");
expect(b.scenarioHref).toContain("objectId=2");
it("degrades to the unfiltered registry hub when no numeric id is resolved", () => {
render(DashboardHeader, {
props: headerProps({ dashboard: null, resolvedDashboardId: null, dashboardRef: "sales-dash" }),
});
expect(scenariosLink().getAttribute("href")).toBe("/dashboard-testing/scenarios");
});
});
describe("DashboardHeader agent-drift removal (UX-11 negative acceptance)", () => {
it("renders no /agent links — only manual, read-only navigation remains", () => {
render(DashboardHeader, { props: headerProps() });
const agentAnchors = Array.from(document.querySelectorAll("a")).filter((a) =>
(a.getAttribute("href") || "").includes("/agent"),
);
expect(agentAnchors).toEqual([]);
// Positive control: the permitted manual registry entry survives next to the removal.
expect(scenariosLink().getAttribute("href")).toBe(
"/dashboard-testing/scenarios?dashboard_id=42",
);
});
it("ordinary AI link (agentHref) preserved alongside scenario", () => {
// The header exposes scenarioHref separately from the ordinary AI href.
const m = withModel("5", "env-prod", 5);
const params = new URLSearchParams(m.scenarioHref.split("?")[1]);
expect(params.get("contextVersion")).toBe("2"); // scenario link is distinct from v1 AI link
it("renders no prompt/textarea/proposal-semantics controls", () => {
render(DashboardHeader, { props: headerProps() });
expect(document.querySelector("textarea")).toBeNull();
expect(document.querySelector("[contenteditable='true']")).toBeNull();
expect(screen.queryByRole("textbox")).toBeNull();
const suspicious = Array.from(
document.querySelectorAll("a, button, [role='button']"),
).filter((el) => /proposal|prompt|generate|agent/i.test(el.textContent || ""));
expect(suspicious).toEqual([]);
});
it("interacting with the rendered header never initiates agent API calls", async () => {
const calls: string[] = [];
const fetchSpy = vi.fn(async (input: RequestInfo | URL) => {
calls.push(String(input instanceof Request ? input.url : input));
return {
ok: true,
status: 200,
// Empty list: BranchModel assigns .branches directly and derives over it.
json: async () => [] as unknown[],
text: async () => "",
};
});
vi.stubGlobal("fetch", fetchSpy);
try {
// hasGitRepo=true mounts BranchSelector so live git traffic proves the recorder works.
render(DashboardHeader, {
props: headerProps({ hasGitRepo: true, gitDashboardRef: "dash-42" }),
});
for (const btn of Array.from(document.querySelectorAll("button"))) btn.click();
await new Promise((resolve) => setTimeout(resolve, 0));
expect(calls.length).toBeGreaterThan(0); // recorder is live (git endpoints flowed)
expect(agentCalls(calls)).toEqual([]);
} finally {
vi.unstubAllGlobals();
}
});
});

View File

@@ -8,6 +8,11 @@
<!-- @UX_STATE error -> Error banner with retry button. -->
<!-- @UX_FEEDBACK Clicking linked dashboard navigates to dashboard detail. -->
<!-- @UX_RECOVERY Refresh button reloads dataset details. -->
<!-- @RATIONALE UX-11 (2026-09-12): the "Ask AI" /agent handoff link (agentHref) was removed —
agent workspace/start entry from a product surface is prohibited by the frontend
boundary 2026-09-08 (044/036/046 refresh); dataset questions belong to external MCP. -->
<!-- @REJECTED Keeping the /agent dataset link as an opt-in entry rejected — the boundary
prohibits the control class in the product frontend, not its visibility. -->
<script lang="ts">
import { onMount } from 'svelte';
import { page } from '$app/state';
@@ -21,16 +26,6 @@
let datasetId = $derived(page.params.id);
let envId = $derived(page.url.searchParams.get('env_id') || '');
const model = new DatasetDetailModel();
let agentHref = $derived.by(() => {
const params = new URLSearchParams({
objectType: "dataset",
objectId: datasetId,
objectName: model.dataset?.table_name || "",
envId,
route: `/datasets/${datasetId}`,
});
return `/agent?${params.toString()}`;
});
$effect(() => { model.datasetId = datasetId; });
$effect(() => { model.envId = envId; });
@@ -57,14 +52,6 @@
{/if}
</div>
<div class="flex items-center gap-2">
<a
href={agentHref}
class="inline-flex items-center justify-center gap-1.5 rounded-lg border border-border-strong bg-surface-card px-4 py-2 text-sm font-medium text-text transition-colors hover:bg-primary hover:text-white"
title={$t.assistant?.ask_ai_dataset || 'Ask AI about the dataset'}
>
<Icon name="aiAssistant" size={16} />
<span>AI</span>
</a>
<Button variant="destructive" onclick={() => model.loadDatasetDetail()}>
{$t.common?.refresh}
</Button>

View File

@@ -137,6 +137,8 @@ Gradio chat retirement is formalized in `specs/050-mcp-interface/spec.md`. Carry
**Frontend boundary (user decision 2026-09-08)**: All agent interaction is external MCP only. Product frontend MUST NOT contain agent chat, prompt/request textarea, assistant editing, typical-operation-to-agent selector, proposal-generation, agent workspace/start or handoff controls/routes. Ordinary manual CRUD/editor, human approval/review, monitoring and read-only evidence/evaluation are permitted. AgentEvaluationCard is read-only, with no prompt/retry-agent/provider controls. Existing agent proposal UI is runtime drift; removal/negative DOM-route-network acceptance remains OPEN in this spec-only change.
**AGSTAB-FR-013 traceability (UX-11, 2026-09-12)**: product-surface `/agent` entries removed (DashboardHeader MCP/Create-scenario links + derived routes, TopNavbar assistant handoff, dataset-detail AI link, DashboardDetailModel.scenarioHref); negative DOM-route-network acceptance in `Test.DashboardTesting.DashboardHeaderEntry`. Route-file removal stays with 050.
**AGSTAB-FR-014 — Evidence owner and promotion**: DraftArtifact ownership MUST remain AgentRun-scoped. Promoting capture into a 037 baseline acquires a durable baseline retention hold and verified review receipt before draft expiry. ScenarioRun uses 044 owner receipts/content API, never fabricated AgentRun ownership. Gates bind candidate/review/catalog-CAS/release/publication intent, and MCP/REST decisions consume the same store under fresh ACL.
Normative contract: [Evidence owner and promotion](../017-llm-analysis-plugin/contracts/scenario-reuse.md). New requirements are specified, **implemented=false / acceptance OPEN** until executable evidence closes the linked tasks/checklist/traceability rows. Historical local tests and the manual inconclusive ss-prod run do not prove browser/capture/baseline/LLM production readiness. The refresh scope is the audited P0/P1/P2 agentic E2E and baseline gaps; an approved ExecutionPerformanceBaseline is not introduced.

View File

@@ -3,6 +3,7 @@
@RATIONALE Persist canonical names in newly compiled descriptors while preserving explicit import aliases for old registries.
@REJECTED Accepting an unsupported registry action and discovering transport failure after browser I/O.
Status: normative target, implemented=false. Existing runtime038.1.0 accepts only an incomplete subset; transport support is not inferred from names. Every enabled descriptor MUST have matching input/output schemas and a canaried driver/reconciler; unsupported capability blocks compilation/admission. Aliases are compile-time migrations with new hashes, never runtime fallback.
Provider readiness update (044 UX-1, 2026-09-12): `apply_native_filter` is implemented in the 044 isolated browser provider/transport as a read-only filter-bar UI interaction — typed input `{filter_id|filter_name|column, values[], wait_state?}` (empty input = current-state apply), bounded chart settle, `filter_applied`/`charts_settled` checkpoints, post-action PNG evidence; locator miss is typed `BROWSER_SELECTOR_NOT_FOUND` → inconclusive without retry. This is a provider readiness closure only: **the registry version/hash is NOT bumped** (the descriptor was already registered in 038.4.0; `ACTION_REGISTRY_VERSION` stays `038.4.0`), the descriptor input/output contracts in this table are unchanged, and the remaining actions in this table remain unimplemented pending their own canaried drivers.
| Canonical action | Legacy import alias | Required bounded input | Output receipt payload |
|---|---|---|---|
| open_dashboard | same | server dashboard_ref | dashboard_ref, state_hash |

View File

@@ -189,7 +189,8 @@ class LiveMcpReplay:
# (baseline capability stays off), so the gated start never needs the unpublished catalog.
# @REJECTED Hand-building the scenario fixture like the 2026-09-07 run was rejected — T029m demands the
# derived-capability compile from live context (MCPX-FR-028), not a client-authored graph.
def run_chain(baseline: bool = False) -> dict:
def run_chain(baseline: bool = False, filter_values: list[str] | None = None,
expect_passed: bool = False) -> dict:
replay = LiveMcpReplay()
evidence: dict = {"base": BASE, "environment_id": ENVIRONMENT_ID, "dashboard_id": DASHBOARD_ID}
@@ -284,6 +285,10 @@ def run_chain(baseline: bool = False) -> dict:
"scenario_id": boot["scenario_id"], "revision_id": boot["revision_id"],
"environment_id": ENVIRONMENT_ID, "idempotency_key": start_key,
}
if filter_values:
# UX-6: launch params are part of the derived program — RunnerPlan derivation binds
# params.filter_values onto the pinned apply_native_filter step (param_binding).
start_request["params"] = {"filter_values": list(filter_values)}
if baseline:
# Explicit selector: the pin must resolve from the published Gitea envelope, not request bytes.
start_request["baseline_set"] = BASELINE_SET
@@ -317,13 +322,22 @@ def run_chain(baseline: bool = False) -> dict:
time.sleep(5)
if status not in _TERMINAL_STATUSES:
raise ReplayError(f"run did not reach a terminal status within {POLL_TIMEOUT_SECONDS}s (last={status})")
# Fail-closed evidence guard: the compiled B01 graph's browser action (apply_native_filter) is not
# supported by the isolated transport, so this chain's honest terminal is a typed non-pass. A
# `passed` terminal would mean the unsupported step was silently synthesized into success.
evidence["filter_values"] = list(filter_values or [])
# Fail-closed evidence guard. Since UX-1 (2026-09-12) the isolated transport implements
# apply_native_filter (read-only filter-bar UI flow), so a live B01 run may honestly terminalize
# as passed. Default stays fail-closed: a `passed` terminal is rejected as unverified
# (UNEXPECTED_SYNTHETIC_PASS). With --expect-passed (re-baseline after the live canary) a passed
# terminal is accepted ONLY on durable evidence: the apply_native_filter step passed with a
# non-empty step_outcome sha256 AND capture_screenshot passed — any gap raises ReplayError.
if status == "passed":
raise ReplayError("UNEXPECTED_SYNTHETIC_PASS: the typed non-pass terminal became passed")
if not expect_passed:
raise ReplayError("UNEXPECTED_SYNTHETIC_PASS: the typed non-pass terminal became passed")
unverified = _unverified_passed_evidence(_step_report(run_id))
if unverified is not None:
raise ReplayError(f"UNVERIFIED_PASSED_TERMINAL: {unverified}")
evidence["terminal"] = {"status": status, "phase": detail.get("phase"), "error_code": detail.get("error_code"),
"synthetic_pass_guard": "non_pass_confirmed"}
"synthetic_pass_guard": ("pass_verified_durable_evidence" if status == "passed"
else "non_pass_confirmed")}
evidence["steps"] = _step_report(run_id)
if baseline:
# T046 baseline-pin proof: the plan pin must come from the published catalog at start time.
@@ -342,15 +356,48 @@ def run_chain(baseline: bool = False) -> dict:
def _step_report(run_id: str) -> list[dict]:
with SessionLocal() as db:
steps = db.query(ScenarioStepRun).filter(ScenarioStepRun.run_id == run_id).all()
return [{
"logical_step_id": step.logical_step_id, "status": step.status, "error_code": step.error_code,
"artifact_refs_count": len(step.artifact_refs or []),
"step_outcome": {
key: value for key, value in (step.step_outcome or {}).items()
if key in ("status", "reason_code", "sha256", "checkpoints", "page_url",
"action", "screenshot_count")
},
} for step in steps]
report: list[dict] = []
for step in steps:
outcome = dict(step.step_outcome or {})
# The runner persists the adapter details nested under a same-named inner key.
details = outcome.get("step_outcome") if isinstance(outcome.get("step_outcome"), dict) else {}
merged = {**details, **{key: value for key, value in outcome.items() if key != "step_outcome"}}
report.append({
"logical_step_id": step.logical_step_id, "status": step.status, "error_code": step.error_code,
"artifact_refs_count": len(step.artifact_refs or []),
"step_outcome": {
key: value for key, value in merged.items()
if key in ("status", "reason_code", "sha256", "checkpoints", "page_url",
"action", "screenshot_count", "applied_mode", "applied_values",
"filter_target", "chart_data_observed")
},
})
return report
# #region ScenarioExecution.LiveMcpReplay.PassEvidence [C:3] [TYPE Function] [SEMANTICS replay,evidence,guard,expect-passed]
# @ingroup ScenarioExecution
# @BRIEF Re-baselined pass guard: a passed terminal is accepted only on durable step evidence.
# @POST Returns None when every apply_native_filter step passed with a non-empty step_outcome sha256
# AND capture_screenshot passed; otherwise a short typed rejection reason.
# @RATIONALE Re-baseline after the live canary (2026-09-12): with --expect-passed a passed terminal is
# only accepted on durable provider evidence (stored artifact digest + passed screenshot),
# so a synthetic pass cannot slip through; without the flag the previous fail-closed
# non-pass expectation (UNEXPECTED_SYNTHETIC_PASS) is unchanged.
def _unverified_passed_evidence(steps: list[dict]) -> str | None:
filter_steps = [step for step in steps if str(step.get("logical_step_id", "")).endswith("apply_native_filter")]
if not filter_steps:
return "no apply_native_filter step in the step report"
for step in filter_steps:
if step.get("status") != "passed":
return f"apply_native_filter step {step.get('logical_step_id')} status={step.get('status')}"
if not str((step.get("step_outcome") or {}).get("sha256") or "").strip():
return f"apply_native_filter step {step.get('logical_step_id')} has no durable sha256 evidence"
screenshot_steps = [step for step in steps if str(step.get("logical_step_id", "")).endswith("capture_screenshot")]
if not screenshot_steps or any(step.get("status") != "passed" for step in screenshot_steps):
return "capture_screenshot did not pass"
return None
# #endregion ScenarioExecution.LiveMcpReplay.PassEvidence
# #endregion ScenarioExecution.LiveMcpReplay.Chain
@@ -360,9 +407,14 @@ def main() -> None:
parser = argparse.ArgumentParser(description="T029m live external MCP replay (optionally baseline-pinned).")
parser.add_argument("--baseline", action="store_true",
help="compile with the baseline capability and start with the published-catalog selector")
parser.add_argument("--expect-passed", action="store_true",
help="accept a passed terminal only with durable apply_native_filter + screenshot evidence")
parser.add_argument("--filter-value", action="append", default=[],
help="repeatable launch param filter_values entry (UX-6)")
args = parser.parse_args()
try:
evidence = run_chain(baseline=args.baseline)
evidence = run_chain(baseline=args.baseline, filter_values=args.filter_value or None,
expect_passed=args.expect_passed)
evidence["result"] = "ok"
except ReplayError as exc:
evidence = {"result": "blocked", "error": str(exc)}

View File

@@ -330,6 +330,39 @@ receipts, cancellation/reconciliation, mutation policy, readiness checks and man
not claim runtime implementation: T028-T034, T040-T042, T042b and a real PREPROD canary remain required before
the BrowserProvider can be called production-ready.
**Update — apply_native_filter read-only action (UX-1, 2026-09-12).** The isolated browser transport now
admits `apply_native_filter` as a read-only UI interaction (`ScenarioExecution.BrowserProvider.NativeFilter`,
`browser_native_filter.py` + `browser_transport.py` + provider admission in `browser.py`): after
`open_dashboard` the filter bar is located through multi-strategy locators (pinned `selector_hint`, then
`filter_id|filter_name|column`, then first visible control), typed `values` are selected through the UI and
applied, empty input applies the current filter state (`applied_mode=current_state`), the flow waits for a
bounded chart settle (`_wait_for_charts_stabilized`) and produces `filter_applied`/`charts_settled`
checkpoints with post-action PNG evidence. Fail-closed: any locator miss is a typed
`BROWSER_SELECTOR_NOT_FOUND` → `inconclusive` without retry; unknown `wait_state` →
`BROWSER_WAIT_STATE_INVALID`; invalid input is rejected before any browser I/O; the mutation catalog
(`row_edit`/`bulk_edit`) is unchanged and a mutating `apply_native_filter` descriptor is still
`BROWSER_ACTION_NOT_SUPPORTED`. The 038 registry version/hash is unchanged — the action descriptor was
already registered; only provider/transport readiness closed the live `BROWSER_ACTION_NOT_SUPPORTED` gap
(see `docs/reports/ux-flow-improvement-plan-2026-09-11.md` §UX-1). Live canary re-run and index rebuild
remain the acceptance evidence and are tracked separately.
**Update — run parameter `filter_values` binding (UX-6, 2026-09-12).** Launch params now bind into the
pinned program at RunnerPlan derivation (`ScenarioExecution.RunnerPlan.BindParams`, wired from
`start_run`): when the start request's typed `params` carry `filter_values` (MCP `start_scenario_run`
`params` and REST `POST /api/scenario-runs` `params` — the UI panel submits string-typed rows, so a bare
string binds as a single value), every pinned browser `apply_native_filter` step is stamped with
`param_binding.filter_values` and the provider merges it over descriptor-pinned values
(`resolve_native_filter_input`) before typed validation and any browser I/O. Fail-closed rules: the
parameter is validated by the same provider contract (`validate_native_filter_input` — bounded string
list); a present-but-invalid value rejects the start with `BROWSER_FILTER_VALUES_INVALID` before any
ScenarioRun/gate row exists — the requested filter is never silently dropped, because a silent
current-state fallback would fabricate a PASS for a filter that was never applied. An absent (or empty)
parameter leaves steps unbound and the step runs in the UX-1 `current_state` observe mode. The binding
is part of the derived plan (plan hash covers it, determinism is per revision+params); the 038
`action_descriptor` snapshots and the graph's declarative input refs stay byte-identical
(`validate_pinned_runner_plan` unchanged). `selector_hint` merge authority (UX-1) is unchanged: an
explicit descriptor hint still wins, the pinned description hint still fills first-candidate locators.
## Drift Amendment — MCP Interface (2026-08-24)
- Unaffected structurally: ScenarioRun, executors, capacity and gates never referenced the chat runtime. MCP clients author scenarios before runs (038 chain) and investigate after terminal signals (047 cases) through governed tools only; `manual_run_only` and PROD gating apply regardless of the actor.
@@ -375,8 +408,81 @@ Exploratory Playwright/code sandbox activity is authoring-only and must remain i
**Frontend boundary (user decision 2026-09-08)**: All agent interaction is external MCP only. Product frontend MUST NOT contain agent chat, prompt/request textarea, assistant editing, typical-operation-to-agent selector, proposal-generation, agent workspace/start or handoff controls/routes. Ordinary manual CRUD/editor, human approval/review, monitoring and read-only evidence/evaluation are permitted. AgentEvaluationCard is read-only, with no prompt/retry-agent/provider controls. Existing agent proposal UI is runtime drift; removal/negative DOM-route-network acceptance remains OPEN in this spec-only change.
**Removal note (UX-11, 2026-09-12)**: product-surface agent entries to `/agent` removed — `DashboardHeader.svelte` (`agentHref`/`scenarioHref` derived routes + «MCP» / «Create scenario» buttons), `TopNavbar.svelte` assistant handoff button, `routes/datasets/[id]` "Ask AI" link, `DashboardDetailModel.scenarioHref`. Read-only UX-5 registry link (`/dashboard-testing/scenarios?dashboard_id=`) preserved. Negative DOM-route-network acceptance enforced by `Test.DashboardTesting.DashboardHeaderEntry`: no `a[href*="/agent"]`, no prompt/textarea/proposal semantics, zero fetches to `/agent`|`/api/agent*`|`/api/assistant*` from the rendered header (UX-5 positive tests remain). Full route-level removal (`/agent` route files, `ROUTES.agent` registry entry, legacy chat surfaces and their pre-existing orphan i18n keys) remains OPEN — dedicated 050-legacy task.
**SCEX-FR-028 — Production baseline-backed evaluation**: Run admission, canonical request hash, RunnerPlan, artifacts, immutable comparisons/evaluations, policy outcomes and result/SSE MUST satisfy production-chain.md, artifact-content.openapi.yaml and result-evidence.schema.json. Provider loop startup/shutdown, protected GET/HEAD, baseline resolution/pinning and cancellation/reconcile are mandatory; screenshots currently exist as reusable adapters but missing deployment/evaluation/content integration is not complete. 038 DecisionPolicy is sole semantic outcome mapper.
Normative contract: [Production baseline-backed evaluation](contracts/production-chain.md). New requirements are specified, **implemented=false / acceptance OPEN** until executable evidence closes the linked tasks/checklist/traceability rows. Historical local tests and the manual inconclusive ss-prod run do not prove browser/capture/baseline/LLM production readiness. The refresh scope is the audited P0/P1/P2 agentic E2E and baseline gaps; an approved ExecutionPerformanceBaseline is not introduced.
## Field-run Amendment — comparison executor typed failure boundary (2026-09-12)
> Источник: полевой ss-prod B01-прогон, run `6de8d0d9-ab9e-4189-93af-b23318421002` (dashboard 11,
> baseline_pin ss-prod-visual/v1), evidence `/tmp/kilo/live_ux_baseline.json` (UX-2 residual).
> Статус: **IMPLEMENTED 2026-09-12** — `ScenarioExecution.OfflineExecutors.Assertion`,
> `ScenarioExecution.Dispatch.Step`; тесты `test_scenario_compare_to_baseline_typed_failure`,
> `test_scenario_queued_dispatch`.
1. **Root cause (полевой факт):** шаг `compare_to_baseline` компилируется с
`Expected(kind="baseline_ref", ref="baseline.default", ...)` (`ScenarioGraph.Compiler.BuildStep`).
Executor `assertion` передавал этот dict в `_as_normalized` → `NormalizedValue.model_validate` →
pydantic `ValidationError` (`kind` не входит в `ValueKind`, `extra="forbid"`). Исключение вышло
за `dispatch_step` и попало в blanket-`except` диспетчера очереди (`dispatch_queued_runs`),
который закрыл ВСЕ шаги и ран как `inconclusive QUEUED_DISPATCH_ERROR` без диагностики. Прогон:
`apply_native_filter` passed, затем шаг сравнения — typed `inconclusive QUEUED_DISPATCH_ERROR`
без деталей; ран целиком `inconclusive QUEUED_DISPATCH_ERROR`.
2. **Фикс — типизированный fail-closed на границе шага (без синтетики):** (а) executor `assertion`
распознаёт скомпилированный `baseline_ref` и возвращает typed inconclusive
`BASELINE_EVIDENCE_UNAVAILABLE` (существующая D11-таксономия кодов baseline из
`ScenarioExecution.BaselineResolver`); 037-сравнение с опубликованным baseline-артефактом
остаётся отдельным срезом — executor-граница не получает baseline-байты, фабрикация
PASS/FAIL из голой ссылки запрещена. (б) `ScenarioExecution.Dispatch.Step` содержит ЛЮБОЕ
исключение executor-вызова на границе шага: typed inconclusive `EXECUTOR_STEP_ERROR`
(payload несёт `exception_type`); зависимые шаги следуют существующей failure policy
(`inconclusive`-продюсер её не блокирует, `failed/blocked` — блокируют), ран честно
терминалируется агрегацией, повторный dispatch-тик — durable no-op.
3. **Инварианты не тронуты:** `_close_queued_dispatch_error` (`QUEUED_DISPATCH_ERROR`) остаётся
только для реальных ошибок инфраструктуры диспетчера; ошибки разрешения исполнителя
(`registry.resolve`) и preflight-валидация плана идут прежними путями
(`_reject_malformed_plan` / инфраструктурное закрытие). `NormalizedValue` остаётся строгим —
нормализация не «смягчена» ради обхода дефекта.
4. **Наблюдаемое поведение после фикса:** недоступное/невозможное сравнение `compare_to_baseline`
→ шаг `inconclusive BASELINE_EVIDENCE_UNAVAILABLE`, DecisionPolicy row 7
(`COMPARISON_INCONCLUSIVE`, штамп policy в step_outcome сохраняется), ран терминален
(`inconclusive`, `run.error_code` пуст). Тесты: `test_scenario_compare_to_baseline_typed_failure`
(unit executor, dispatch-терминация, граница исключения, неизменность literal-сравнения);
`test_scenario_queued_dispatch::test_dispatch_executor_exception_is_typed_inconclusive_not_dispatch_error`
(обновлён с crash-поведения на typed-границу).
## Decision Amendment — BrowserProvider run-scoped session & filter-state continuity (2026-09-12, DG-1)
> Источник: решение пользователя по DG-1 (`docs/reports/ux10-production-ops-plan-2026-09-12.md`).
> Формулировка цели (verbatim): «мне нужно чтобы native filters сохранялись — в этом вся суть
> проверки дашборда — зафиксировать эталонное состояние».
1. **Решение (принято):** BrowserProvider работает в модели **run-scoped session**: один
изолированный контекст на (run, lease) живёт на shared provider event loop через ВСЕ browser-шаги
рана; lease — per-run (pinned defaults 2 DEV/PREPROD, 1 PROD). Это реализация T034 дословно,
а не его ослабление.
2. **Назначение:** native filter state, применённый любым шагом рана, **сохраняется для всех
последующих шагов** — screenshot/extract/comparison-шаги наблюдают именно отфильтрованное
эталонное состояние дашборда. Per-step изоляция (контекст на каждый шаг) разрывала эту
непрерывность (доказано live: шаг `capture_screenshot` открывал дашборд заново без фильтра)
и создавала риск false-PASS для зависимых цепочек «фильтр → наблюдение». Эталонное
отфильтрованное состояние — также то, что baseline capture (037) и сравнение фиксируют.
3. **Crash discipline (SCEX-FR-007 без изменений):** после каждого browser-шага ран пинит
**browser-safe checkpoint** — полный reconstructible срез: `{dashboard_id, native_filter_state
(все применённые фильтры+значения), active_tab, wait_states}`. Мёртвый контекст НИКОГДА не
оживляется: recovery — новый контекст, `open_dashboard` + replay checkpoint, продолжение с
упавшего шага. Ран без declared checkpoint на момент сбоя → `inconclusive` (fail-closed,
текущая семантика сохраняется).
4. **Сериализация:** параллельные browser-шаги одного рана сериализуются на контексте рана
(детерминизм порядка; параллельные UI-действия в одной сессии — источник гонок).
5. **@REJECTED:** per-step изоляция как норма — отклонена (разрыв filter continuity, оверхед
login/launch на каждый шаг несовместим с ops-v1 SLO); warm context pool — отклонён T034
(утечка состояния между ранами). Per-step контекст остаётся только как recover-fallback
для ранов без browser-шагов в истории.
6. **Связь задач:** реализация — Wave C плана UX-10 (C1 раунды 1–4: session manager → каталог →
limits → mutation lease); приёмка — T042b PREPROD canary включая safe-checkpoint reconstruction
trace.
#endregion ScenarioExecution.Spec

View File

@@ -99,29 +99,72 @@
`ProviderExecutionResult` schemas in `backend/src/services/dashboard_testing/execution/provider_protocol.py`.
- [x] T029 [P] [US2] Add provider ownership receipts and atomic evidence commit contract in
`backend/src/services/dashboard_testing/execution/provider_evidence.py`.
- [ ] T030 [P] [US2] Add provider operation lifecycle, cancellation and reconciliation contracts in
- [x] T030 [P] [US2] Add provider operation lifecycle, cancellation and reconciliation contracts in
`backend/src/services/dashboard_testing/execution/provider_operations.py`.
**Status (2026-09-12): done.** `ScenarioExecution.ProviderOperations.Service.Cancel`:
`cancel_provider_operation(db, operation_id, reason, acknowledge)` — durable cancel request,
acknowledge ∈ {stopped, completed, unknown}; terminal receipts immutable (late cancel is a
no-op); `unknown` → `reconciliation_required` (retry/PASS blocked until ReconcileWorker,
escalation unknown→stopped allowed); idempotent repeat (single cancel_request history entry);
`cancellation_requested` + `cancellation_deadline_at` stamped. Receipts coverage extended:
screenshot provider (open-before-I/O inside capacity claim, typed terminal finalize on all
paths) and superset/sql_evidence via `live_binding.py` (open best-effort, never masks the
adapter outcome). Tests: `test_provider_operations.py` (19, incl. ReconcileWorker integration),
`test_provider_screenshot.py` (14), `test_live_execution_binding.py` (13, binding receipts
region). Remaining: browser cancel I/O hook — B2 phase (after C1 rounds).
- [x] T031 [P] [US2] Add provider liveness/readiness/dependency health and redacted telemetry contract in
`backend/src/services/dashboard_testing/execution/provider_health.py`.
- [ ] T032 [US2] Integrate atomic shared `ExecutionCapacityManager` admission, heartbeat, expiry, release
- [x] T032 [US2] Integrate atomic shared `ExecutionCapacityManager` admission, heartbeat, expiry, release
and reconciliation with dispatcher claims in `backend/src/services/dashboard_testing/execution/capacity.py`.
@INVARIANT: no provider I/O without a capacity lease; retries claim a new lease.
Include the `ProviderRuntime` contract (`provider_runtime.py`): one application-owned long-lived
event-loop thread for async transports, bounded submissions, and the pinned concurrency defaults
(2 browser contexts DEV/PREPROD, 1 PROD; screenshot capture on the same lease accounting).
**Status (2026-09-13): done.** Дельта закрыта на существующей step-level модели: (1) dispatcher tick
drives `reconcile_expired_leases` перед admission (упавший воркер освобождает квоту каждым тиком);
(2) heartbeat перед loop-submission в browser/screenshot провайдерах — потерянный/истёкший lease →
typed `*_CAPACITY_UNAVAILABLE` БЕЗ I/O (walker паркит run через CAPACITY_BLOCKED); (3) DG-1/B-wave
финализатор: `close_run_sessions` — терминальный ран (dispatch) и cancelled ран (cancel_lifecycle)
закрывают run-scoped browser session; weak manager registry в `browser_session.py`. Run-level lease
НЕ введён: env-квота per workload_class уже даёт pinned defaults (browser PROD=1/DEV=2); дублирующий
run-slot уровень отклонён как конфликтующий с этой моделью (@REJECTED в capacity intake).
Тесты: `test_dispatch_capacity_lifecycle.py` (6: expiry reconcile, heartbeat order+refusal оба
провайдера, terminal finalizer, waiting_human skip, cancel close). Регрессия 697 passed.
- [x] T033 [P] [US2] Add per-provider input/output schemas, limits and error taxonomy in
`backend/src/services/dashboard_testing/execution/provider_contracts.py` for browser, Superset API,
SQL evidence, XLSX, assertion, transform, screenshot, report and artifact.
Include bounded verification response bytes/rows/cells/canonicalization time: oversized output is
`RESULT_TOO_LARGE` + inconclusive and can never be truncated into PASS or a baseline update.
- [ ] T034 [US2] Implement BrowserProvider resource ownership, safe-checkpoint replay, cancel and
- [x] T034 [US2] Implement BrowserProvider resource ownership, safe-checkpoint replay, cancel and
reconciliation adapter in `backend/src/services/dashboard_testing/execution/providers/browser.py`.
Required action catalog: open_dashboard, navigate_tab, apply_native_filter, inspect_filter_state,
apply_table_filter, extract_table, scroll_to, inspect_columns, click, select_rows, edit_row,
bulk_edit, download, refresh, wait_for_state. Required limits: 120s context/auth, 30s action, 3
pages, 25 MiB downloads, 10 MiB screenshots; mutation actions require fixture lease and cleanup.
Context/session objects are bound to the shared provider event loop and stay live across all steps
of one run; no per-step loop creation and no warm context pool.
**Status (2026-09-13): done (rounds 1–4).**
Round 1 (2026-09-12): run-scoped session (`browser_session.py` BrowserSessionManager:
acquire/reuse/replay/close, leak-guards), admission split (`browser_admission.py`), receipts
(`browser_receipt.py`), checkpoint stamped into step_outcome (`browser_checkpoint` with
native_filter_state — DG-1), fail-closed `BROWSER_CHECKPOINT_MISSING`.
Round 2 (2026-09-12): полный read-only каталог — navigate_tab, inspect_filter_state,
apply_table_filter, extract_table, scroll_to, inspect_columns, click, select_rows, download
(`browser_readonly_actions.py`, typed `BROWSER_*_INVALID`, selector miss typed без retry,
extract bounded 10k×100/10 MiB, download bounded 25 MiB дважды). Дескрипторы 038 уже
существовали — registry не менялся.
Round 3 (2026-09-13): limits enforcement — context/auth bound 120s (`asyncio.wait_for` в
open_session → typed BROWSER_ACTION_TIMEOUT), session page bound 3 (extras закрываются,
потерянная driving page → typed `BROWSER_PAGE_LOST`); 30s action / 25 MiB download / 10 MiB
screenshot уже были. Тесты: `test_browser_limits_cleanup.py`.
Round 4 (2026-09-13): mutation fixture cleanup — закрытый словарь cleanup_policy
{restore_fixture, retain} (валидация в `validate_mutation_contract`); restore_fixture
исполняется в той же сессии (`build_cleanup_script` — revert из pre-image, verify
restored_hash == precondition_hash, checkpoint `fixture_restored`); неудача → typed
`BROWSER_MUTATION_CLEANUP_FAILED`, inconclusive, receipt `reconciliation_required` с
effect_state=completed (мутация известна, окружение грязное; retry только после reconcile).
Cancel: dispatch-финализатор `close_run_sessions` на терминале + cancel_lifecycle (T032 wave).
Original text: Required action catalog: open_dashboard, navigate_tab, apply_native_filter,
inspect_filter_state, apply_table_filter, extract_table, scroll_to, inspect_columns, click,
select_rows, edit_row, bulk_edit, download, refresh, wait_for_state. Required limits: 120s
context/auth, 30s action, 3 pages, 25 MiB downloads, 10 MiB screenshots; mutation actions
require fixture lease and cleanup. Context/session objects are bound to the shared provider
event loop and stay live across all steps of one run; no per-step loop creation and no warm
context pool.
- [x] T035 [US2] Implement ScreenshotProvider atomic durable evidence adapter with cleanup and ownership
receipts in `backend/src/services/dashboard_testing/execution/providers/screenshot.py`, wrapping the
existing llm_analysis Playwright capture stack; capture runs on the shared provider event loop under
@@ -141,23 +184,36 @@
Evidence: live canary v2 run `4eebfab3` — real qwen3.8-flash evaluation, `AgentEvaluation` row
persisted, DecisionPolicy row 11 `LOW_CONFIDENCE` (rows 9–14 reachable); trace
`docs/reports/agentic-runtime-live-canary-v2-2026-09-10.md`.
- [~] T040 [US2] Add startup deployment registration and readiness preflight for all provider capabilities.
- [x] T040 [US2] Add startup deployment registration and readiness preflight for all provider capabilities.
**Status (2026-09-10): implemented and live-verified; deployment-evidence rule partially met.**
**Status (2026-09-12): deployment-evidence rule closed.** Readiness payload now carries
provider/version (`ACTION_REGISTRY_VERSION`), capability fingerprint
(`action_registry_fingerprint()` + sorted feature list) and redacted dependency diagnostics
(booleans/binding counts only — no credentials/paths/SQL, SCEX-FR-022) for superset/screenshot/
browser entries in `execution/providers/preflight.py`; fail-closed admission of unready providers
proven by `test_unready_provider_admits_zero_new_operations` +
`test_degraded_readiness_blocks_browser_capability_admission`
(`tests/services/dashboard_testing/registry/test_provider_preflight.py`); `/api/ready` schema
stability pinned in `tests/api/test_ready.py::test_ready_providers_block_schema_stable`.
Actual modules: `execution/live_composition.py` + `execution/providers/preflight.py` + `app.py`
lifespan (the `provider_bootstrap.py` path is stale). Live `/api/ready`: provider_loop/evidence_storage/
browser/screenshot = ready, bindings registered = 1; the browser-probe startup deadlock is fixed
(`ba2f1f45`). Remaining per the deployment-evidence rule: provider/version + capability fingerprint +
redacted dependency diagnostics are not yet part of the readiness payload.
(`ba2f1f45`).
- [x] T041 [US2] Add common provider contract tests for unavailable/dependency failure/capacity
exhaustion/timeout/cancel/duplicate/late response/ownership mismatch/malformed result/cleanup and
reconciliation in `backend/tests/services/dashboard_testing/registry/test_provider_contract.py`.
- [~] T042 [US2] Add provider-specific contract tests in `test_provider_*.py` and real deployment health
checks for Superset, Browser and Screenshot bindings.
**Status (2026-09-10): partial.** Browser/Screenshot/preflight/contract tests present and green
(`test_provider_browser.py`, `test_provider_screenshot.py`, `test_provider_preflight.py`,
`test_provider_contract.py`); live health proven for browser/screenshot plus Superset-provider
composition (canary v1 `597274d3`, canary v2 `4eebfab3`). Missing: a Superset-specific contract test
and one live Superset query through the binding (canary graphs used browser+screenshot+evaluation).
**Status (2026-09-12): offline contract suites complete; live Superset query remains.** NEW
`test_provider_superset.py` (14 tests, `Test.ScenarioExecution.SupersetProviderContract`): typed
success (receipt completed, principal/RLS fingerprints on receipt identity, owned
`draft:{run}:{sha256}` evidence ref), identity fail-closed matrix (missing/invalid/mismatch/
unavailable → typed inconclusive, spy client never called, no receipt), external taxonomy
(SupersetAPIError→failed, exception→inconclusive+receipt failed/unknown, timeout→re-raise with
receipt finalized first), evidence digest/ref mismatch → inconclusive nothing stored, receipt
open-before-I/O ordering, duplicate open never masks outcome, sql_evidence shared bound path.
Remaining: one live Superset query through the binding (D-wave live, ss-prod binding exists) и
явный RESULT_TOO_LARGE bound для query responses — gap зафиксирован (см. ux10 план, remaining).
- [x] T042c [US2] Add dispatcher policy/binding revalidation tests in
`backend/tests/services/dashboard_testing/registry/test_provider_contract.py`: reclassification to PROD
or a changed provider/security fingerprint after approval prevents all provider I/O and returns typed

View File

@@ -4,7 +4,7 @@
|-------|-------------|-------|------------------|----------|------|------|---------------------|
| US1 Start | SCEX-FR-001/008 | ScenarioRun | scenarioRun.start | Execution.Start, Execution.Runner.QueuedDispatch | T006-T008, T024 | test_runner, test_scenario_queued_dispatch, test_scenario_scheduler_callbacks, test_scenario_runs_api, test_scenario_automation_api, test_live_execution_binding | `[~]` HTTP/automation start and replay persist queued/pending rows without request-time dispatch; only the scheduler composition's durable queued->running CAS walks its winner. Fixed scheduler callback registration, database-edge containment, and repeat-tick terminal side-effect idempotency are unit-proven. Automated human plans are rejected before they reach CAS/walker; a manual human graph reaches HumanCheckpoint only after that dispatcher claim. Approval-to-real live dispatch remains unproven. |
| US2 Dispatch | SCEX-FR-002/006/009 | ScenarioStepRun, LiveExecutionBinding | scenarioRun.step | Execution.Dispatch, Execution.LiveCompositionRoot | T009-T011, T024 | test_dispatch, test_scenario_executors, test_live_execution_binding | `[~]` Browser/Superset/Screenshot use fail-safe typed adapter boundaries: no explicit adapter success means no PASS; invalid evidence digest/ref remains inconclusive. Lifespan bootstraps `settings.scenario_live_execution_bindings` through the existing `SupersetClient`, exact model and durable storage, so configured Superset dispatch invokes 037 and stores the exact raw-byte digest/ref; mismatched/unavailable providers make no I/O call. Browser safe-checkpoint and Screenshot durable-evidence registration are supported but no provider is deployed, so enabled bindings return stable configured-unavailable codes. |
| BrowserProvider | SCEX-FR-016..023 | BrowserProviderActionContract, BrowserOperationReceipt, BrowserEvidenceReceipt | scenarioRun.step | ScenarioExecution.BrowserProvider | T028-T034, T040-T042, T042b | test_provider_browser_*, Browser PREPROD canaries | `[~]` Contract completeness is **90/100**: all actions, risk classes, limits, lifecycle, checkpoint/recovery, ownership, cancellation/reconciliation, readiness and canary gates are specified. Runtime provider, shared capacity integration, deployment registration and real canary evidence remain open. |
| BrowserProvider | SCEX-FR-016..023 | BrowserProviderActionContract, BrowserOperationReceipt, BrowserEvidenceReceipt | scenarioRun.step | ScenarioExecution.BrowserProvider | T028-T034, T040-T042, T042b | test_provider_browser_*, test_browser_native_filter, Browser PREPROD canaries | `[~]` Contract completeness is **90/100**: all actions, risk classes, limits, lifecycle, checkpoint/recovery, ownership, cancellation/reconciliation, readiness and canary gates are specified. `apply_native_filter` is implemented read-only in the provider/transport (UX-1, 2026-09-12): typed input merge/validation before I/O, filter-bar UI automation with `filter_applied`/`charts_settled` checkpoints and PNG evidence, typed `BROWSER_SELECTOR_NOT_FOUND`/`BROWSER_WAIT_STATE_INVALID` inconclusive without retry, mutation catalog unchanged (`test_provider_browser.py`, `test_browser_native_filter.py`). Launch-param binding `filter_values` -> `apply_native_filter` input `values` is closed at RunnerPlan derivation (UX-6, 2026-09-12, `ScenarioExecution.RunnerPlan.BindParams`; fail-closed typed reject on a present-but-invalid param, absent param stays `current_state`). Runtime provider for the remaining catalog, shared capacity integration, deployment registration, live canary evidence and PREPROD canaries remain open. |
| US3 Human | SCEX-FR-004/010 | ScenarioRun(waiting_human) | scenarioRun.humanDecision | Execution.SuspendForHuman, Execution.Resume | T012-T014, T025 | test_human_resume, test_scenario_runs_api | `[~]` persisted HumanCheckpoint and infrastructure-resume continuations advance only the missing DAG frontier; completed steps are not re-run. Full live-composition closure remains pending. |
| Manual-only boundary | SCEX-FR-004a | ScenarioRun, HumanCheckpoint | — | Execution.Runner.Start, RunnerPlan.Derive | T027 | test_scenario_manual_run_only, test_scenario_automation_api | `[x]` Trusted scheduled/deploy/release/ETL/API origins reject persisted human revisions before idempotency or any run/gate/notification/queue side effect. Manual origin remains eligible; HumanCheckpoint is not an approval gate. |
| Automated human prohibition | SCEX-FR-004a/013, SCAUTO-FR-001/007 | RunnerPlan.manual_run_only | — | Execution.Runner.TriggerSource, Automation.Trigger | T027, 046 T016/T019 | test_scenario_manual_run_only, test_scenario_automation_api, test_scenario_automation_trigger | `[x]` Every trusted non-manual origin, including scheduled, deploy, release, ETL, API and background recovery, is rejected before idempotency/run/gate/notification/queue/dispatch side effects. Only authenticated manual origin may create a human-containing run. |
@@ -18,11 +18,12 @@
| Success criteria | SC-001 | RunnerPlan, ScenarioStepRun | — | ScenarioExecution.RunnerPlan.Derive, ScenarioExecution.Dispatch | T004-T011, T041 | canonical fixture order and duplicate-dispatch assertions | `[~]` Existing fixture/lifecycle coverage passes; exact 100% canonical dispatch evidence remains a release gate. |
| Success criteria | SC-002 | HumanCheckpoint, ScenarioRun | scenarioRun.humanDecision | ScenarioExecution.HumanCheckpoint, ScenarioExecution.Resume | T012-T014, T025 | human CAS/frontier tests | `[x]` Human checkpoint CAS and missing-frontier resume are verified; live provider closure remains separate. |
| Success criteria | SC-003 | ScenarioRun.cancel_drain_deadline_at | scenarioRun.cancel | ScenarioExecution.Cancel | T015-T016, T025, T041 | 100 cancellation trials plus scheduler finalizer | `[~]` Bounded drain is unit-proven; required 100-trial production evidence is open. |
| Success criteria | SC-004 | RunnerPlan, worker lease, operation receipt | scenarioRun.detail | ScenarioExecution.Runner.CrashRecovery, ScenarioExecution.ProviderOperations | T014c, T025, T030, T041 | safe/unsafe recovery and reconciliation tests | `[~]` Safe/unsafe persisted recovery is verified; provider operation reconciliation is not implemented. |
| Success criteria | SC-004 | RunnerPlan, worker lease, operation receipt | scenarioRun.detail | ScenarioExecution.Runner.CrashRecovery, ScenarioExecution.ProviderOperations | T014c, T025, T030, T041 | safe/unsafe recovery and reconciliation tests | `[x]` (2026-09-12) Safe/unsafe persisted recovery verified; provider operation reconciliation + operation-aware cancellation implemented (`cancel_provider_operation`: stopped/completed/unknown, unknown→reconciliation_required блокирует retry/PASS; идемпотентный; terminal receipts immutable) — `test_provider_operations.py` (19), `test_provider_screenshot.py` (14), `test_live_execution_binding.py` (13). |
| Success criteria | SC-005 | immutable run provenance | scenarioRun.result | Execution.RunnerPlan, Execution.Result | T017, T020, T041 | revision-edit immutability and evidence receipt tests | `[~]` Snapshot fields exist; complete receipt-level byte identity evidence is open. |
| Success criteria | SC-006 | ActionApprovalGate, ExecutorRegistry | scenarioRun.start | Execution.ActionApprovalGate, ScenarioExecution.ExecutorRegistry | T014d, T019, T041 | PROD no-call and human-not-executor tests | `[x]` Current no-call and registry rejection tests pass. |
| Success criteria | SC-007..010 | Provider protocol/health/capacity | — | ScenarioExecution.ProviderProtocol, ProviderOperations.Observability, CapacityManager | T028-T042 | common/provider-specific contract and deployment profiles | `[ ]` New production gate not implemented. |
| Success criteria | SC-007..010 | Provider protocol/health/capacity | — | ScenarioExecution.ProviderProtocol, ProviderOperations.Observability, CapacityManager | T028-T042 | common/provider-specific contract and deployment profiles | `[~]` New production gate partially closed. SC-010 closed (2026-09-12, T040): readiness snapshot with provider/version + capability fingerprint + redacted diagnostics. SC-008 closed (2026-09-13, T032): dispatcher tick reconciles expired leases; browser/screenshot heartbeat before loop submission with typed `*_CAPACITY_UNAVAILABLE` refusal (no I/O without a live lease); run-terminal/cancel session finalizer — `test_dispatch_capacity_lifecycle.py` (6), regression 697 passed. SC-009 closed (T039). SC-007 open: contract suites exist for all providers; T034 catalog rounds 3–4 (limits remainder, mutation fixture lease) pending. |
| Success criteria | SC-011 | release evidence record | — | ScenarioExecution.ProviderProtocol | T022, T041-T042 | full profiles, PostgreSQL and semantic audit outputs | `[ ]` Release evidence package is incomplete. |
| UX-2 Dispatch resilience | SCEX-FR-002/028 | ScenarioStepRun, DecisionPolicy | scenarioRun.step | Execution.Dispatch.Step, ScenarioExecution.OfflineExecutors.Assertion | T009-T011 | test_scenario_compare_to_baseline_typed_failure, test_scenario_queued_dispatch | `[x]` Field-run fix (2026-09-12, live ss-prod B01 run 6de8d0d9): a compiled `baseline_ref` expectation resolves typed inconclusive `BASELINE_EVIDENCE_UNAVAILABLE` (DecisionPolicy row 7 `COMPARISON_INCONCLUSIVE`), and any executor exception is contained at the dispatch step boundary as `EXECUTOR_STEP_ERROR`; the run terminalizes honestly, replay ticks are no-ops, and `QUEUED_DISPATCH_ERROR` stays infrastructure-only. 037 baseline-vs-pin comparison remains a later slice. |
N/A: Registry (042), Editor (043), Monitor UX (045), Automation (046), Analytics (047).
@@ -33,6 +34,7 @@ N/A: Registry (042), Editor (043), Monitor UX (045), Automation (046), Analytics
| persisted revision is the only launch program input | `ScenarioExecution.RunPreflight` | 042 `ScenarioRevision` -> preflight handle | `test_scenario_runner`, `test_scenario_automation_api` |
| server-owned preflight digest and failure boundary | `ScenarioExecution.RunPreflight` | blocks before plan, lease, dispatch or provider I/O | `test_scenario_runner`, policy/binding tests |
| deterministic plan and plan digest | `ScenarioExecution.RunnerPlan.Derive` | preflight -> pinned `RunnerPlan` -> run | `test_scenario_runner_plan`, `test_scenario_dispatch` |
| launch param `filter_values` binds into pinned `apply_native_filter` (UX-6) | `ScenarioExecution.RunnerPlan.BindParams` | start `params` -> pinned plan `param_binding` -> provider `resolve_native_filter_input` -> typed `values`; absent param -> `current_state`; invalid param -> typed start reject | `test_scenario_runner_plan`, `test_browser_native_filter`, `test_provider_browser` |
| queued/pending approval continuation | `ScenarioExecution.Start`, `ScenarioExecution.ActionApprovalGate` | approved pending run -> queued; denial/expiry -> blocked | `test_scenario_queued_dispatch`, `test_scenario_automation_api` |
| ScenarioRun snapshot and idempotency | `ScenarioExecution.Start` / ScenarioRun model | exact revision/content/target/principal/binding snapshot | `test_scenario_runs_api`, `test_scenario_scheduler_callbacks` |
| evidence ownership and authoritative StepOutcome | `ScenarioExecution.ProviderProtocol`, `ScenarioStepRun` | provider receipt -> `Artifact(owner_type=scenario_run)` -> StepOutcome | `test_provider_contract`, `test_scenario_terminal_signals` |

View File

@@ -17,6 +17,9 @@ paths:
- { name: owner, in: query, schema: { type: string } }
- { name: trigger, in: query, schema: { type: string } }
- { name: waiting_for_me, in: query, schema: { type: boolean } }
# UX-7 amendment 2026-09-12: read-only pending_approval projection (badge total);
# same listing visibility as waiting_for_me — approval decisions still require RUN_PROD.
- { name: pending_approval, in: query, schema: { type: boolean } }
- { name: page, in: query, schema: { type: integer, default: 1 } }
- { name: page_size, in: query, schema: { type: integer, default: 50 } }
responses:

View File

@@ -38,6 +38,17 @@ RunConfiguration sends explicit baseline_set_id/version selectors; server pin is
Normative detail: [Unified result projection](contracts/evidence-ui.md). New fields, CAS transitions and cross-record integrity checks are implemented=false until [production tasks](tasks.md) and [traceability](traceability.md) close with executable evidence. Existing shorter field lists are legacy compatibility projections, not permission to omit the production identity fields.
## Waiting badge projection — 2026-09-12 (plan UX-7)
`GET /api/scenario-runs` gains `pending_approval=true` — a server-computed projection narrowing the
same listing to runs in status `pending_approval` (the durable PROD gate created atomically with
the run, 044). It composes with the existing filters and returns the standard `{items, total}`
envelope. `waiting_for_me` semantics are unchanged (pending HumanCheckpoint only). Visibility
follows the listing permission (`scenario:RUN`): the projection is a read-only aggregate; approval
mutations remain behind `scenario:RUN_PROD`. The sidebar store
(`Stores.ScenarioRuns.WaitingStore`) polls both projections bounded (`page_size=1`, total-only),
sums them for the badge, and fails closed (leave-last-value, disable on 401/403).
Frontend contains no agent interaction state, prompt, proposal-generation or workspace/start controls. Human manual editor/review state and read-only evaluation/result artifacts are separate from external MCP authoring. Approved performance baseline is outside scope.
#endregion ScenarioRunMonitor.DataModel

View File

@@ -37,6 +37,38 @@ Analyst opens the persistent launch panel, picks PREPROD + r17 + v31 baseline +
- Queue is a navigation destination from the result and Operations Center; it does not become a modal or start case work before analyst selection.
- Compare different revisions → warning surfaced.
## 6. Terminal Reason Banner (amendment 2026-09-12, plan UX-4)
**@UX_STATE**: terminal (failed/cancelled/blocked/inconclusive) → typed reason banner rendered
above ScenarioResultView. Source: `ScenarioRunMonitor.TerminalReasons` (`terminal-reasons.ts`) —
display-only mapping of persisted codes, no new queries.
- Known product gaps (`BROWSER_ACTION_NOT_SUPPORTED`, `AUTOMATION_INELIGIBLE_HUMAN_STEP`) →
info severity, body explains the gap, next step references the improvement plan / manual-run rule.
- Configuration family `*_BINDING_(MISSING|UNAVAILABLE|INVALID|MISMATCH)` (prefix regex) →
warning severity, next step: administrator / `scenario-live-bindings`.
- Unlisted code → fallback: code verbatim in `<code>` + "typed non-pass — результат не синтезирован";
no fabricated next step (`nextKey` optional).
- Destructive tokens are never used for the banner (a typed non-pass is honest status, not an error
of the user). Audit continuity: banner reads `run.error_code`, else first `failures[].error_code`;
it never mutates or masks the persisted codes.
## 7. Sidebar Decision Badge (amendment 2026-09-12, plan UX-7)
The RUNMON-FR-009 sidebar badge «Ожидают моего решения» now sums two server-computed projections
of the Global Run Operations Center listing (`GET /api/scenario-runs`, both bounded `page_size=1`,
only `total` consumed — `Stores.ScenarioRuns.WaitingStore`):
- `waiting_for_me=true` — runs with a pending HumanCheckpoint (semantics unchanged);
- `pending_approval=true` — runs in status `pending_approval` (PROD gate awaiting approve,
including scheduled runs).
The label stays semantically correct: both projections are decisions the analyst owes. Visibility
follows the listing permission (`scenario:RUN`); the approve decision itself still requires
`scenario:RUN_PROD` (MCP `list_pending_approvals`/`decide_approval`, `service_allowed=False`) —
the badge grants nothing, it makes the gate discoverable. This is a UI projection only and does
not duplicate the 046 T017 notification contour.
## 5. Tone & Voice
- **Style**: Concise, technical. **Terminology**: "Scenario run" distinct from "Release verification" and "Load tests".

View File

@@ -2,12 +2,19 @@
@BRIEF Identical REST/MCP admission, durable trigger identity, retention holds and measured rollout.
@RATIONALE Disabled invalid rules are still durable automation configuration and must pass the same eligibility checks as enabled rules.
@REJECTED REST disabled-config bypass, anonymous operational reads, independent provider quotas and deletion of baseline-held artifacts.
Status: normative target, implemented=false; DEF-02/SEC-01 remain open.
Status: normative target, partially implemented; DEF-02/SEC-01 CLOSED 2026-09-12 (T019, DG-2):
reads require scenario:automation READ + per-object ownership ACL with REST/MCP semantic parity
(401 unauthenticated / 403 permission_denied / 404 not_found == missing); disabled HumanStep
schedule rejects identically enabled/disabled on both surfaces with zero side-effect rows.
Evidence: backend/tests/api/test_scenario_automation_api.py (TestScenarioAutomationReadAcl,
TestScenarioAutomationAnonymousReads), backend/tests/test_mcp_automation_parity.py.
SLO profile and rollout gates below remain open.
All schedule/trigger create/update/enable/disable and direct triggers invoke one service policy; user/MCP/service principals are resolved live. A human-containing, candidate/unactivated, archived, invalid-context or unsupported-provider revision is ineligible even enabled=false; rejection precedes any schedule/gate/run/notification/queue write. Delete of an owned existing invalid rule remains allowed for cleanup. REST and MCP return identical semantic code, field errors and side-effect counts.
Six reads (schedules, trigger rules, policies, notifications, metrics, retention) require authenticated VIEW/READ and per-object scenario/dashboard/environment ACL; anonymous→401 even empty, lacking permission→403, foreign object→404. Pagination never leaks totals of inaccessible objects.
Rules pin baseline_set_id/version and target/revision policy. A due event atomically resolves activated revision+BaselineSelectionPin, verifies eligibility and stores immutable canonical request hash plus source_event_id; duplicate events/restarts reuse the same run/gate. Changed baseline set or canonical request under same idempotency key→409. Candidate save never changes a schedule target. PROD eligible run is pending_approval until exact request/binding/policy approval. Transition record, queue and notification outbox are one transaction; dispatcher queued→running CAS is sole I/O authority.
Note (2026-09-12, T020): due-admission atomics verified executable. BaselineSelectionPin resolves inside the start boundary before the idempotency lookup and is stamped atomically into target_snapshot/runner_plan; the PROD gate binds the same canonical request hash; a changed resolved pin under one idempotency key raises typed IDEMPOTENCY_KEY_REUSED (REST 409). Crash/replay/concurrent due events converge to exactly one run+gate (unique idempotency key + single caller-owned admission transaction; replay-side gate repair rejected as a side-effecting replay). Run transition+047 queue signal+notification outbox are flush-only effects of the tick's one transaction. A pending_approval run is never a dispatcher CAS candidate before gate approval. Schedule/rule source identity remains the deterministic idempotency key (no new columns per D11). Evidence: backend tests Test.ScenarioExecution.ScheduledStart and Test.ScenarioExecution.DueAdmissionAtomicity (registry suite 466 passed).
Use existing APScheduler semantics; no second scheduler. Persistent retries and restart must preserve source identity. Capacity is shared across ScenarioRun/AgentRun/VerificationRun/LoadRun and bounded globally per environment.
Retention defaults: screenshots/heavy artifacts30d, raw VLM7d, step metrics90d, run metadata180d, audit365d; approved-baseline references and active operations add holds. Deletion is mark→eligible-after-all-holds→delete bytes→verify absent→tombstone/audit; retry same deletion ID idempotently. Failure remains deletion_pending; never report deleted while bytes survive. Preserve analytics minimum window independent of metadata pruning.

View File

@@ -158,7 +158,13 @@ the automation workflow is not wired end-to-end.
manual-run-only pre-create with no run-side effect. HTTP/scheduler trigger paths persist only
queued or server-gated pending_approval rows; separate 044 queued->running CAS dispatch is the
sole initial adapter authority and excludes pending gates. No long-running event-subscriber proof
is yet available.
is yet available. **Update (2026-09-12):** the deploy/release/ETL event path is now pinned by
hardcoded-fixture tests over the real 044 start boundary (18 passed in
`test_scenario_automation_trigger.py`): pinned/current revision runs, disabled/mismatch skip with
zero durable rows, capacity (incl. shared across rules of one event), PROD event →
pending_approval + durable gate, dedup window across recent completed runs, and idempotent event
redelivery. Two supply defects found by these tests (completed-run dedup invisibility;
same-dispatch capacity blindness) are fixed in `automation/trigger.py`.
- `[ ]` `notify()` only appends to a caller-provided list and is not connected to run, stale or repeated
failure lifecycle events.
- `[~]` Scheduler startup reloads enabled `ScenarioSchedule` rows and registration forwards their

View File

@@ -15,13 +15,20 @@
## Phase 2 — US1 Schedule & Trigger
- [~] T003 [US1] Write failing schedule/trigger tests (trigger semantics contract coverage:
- [x] T003 [US1] Write failing schedule/trigger tests (trigger semantics contract coverage:
`backend/tests/services/dashboard_testing/registry/test_scenario_automation_trigger.py` —
**Status (2026-09-11):** cron-путь закрыт live+тестами (T018 CLOSED; `test_scheduled_scenario_start.py`
6 passed: double-fire → 1 run+gate, human-revision reject, superseded-pinned reject, сигнатурный
parity REST/MCP); trigger-rules (deploy/release/ETL-события) — остаток этой задачи.
filename differs from plan; coverage closes the task: event->run, release_create->run,
ETL->run, disabled/mismatch skip, capacity/PROG/dedup gates)
**Status (2026-09-12): CLOSED** — trigger-rules покрыты на реальной 044 start-boundary
(`test_scenario_automation_trigger.py`, 18 passed): deploy_to_preprod→pinned queued run,
release_created→current-revision run, etl_completed→run, disabled/mismatch skip без durable
rows, capacity gate (существующий active run + shared-capacity внутри одного dispatch),
PROD event→pending_approval+durable gate, dedup window против recent completed run,
идемпотентная повторная доставка события. manual_run_only reject — ссылка на
`test_scenario_manual_run_only.py` (не дублируется).
- [~] T004 [US1] Implement `upsert_schedule` schedule semantics in `automation/schedule.py`
- [~] T005 [US1] Implement `handle_trigger_event` policy-bound event mapping in `automation/trigger.py`
@POST: matched rules start runs via 044; pinned revision+env
@@ -87,8 +94,18 @@
due firing and subscriber integration remain open. Add a dedicated real scheduled-PROD
integration test proving server policy produces `pending_approval` plus one durable gate
before dispatcher CAS; this remaining coverage is not a gate bypass.
**Status (2026-09-12):** event-path wiring закрыт тестами
(`test_scenario_automation_trigger.py`, 18 passed): provenance в start_run, PROD gate на
событиях (pending_approval+gate), dedup window, redelivery-идемпотентность. Дыра в dedup
supply (completed-раны внутри окна были невидимы) и дыра shared-capacity между правилами
одного события закрыты минимальным диффом в `automation/trigger.py`. Остаётся открытым:
dedicated live APScheduler scheduled-PROD integration evidence (процессный прогон, не
callback-unit) и subscriber integration.
- [ ] T017 Emit persisted lifecycle notification events and canonical InvestigationSignals for required
run/stale/repeated-failure transitions; prove no auto-started agent work.
**Status (2026-09-12): remaining** — `notify()`/`persist_notification()` still require explicit
callers; wiring them into 044 lifecycle/stale/repeated-failure transitions is a cross-cutting
change outside the trigger-semantics package and was not forced here.
- [x] T018 Load enabled ScenarioSchedule rows at scheduler startup and map timezone, misfire grace,
coalesce and max instances into APScheduler. **Status (2026-09-11): CLOSED live** — cron due-job
доказан end-to-end (`docs/reports/agentic-runtime-scheduled-happy-path-2026-09-11.md`):
@@ -108,9 +125,37 @@ Historical [x] rows above retain only their dated local/transport evidence; they
Contract: [Parity, retention and rollout](contracts/production-operations.md).
- [ ] T019 [P0/P1/P2] Disabled and enabled HumanStep schedules reject equally via REST/MCP; all six automation reads reject anonymous/cross-owner principals. Implement at the existing 046 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
- [ ] T020 [P0/P1/P2] Crash/replay/concurrent due events create one pinned run/gate/outbox and preserve approval before dispatcher CAS. Implement at the existing 046 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
- [ ] T021 [P0/P1/P2] Retention honors baseline/case holds; deletion receipt proves bytes removed; 5/15/50-tab cost/load/timeout/cancel canaries meet versioned SLO thresholds. Implement at the existing 046 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
- [x] T019 [P0/P1/P2] Disabled and enabled HumanStep schedules reject equally via REST/MCP; all six automation reads reject anonymous/cross-owner principals. Implement at the existing 046 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
**Status (2026-09-12): CLOSED.** `("scenario:automation","READ")` grant + 6 REST reads с per-object
scenario-ownership ACL (admin bypass; foreign→filtered, pagination no-leak) — `scenario_automation.py`
(`Api.ScenarioAutomation.ReadAcl`); MCP reads → `_viewer()` (human: live RBAC READ + admin; service:
mcp:read scope) + ACL, scenario-addressed foreign==missing `not_found` (no existence leak); mutations
human-only без изменений; catalog 2.2.0→2.3.0 (additive MINOR). DEF-02 disabled==enabled: create/patch
HumanStep schedule при enabled=False → идентичный 409 `AUTOMATION_INELIGIBLE_HUMAN_STEP`, 0 side-effect
rows. Evidence: `tests/api/test_scenario_automation_api.py` (TestScenarioAutomationReadAcl, 15 passed),
`tests/test_mcp_automation_parity.py` (NEW, 7 passed), `pytest -q tests -k "automation"` → 68 passed;
production-operations.md status note SEC-01/DEF-02 CLOSED.
- [x] T020 [P0/P1/P2] Crash/replay/concurrent due events create one pinned run/gate/outbox and preserve approval before dispatcher CAS. Implement at the existing 046 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
- [~] T021 [P0/P1/P2] Retention honors baseline/case holds; deletion receipt proves bytes removed; 5/15/50-tab cost/load/timeout/cancel canaries meet versioned SLO thresholds. Implement at the existing 046 domain boundary; verify with independent hardcoded fixtures and retain command/evidence references in traceability.md.
**Status (2026-09-12): offline holds + durable deletion pipeline DONE; canaries OPEN (need
T034 catalog + live stand).** Migration id fix (2026-09-14): revision renamed
`0024_retention_deletions` — the original `0024_scenario_retention_deletions` (33 chars)
exceeded `alembic_version.version_num VARCHAR(32)` and failed the real PostgreSQL upgrade at
the stamp step (transactional DDL rolled the table back cleanly). Static guard added:
`tests/test_alembic_migrations.py::test_revision_ids_fit_varchar32` (SQLite cannot catch this
class). Verified live: `alembic upgrade head` on PostgreSQL → 0024 head, 16 columns,
PK + 3 indexes + unique (target_type, target_id), second upgrade no-op. Delivered: `automation/retention.py` `apply_holds` (approved-
baseline pin holds — fail-closed when the catalog collector is unwired; active operations;
analytics-window hold applies to `run_metadata` only so the 30d artifact tiers stay live);
`automation/deletions.py` — durable `ScenarioRetentionDeletion` receipts (migration
`0024_retention_deletions`), pipeline mark→eligible→delete-bytes→verify-absent→
tombstone, idempotent retry (tombstoned = no-op, pending re-attempts), storage failure stays
`deletion_pending` and never reports deleted while bytes survive; daily sweep registered
(`scenario_retention_sweep`, CronTrigger 03:45 UTC); `GET /api/scenario-automation/retention`
projects ACL-filtered receipts (`deletions`/`deletions_total`, null-scenario rows admin-only).
Tests: `test_scenario_retention_holds.py` (11) + `test_scenario_automation_api.py`
RetentionReceipts ACL (16 passed in file). Remaining: 5/15/50-tab canaries vs versioned ops-v1
SLO thresholds (D-wave, after T034 catalog).
Frontend boundary for this package: manual CRUD/editor, human review/approval, monitoring and read-only evidence/evaluation only; all agent interaction is external MCP. No agent chat/prompt/assistant editing/proposal generation/workspace/start/handoff controls. Runtime removal is OPEN, not performed by this spec refresh. Optional approved performance baseline is outside scope.

View File

@@ -16,9 +16,9 @@ Historical rows above identify prior tests/code only; removed agent UI paths are
| Requirement | Domain contract / DTO | Task | Falsifiable acceptance | State |
|---|---|---|---|---|
| SCAUTO-FR-019 | [Parity, retention and rollout](contracts/production-operations.md); [data model](data-model.md) | [T019](tasks.md) | Disabled and enabled HumanStep schedules reject equally via REST/MCP; all six automation reads reject anonymous/cross-owner principals. | OPEN |
| SCAUTO-FR-019 | [Parity, retention and rollout](contracts/production-operations.md); [data model](data-model.md) | [T020](tasks.md) | Crash/replay/concurrent due events create one pinned run/gate/outbox and preserve approval before dispatcher CAS. | OPEN |
| SCAUTO-FR-019 | [Parity, retention and rollout](contracts/production-operations.md); [data model](data-model.md) | [T021](tasks.md) | Retention honors baseline/case holds; deletion receipt proves bytes removed; 5/15/50-tab cost/load/timeout/cancel canaries meet versioned SLO thresholds. | OPEN |
| SCAUTO-FR-019 | [Parity, retention and rollout](contracts/production-operations.md); [data model](data-model.md) | [T019](tasks.md) | Disabled and enabled HumanStep schedules reject equally via REST/MCP; all six automation reads reject anonymous/cross-owner principals. | DONE (2026-09-12): READ grant + per-object scenario ACL на 6 REST reads (`Api.ScenarioAutomation.ReadAcl`), MCP `_viewer()` parity (foreign==missing not_found, service mcp:read), mutations human-only, catalog 2.3.0; DEF-02 disabled==enabled 409 `AUTOMATION_INELIGIBLE_HUMAN_STEP` 0 rows — `tests/api/test_scenario_automation_api.py::TestScenarioAutomationReadAcl`, `tests/test_mcp_automation_parity.py`; `pytest -q tests -k "automation"` → 68 passed. |
| SCAUTO-FR-019 | [Parity, retention and rollout](contracts/production-operations.md); [data model](data-model.md) | [T020](tasks.md) | Crash/replay/concurrent due events create one pinned run/gate/outbox and preserve approval before dispatcher CAS. | DONE (2026-09-12): baseline pin in due identity + typed conflict on changed canonical request — `test_scheduled_scenario_start.py` (BaselineStamp/BaselineConflict/BaselineNextDue); concurrent double-callback convergence, crash-between-run-and-gate rollback+retry, terminal transition+queue+outbox single transaction, approval-before-dispatcher-CAS — `test_due_admission_atomicity.py`; evidence: `python -m pytest -q tests/services/dashboard_testing/registry/` (466 passed). |
| SCAUTO-FR-019 | [Parity, retention and rollout](contracts/production-operations.md); [data model](data-model.md) | [T021](tasks.md) | Retention honors baseline/case holds; deletion receipt proves bytes removed; 5/15/50-tab cost/load/timeout/cancel canaries meet versioned SLO thresholds. | PARTIAL (2026-09-12): holds (`apply_holds` — approved-baseline fail-closed, active ops, analytics window на run_metadata) + durable deletion pipeline (`automation/deletions.py`, модель + миграция `0024`, mark→eligible→delete-bytes→verify→tombstone, идемпотентный retry, never-deleted-while-bytes-survive) + sweep job + ACL-проекция в `GET /retention` — `test_scenario_retention_holds.py` (11), `test_scenario_automation_api.py` (16). OPEN: canaries vs ops-v1 SLO (нужен T034 каталог + live стенд). |
| SCAUTO-FR-019; external-MCP-only UI | manual editor/review; read-only evidence | [production tasks](tasks.md) | No frontend agent prompt/chat/assistant editing/proposal generation/workspace/start/handoff routes or requests; human approval remains usable. | OPEN |
Sources: [production gap](../../docs/reports/ss-prod-agentic-e2e-production-gap-2026-09-08.md), [coverage gap](../../docs/reports/ss-prod-agentic-e2e-spec-coverage-2026-09-08.md), [baseline gap](../../docs/reports/ss-prod-agentic-e2e-baseline-gap-2026-09-08.md). Spec schema/static checks prove contract structure only; live canary/runtime closure and optional approved performance baseline are not claimed.