feat(scenario): live eval trust — MIME sniffing, multimodal evidence, baseline pin stamping, binding admin surface

This commit is contained in:
2026-09-11 08:40:19 +03:00
parent bac002bcbc
commit 792bb1251d
19 changed files with 952 additions and 54 deletions

View File

@@ -24,6 +24,9 @@ from src.api.routes.dashboard_testing.scenario_run_center import router as scena
from src.api.routes.dashboard_testing.scenario_artifact_content import (
router as scenario_artifact_content_router,
)
from src.api.routes.dashboard_testing.scenario_live_bindings import (
router as scenario_live_bindings_router,
)
from src.api.routes.dashboard_testing.scenario_runs import (
history_router as scenario_runs_history_router,
runs_router as scenario_runs_router,
@@ -48,6 +51,7 @@ router.include_router(scenario_router)
router.include_router(scenarios_router)
router.include_router(scenario_runs_router)
router.include_router(scenario_artifact_content_router)
router.include_router(scenario_live_bindings_router)
router.include_router(scenario_runs_history_router)
router.include_router(scenario_run_center_router)
router.include_router(scenario_analytics_router)

View File

@@ -0,0 +1,164 @@
# #region Api.ScenarioLiveBindings [C:4] [TYPE Module] [SEMANTICS scenario,execution,live-binding,admin,api,config]
# @defgroup Api Trusted admin CRUD for deployment-configured 044 live-execution bindings.
# @BRIEF Admin-only list/register/toggle of `settings.scenario_live_execution_bindings` via ConfigManager.
# @RELATION DEPENDS_ON -> [Core.ConfigManager]
# @RELATION DEPENDS_ON -> [ScenarioExecution.LiveBinding.Identity]
# @INVARIANT Writes go through ConfigManager.update_global_settings so the persisted settings remain the
# single authority that ScenarioExecution.Runner.ConfiguredBinding reads at start.
# @INVARIANT A binding snapshot is validated by LiveExecutionBinding.from_snapshot; a malformed snapshot
# never enters the configuration. No credentials exist in a snapshot (identity fingerprints only).
# @RATIONALE Track B (2026-09-10): server-side binding resolution existed but the only registration path
# was a direct DB/config write. A reviewed admin surface closes that gap without exposing a
# client-facing binding selector (the request body still never carries a binding).
# @REJECTED Exposing bindings through MCP was rejected for this change — the live-binding identity is a
# deployment trust boundary, not agent-authorable state; a separate architect decision is required.
from __future__ import annotations
from typing import Any
from fastapi import APIRouter, Depends, HTTPException
from pydantic import BaseModel, ConfigDict, ValidationError
from src.core.config_manager import ConfigManager
from src.core.config_models import ScenarioLiveExecutionBindingConfig
from src.dependencies import get_config_manager, has_permission
from src.schemas.dashboard_testing import DashboardQueryModel
from src.services.dashboard_testing.execution.live_binding import LiveExecutionBinding
router = APIRouter(prefix="/api/scenario-live-bindings", tags=["scenario-live-bindings"])
# #region Api.ScenarioLiveBindings.Schemas [C:2] [TYPE Class] [SEMANTICS scenario,execution,live-binding,dto]
# @BRIEF Bounded request/response DTOs; the snapshot identity is echoed without any secret material.
class LiveBindingUpsertRequest(BaseModel):
model_config = ConfigDict(extra="forbid")
enabled: bool = False
binding_snapshot: dict[str, Any]
query_model_snapshot: dict[str, Any]
class LiveBindingToggleRequest(BaseModel):
model_config = ConfigDict(extra="forbid")
enabled: bool
class LiveBindingView(BaseModel):
model_config = ConfigDict(extra="forbid")
binding_ref: str | None
enabled: bool
environment_id: str | None = None
dashboard_id: int | None = None
dashboard_release_id: str | None = None
execution_principal_fingerprint: str | None = None
query_model_fingerprint: str | None = None
rls_security_fingerprint: str | None = None
# #endregion Api.ScenarioLiveBindings.Schemas
# #region Api.ScenarioLiveBindings.Project [C:2] [TYPE Function]
# @BRIEF Project one config record to a view; a malformed snapshot is surfaced, never hidden or trusted.
def _project(record: ScenarioLiveExecutionBindingConfig) -> LiveBindingView:
snapshot = record.binding_snapshot if isinstance(record.binding_snapshot, dict) else {}
dashboard_id = snapshot.get("dashboard_id")
return LiveBindingView(
binding_ref=snapshot.get("binding_ref") if isinstance(snapshot.get("binding_ref"), str) else None,
enabled=bool(record.enabled),
environment_id=snapshot.get("environment_id") if isinstance(snapshot.get("environment_id"), str) else None,
dashboard_id=dashboard_id if isinstance(dashboard_id, int) else None,
dashboard_release_id=snapshot.get("dashboard_release_id") if isinstance(snapshot.get("dashboard_release_id"), str) else None,
execution_principal_fingerprint=snapshot.get("execution_principal_fingerprint") if isinstance(snapshot.get("execution_principal_fingerprint"), str) else None,
query_model_fingerprint=snapshot.get("query_model_fingerprint") if isinstance(snapshot.get("query_model_fingerprint"), str) else None,
rls_security_fingerprint=snapshot.get("rls_security_fingerprint") if isinstance(snapshot.get("rls_security_fingerprint"), str) else None,
)
# #endregion Api.ScenarioLiveBindings.Project
# #region Api.ScenarioLiveBindings.Store [C:2] [TYPE Function]
# @BRIEF Replace the settings list atomically through ConfigManager (single authority at start).
def _store(config_manager: ConfigManager, records: list[ScenarioLiveExecutionBindingConfig]) -> None:
settings = config_manager.get_config().settings
updated = settings.model_copy(update={"scenario_live_execution_bindings": records})
config_manager.update_global_settings(updated)
# #endregion Api.ScenarioLiveBindings.Store
# #region Api.ScenarioLiveBindings.Ref [C:1] [TYPE Function]
# @BRIEF Best-effort binding_ref of a stored record; malformed snapshots are skipped by identity.
def _record_ref(record: ScenarioLiveExecutionBindingConfig) -> str | None:
snapshot = record.binding_snapshot
ref = snapshot.get("binding_ref") if isinstance(snapshot, dict) else None
return ref if isinstance(ref, str) and ref else None
# #endregion Api.ScenarioLiveBindings.Ref
# #region Api.ScenarioLiveBindings.List [C:2] [TYPE Function] [SEMANTICS scenario,execution,live-binding,list]
# @BRIEF Admin list of configured live bindings with identity fields only.
@router.get("", response_model=list[LiveBindingView])
def list_live_bindings(
config_manager: ConfigManager = Depends(get_config_manager),
_=Depends(has_permission("admin:settings", "READ")),
):
records = config_manager.get_config().settings.scenario_live_execution_bindings or []
return [_project(record) for record in records]
# #endregion Api.ScenarioLiveBindings.List
# #region Api.ScenarioLiveBindings.Upsert [C:3] [TYPE Function] [SEMANTICS scenario,execution,live-binding,upsert]
# @BRIEF Validate and create-or-replace one binding (canonicalized snapshot) in the config authority.
# @INVARIANT A snapshot that fails LiveExecutionBinding.from_snapshot, or whose binding_ref differs
# from the path, is rejected with no persistence.
@router.put("/{binding_ref}", response_model=LiveBindingView)
def upsert_live_binding(
binding_ref: str,
request: LiveBindingUpsertRequest,
config_manager: ConfigManager = Depends(get_config_manager),
_=Depends(has_permission("admin:settings", "WRITE")),
):
try:
binding = LiveExecutionBinding.from_snapshot(request.binding_snapshot)
except ValueError:
raise HTTPException(status_code=422, detail="LIVE_BINDING_SNAPSHOT_INVALID") from None
if binding.binding_ref != binding_ref:
raise HTTPException(status_code=422, detail="LIVE_BINDING_REF_MISMATCH")
try:
query_model = DashboardQueryModel.model_validate(request.query_model_snapshot)
except ValidationError:
raise HTTPException(status_code=422, detail="LIVE_BINDING_QUERY_MODEL_INVALID") from None
if query_model.environment_id != binding.environment_id or query_model.dashboard_id != binding.dashboard_id:
raise HTTPException(status_code=422, detail="LIVE_BINDING_QUERY_MODEL_MISMATCH")
record = ScenarioLiveExecutionBindingConfig(
enabled=request.enabled,
binding_snapshot=binding.snapshot(),
query_model_snapshot=query_model.model_dump(mode="json"),
)
records = list(config_manager.get_config().settings.scenario_live_execution_bindings or [])
for index, existing in enumerate(records):
if _record_ref(existing) == binding_ref:
records[index] = record
break
else:
records.append(record)
_store(config_manager, records)
return _project(record)
# #endregion Api.ScenarioLiveBindings.Upsert
# #region Api.ScenarioLiveBindings.Toggle [C:2] [TYPE Function] [SEMANTICS scenario,execution,live-binding,toggle]
# @BRIEF Enable or disable an existing binding by ref; an unknown ref is 404.
@router.patch("/{binding_ref}", response_model=LiveBindingView)
def toggle_live_binding(
binding_ref: str,
request: LiveBindingToggleRequest,
config_manager: ConfigManager = Depends(get_config_manager),
_=Depends(has_permission("admin:settings", "WRITE")),
):
records = list(config_manager.get_config().settings.scenario_live_execution_bindings or [])
for index, existing in enumerate(records):
if _record_ref(existing) == binding_ref:
updated = existing.model_copy(update={"enabled": request.enabled})
records[index] = updated
_store(config_manager, records)
return _project(updated)
raise HTTPException(status_code=404, detail="LIVE_BINDING_NOT_FOUND")
# #endregion Api.ScenarioLiveBindings.Toggle
# #endregion Api.ScenarioLiveBindings

View File

@@ -5,6 +5,7 @@
# @REJECTED A permissive dict passthrough was rejected because it allows findings to claim undeclared deterministic authority.
from __future__ import annotations
import base64
from datetime import UTC, datetime
from typing import Any, Literal
import uuid
@@ -12,6 +13,7 @@ import uuid
from pydantic import BaseModel, ConfigDict, Field, model_validator
from sqlalchemy.orm import Session
from src.core.logger import logger
from src.models.scenario_evaluation import AgentEvaluation as AgentEvaluationRow
from src.services.dashboard_testing.scenario.models import AgentEvaluationSpec
from src.models.scenario_artifact import ScenarioArtifact
@@ -206,8 +208,29 @@ def validate_evaluation_evidence(
# #endregion ScenarioExecution.AgentEvaluation.Evidence
# #region ScenarioExecution.AgentEvaluation.Messages [C:3] [TYPE Function] [SEMANTICS evaluation,multimodal,messages,content-parts]
# @BRIEF Compose the OpenAI chat message: text-only, or the prompt plus image data URLs.
# @PRE images, when present, are (bytes, allowlisted-image-MIME) pairs already bounded by the adapter.
# @POST Returns one user message whose content parts carry the prompt and base64 image URLs; no
# image parts are produced for an empty/None list, preserving the text-only transport.
# @INVARIANT Image bytes are inlined only as data URLs; MIME comes from server-verified manifest bytes.
# @RATIONALE A text-only prompt left the live judge blind (canary v2: honest low-confidence).
# Attaching the captured evidence lets a multimodal model reason over real pixels.
# @REJECTED Uploading evidence to an external file host was rejected — bytes never leave the
# bounded provider call and no third-party URL enters the evaluation record.
def _evaluation_messages(prompt: str, images: list[tuple[bytes, str]] | None) -> list[dict[str, Any]]:
if not images:
return [{"role": "user", "content": prompt}]
parts: list[dict[str, Any]] = [{"type": "text", "text": prompt}]
for data, content_type in images:
encoded = base64.b64encode(data).decode("ascii")
parts.append({"type": "image_url", "image_url": {"url": f"data:{content_type};base64,{encoded}"}})
return [{"role": "user", "content": parts}]
# #endregion ScenarioExecution.AgentEvaluation.Messages
async def submit_evaluation(
db: Session, *, spec: AgentEvaluationSpec, prompt: str, artifacts: list[bytes],
db: Session, *, spec: AgentEvaluationSpec, prompt: str, images: list[tuple[bytes, str]] | None = None,
environment_id: str, environment_class: str, run_id: str, logical_step_id: str,
client: Any = None,
) -> dict[str, Any]:
@@ -229,7 +252,16 @@ async def submit_evaluation(
if not api_key:
raise RuntimeError("EVALUATION_PROVIDER_MISSING")
llm = client or LLMClient(LLMProviderType(provider.provider_type), api_key, provider.base_url, spec.model_id)
result = await llm.get_json_completion([{"role": "user", "content": prompt}])
# Images reach the model only through a multimodal provider; a text-only provider
# keeps the honest text-only manifest (never a fabricated visual verdict).
attached = images if bool(provider.is_multimodal) else None
if attached:
logger.reflect(
"Multimodal images attached to the evaluation prompt",
src="ScenarioExecution.AgentEvaluation",
payload={"image_count": len(attached), "total_bytes": sum(len(data) for data, _ in attached)},
)
result = await llm.get_json_completion(_evaluation_messages(prompt, attached))
if not isinstance(result, dict):
raise RuntimeError("EVALUATION_RESPONSE_INVALID")
return result

View File

@@ -29,6 +29,8 @@ from src.models.scenario_artifact import ScenarioArtifact
from src.models.scenario_run import ScenarioRun
from src.services.agent_runs.artifacts import get_draft_storage
from .mime_sniff import sniff_mime
IMAGE_MAX_BYTES = 10_485_760
OTHER_MAX_BYTES = 26_214_400
IMAGE_MIME = frozenset({"image/jpeg", "image/png", "image/webp"})
@@ -94,23 +96,11 @@ def _has_result_view(user: object) -> bool:
# #endregion ScenarioExecution.ArtifactContent.HasView
# #region ScenarioExecution.ArtifactContent.SniffMime [C:2] [TYPE Function] [SEMANTICS scenario,artifact,mime,magic]
# @BRIEF Detect allowlisted MIME from complete-object magic bytes; JSON is BOM/whitespace then { or [.
def sniff_mime(data: bytes) -> str | None:
if data.startswith(b"\xff\xd8\xff"):
return "image/jpeg"
if data.startswith(b"\x89PNG\r\n\x1a\n"):
return "image/png"
if len(data) >= 12 and data.startswith(b"RIFF") and data[8:12] == b"WEBP":
return "image/webp"
if data.startswith(b"%PDF-"):
return "application/pdf"
if data.startswith(b"PK\x03\x04"):
return "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"
stripped = data.lstrip(b"\xef\xbb\xbf \t\r\n")
if stripped[:1] in (b"{", b"["):
return "application/json"
return None
# #region ScenarioExecution.ArtifactContent.SniffMime [C:1] [TYPE Tombstone] [SEMANTICS scenario,artifact,mime,magic]
# @STATUS DEPRECATED
# @REPLACED_BY -> [ScenarioExecution.MimeSniff.Detect]
# @BRIEF Detection moved to ScenarioExecution.MimeSniff; the module still re-exports sniff_mime
# at import time so existing callers and tests resolve unchanged.
# #endregion ScenarioExecution.ArtifactContent.SniffMime

View File

@@ -126,6 +126,26 @@ def attach_baseline_pin(plan: dict[str, Any], pin: dict[str, Any] | None) -> dic
# #endregion ScenarioExecution.BaselineResolver.Attach
# #region ScenarioExecution.BaselineResolver.StampEvaluation [C:3] [TYPE Function] [SEMANTICS baseline,pin,evaluation,record,stamp]
# @ingroup ScenarioExecution
# @BRIEF Stamp the plan's resolved pin as the record's server-owned baseline_pin.
# @PRE record is the adapter-built AgentEvaluation payload; plan is the pinned ScenarioRun.runner_plan.
# @POST Returns the record with baseline_pin = plan pin, or {} when the plan resolved no pin.
# @INVARIANT baseline_pin is server identity: any provider-claimed value is overwritten, so a record
# can never carry a pin the server did not resolve. A null plan pin stamps {}.
# @RATIONALE Live canary v2 persisted baseline_pin={} on a baseline-backed plan because the provider
# has no pin authority. Stamping at the persist boundary keeps provider text out of identity
# and makes the record whole; forcing {} also defeats a model-supplied fake pin.
# @REJECTED Threading the pin through the executor step payload was rejected — it widens the executor
# contract for a purely server-side identity concern handled at persistence.
def stamp_baseline_pin(record: dict[str, Any] | None, plan: dict[str, Any] | None) -> dict[str, Any] | None:
if not isinstance(record, dict):
return record
pin = (plan or {}).get("baseline_pin")
return {**record, "baseline_pin": dict(pin) if isinstance(pin, dict) else {}}
# #endregion ScenarioExecution.BaselineResolver.StampEvaluation
def _blank(value: Any) -> str | None:
if value is None:
return None

View File

@@ -17,10 +17,12 @@ from typing import Any
import uuid
from src.core.database import SessionLocal
from src.core.logger import logger
from src.services.dashboard_testing.scenario.models import AgentEvaluationSpec
from .agent_evaluation import AgentEvaluation, parse_evaluation_response, submit_evaluation
from .artifacts import is_valid_sha256
from .evaluation_images import image_payloads_from_manifest
from .live_binding import EvidenceStorage
_ALLOWED_CONTENT_TYPES = frozenset({"image/jpeg", "image/png", "image/webp", "application/json"})
@@ -173,7 +175,9 @@ def _normalize_provider_response(raw: dict[str, Any], spec: AgentEvaluationSpec)
payload["output_schema_hash"] = _canonical_hash(spec.output_schema)
payload["trust_policy_hash"] = spec.trust_policy_hash
payload["comparison_ids"] = list(spec.comparison_refs)
payload.setdefault("baseline_pin", {})
# baseline_pin is server-owned identity: never accept a provider-claimed pin. The walker
# stamps the resolved plan pin (ScenarioExecution.BaselineResolver.StampEvaluation).
payload["baseline_pin"] = {}
payload.setdefault("reason_codes", [])
# EvaluationUsage is required by the pinned record schema; providers omit it or send
# partial token counts — the server normalizes rather than trusting provider cost text.
@@ -300,6 +304,15 @@ def evaluation_adapter_from(
raise RuntimeError("EVALUATION_RESPONSE_INVALID")
attempt = int(step.get("attempt") or 1)
evidence = storage if storage is not None else _default_storage()
images = image_payloads_from_manifest(manifest, evidence, max_images=spec.limits.max_images)
if images:
# Candidate evidence loaded; actual attachment is gated on provider multimodality
# inside submit_evaluation, which logs the attached count.
logger.reflect(
"Evaluation image evidence loaded from manifest",
src="ScenarioExecution.EvaluationAdapter",
payload={"image_count": len(images), "total_bytes": sum(len(data) for data, _ in images)},
)
prompt = json.dumps(
{
# Live canary v2 (2026-09-10): the model echoed spec.evidence_refs into
@@ -319,6 +332,8 @@ def evaluation_adapter_from(
"server-owned identity fields (ids, hashes, template and policy metadata) are "
"filled by the server; do not emit them",
"verdict, confidence (0..1), findings and usage must always be present",
"verdict MUST be exactly one of pass, fail, inconclusive — status is a separate "
"server-owned field and 'succeeded' is never a verdict value",
],
},
"evaluation_spec": spec.model_dump(mode="json"),
@@ -327,7 +342,7 @@ def evaluation_adapter_from(
sort_keys=True, separators=(",", ":"), default=str,
)
if submit is not None:
raw = submit(spec=spec, prompt=prompt, step=step, completed=completed)
raw = submit(spec=spec, prompt=prompt, step=step, completed=completed, images=images)
else:
target = step.get("target_snapshot") if isinstance(step.get("target_snapshot"), dict) else {}
meta = step.get("step_meta") if isinstance(step.get("step_meta"), dict) else {}
@@ -338,7 +353,7 @@ def evaluation_adapter_from(
db = (db_factory or SessionLocal)()
try:
coro = submit_evaluation(
db, spec=spec, prompt=prompt, artifacts=[], environment_id=environment_id,
db, spec=spec, prompt=prompt, images=images, environment_id=environment_id,
environment_class=environment_class, run_id=run_id, logical_step_id=logical_step_id,
client=client,
)

View File

@@ -0,0 +1,69 @@
# #region ScenarioExecution.EvaluationImages [C:3] [TYPE Module] [SEMANTICS scenario,execution,evaluation,multimodal,images,evidence,store]
# @defgroup ScenarioExecution Bounded image-evidence loading for the multimodal evaluation judge.
# @BRIEF Read manifest image refs from owned storage into (bytes, sniffed-MIME) pairs within budget.
# @RELATION DEPENDS_ON -> [ScenarioExecution.MimeSniff.Detect]
# @RELATION CALLED_BY -> [ScenarioExecution.EvaluationAdapter]
# @INVARIANT The model can only see artifacts that already passed the manifest digest/content_type/
# byte_length gate; unreadable, non-image, or over-budget items are skipped, never invented.
# @RATIONALE Canary v2 proved the text-only judge blind to pixels. Loading the real captures here keeps
# the adapter module under INV_7 and separates evidence loading from adapter orchestration.
# @REJECTED Querying ScenarioArtifact in a fresh SessionLocal was rejected (D1): the walker has only
# flushed prior-step rows, so the completed outcomes stay the only lawful manifest source.
from __future__ import annotations
from hashlib import sha256
from typing import Any
from .mime_sniff import sniff_mime
IMAGE_CONTENT_TYPES = frozenset({"image/jpeg", "image/png", "image/webp"})
MAX_EVALUATION_IMAGES = 8
MAX_EVALUATION_IMAGE_BYTES = 10 * 1024 * 1024
# #region ScenarioExecution.EvaluationImages.Load [C:4] [TYPE Function] [SEMANTICS evaluation,multimodal,images,budget]
# @BRIEF Load readable image evidence from manifest refs, bounded by count and total byte budget.
# @PRE manifest items are the adapter's own server-built entries; storage may or may not expose retrieve.
# @POST Returns (bytes, MIME) pairs in manifest order; MIME is the sniffed signature, so a mislabeled
# ref cannot reach the model. Missing retrieve or any per-item failure degrades to fewer images.
def image_payloads_from_manifest(
manifest: list[dict[str, Any]], storage: Any, *, max_images: int,
) -> list[tuple[bytes, str]]:
retrieve = getattr(storage, "retrieve", None)
if not callable(retrieve):
return []
limit = max(0, min(MAX_EVALUATION_IMAGES, int(max_images)))
images: list[tuple[bytes, str]] = []
total_bytes = 0
for item in manifest:
if len(images) >= limit:
break
if item.get("content_type") not in IMAGE_CONTENT_TYPES:
continue
ref = item.get("artifact_id")
if not isinstance(ref, str) or not ref:
continue
# Trusted manifest length lets an over-budget item be rejected without reading its bytes.
declared_length = item.get("byte_length")
if isinstance(declared_length, int) and declared_length > MAX_EVALUATION_IMAGE_BYTES - total_bytes:
continue
try:
data = retrieve(ref)
except Exception:
continue
if not isinstance(data, (bytes, bytearray)) or not data:
continue
# Defense in depth: the loaded bytes must match the manifest digest before reaching the model.
expected_digest = item.get("sha256")
if isinstance(expected_digest, str) and sha256(bytes(data)).hexdigest() != expected_digest.lower():
continue
content_type = sniff_mime(bytes(data))
if content_type not in IMAGE_CONTENT_TYPES:
continue
if total_bytes + len(data) > MAX_EVALUATION_IMAGE_BYTES:
continue
images.append((bytes(data), content_type))
total_bytes += len(data)
return images
# #endregion ScenarioExecution.EvaluationImages.Load
# #endregion ScenarioExecution.EvaluationImages

View File

@@ -58,6 +58,7 @@ class LiveExecutionBinding:
binding.execution_principal_fingerprint, binding.rls_security_fingerprint,
))
or not isinstance(binding.dashboard_id, int)
or isinstance(binding.dashboard_id, bool)
or binding.dashboard_id <= 0
or binding.evidence_owner_type != "scenario_run"
or binding.evidence_ref_policy != "draft_storage_raw_response"

View File

@@ -0,0 +1,31 @@
# #region ScenarioExecution.MimeSniff [C:3] [TYPE Module] [SEMANTICS scenario,execution,mime,magic,bytes,signature]
# @defgroup ScenarioExecution Pure magic-byte MIME detection shared by providers and artifact delivery.
# @BRIEF Detect an allowlisted MIME from a byte-object signature; returns None when unrecognized.
# @RELATION CALLED_BY -> [ScenarioExecution.ArtifactContent]
# @RELATION CALLED_BY -> [ScenarioExecution.ScreenshotProvider]
# @RELATION CALLED_BY -> [ScenarioExecution.BrowserProvider]
# @INVARIANT Detection is byte-sniffing only (no path, name, or provider hint) so stored MIME can
# never drift from the actual bytes (e.g. a WebP archive mislabeled image/jpeg).
# @RATIONALE Extracted from ArtifactContent so evidence providers can stamp per-ref MIME from the
# captured bytes without depending on the HTTP artifact-delivery service layer.
# @REJECTED Trusting the provider's nominal format (screenshot→jpeg, browser→png) was rejected: the
# capture service can emit WebP, which then fails the evaluation evidence MIME match.
# #region ScenarioExecution.MimeSniff.Detect [C:2] [TYPE Function] [SEMANTICS mime,magic]
# @BRIEF Detect allowlisted MIME from complete-object magic bytes; JSON is BOM/whitespace then { or [.
def sniff_mime(data: bytes) -> str | None:
if data.startswith(b"\xff\xd8\xff"):
return "image/jpeg"
if data.startswith(b"\x89PNG\r\n\x1a\n"):
return "image/png"
if len(data) >= 12 and data.startswith(b"RIFF") and data[8:12] == b"WEBP":
return "image/webp"
if data.startswith(b"%PDF-"):
return "application/pdf"
if data.startswith(b"PK\x03\x04"):
return "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"
stripped = data.lstrip(b"\xef\xbb\xbf \t\r\n")
if stripped[:1] in (b"{", b"["):
return "application/json"
return None
# #endregion ScenarioExecution.MimeSniff.Detect
# #endregion ScenarioExecution.MimeSniff

View File

@@ -28,7 +28,6 @@
# on the single application-owned provider loop.
from __future__ import annotations
import json
import uuid
from collections.abc import Callable
from hashlib import sha256
@@ -43,8 +42,8 @@ from src.services.dashboard_testing.execution.capacity import (
)
from src.services.dashboard_testing.execution.live_adapter import LiveAdapterResult
from src.services.dashboard_testing.execution.live_binding import binding_from_step
from src.services.dashboard_testing.execution.mime_sniff import sniff_mime
from src.services.dashboard_testing.execution.provider_operations import (
complete_provider_operation,
open_provider_operation,
)
from src.services.dashboard_testing.execution.provider_runtime import (
@@ -62,6 +61,10 @@ from src.services.dashboard_testing.execution.providers.browser_mutation import
validate_mutation_contract,
validate_mutation_inputs,
)
from src.services.dashboard_testing.execution.providers.browser_receipt import (
descriptor_fingerprint,
finalize_provider_receipt,
)
_SRC = "ScenarioExecution.BrowserProvider"
_DEFAULT_ACTION_TIMEOUT_SECONDS = 30
@@ -237,7 +240,7 @@ def build_browser_provider(
provider_id="browser",
provider_version=str(descriptor.get("provider_version") or "unset"),
action=action,
descriptor_fingerprint=_descriptor_fingerprint(descriptor),
descriptor_fingerprint=descriptor_fingerprint(descriptor),
binding_ref=binding.binding_ref,
execution_principal_fingerprint=binding.execution_principal_fingerprint,
idempotency_key=f"{run_id}:{metadata.get('logical_step_id')}:{int(metadata.get('attempt') or 1)}",
@@ -280,7 +283,7 @@ def build_browser_provider(
except TimeoutError:
if mutating:
logger.explore("Mutating action deadline expired with unknown effect", src=_SRC, payload={"run_id": run_id}, error_code="BROWSER_MUTATION_RECONCILE_REQUIRED")
_finalize_provider_receipt(operation_id, "reconciliation_required", "unknown", summary={"phase": "timeout"})
finalize_provider_receipt(operation_id, "reconciliation_required", "unknown", summary={"phase": "timeout"})
return LiveAdapterResult(
status="inconclusive",
reason_code="BROWSER_MUTATION_RECONCILE_REQUIRED",
@@ -293,7 +296,7 @@ def build_browser_provider(
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_LOOP_OVERFLOW")
except BrowserTransportPreconditionMismatch:
logger.explore("Mutation precondition mismatch; nothing mutated", src=_SRC, payload={"run_id": run_id}, error_code="BROWSER_MUTATION_PRECONDITION_MISMATCH")
_finalize_provider_receipt(operation_id, "failed", "not_started", summary={"phase": "precondition_mismatch"})
finalize_provider_receipt(operation_id, "failed", "not_started", summary={"phase": "precondition_mismatch"})
return LiveAdapterResult(
status="inconclusive",
reason_code="BROWSER_MUTATION_PRECONDITION_MISMATCH",
@@ -301,7 +304,7 @@ def build_browser_provider(
)
except BrowserTransportUnsupported:
logger.explore("Transport cannot execute the action; nothing started", src=_SRC, payload={"action": action}, error_code="BROWSER_ACTION_NOT_SUPPORTED")
_finalize_provider_receipt(operation_id, "failed", "not_started", summary={"phase": "unsupported"})
finalize_provider_receipt(operation_id, "failed", "not_started", summary={"phase": "unsupported"})
return LiveAdapterResult(
status="inconclusive",
reason_code="BROWSER_ACTION_NOT_SUPPORTED",
@@ -313,7 +316,7 @@ def build_browser_provider(
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_LOOP_UNAVAILABLE")
if mutating:
logger.explore("Mutating action failed with unknown effect", src=_SRC, payload={"run_id": run_id}, error_code="BROWSER_MUTATION_RECONCILE_REQUIRED")
_finalize_provider_receipt(operation_id, "reconciliation_required", "unknown", summary={"phase": "runtime_error"})
finalize_provider_receipt(operation_id, "reconciliation_required", "unknown", summary={"phase": "runtime_error"})
return LiveAdapterResult(
status="inconclusive",
reason_code="BROWSER_MUTATION_RECONCILE_REQUIRED",
@@ -324,17 +327,17 @@ def build_browser_provider(
evidence = outcome.evidence_png
if not evidence:
logger.explore("Transport produced no evidence", src=_SRC, payload={"run_id": run_id}, error_code="BROWSER_EVIDENCE_REQUIRED")
_finalize_provider_receipt(operation_id, "reconciliation_required" if mutating else "failed", "unknown" if mutating else "not_started", summary={"phase": "evidence_missing"})
finalize_provider_receipt(operation_id, "reconciliation_required" if mutating else "failed", "unknown" if mutating else "not_started", summary={"phase": "evidence_missing"})
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_EVIDENCE_REQUIRED")
rejection, artifact_refs, artifact_digests = _store_browser_evidence(
evidence, storage, run_id, max_screenshot_bytes=max_screenshot_bytes,
)
if rejection is not None:
_finalize_provider_receipt(operation_id, "reconciliation_required" if mutating else "failed", "unknown" if mutating else "not_started", summary={"phase": "evidence_store"})
finalize_provider_receipt(operation_id, "reconciliation_required" if mutating else "failed", "unknown" if mutating else "not_started", summary={"phase": "evidence_store"})
return rejection
effect_state = outcome.effect_state if mutating else "none"
if mutating:
_finalize_provider_receipt(operation_id, "completed", effect_state, summary={"checkpoints": list(outcome.checkpoints), "page_url": outcome.page_url})
finalize_provider_receipt(operation_id, "completed", effect_state, summary={"checkpoints": list(outcome.checkpoints), "page_url": outcome.page_url})
logger.reflect(
"Browser action completed with evidence", src=_SRC,
payload={"run_id": run_id, "checkpoints": list(outcome.checkpoints), "bytes": len(evidence), "effect_state": effect_state, "operation_id": operation_id},
@@ -354,7 +357,10 @@ def build_browser_provider(
"artifact_byte_lengths": {
ref: len(evidence) for ref in artifact_refs
},
"artifact_content_types": {ref: "image/png" for ref in artifact_refs},
# Sniff MIME from the stored bytes (a WebP payload must not be labelled png).
"artifact_content_types": {
ref: sniff_mime(evidence) or "image/png" for ref in artifact_refs
},
**outcome.details,
},
artifact_refs=artifact_refs,
@@ -363,7 +369,7 @@ def build_browser_provider(
except Exception as exc:
logger.explore("Browser provider failed", src=_SRC, payload={"run_id": run_id if isinstance(run_id, str) else None}, error=repr(exc))
if mutating:
_finalize_provider_receipt(operation_id, "reconciliation_required", "unknown", summary={"phase": "provider_error"})
finalize_provider_receipt(operation_id, "reconciliation_required", "unknown", summary={"phase": "provider_error"})
return LiveAdapterResult(
status="inconclusive",
reason_code="BROWSER_MUTATION_RECONCILE_REQUIRED",
@@ -382,23 +388,4 @@ def build_browser_provider(
return provider
# #endregion ScenarioExecution.BrowserProvider.Factory
# #region ScenarioExecution.BrowserProvider.Receipt [C:3] [TYPE Function] [SEMANTICS provider,operation,receipt,finalize]
# @ingroup ScenarioExecution
# @BRIEF Finalize the durable receipt outside the run transaction; failures never mask the outcome.
def _descriptor_fingerprint(descriptor: dict[str, Any]) -> str:
canonical = json.dumps(descriptor, sort_keys=True, separators=(",", ":"))
return sha256(canonical.encode()).hexdigest()
def _finalize_provider_receipt(operation_id: str | None, status: str, effect_state: str, summary: dict[str, Any] | None = None) -> None:
if operation_id is None:
return
try:
with SessionLocal() as db:
complete_provider_operation(db, operation_id, status=status, effect_state=effect_state, summary=summary)
db.commit()
except Exception as exc:
logger.explore("Receipt finalization failed", src=_SRC, payload={"operation_id": operation_id, "status": status}, error=repr(exc))
# #endregion ScenarioExecution.BrowserProvider.Receipt
# #endregion ScenarioExecution.BrowserProvider

View File

@@ -0,0 +1,43 @@
# #region ScenarioExecution.BrowserReceipt [C:3] [TYPE Module] [SEMANTICS scenario,execution,provider,browser,operation,receipt,finalize]
# @defgroup ScenarioExecution Durable provider-operation receipt helpers for the browser provider.
# @BRIEF Fingerprint a descriptor and finalize operation receipts outside the run transaction.
# @RELATION DEPENDS_ON -> [ScenarioExecution.ProviderOperations.Service]
# @INVARIANT Receipt finalization never masks the provider outcome: a failure is logged, not raised.
# @RATIONALE Extracted from BrowserProvider (Front 6 INV_7): receipt finalization is a distinct
# concern from action admission/execution and keeps the provider module under 400 LOC.
from __future__ import annotations
import json
from hashlib import sha256
from typing import Any
from src.core.database import SessionLocal
from src.core.logger import logger
from src.services.dashboard_testing.execution.provider_operations import complete_provider_operation
_SRC = "ScenarioExecution.BrowserProvider"
# #region ScenarioExecution.BrowserReceipt.Descriptor [C:2] [TYPE Function] [SEMANTICS provider,operation,descriptor,fingerprint]
# @BRIEF Canonical SHA-256 of an action descriptor for the operation receipt.
def descriptor_fingerprint(descriptor: dict[str, Any]) -> str:
canonical = json.dumps(descriptor, sort_keys=True, separators=(",", ":"))
return sha256(canonical.encode()).hexdigest()
# #endregion ScenarioExecution.BrowserReceipt.Descriptor
# #region ScenarioExecution.BrowserReceipt.Finalize [C:3] [TYPE Function] [SEMANTICS provider,operation,receipt,finalize]
# @BRIEF Finalize the durable receipt outside the run transaction; failures never mask the outcome.
def finalize_provider_receipt(
operation_id: str | None, status: str, effect_state: str, summary: dict[str, Any] | None = None,
) -> None:
if operation_id is None:
return
try:
with SessionLocal() as db:
complete_provider_operation(db, operation_id, status=status, effect_state=effect_state, summary=summary)
db.commit()
except Exception as exc:
logger.explore("Receipt finalization failed", src=_SRC, payload={"operation_id": operation_id, "status": status}, error=repr(exc))
# #endregion ScenarioExecution.BrowserReceipt.Finalize
# #endregion ScenarioExecution.BrowserReceipt

View File

@@ -38,6 +38,7 @@ from src.services.dashboard_testing.execution.live_adapter import LiveAdapterRes
from src.services.dashboard_testing.execution.live_binding import (
binding_from_step,
)
from src.services.dashboard_testing.execution.mime_sniff import sniff_mime
from src.services.dashboard_testing.execution.provider_runtime import (
ProviderEventLoop,
ProviderSubmissionOverflow,
@@ -165,6 +166,7 @@ def build_screenshot_provider(
artifact_refs: list[str] = []
artifact_digests: dict[str, str] = {}
artifact_byte_lengths: dict[str, int] = {}
artifact_content_types: dict[str, str] = {}
for index, path in enumerate(jpeg_paths):
raw = Path(path).read_bytes()
if len(raw) > max_screenshot_bytes:
@@ -182,6 +184,8 @@ def build_screenshot_provider(
artifact_refs.append(content_ref)
artifact_digests[content_ref] = digest
artifact_byte_lengths[content_ref] = len(raw)
# Sniff MIME from the captured bytes (a WebP capture must not be labelled jpeg).
artifact_content_types[content_ref] = sniff_mime(raw) or "image/jpeg"
logger.reflect(
"Screenshot evidence stored", src=_SRC,
payload={"run_id": run_id, "artifact_index": index, "bytes": len(raw)},
@@ -199,7 +203,7 @@ def build_screenshot_provider(
# ScenarioExecution.EvaluationAdapter.Manifest admit screenshot
# evidence (the default byte_length is absent for capture outcomes).
"artifact_byte_lengths": artifact_byte_lengths,
"artifact_content_types": {ref: "image/jpeg" for ref in artifact_refs},
"artifact_content_types": artifact_content_types,
},
artifact_refs=artifact_refs,
artifact_digests=artifact_digests,

View File

@@ -17,6 +17,7 @@ from src.models.scenario_run import ScenarioRun, ScenarioStepRun
from .agent_evaluation import AgentEvaluation, persist_agent_evaluation, validate_evaluation_evidence
from .artifacts import register_step_evidence
from .baseline_resolver import stamp_baseline_pin
from .decision_policy import decide_step_outcome, policy_inputs_from_outcome, verified_evidence_refs
from .dispatch import dispatch_step
from .executor_registry import ScenarioExecutorRegistry
@@ -329,6 +330,7 @@ def _advance_run(db: Session, run: ScenarioRun, registry: ScenarioExecutorRegist
evaluation_record = _evaluation_record_from_outcome(step.step_outcome)
evaluation_publish_failed = False
if step_meta.get("tool") == "agent_evaluation" and isinstance(evaluation_record, dict):
evaluation_record = stamp_baseline_pin(evaluation_record, run.runner_plan or {})
try:
record = AgentEvaluation.model_validate(evaluation_record)
validate_evaluation_evidence(

View File

@@ -0,0 +1,200 @@
# #region Test.Api.ScenarioLiveBindings [C:3] [TYPE Module] [SEMANTICS test,api,scenario,execution,live-binding,admin]
# @BRIEF Admin CRUD for deployment live bindings via an isolated FastAPI app + fake ConfigManager.
# @RELATION BINDS_TO -> [Api.ScenarioLiveBindings]
# @RELATION BINDS_TO -> [ScenarioExecution.Runner.ConfiguredBinding]
# @TEST_CONTRACT: GET/PUT/PATCH /api/scenario-live-bindings -> validated settings.scenario_live_execution_bindings
# @TEST_EDGE: invalid_snapshot -> 422 before persistence
# @TEST_EDGE: binding_ref_mismatch -> 422
# @TEST_EDGE: unknown_toggle -> 404
# @TEST_INVARIANT Api.ScenarioLiveBindings: a registered, enabled binding is resolved server-side by
# ScenarioExecution.Runner.ConfiguredBinding; disabled/unregistered is not.
# -> VERIFIED_BY: test_registered_binding_resolves_server_side
# @TEST_INVARIANT Api.ScenarioLiveBindings: non-admin is denied 403 by the real has_permission checker.
# -> VERIFIED_BY: test_non_admin_denied_403
from __future__ import annotations
from fastapi import FastAPI
from fastapi.testclient import TestClient
from src.api.routes.dashboard_testing.scenario_live_bindings import router
from src.core.config_models import GlobalSettings
from src.dependencies import get_config_manager, get_current_user
from src.models.auth import Role, User
_BINDING_REF = "ss-prod-d11-live-001"
_SNAPSHOT = {
"binding_ref": _BINDING_REF,
"environment_id": "env-prod-01",
"dashboard_release_id": "release-1",
"release_fingerprint": "f" * 64,
"dashboard_id": 11,
"query_model_fingerprint": "q" * 64,
"execution_principal_fingerprint": "p" * 64,
"rls_security_fingerprint": "r" * 64,
"browser_safe_checkpoint_ref": None,
"browser_action_binding_ref": None,
"evidence_owner_type": "scenario_run",
"evidence_ref_policy": "draft_storage_raw_response",
}
_QUERY_MODEL = {
"schema_version": 1,
"environment_id": "env-prod-01",
"dashboard_id": 11,
"title": "Sales Dashboard",
"query_model_fingerprint": "q" * 64,
}
class _FakeConfig:
def __init__(self) -> None:
self.settings = GlobalSettings()
class _FakeConfigManager:
def __init__(self) -> None:
self._config = _FakeConfig()
def get_config(self):
return self._config
def update_global_settings(self, settings):
self._config.settings = settings
return self._config
def _admin() -> User:
role = Role(id="role-admin", name="admin", is_admin=True)
role.permissions = []
user = User(id="user-admin", username="admin", email="admin@test.com")
user.roles = [role]
return user
def _outsider() -> User:
role = Role(id="role-viewer", name="viewer", is_admin=False)
role.permissions = []
user = User(id="user-viewer", username="viewer", email="viewer@test.com")
user.roles = [role]
return user
class _Env:
def __init__(self) -> None:
self.config_manager = _FakeConfigManager()
self.app = FastAPI()
self.app.include_router(router)
self.app.dependency_overrides[get_config_manager] = lambda: self.config_manager
self.client = TestClient(self.app)
def as_user(self, user: User) -> None:
self.app.dependency_overrides[get_current_user] = lambda: user
def _put_body(**overrides) -> dict:
body = {"enabled": True, "binding_snapshot": dict(_SNAPSHOT), "query_model_snapshot": dict(_QUERY_MODEL)}
body.update(overrides)
return body
# #region Test.Api.ScenarioLiveBindings.Crud [C:2] [TYPE Function]
def test_list_empty_then_register_and_toggle():
env = _Env()
env.as_user(_admin())
assert env.client.get("/api/scenario-live-bindings").json() == []
created = env.client.put(f"/api/scenario-live-bindings/{_BINDING_REF}", json=_put_body())
assert created.status_code == 200
assert created.json()["binding_ref"] == _BINDING_REF
assert created.json()["enabled"] is True
listed = env.client.get("/api/scenario-live-bindings").json()
assert [item["binding_ref"] for item in listed] == [_BINDING_REF]
toggled = env.client.patch(f"/api/scenario-live-bindings/{_BINDING_REF}", json={"enabled": False})
assert toggled.status_code == 200
assert toggled.json()["enabled"] is False
def test_put_rejects_invalid_snapshot_without_persisting():
env = _Env()
env.as_user(_admin())
bad = {**_SNAPSHOT, "evidence_owner_type": "draft"}
response = env.client.put(f"/api/scenario-live-bindings/{_BINDING_REF}", json=_put_body(binding_snapshot=bad))
assert response.status_code == 422
assert env.config_manager.get_config().settings.scenario_live_execution_bindings == []
def test_put_rejects_binding_ref_mismatch():
env = _Env()
env.as_user(_admin())
response = env.client.put("/api/scenario-live-bindings/other-ref", json=_put_body())
assert response.status_code == 422
assert env.config_manager.get_config().settings.scenario_live_execution_bindings == []
def test_put_rejects_invalid_and_mismatched_query_model():
env = _Env()
env.as_user(_admin())
invalid = env.client.put(
f"/api/scenario-live-bindings/{_BINDING_REF}",
json=_put_body(query_model_snapshot={"totally": "bogus"}),
)
assert invalid.status_code == 422
mismatched = env.client.put(
f"/api/scenario-live-bindings/{_BINDING_REF}",
json=_put_body(query_model_snapshot={**_QUERY_MODEL, "environment_id": "other-env"}),
)
assert mismatched.status_code == 422
assert env.config_manager.get_config().settings.scenario_live_execution_bindings == []
def test_toggle_unknown_binding_is_404():
env = _Env()
env.as_user(_admin())
response = env.client.patch("/api/scenario-live-bindings/missing-ref", json={"enabled": True})
assert response.status_code == 404
def test_non_admin_denied_403():
env = _Env()
env.as_user(_outsider())
assert env.client.get("/api/scenario-live-bindings").status_code == 403
assert env.client.put(f"/api/scenario-live-bindings/{_BINDING_REF}", json=_put_body()).status_code == 403
assert env.client.patch(f"/api/scenario-live-bindings/{_BINDING_REF}", json={"enabled": True}).status_code == 403
# #endregion Test.Api.ScenarioLiveBindings.Crud
# #region Test.Api.ScenarioLiveBindings.Resolve [C:2] [TYPE Function]
def test_registered_binding_resolves_server_side():
from src.services.dashboard_testing.execution.start_run import _resolve_configured_live_binding
env = _Env()
env.as_user(_admin())
env.client.put(f"/api/scenario-live-bindings/{_BINDING_REF}", json=_put_body())
plan = {"steps": [{"dashboard_id": 11}]}
principal = _SNAPSHOT["execution_principal_fingerprint"]
resolved = _resolve_configured_live_binding(
environment_id="env-prod-01", plan=plan, principal_fingerprint=principal,
config_manager=env.config_manager,
)
assert resolved is not None
assert resolved.binding_ref == _BINDING_REF
# A different principal or environment never adopts the binding.
assert _resolve_configured_live_binding(
environment_id="env-prod-01", plan=plan, principal_fingerprint="z" * 64,
config_manager=env.config_manager,
) is None
env.client.patch(f"/api/scenario-live-bindings/{_BINDING_REF}", json={"enabled": False})
assert _resolve_configured_live_binding(
environment_id="env-prod-01", plan=plan, principal_fingerprint=principal,
config_manager=env.config_manager,
) is None
# #endregion Test.Api.ScenarioLiveBindings.Resolve
# #endregion Test.Api.ScenarioLiveBindings

View File

@@ -8,6 +8,7 @@ from __future__ import annotations
import uuid
from datetime import UTC, datetime
from hashlib import sha256
import pytest
@@ -307,6 +308,29 @@ def test_omitted_criterion_id_maps_to_the_single_pinned_criterion():
# #endregion Test.ScenarioExecution.EvaluationAdapter.Normalize
# #region Test.ScenarioExecution.EvaluationAdapter.Identity [C:2] [TYPE Function]
# @BRIEF Provider text cannot author baseline identity; the adapter zeroes any provider baseline_pin.
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter.Normalize: a provider-supplied baseline_pin is
# discarded; only the walker's server-resolved plan pin is persisted.
# -> VERIFIED_BY: test_adapter_drops_provider_supplied_baseline_pin
def test_adapter_drops_provider_supplied_baseline_pin():
record = _run_adapter_with_response({
"status": "succeeded",
"verdict": "inconclusive",
"confidence": 0.2,
"findings": [{
"criterion_id": "crit-visual",
"evidence_artifact_ids": ["draft:run-norm:" + "e" * 64],
"message": "claimed pin",
}],
"reason_codes": [],
"baseline_pin": {"catalog_revision_id": "provider-claimed"},
"usage": {"input_tokens": 1, "output_tokens": 1},
})
assert record["baseline_pin"] == {}
# #endregion Test.ScenarioExecution.EvaluationAdapter.Identity
# #region Test.ScenarioExecution.EvaluationAdapter.Empty [C:2] [TYPE Function]
# @BRIEF Missing prior-step evidence is fail-closed; the adapter never synthesizes PASS.
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter: empty completed fails closed. -> VERIFIED_BY: adapter_empty_completed_fail_closed
@@ -343,4 +367,197 @@ def test_composition_root_evaluation_adapter_is_composed_and_fail_closed():
assert outcome["status"] == "inconclusive"
assert outcome["error_code"] is not None
# #endregion Test.ScenarioExecution.EvaluationAdapter.Composition
# #region Test.ScenarioExecution.EvaluationAdapter.Multimodal [C:3] [TYPE Function]
# @BRIEF Captured images reach the judge as sniffed content-parts; text-only providers stay text-only.
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter.Images: only manifest image items whose bytes
# sniff to an allowlisted image MIME are attached; unreadable/non-image items are skipped.
# -> VERIFIED_BY: adapter_attaches_sniffed_images, adapter_skips_invalid_evidence
# @TEST_INVARIANT ScenarioExecution.AgentEvaluation.Messages: images become base64 data URLs only for a
# multimodal provider. -> VERIFIED_BY: submit_evaluation_multimodal_gate
_PNG_BYTES = b"\x89PNG\r\n\x1a\n" + b"evidence"
_JPEG_BYTES = b"\xff\xd8\xff\xe0" + b"evidence"
class _RetrievingStore(_EvidenceStore):
def __init__(self, payloads: dict[str, bytes]) -> None:
super().__init__()
self.payloads = payloads
def retrieve(self, content_ref: str):
return self.payloads.get(content_ref)
def _image_outcome(ref: str, digest: str, payload: bytes, **nested_overrides) -> dict:
nested = {
"tool": "screenshot",
"artifact_digests": {ref: digest},
"artifact_content_types": {ref: "image/png"},
"artifact_byte_lengths": {ref: len(payload)},
}
nested.update(nested_overrides)
return {"status": "passed", "artifact_refs": [ref], "step_outcome": nested}
def test_adapter_attaches_sniffed_images_from_manifest():
from src.services.dashboard_testing.execution.evaluation_adapter import evaluation_adapter_from
png_digest = sha256(_PNG_BYTES).hexdigest()
prior_ref = f"draft:run-mm:{png_digest}"
run_id = str(uuid.uuid4())
captured: dict = {}
def _submit(**kwargs):
captured.update(kwargs)
return _eval_dict()
store = _RetrievingStore({prior_ref: _PNG_BYTES})
adapter = evaluation_adapter_from(storage=store, submit=_submit)
adapter(
{
"logical_step_id": "step-eval-1",
"scenario_run_id": run_id,
"attempt": 1,
"agent_evaluation_spec": _spec().model_dump(mode="json"),
},
{"s1-shot": _image_outcome(prior_ref, png_digest, _PNG_BYTES)},
)
assert captured["images"] == [(_PNG_BYTES, "image/png")]
def test_adapter_skips_invalid_evidence_before_attaching():
from src.services.dashboard_testing.execution.evaluation_adapter import evaluation_adapter_from
json_ref = "draft:run-mm-json:" + ("a" * 64)
bad_ref = "draft:run-mm-bad:" + ("b" * 64)
unreadable_ref = "draft:run-mm-missing:" + ("c" * 64)
run_id = str(uuid.uuid4())
captured: dict = {}
def _submit(**kwargs):
captured.update(kwargs)
return _eval_dict()
store = _RetrievingStore({bad_ref: b"not-an-image-signature"})
adapter = evaluation_adapter_from(storage=store, submit=_submit)
adapter(
{
"logical_step_id": "step-eval-1",
"scenario_run_id": run_id,
"attempt": 1,
"agent_evaluation_spec": _spec().model_dump(mode="json"),
},
{
"s1-json": {
"status": "passed",
"artifact_refs": [json_ref],
"step_outcome": {
"tool": "superset_api",
"artifact_digests": {json_ref: "a" * 64},
"artifact_content_types": {json_ref: "application/json"},
"artifact_byte_lengths": {json_ref: 10},
},
},
"s2-bad": _image_outcome(bad_ref, "b" * 64, b"not-an-image-signature"),
"s3-missing": _image_outcome(unreadable_ref, "c" * 64, b"\x89PNG-still"),
},
)
assert captured["images"] == []
def test_evaluation_messages_text_only_and_multimodal():
from src.services.dashboard_testing.execution.agent_evaluation import _evaluation_messages
assert _evaluation_messages("P", None) == [{"role": "user", "content": "P"}]
parts = _evaluation_messages("P", [(b"abc", "image/png")])
content = parts[0]["content"]
assert content[0] == {"type": "text", "text": "P"}
assert content[1]["image_url"]["url"] == "data:image/png;base64,YWJj"
@pytest.mark.parametrize("is_multimodal, expects_parts", [(True, True), (False, False)])
def test_submit_evaluation_gates_images_on_multimodal_provider(monkeypatch, is_multimodal, expects_parts):
import asyncio
import src.plugins.llm_analysis.service as llm_service_module
import src.services.llm_provider as llm_provider_module
from src.services.dashboard_testing.execution import agent_evaluation as ae
class _Provider:
provider_type = "openai"
base_url = "http://local-llm"
is_multimodal = False
_Provider.is_multimodal = is_multimodal
seen: dict = {}
class _Client:
def __init__(self, *args, **kwargs):
pass
async def get_json_completion(self, messages):
seen["messages"] = messages
return {"status": "succeeded"}
class _ProviderService:
def __init__(self, db):
pass
def get_provider(self, provider_id):
return _Provider()
def get_decrypted_api_key(self, provider_id):
return "key"
monkeypatch.setattr(llm_provider_module, "LLMProviderService", _ProviderService)
monkeypatch.setattr(llm_service_module, "LLMClient", _Client)
monkeypatch.setattr(ae, "claim_capacity", lambda *args, **kwargs: {"lease_id": "lease-1"})
monkeypatch.setattr(ae, "release_capacity", lambda *args, **kwargs: None)
result = asyncio.run(ae.submit_evaluation(
object(), spec=_spec(), prompt="P", images=[(b"abc", "image/png")],
environment_id="env", environment_class="DEV", run_id="run", logical_step_id="step",
))
assert result == {"status": "succeeded"}
content = seen["messages"][0]["content"]
if expects_parts:
assert isinstance(content, list)
assert content[1]["image_url"]["url"] == "data:image/png;base64,YWJj"
else:
assert content == "P"
def _image_manifest_item(ref: str, data: bytes) -> dict:
return {
"artifact_id": ref, "sha256": sha256(data).hexdigest(),
"content_type": "image/png", "byte_length": len(data), "role": "actual",
}
def test_image_payloads_respect_count_and_byte_budget(monkeypatch):
from src.services.dashboard_testing.execution import evaluation_images as ev
payload_a = b"\x89PNG\r\n\x1a\n" + b"a" * 92
payload_b = b"\x89PNG\r\n\x1a\n" + b"b" * 92
store = _RetrievingStore({"ref-a": payload_a, "ref-b": payload_b})
manifest = [_image_manifest_item("ref-a", payload_a), _image_manifest_item("ref-b", payload_b)]
assert ev.image_payloads_from_manifest(manifest, store, max_images=1) == [(payload_a, "image/png")]
monkeypatch.setattr(ev, "MAX_EVALUATION_IMAGE_BYTES", 150)
assert ev.image_payloads_from_manifest(manifest, store, max_images=8) == [(payload_a, "image/png")]
def test_image_payloads_skip_digest_mismatch():
from src.services.dashboard_testing.execution.evaluation_images import image_payloads_from_manifest
store = _RetrievingStore({"ref-a": _PNG_BYTES})
tampered = {**_image_manifest_item("ref-a", _PNG_BYTES), "sha256": "0" * 64}
assert image_payloads_from_manifest([tampered], store, max_images=8) == []
# #endregion Test.ScenarioExecution.EvaluationAdapter.Multimodal
# #endregion Test.ScenarioExecution.AgentEvaluation

View File

@@ -32,6 +32,7 @@ from src.services.dashboard_testing.execution.baseline_resolver import (
graph_has_baseline_refs,
load_published_catalog,
resolve_baseline_pin,
stamp_baseline_pin,
)
from src.services.dashboard_testing.execution.runner import start_run
from src.services.dashboard_testing.scenario.templates import (
@@ -295,4 +296,23 @@ def test_idempotency_rejects_different_pin():
session.close()
# #endregion Test.ScenarioExecution.BaselineResolver.Idempotency
# #region Test.ScenarioExecution.BaselineResolver.Stamp [C:2] [TYPE Function]
# @BRIEF baseline_pin is server-owned: the plan pin always wins; a provider-claimed pin is discarded.
# @TEST_INVARIANT ScenarioExecution.BaselineResolver.StampEvaluation: any non-plan pin, including one
# supplied by the provider, is replaced by the resolved plan pin (or {}). -> VERIFIED_BY: test_stamp_baseline_pin_enforces_server_authority
def test_stamp_baseline_pin_enforces_server_authority():
plan_pin = {"catalog_revision_id": "r-1", "baseline_set_id": "ss-prod-visual", "baseline_set_version": "1"}
filled = stamp_baseline_pin({"baseline_pin": {}}, {"baseline_pin": plan_pin})
assert filled["baseline_pin"] == plan_pin
# A provider-claimed pin must never survive: the server plan pin is authoritative.
claimed = {"catalog_revision_id": "provider-claimed"}
overwritten = stamp_baseline_pin({"baseline_pin": claimed}, {"baseline_pin": plan_pin})
assert overwritten["baseline_pin"] == plan_pin
assert stamp_baseline_pin({"baseline_pin": claimed}, {"baseline_pin": None})["baseline_pin"] == {}
assert stamp_baseline_pin({"baseline_pin": {}}, {}) == {"baseline_pin": {}}
assert stamp_baseline_pin(None, {"baseline_pin": plan_pin}) is None
# #endregion Test.ScenarioExecution.BaselineResolver.Stamp
# #endregion Test.ScenarioExecution.BaselineResolver

View File

@@ -419,4 +419,33 @@ def test_read_only_pass_reports_effect_state_none(loop_runtime, clean_env):
assert result.details["effect_state"] == "none"
assert "operation_id" not in result.details
# #endregion Test.ScenarioExecution.BrowserProvider.MutationOutcome
# #region Test.ScenarioExecution.BrowserProvider.MimeSniff [C:2] [TYPE Function] [SEMANTICS test,provider,browser,mime]
# @BRIEF Content type is sniffed from stored evidence bytes, never assumed image/png.
# @TEST_INVARIANT A WebP evidence payload must not be labelled image/png (evaluation MIME match).
def test_browser_evidence_content_type_is_sniffed(loop_runtime, clean_env):
binding = clean_env
webp = b"RIFF\x00\x00\x00\x00WEBP" + b"payload"
provider = build_browser_provider(transport=FakeTransport(evidence=webp), storage=FakeStorage(), loop=loop_runtime)
result = provider(LiveProviderContext(binding=binding, step=build_step(binding), completed={}))
assert result.status == "passed"
ref = result.artifact_refs[0]
assert result.details["artifact_content_types"][ref] == "image/webp"
def test_browser_unrecognized_evidence_falls_back_to_png(loop_runtime, clean_env):
binding = clean_env
provider = build_browser_provider(
transport=FakeTransport(evidence=b"raw-evidence-bytes"), storage=FakeStorage(), loop=loop_runtime,
)
result = provider(LiveProviderContext(binding=binding, step=build_step(binding), completed={}))
assert result.status == "passed"
ref = result.artifact_refs[0]
assert result.details["artifact_content_types"][ref] == "image/png"
# #endregion Test.ScenarioExecution.BrowserProvider.MimeSniff
# #endregion Test.ScenarioExecution.BrowserProvider

View File

@@ -270,5 +270,39 @@ def test_capture_running_on_shared_loop_is_verified(loop_runtime, clean_env):
assert result.status == "passed"
assert observed["loop"] == id(loop_runtime._loop)
# #region Test.ScenarioExecution.ScreenshotProvider.MimeSniff [C:2] [TYPE Function] [SEMANTICS test,provider,screenshot,mime]
# @BRIEF Content types are sniffed per ref from the captured bytes, never assumed image/jpeg.
# @TEST_INVARIANT WebP or PNG captures must not be labelled image/jpeg (evaluation MIME match).
def test_capture_content_types_are_sniffed_per_ref(loop_runtime, clean_env, tmp_path):
binding = clean_env
webp = b"RIFF\x00\x00\x00\x00WEBP" + b"payload"
jpeg = b"\xff\xd8\xff\xe0" + b"payload"
first, second = tmp_path / "a.webp", tmp_path / "b.jpg"
first.write_bytes(webp)
second.write_bytes(jpeg)
service = FakeCaptureService(paths=[str(first), str(second)])
provider = build_screenshot_provider(service=service, storage=FakeStorage(), loop=loop_runtime)
result = provider(LiveProviderContext(binding=binding, step=build_step(binding), completed={}))
assert result.status == "passed"
types = result.details["artifact_content_types"]
refs = result.artifact_refs
assert [types[refs[0]], types[refs[1]]] == ["image/webp", "image/jpeg"]
def test_capture_unrecognized_bytes_fall_back_to_jpeg(loop_runtime, clean_env):
binding = clean_env
service = FakeCaptureService(payload=b"not-an-image-signature")
provider = build_screenshot_provider(service=service, storage=FakeStorage(), loop=loop_runtime)
result = provider(LiveProviderContext(binding=binding, step=build_step(binding), completed={}))
assert result.status == "passed"
ref = result.artifact_refs[0]
assert result.details["artifact_content_types"][ref] == "image/jpeg"
# #endregion Test.ScenarioExecution.ScreenshotProvider.MimeSniff
# #endregion Test.ScenarioExecution.ScreenshotProvider.Evidence
# #endregion Test.ScenarioExecution.ScreenshotProvider

View File

@@ -120,6 +120,12 @@ def _mock_eval_adapter(step: dict, _completed: dict) -> dict:
# #endregion Test.ScenarioExecution.Walker.MockEvalAdapter
def _mock_eval_adapter_empty_pin(step: dict, completed: dict) -> dict:
envelope = _mock_eval_adapter(step, completed)
envelope["evaluation_record"] = {**envelope["evaluation_record"], "baseline_pin": {}}
return envelope
def test_walker_runs_fixture_to_completion(seeded_execution):
run = start_run(
seeded_execution,
@@ -436,6 +442,36 @@ def test_walker_evaluation_baseline_and_semantic_pass(seeded_execution):
# #endregion Test.ScenarioExecution.Walker.EvaluationPass
# #region Test.ScenarioExecution.Walker.EvaluationPin [C:2] [TYPE Function]
# @BRIEF The walker stamps the plan's resolved baseline pin into a record that carries none.
# @TEST_INVARIANT ScenarioExecution.BaselineResolver.StampEvaluation: a baseline-backed plan's pin is
# persisted into the AgentEvaluation record. -> VERIFIED_BY: walker_stamps_plan_baseline_pin
def test_walker_stamps_plan_baseline_pin(seeded_execution):
run = start_run(
seeded_execution,
"aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaa1",
"bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbb1",
{},
"env-preprod-02",
actor="test-walker",
idempotency_key="walker-eval-pin-001",
auto_advance=False,
)
resolved_pin = {"catalog_revision_id": str(uuid.uuid4()), "baseline_set_id": "ss-prod-visual", "baseline_set_version": "1"}
plan = _required_eval_plan("eval-pin-044")
plan["baseline_pin"] = resolved_pin
run.runner_plan = plan
seeded_execution.flush()
registry = ScenarioExecutorRegistry()
_register_default_executors(registry, agent_evaluation_adapter=_mock_eval_adapter_empty_pin)
_advance_run(seeded_execution, run, registry, worker_id="test-walker")
row = seeded_execution.query(AgentEvaluationRow).filter_by(scenario_run_id=run.id).one()
assert row.baseline_pin == resolved_pin
# #endregion Test.ScenarioExecution.Walker.EvaluationPin
# #region Test.ScenarioExecution.Walker.EvaluationUnavailable [C:2] [TYPE Function]
# @BRIEF Required evaluation with no adapter is inconclusive EVALUATION_UNAVAILABLE.
# @TEST_INVARIANT ScenarioExecution.AgentEvaluation: required + missing adapter -> EVALUATION_UNAVAILABLE. -> VERIFIED_BY: walker_required_missing_adapter_unavailable