feat(scenario): live eval trust — MIME sniffing, multimodal evidence, baseline pin stamping, binding admin surface
This commit is contained in:
@@ -24,6 +24,9 @@ from src.api.routes.dashboard_testing.scenario_run_center import router as scena
|
||||
from src.api.routes.dashboard_testing.scenario_artifact_content import (
|
||||
router as scenario_artifact_content_router,
|
||||
)
|
||||
from src.api.routes.dashboard_testing.scenario_live_bindings import (
|
||||
router as scenario_live_bindings_router,
|
||||
)
|
||||
from src.api.routes.dashboard_testing.scenario_runs import (
|
||||
history_router as scenario_runs_history_router,
|
||||
runs_router as scenario_runs_router,
|
||||
@@ -48,6 +51,7 @@ router.include_router(scenario_router)
|
||||
router.include_router(scenarios_router)
|
||||
router.include_router(scenario_runs_router)
|
||||
router.include_router(scenario_artifact_content_router)
|
||||
router.include_router(scenario_live_bindings_router)
|
||||
router.include_router(scenario_runs_history_router)
|
||||
router.include_router(scenario_run_center_router)
|
||||
router.include_router(scenario_analytics_router)
|
||||
|
||||
@@ -0,0 +1,164 @@
|
||||
# #region Api.ScenarioLiveBindings [C:4] [TYPE Module] [SEMANTICS scenario,execution,live-binding,admin,api,config]
|
||||
# @defgroup Api Trusted admin CRUD for deployment-configured 044 live-execution bindings.
|
||||
# @BRIEF Admin-only list/register/toggle of `settings.scenario_live_execution_bindings` via ConfigManager.
|
||||
# @RELATION DEPENDS_ON -> [Core.ConfigManager]
|
||||
# @RELATION DEPENDS_ON -> [ScenarioExecution.LiveBinding.Identity]
|
||||
# @INVARIANT Writes go through ConfigManager.update_global_settings so the persisted settings remain the
|
||||
# single authority that ScenarioExecution.Runner.ConfiguredBinding reads at start.
|
||||
# @INVARIANT A binding snapshot is validated by LiveExecutionBinding.from_snapshot; a malformed snapshot
|
||||
# never enters the configuration. No credentials exist in a snapshot (identity fingerprints only).
|
||||
# @RATIONALE Track B (2026-09-10): server-side binding resolution existed but the only registration path
|
||||
# was a direct DB/config write. A reviewed admin surface closes that gap without exposing a
|
||||
# client-facing binding selector (the request body still never carries a binding).
|
||||
# @REJECTED Exposing bindings through MCP was rejected for this change — the live-binding identity is a
|
||||
# deployment trust boundary, not agent-authorable state; a separate architect decision is required.
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from fastapi import APIRouter, Depends, HTTPException
|
||||
from pydantic import BaseModel, ConfigDict, ValidationError
|
||||
|
||||
from src.core.config_manager import ConfigManager
|
||||
from src.core.config_models import ScenarioLiveExecutionBindingConfig
|
||||
from src.dependencies import get_config_manager, has_permission
|
||||
from src.schemas.dashboard_testing import DashboardQueryModel
|
||||
from src.services.dashboard_testing.execution.live_binding import LiveExecutionBinding
|
||||
|
||||
router = APIRouter(prefix="/api/scenario-live-bindings", tags=["scenario-live-bindings"])
|
||||
|
||||
|
||||
# #region Api.ScenarioLiveBindings.Schemas [C:2] [TYPE Class] [SEMANTICS scenario,execution,live-binding,dto]
|
||||
# @BRIEF Bounded request/response DTOs; the snapshot identity is echoed without any secret material.
|
||||
class LiveBindingUpsertRequest(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
enabled: bool = False
|
||||
binding_snapshot: dict[str, Any]
|
||||
query_model_snapshot: dict[str, Any]
|
||||
|
||||
|
||||
class LiveBindingToggleRequest(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
enabled: bool
|
||||
|
||||
|
||||
class LiveBindingView(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
binding_ref: str | None
|
||||
enabled: bool
|
||||
environment_id: str | None = None
|
||||
dashboard_id: int | None = None
|
||||
dashboard_release_id: str | None = None
|
||||
execution_principal_fingerprint: str | None = None
|
||||
query_model_fingerprint: str | None = None
|
||||
rls_security_fingerprint: str | None = None
|
||||
# #endregion Api.ScenarioLiveBindings.Schemas
|
||||
|
||||
|
||||
# #region Api.ScenarioLiveBindings.Project [C:2] [TYPE Function]
|
||||
# @BRIEF Project one config record to a view; a malformed snapshot is surfaced, never hidden or trusted.
|
||||
def _project(record: ScenarioLiveExecutionBindingConfig) -> LiveBindingView:
|
||||
snapshot = record.binding_snapshot if isinstance(record.binding_snapshot, dict) else {}
|
||||
dashboard_id = snapshot.get("dashboard_id")
|
||||
return LiveBindingView(
|
||||
binding_ref=snapshot.get("binding_ref") if isinstance(snapshot.get("binding_ref"), str) else None,
|
||||
enabled=bool(record.enabled),
|
||||
environment_id=snapshot.get("environment_id") if isinstance(snapshot.get("environment_id"), str) else None,
|
||||
dashboard_id=dashboard_id if isinstance(dashboard_id, int) else None,
|
||||
dashboard_release_id=snapshot.get("dashboard_release_id") if isinstance(snapshot.get("dashboard_release_id"), str) else None,
|
||||
execution_principal_fingerprint=snapshot.get("execution_principal_fingerprint") if isinstance(snapshot.get("execution_principal_fingerprint"), str) else None,
|
||||
query_model_fingerprint=snapshot.get("query_model_fingerprint") if isinstance(snapshot.get("query_model_fingerprint"), str) else None,
|
||||
rls_security_fingerprint=snapshot.get("rls_security_fingerprint") if isinstance(snapshot.get("rls_security_fingerprint"), str) else None,
|
||||
)
|
||||
# #endregion Api.ScenarioLiveBindings.Project
|
||||
|
||||
|
||||
# #region Api.ScenarioLiveBindings.Store [C:2] [TYPE Function]
|
||||
# @BRIEF Replace the settings list atomically through ConfigManager (single authority at start).
|
||||
def _store(config_manager: ConfigManager, records: list[ScenarioLiveExecutionBindingConfig]) -> None:
|
||||
settings = config_manager.get_config().settings
|
||||
updated = settings.model_copy(update={"scenario_live_execution_bindings": records})
|
||||
config_manager.update_global_settings(updated)
|
||||
# #endregion Api.ScenarioLiveBindings.Store
|
||||
|
||||
|
||||
# #region Api.ScenarioLiveBindings.Ref [C:1] [TYPE Function]
|
||||
# @BRIEF Best-effort binding_ref of a stored record; malformed snapshots are skipped by identity.
|
||||
def _record_ref(record: ScenarioLiveExecutionBindingConfig) -> str | None:
|
||||
snapshot = record.binding_snapshot
|
||||
ref = snapshot.get("binding_ref") if isinstance(snapshot, dict) else None
|
||||
return ref if isinstance(ref, str) and ref else None
|
||||
# #endregion Api.ScenarioLiveBindings.Ref
|
||||
|
||||
|
||||
# #region Api.ScenarioLiveBindings.List [C:2] [TYPE Function] [SEMANTICS scenario,execution,live-binding,list]
|
||||
# @BRIEF Admin list of configured live bindings with identity fields only.
|
||||
@router.get("", response_model=list[LiveBindingView])
|
||||
def list_live_bindings(
|
||||
config_manager: ConfigManager = Depends(get_config_manager),
|
||||
_=Depends(has_permission("admin:settings", "READ")),
|
||||
):
|
||||
records = config_manager.get_config().settings.scenario_live_execution_bindings or []
|
||||
return [_project(record) for record in records]
|
||||
# #endregion Api.ScenarioLiveBindings.List
|
||||
|
||||
|
||||
# #region Api.ScenarioLiveBindings.Upsert [C:3] [TYPE Function] [SEMANTICS scenario,execution,live-binding,upsert]
|
||||
# @BRIEF Validate and create-or-replace one binding (canonicalized snapshot) in the config authority.
|
||||
# @INVARIANT A snapshot that fails LiveExecutionBinding.from_snapshot, or whose binding_ref differs
|
||||
# from the path, is rejected with no persistence.
|
||||
@router.put("/{binding_ref}", response_model=LiveBindingView)
|
||||
def upsert_live_binding(
|
||||
binding_ref: str,
|
||||
request: LiveBindingUpsertRequest,
|
||||
config_manager: ConfigManager = Depends(get_config_manager),
|
||||
_=Depends(has_permission("admin:settings", "WRITE")),
|
||||
):
|
||||
try:
|
||||
binding = LiveExecutionBinding.from_snapshot(request.binding_snapshot)
|
||||
except ValueError:
|
||||
raise HTTPException(status_code=422, detail="LIVE_BINDING_SNAPSHOT_INVALID") from None
|
||||
if binding.binding_ref != binding_ref:
|
||||
raise HTTPException(status_code=422, detail="LIVE_BINDING_REF_MISMATCH")
|
||||
try:
|
||||
query_model = DashboardQueryModel.model_validate(request.query_model_snapshot)
|
||||
except ValidationError:
|
||||
raise HTTPException(status_code=422, detail="LIVE_BINDING_QUERY_MODEL_INVALID") from None
|
||||
if query_model.environment_id != binding.environment_id or query_model.dashboard_id != binding.dashboard_id:
|
||||
raise HTTPException(status_code=422, detail="LIVE_BINDING_QUERY_MODEL_MISMATCH")
|
||||
record = ScenarioLiveExecutionBindingConfig(
|
||||
enabled=request.enabled,
|
||||
binding_snapshot=binding.snapshot(),
|
||||
query_model_snapshot=query_model.model_dump(mode="json"),
|
||||
)
|
||||
records = list(config_manager.get_config().settings.scenario_live_execution_bindings or [])
|
||||
for index, existing in enumerate(records):
|
||||
if _record_ref(existing) == binding_ref:
|
||||
records[index] = record
|
||||
break
|
||||
else:
|
||||
records.append(record)
|
||||
_store(config_manager, records)
|
||||
return _project(record)
|
||||
# #endregion Api.ScenarioLiveBindings.Upsert
|
||||
|
||||
|
||||
# #region Api.ScenarioLiveBindings.Toggle [C:2] [TYPE Function] [SEMANTICS scenario,execution,live-binding,toggle]
|
||||
# @BRIEF Enable or disable an existing binding by ref; an unknown ref is 404.
|
||||
@router.patch("/{binding_ref}", response_model=LiveBindingView)
|
||||
def toggle_live_binding(
|
||||
binding_ref: str,
|
||||
request: LiveBindingToggleRequest,
|
||||
config_manager: ConfigManager = Depends(get_config_manager),
|
||||
_=Depends(has_permission("admin:settings", "WRITE")),
|
||||
):
|
||||
records = list(config_manager.get_config().settings.scenario_live_execution_bindings or [])
|
||||
for index, existing in enumerate(records):
|
||||
if _record_ref(existing) == binding_ref:
|
||||
updated = existing.model_copy(update={"enabled": request.enabled})
|
||||
records[index] = updated
|
||||
_store(config_manager, records)
|
||||
return _project(updated)
|
||||
raise HTTPException(status_code=404, detail="LIVE_BINDING_NOT_FOUND")
|
||||
# #endregion Api.ScenarioLiveBindings.Toggle
|
||||
# #endregion Api.ScenarioLiveBindings
|
||||
@@ -5,6 +5,7 @@
|
||||
# @REJECTED A permissive dict passthrough was rejected because it allows findings to claim undeclared deterministic authority.
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
from datetime import UTC, datetime
|
||||
from typing import Any, Literal
|
||||
import uuid
|
||||
@@ -12,6 +13,7 @@ import uuid
|
||||
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from src.core.logger import logger
|
||||
from src.models.scenario_evaluation import AgentEvaluation as AgentEvaluationRow
|
||||
from src.services.dashboard_testing.scenario.models import AgentEvaluationSpec
|
||||
from src.models.scenario_artifact import ScenarioArtifact
|
||||
@@ -206,8 +208,29 @@ def validate_evaluation_evidence(
|
||||
# #endregion ScenarioExecution.AgentEvaluation.Evidence
|
||||
|
||||
|
||||
# #region ScenarioExecution.AgentEvaluation.Messages [C:3] [TYPE Function] [SEMANTICS evaluation,multimodal,messages,content-parts]
|
||||
# @BRIEF Compose the OpenAI chat message: text-only, or the prompt plus image data URLs.
|
||||
# @PRE images, when present, are (bytes, allowlisted-image-MIME) pairs already bounded by the adapter.
|
||||
# @POST Returns one user message whose content parts carry the prompt and base64 image URLs; no
|
||||
# image parts are produced for an empty/None list, preserving the text-only transport.
|
||||
# @INVARIANT Image bytes are inlined only as data URLs; MIME comes from server-verified manifest bytes.
|
||||
# @RATIONALE A text-only prompt left the live judge blind (canary v2: honest low-confidence).
|
||||
# Attaching the captured evidence lets a multimodal model reason over real pixels.
|
||||
# @REJECTED Uploading evidence to an external file host was rejected — bytes never leave the
|
||||
# bounded provider call and no third-party URL enters the evaluation record.
|
||||
def _evaluation_messages(prompt: str, images: list[tuple[bytes, str]] | None) -> list[dict[str, Any]]:
|
||||
if not images:
|
||||
return [{"role": "user", "content": prompt}]
|
||||
parts: list[dict[str, Any]] = [{"type": "text", "text": prompt}]
|
||||
for data, content_type in images:
|
||||
encoded = base64.b64encode(data).decode("ascii")
|
||||
parts.append({"type": "image_url", "image_url": {"url": f"data:{content_type};base64,{encoded}"}})
|
||||
return [{"role": "user", "content": parts}]
|
||||
# #endregion ScenarioExecution.AgentEvaluation.Messages
|
||||
|
||||
|
||||
async def submit_evaluation(
|
||||
db: Session, *, spec: AgentEvaluationSpec, prompt: str, artifacts: list[bytes],
|
||||
db: Session, *, spec: AgentEvaluationSpec, prompt: str, images: list[tuple[bytes, str]] | None = None,
|
||||
environment_id: str, environment_class: str, run_id: str, logical_step_id: str,
|
||||
client: Any = None,
|
||||
) -> dict[str, Any]:
|
||||
@@ -229,7 +252,16 @@ async def submit_evaluation(
|
||||
if not api_key:
|
||||
raise RuntimeError("EVALUATION_PROVIDER_MISSING")
|
||||
llm = client or LLMClient(LLMProviderType(provider.provider_type), api_key, provider.base_url, spec.model_id)
|
||||
result = await llm.get_json_completion([{"role": "user", "content": prompt}])
|
||||
# Images reach the model only through a multimodal provider; a text-only provider
|
||||
# keeps the honest text-only manifest (never a fabricated visual verdict).
|
||||
attached = images if bool(provider.is_multimodal) else None
|
||||
if attached:
|
||||
logger.reflect(
|
||||
"Multimodal images attached to the evaluation prompt",
|
||||
src="ScenarioExecution.AgentEvaluation",
|
||||
payload={"image_count": len(attached), "total_bytes": sum(len(data) for data, _ in attached)},
|
||||
)
|
||||
result = await llm.get_json_completion(_evaluation_messages(prompt, attached))
|
||||
if not isinstance(result, dict):
|
||||
raise RuntimeError("EVALUATION_RESPONSE_INVALID")
|
||||
return result
|
||||
|
||||
@@ -29,6 +29,8 @@ from src.models.scenario_artifact import ScenarioArtifact
|
||||
from src.models.scenario_run import ScenarioRun
|
||||
from src.services.agent_runs.artifacts import get_draft_storage
|
||||
|
||||
from .mime_sniff import sniff_mime
|
||||
|
||||
IMAGE_MAX_BYTES = 10_485_760
|
||||
OTHER_MAX_BYTES = 26_214_400
|
||||
IMAGE_MIME = frozenset({"image/jpeg", "image/png", "image/webp"})
|
||||
@@ -94,23 +96,11 @@ def _has_result_view(user: object) -> bool:
|
||||
# #endregion ScenarioExecution.ArtifactContent.HasView
|
||||
|
||||
|
||||
# #region ScenarioExecution.ArtifactContent.SniffMime [C:2] [TYPE Function] [SEMANTICS scenario,artifact,mime,magic]
|
||||
# @BRIEF Detect allowlisted MIME from complete-object magic bytes; JSON is BOM/whitespace then { or [.
|
||||
def sniff_mime(data: bytes) -> str | None:
|
||||
if data.startswith(b"\xff\xd8\xff"):
|
||||
return "image/jpeg"
|
||||
if data.startswith(b"\x89PNG\r\n\x1a\n"):
|
||||
return "image/png"
|
||||
if len(data) >= 12 and data.startswith(b"RIFF") and data[8:12] == b"WEBP":
|
||||
return "image/webp"
|
||||
if data.startswith(b"%PDF-"):
|
||||
return "application/pdf"
|
||||
if data.startswith(b"PK\x03\x04"):
|
||||
return "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"
|
||||
stripped = data.lstrip(b"\xef\xbb\xbf \t\r\n")
|
||||
if stripped[:1] in (b"{", b"["):
|
||||
return "application/json"
|
||||
return None
|
||||
# #region ScenarioExecution.ArtifactContent.SniffMime [C:1] [TYPE Tombstone] [SEMANTICS scenario,artifact,mime,magic]
|
||||
# @STATUS DEPRECATED
|
||||
# @REPLACED_BY -> [ScenarioExecution.MimeSniff.Detect]
|
||||
# @BRIEF Detection moved to ScenarioExecution.MimeSniff; the module still re-exports sniff_mime
|
||||
# at import time so existing callers and tests resolve unchanged.
|
||||
# #endregion ScenarioExecution.ArtifactContent.SniffMime
|
||||
|
||||
|
||||
|
||||
@@ -126,6 +126,26 @@ def attach_baseline_pin(plan: dict[str, Any], pin: dict[str, Any] | None) -> dic
|
||||
# #endregion ScenarioExecution.BaselineResolver.Attach
|
||||
|
||||
|
||||
# #region ScenarioExecution.BaselineResolver.StampEvaluation [C:3] [TYPE Function] [SEMANTICS baseline,pin,evaluation,record,stamp]
|
||||
# @ingroup ScenarioExecution
|
||||
# @BRIEF Stamp the plan's resolved pin as the record's server-owned baseline_pin.
|
||||
# @PRE record is the adapter-built AgentEvaluation payload; plan is the pinned ScenarioRun.runner_plan.
|
||||
# @POST Returns the record with baseline_pin = plan pin, or {} when the plan resolved no pin.
|
||||
# @INVARIANT baseline_pin is server identity: any provider-claimed value is overwritten, so a record
|
||||
# can never carry a pin the server did not resolve. A null plan pin stamps {}.
|
||||
# @RATIONALE Live canary v2 persisted baseline_pin={} on a baseline-backed plan because the provider
|
||||
# has no pin authority. Stamping at the persist boundary keeps provider text out of identity
|
||||
# and makes the record whole; forcing {} also defeats a model-supplied fake pin.
|
||||
# @REJECTED Threading the pin through the executor step payload was rejected — it widens the executor
|
||||
# contract for a purely server-side identity concern handled at persistence.
|
||||
def stamp_baseline_pin(record: dict[str, Any] | None, plan: dict[str, Any] | None) -> dict[str, Any] | None:
|
||||
if not isinstance(record, dict):
|
||||
return record
|
||||
pin = (plan or {}).get("baseline_pin")
|
||||
return {**record, "baseline_pin": dict(pin) if isinstance(pin, dict) else {}}
|
||||
# #endregion ScenarioExecution.BaselineResolver.StampEvaluation
|
||||
|
||||
|
||||
def _blank(value: Any) -> str | None:
|
||||
if value is None:
|
||||
return None
|
||||
|
||||
@@ -17,10 +17,12 @@ from typing import Any
|
||||
import uuid
|
||||
|
||||
from src.core.database import SessionLocal
|
||||
from src.core.logger import logger
|
||||
from src.services.dashboard_testing.scenario.models import AgentEvaluationSpec
|
||||
|
||||
from .agent_evaluation import AgentEvaluation, parse_evaluation_response, submit_evaluation
|
||||
from .artifacts import is_valid_sha256
|
||||
from .evaluation_images import image_payloads_from_manifest
|
||||
from .live_binding import EvidenceStorage
|
||||
|
||||
_ALLOWED_CONTENT_TYPES = frozenset({"image/jpeg", "image/png", "image/webp", "application/json"})
|
||||
@@ -173,7 +175,9 @@ def _normalize_provider_response(raw: dict[str, Any], spec: AgentEvaluationSpec)
|
||||
payload["output_schema_hash"] = _canonical_hash(spec.output_schema)
|
||||
payload["trust_policy_hash"] = spec.trust_policy_hash
|
||||
payload["comparison_ids"] = list(spec.comparison_refs)
|
||||
payload.setdefault("baseline_pin", {})
|
||||
# baseline_pin is server-owned identity: never accept a provider-claimed pin. The walker
|
||||
# stamps the resolved plan pin (ScenarioExecution.BaselineResolver.StampEvaluation).
|
||||
payload["baseline_pin"] = {}
|
||||
payload.setdefault("reason_codes", [])
|
||||
# EvaluationUsage is required by the pinned record schema; providers omit it or send
|
||||
# partial token counts — the server normalizes rather than trusting provider cost text.
|
||||
@@ -300,6 +304,15 @@ def evaluation_adapter_from(
|
||||
raise RuntimeError("EVALUATION_RESPONSE_INVALID")
|
||||
attempt = int(step.get("attempt") or 1)
|
||||
evidence = storage if storage is not None else _default_storage()
|
||||
images = image_payloads_from_manifest(manifest, evidence, max_images=spec.limits.max_images)
|
||||
if images:
|
||||
# Candidate evidence loaded; actual attachment is gated on provider multimodality
|
||||
# inside submit_evaluation, which logs the attached count.
|
||||
logger.reflect(
|
||||
"Evaluation image evidence loaded from manifest",
|
||||
src="ScenarioExecution.EvaluationAdapter",
|
||||
payload={"image_count": len(images), "total_bytes": sum(len(data) for data, _ in images)},
|
||||
)
|
||||
prompt = json.dumps(
|
||||
{
|
||||
# Live canary v2 (2026-09-10): the model echoed spec.evidence_refs into
|
||||
@@ -319,6 +332,8 @@ def evaluation_adapter_from(
|
||||
"server-owned identity fields (ids, hashes, template and policy metadata) are "
|
||||
"filled by the server; do not emit them",
|
||||
"verdict, confidence (0..1), findings and usage must always be present",
|
||||
"verdict MUST be exactly one of pass, fail, inconclusive — status is a separate "
|
||||
"server-owned field and 'succeeded' is never a verdict value",
|
||||
],
|
||||
},
|
||||
"evaluation_spec": spec.model_dump(mode="json"),
|
||||
@@ -327,7 +342,7 @@ def evaluation_adapter_from(
|
||||
sort_keys=True, separators=(",", ":"), default=str,
|
||||
)
|
||||
if submit is not None:
|
||||
raw = submit(spec=spec, prompt=prompt, step=step, completed=completed)
|
||||
raw = submit(spec=spec, prompt=prompt, step=step, completed=completed, images=images)
|
||||
else:
|
||||
target = step.get("target_snapshot") if isinstance(step.get("target_snapshot"), dict) else {}
|
||||
meta = step.get("step_meta") if isinstance(step.get("step_meta"), dict) else {}
|
||||
@@ -338,7 +353,7 @@ def evaluation_adapter_from(
|
||||
db = (db_factory or SessionLocal)()
|
||||
try:
|
||||
coro = submit_evaluation(
|
||||
db, spec=spec, prompt=prompt, artifacts=[], environment_id=environment_id,
|
||||
db, spec=spec, prompt=prompt, images=images, environment_id=environment_id,
|
||||
environment_class=environment_class, run_id=run_id, logical_step_id=logical_step_id,
|
||||
client=client,
|
||||
)
|
||||
|
||||
@@ -0,0 +1,69 @@
|
||||
# #region ScenarioExecution.EvaluationImages [C:3] [TYPE Module] [SEMANTICS scenario,execution,evaluation,multimodal,images,evidence,store]
|
||||
# @defgroup ScenarioExecution Bounded image-evidence loading for the multimodal evaluation judge.
|
||||
# @BRIEF Read manifest image refs from owned storage into (bytes, sniffed-MIME) pairs within budget.
|
||||
# @RELATION DEPENDS_ON -> [ScenarioExecution.MimeSniff.Detect]
|
||||
# @RELATION CALLED_BY -> [ScenarioExecution.EvaluationAdapter]
|
||||
# @INVARIANT The model can only see artifacts that already passed the manifest digest/content_type/
|
||||
# byte_length gate; unreadable, non-image, or over-budget items are skipped, never invented.
|
||||
# @RATIONALE Canary v2 proved the text-only judge blind to pixels. Loading the real captures here keeps
|
||||
# the adapter module under INV_7 and separates evidence loading from adapter orchestration.
|
||||
# @REJECTED Querying ScenarioArtifact in a fresh SessionLocal was rejected (D1): the walker has only
|
||||
# flushed prior-step rows, so the completed outcomes stay the only lawful manifest source.
|
||||
from __future__ import annotations
|
||||
|
||||
from hashlib import sha256
|
||||
from typing import Any
|
||||
|
||||
from .mime_sniff import sniff_mime
|
||||
|
||||
IMAGE_CONTENT_TYPES = frozenset({"image/jpeg", "image/png", "image/webp"})
|
||||
MAX_EVALUATION_IMAGES = 8
|
||||
MAX_EVALUATION_IMAGE_BYTES = 10 * 1024 * 1024
|
||||
|
||||
|
||||
# #region ScenarioExecution.EvaluationImages.Load [C:4] [TYPE Function] [SEMANTICS evaluation,multimodal,images,budget]
|
||||
# @BRIEF Load readable image evidence from manifest refs, bounded by count and total byte budget.
|
||||
# @PRE manifest items are the adapter's own server-built entries; storage may or may not expose retrieve.
|
||||
# @POST Returns (bytes, MIME) pairs in manifest order; MIME is the sniffed signature, so a mislabeled
|
||||
# ref cannot reach the model. Missing retrieve or any per-item failure degrades to fewer images.
|
||||
def image_payloads_from_manifest(
|
||||
manifest: list[dict[str, Any]], storage: Any, *, max_images: int,
|
||||
) -> list[tuple[bytes, str]]:
|
||||
retrieve = getattr(storage, "retrieve", None)
|
||||
if not callable(retrieve):
|
||||
return []
|
||||
limit = max(0, min(MAX_EVALUATION_IMAGES, int(max_images)))
|
||||
images: list[tuple[bytes, str]] = []
|
||||
total_bytes = 0
|
||||
for item in manifest:
|
||||
if len(images) >= limit:
|
||||
break
|
||||
if item.get("content_type") not in IMAGE_CONTENT_TYPES:
|
||||
continue
|
||||
ref = item.get("artifact_id")
|
||||
if not isinstance(ref, str) or not ref:
|
||||
continue
|
||||
# Trusted manifest length lets an over-budget item be rejected without reading its bytes.
|
||||
declared_length = item.get("byte_length")
|
||||
if isinstance(declared_length, int) and declared_length > MAX_EVALUATION_IMAGE_BYTES - total_bytes:
|
||||
continue
|
||||
try:
|
||||
data = retrieve(ref)
|
||||
except Exception:
|
||||
continue
|
||||
if not isinstance(data, (bytes, bytearray)) or not data:
|
||||
continue
|
||||
# Defense in depth: the loaded bytes must match the manifest digest before reaching the model.
|
||||
expected_digest = item.get("sha256")
|
||||
if isinstance(expected_digest, str) and sha256(bytes(data)).hexdigest() != expected_digest.lower():
|
||||
continue
|
||||
content_type = sniff_mime(bytes(data))
|
||||
if content_type not in IMAGE_CONTENT_TYPES:
|
||||
continue
|
||||
if total_bytes + len(data) > MAX_EVALUATION_IMAGE_BYTES:
|
||||
continue
|
||||
images.append((bytes(data), content_type))
|
||||
total_bytes += len(data)
|
||||
return images
|
||||
# #endregion ScenarioExecution.EvaluationImages.Load
|
||||
# #endregion ScenarioExecution.EvaluationImages
|
||||
@@ -58,6 +58,7 @@ class LiveExecutionBinding:
|
||||
binding.execution_principal_fingerprint, binding.rls_security_fingerprint,
|
||||
))
|
||||
or not isinstance(binding.dashboard_id, int)
|
||||
or isinstance(binding.dashboard_id, bool)
|
||||
or binding.dashboard_id <= 0
|
||||
or binding.evidence_owner_type != "scenario_run"
|
||||
or binding.evidence_ref_policy != "draft_storage_raw_response"
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
# #region ScenarioExecution.MimeSniff [C:3] [TYPE Module] [SEMANTICS scenario,execution,mime,magic,bytes,signature]
|
||||
# @defgroup ScenarioExecution Pure magic-byte MIME detection shared by providers and artifact delivery.
|
||||
# @BRIEF Detect an allowlisted MIME from a byte-object signature; returns None when unrecognized.
|
||||
# @RELATION CALLED_BY -> [ScenarioExecution.ArtifactContent]
|
||||
# @RELATION CALLED_BY -> [ScenarioExecution.ScreenshotProvider]
|
||||
# @RELATION CALLED_BY -> [ScenarioExecution.BrowserProvider]
|
||||
# @INVARIANT Detection is byte-sniffing only (no path, name, or provider hint) so stored MIME can
|
||||
# never drift from the actual bytes (e.g. a WebP archive mislabeled image/jpeg).
|
||||
# @RATIONALE Extracted from ArtifactContent so evidence providers can stamp per-ref MIME from the
|
||||
# captured bytes without depending on the HTTP artifact-delivery service layer.
|
||||
# @REJECTED Trusting the provider's nominal format (screenshot→jpeg, browser→png) was rejected: the
|
||||
# capture service can emit WebP, which then fails the evaluation evidence MIME match.
|
||||
# #region ScenarioExecution.MimeSniff.Detect [C:2] [TYPE Function] [SEMANTICS mime,magic]
|
||||
# @BRIEF Detect allowlisted MIME from complete-object magic bytes; JSON is BOM/whitespace then { or [.
|
||||
def sniff_mime(data: bytes) -> str | None:
|
||||
if data.startswith(b"\xff\xd8\xff"):
|
||||
return "image/jpeg"
|
||||
if data.startswith(b"\x89PNG\r\n\x1a\n"):
|
||||
return "image/png"
|
||||
if len(data) >= 12 and data.startswith(b"RIFF") and data[8:12] == b"WEBP":
|
||||
return "image/webp"
|
||||
if data.startswith(b"%PDF-"):
|
||||
return "application/pdf"
|
||||
if data.startswith(b"PK\x03\x04"):
|
||||
return "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"
|
||||
stripped = data.lstrip(b"\xef\xbb\xbf \t\r\n")
|
||||
if stripped[:1] in (b"{", b"["):
|
||||
return "application/json"
|
||||
return None
|
||||
# #endregion ScenarioExecution.MimeSniff.Detect
|
||||
# #endregion ScenarioExecution.MimeSniff
|
||||
@@ -28,7 +28,6 @@
|
||||
# on the single application-owned provider loop.
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import uuid
|
||||
from collections.abc import Callable
|
||||
from hashlib import sha256
|
||||
@@ -43,8 +42,8 @@ from src.services.dashboard_testing.execution.capacity import (
|
||||
)
|
||||
from src.services.dashboard_testing.execution.live_adapter import LiveAdapterResult
|
||||
from src.services.dashboard_testing.execution.live_binding import binding_from_step
|
||||
from src.services.dashboard_testing.execution.mime_sniff import sniff_mime
|
||||
from src.services.dashboard_testing.execution.provider_operations import (
|
||||
complete_provider_operation,
|
||||
open_provider_operation,
|
||||
)
|
||||
from src.services.dashboard_testing.execution.provider_runtime import (
|
||||
@@ -62,6 +61,10 @@ from src.services.dashboard_testing.execution.providers.browser_mutation import
|
||||
validate_mutation_contract,
|
||||
validate_mutation_inputs,
|
||||
)
|
||||
from src.services.dashboard_testing.execution.providers.browser_receipt import (
|
||||
descriptor_fingerprint,
|
||||
finalize_provider_receipt,
|
||||
)
|
||||
|
||||
_SRC = "ScenarioExecution.BrowserProvider"
|
||||
_DEFAULT_ACTION_TIMEOUT_SECONDS = 30
|
||||
@@ -237,7 +240,7 @@ def build_browser_provider(
|
||||
provider_id="browser",
|
||||
provider_version=str(descriptor.get("provider_version") or "unset"),
|
||||
action=action,
|
||||
descriptor_fingerprint=_descriptor_fingerprint(descriptor),
|
||||
descriptor_fingerprint=descriptor_fingerprint(descriptor),
|
||||
binding_ref=binding.binding_ref,
|
||||
execution_principal_fingerprint=binding.execution_principal_fingerprint,
|
||||
idempotency_key=f"{run_id}:{metadata.get('logical_step_id')}:{int(metadata.get('attempt') or 1)}",
|
||||
@@ -280,7 +283,7 @@ def build_browser_provider(
|
||||
except TimeoutError:
|
||||
if mutating:
|
||||
logger.explore("Mutating action deadline expired with unknown effect", src=_SRC, payload={"run_id": run_id}, error_code="BROWSER_MUTATION_RECONCILE_REQUIRED")
|
||||
_finalize_provider_receipt(operation_id, "reconciliation_required", "unknown", summary={"phase": "timeout"})
|
||||
finalize_provider_receipt(operation_id, "reconciliation_required", "unknown", summary={"phase": "timeout"})
|
||||
return LiveAdapterResult(
|
||||
status="inconclusive",
|
||||
reason_code="BROWSER_MUTATION_RECONCILE_REQUIRED",
|
||||
@@ -293,7 +296,7 @@ def build_browser_provider(
|
||||
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_LOOP_OVERFLOW")
|
||||
except BrowserTransportPreconditionMismatch:
|
||||
logger.explore("Mutation precondition mismatch; nothing mutated", src=_SRC, payload={"run_id": run_id}, error_code="BROWSER_MUTATION_PRECONDITION_MISMATCH")
|
||||
_finalize_provider_receipt(operation_id, "failed", "not_started", summary={"phase": "precondition_mismatch"})
|
||||
finalize_provider_receipt(operation_id, "failed", "not_started", summary={"phase": "precondition_mismatch"})
|
||||
return LiveAdapterResult(
|
||||
status="inconclusive",
|
||||
reason_code="BROWSER_MUTATION_PRECONDITION_MISMATCH",
|
||||
@@ -301,7 +304,7 @@ def build_browser_provider(
|
||||
)
|
||||
except BrowserTransportUnsupported:
|
||||
logger.explore("Transport cannot execute the action; nothing started", src=_SRC, payload={"action": action}, error_code="BROWSER_ACTION_NOT_SUPPORTED")
|
||||
_finalize_provider_receipt(operation_id, "failed", "not_started", summary={"phase": "unsupported"})
|
||||
finalize_provider_receipt(operation_id, "failed", "not_started", summary={"phase": "unsupported"})
|
||||
return LiveAdapterResult(
|
||||
status="inconclusive",
|
||||
reason_code="BROWSER_ACTION_NOT_SUPPORTED",
|
||||
@@ -313,7 +316,7 @@ def build_browser_provider(
|
||||
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_LOOP_UNAVAILABLE")
|
||||
if mutating:
|
||||
logger.explore("Mutating action failed with unknown effect", src=_SRC, payload={"run_id": run_id}, error_code="BROWSER_MUTATION_RECONCILE_REQUIRED")
|
||||
_finalize_provider_receipt(operation_id, "reconciliation_required", "unknown", summary={"phase": "runtime_error"})
|
||||
finalize_provider_receipt(operation_id, "reconciliation_required", "unknown", summary={"phase": "runtime_error"})
|
||||
return LiveAdapterResult(
|
||||
status="inconclusive",
|
||||
reason_code="BROWSER_MUTATION_RECONCILE_REQUIRED",
|
||||
@@ -324,17 +327,17 @@ def build_browser_provider(
|
||||
evidence = outcome.evidence_png
|
||||
if not evidence:
|
||||
logger.explore("Transport produced no evidence", src=_SRC, payload={"run_id": run_id}, error_code="BROWSER_EVIDENCE_REQUIRED")
|
||||
_finalize_provider_receipt(operation_id, "reconciliation_required" if mutating else "failed", "unknown" if mutating else "not_started", summary={"phase": "evidence_missing"})
|
||||
finalize_provider_receipt(operation_id, "reconciliation_required" if mutating else "failed", "unknown" if mutating else "not_started", summary={"phase": "evidence_missing"})
|
||||
return LiveAdapterResult(status="inconclusive", reason_code="BROWSER_EVIDENCE_REQUIRED")
|
||||
rejection, artifact_refs, artifact_digests = _store_browser_evidence(
|
||||
evidence, storage, run_id, max_screenshot_bytes=max_screenshot_bytes,
|
||||
)
|
||||
if rejection is not None:
|
||||
_finalize_provider_receipt(operation_id, "reconciliation_required" if mutating else "failed", "unknown" if mutating else "not_started", summary={"phase": "evidence_store"})
|
||||
finalize_provider_receipt(operation_id, "reconciliation_required" if mutating else "failed", "unknown" if mutating else "not_started", summary={"phase": "evidence_store"})
|
||||
return rejection
|
||||
effect_state = outcome.effect_state if mutating else "none"
|
||||
if mutating:
|
||||
_finalize_provider_receipt(operation_id, "completed", effect_state, summary={"checkpoints": list(outcome.checkpoints), "page_url": outcome.page_url})
|
||||
finalize_provider_receipt(operation_id, "completed", effect_state, summary={"checkpoints": list(outcome.checkpoints), "page_url": outcome.page_url})
|
||||
logger.reflect(
|
||||
"Browser action completed with evidence", src=_SRC,
|
||||
payload={"run_id": run_id, "checkpoints": list(outcome.checkpoints), "bytes": len(evidence), "effect_state": effect_state, "operation_id": operation_id},
|
||||
@@ -354,7 +357,10 @@ def build_browser_provider(
|
||||
"artifact_byte_lengths": {
|
||||
ref: len(evidence) for ref in artifact_refs
|
||||
},
|
||||
"artifact_content_types": {ref: "image/png" for ref in artifact_refs},
|
||||
# Sniff MIME from the stored bytes (a WebP payload must not be labelled png).
|
||||
"artifact_content_types": {
|
||||
ref: sniff_mime(evidence) or "image/png" for ref in artifact_refs
|
||||
},
|
||||
**outcome.details,
|
||||
},
|
||||
artifact_refs=artifact_refs,
|
||||
@@ -363,7 +369,7 @@ def build_browser_provider(
|
||||
except Exception as exc:
|
||||
logger.explore("Browser provider failed", src=_SRC, payload={"run_id": run_id if isinstance(run_id, str) else None}, error=repr(exc))
|
||||
if mutating:
|
||||
_finalize_provider_receipt(operation_id, "reconciliation_required", "unknown", summary={"phase": "provider_error"})
|
||||
finalize_provider_receipt(operation_id, "reconciliation_required", "unknown", summary={"phase": "provider_error"})
|
||||
return LiveAdapterResult(
|
||||
status="inconclusive",
|
||||
reason_code="BROWSER_MUTATION_RECONCILE_REQUIRED",
|
||||
@@ -382,23 +388,4 @@ def build_browser_provider(
|
||||
return provider
|
||||
# #endregion ScenarioExecution.BrowserProvider.Factory
|
||||
|
||||
|
||||
# #region ScenarioExecution.BrowserProvider.Receipt [C:3] [TYPE Function] [SEMANTICS provider,operation,receipt,finalize]
|
||||
# @ingroup ScenarioExecution
|
||||
# @BRIEF Finalize the durable receipt outside the run transaction; failures never mask the outcome.
|
||||
def _descriptor_fingerprint(descriptor: dict[str, Any]) -> str:
|
||||
canonical = json.dumps(descriptor, sort_keys=True, separators=(",", ":"))
|
||||
return sha256(canonical.encode()).hexdigest()
|
||||
|
||||
|
||||
def _finalize_provider_receipt(operation_id: str | None, status: str, effect_state: str, summary: dict[str, Any] | None = None) -> None:
|
||||
if operation_id is None:
|
||||
return
|
||||
try:
|
||||
with SessionLocal() as db:
|
||||
complete_provider_operation(db, operation_id, status=status, effect_state=effect_state, summary=summary)
|
||||
db.commit()
|
||||
except Exception as exc:
|
||||
logger.explore("Receipt finalization failed", src=_SRC, payload={"operation_id": operation_id, "status": status}, error=repr(exc))
|
||||
# #endregion ScenarioExecution.BrowserProvider.Receipt
|
||||
# #endregion ScenarioExecution.BrowserProvider
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
# #region ScenarioExecution.BrowserReceipt [C:3] [TYPE Module] [SEMANTICS scenario,execution,provider,browser,operation,receipt,finalize]
|
||||
# @defgroup ScenarioExecution Durable provider-operation receipt helpers for the browser provider.
|
||||
# @BRIEF Fingerprint a descriptor and finalize operation receipts outside the run transaction.
|
||||
# @RELATION DEPENDS_ON -> [ScenarioExecution.ProviderOperations.Service]
|
||||
# @INVARIANT Receipt finalization never masks the provider outcome: a failure is logged, not raised.
|
||||
# @RATIONALE Extracted from BrowserProvider (Front 6 INV_7): receipt finalization is a distinct
|
||||
# concern from action admission/execution and keeps the provider module under 400 LOC.
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from hashlib import sha256
|
||||
from typing import Any
|
||||
|
||||
from src.core.database import SessionLocal
|
||||
from src.core.logger import logger
|
||||
from src.services.dashboard_testing.execution.provider_operations import complete_provider_operation
|
||||
|
||||
_SRC = "ScenarioExecution.BrowserProvider"
|
||||
|
||||
|
||||
# #region ScenarioExecution.BrowserReceipt.Descriptor [C:2] [TYPE Function] [SEMANTICS provider,operation,descriptor,fingerprint]
|
||||
# @BRIEF Canonical SHA-256 of an action descriptor for the operation receipt.
|
||||
def descriptor_fingerprint(descriptor: dict[str, Any]) -> str:
|
||||
canonical = json.dumps(descriptor, sort_keys=True, separators=(",", ":"))
|
||||
return sha256(canonical.encode()).hexdigest()
|
||||
# #endregion ScenarioExecution.BrowserReceipt.Descriptor
|
||||
|
||||
|
||||
# #region ScenarioExecution.BrowserReceipt.Finalize [C:3] [TYPE Function] [SEMANTICS provider,operation,receipt,finalize]
|
||||
# @BRIEF Finalize the durable receipt outside the run transaction; failures never mask the outcome.
|
||||
def finalize_provider_receipt(
|
||||
operation_id: str | None, status: str, effect_state: str, summary: dict[str, Any] | None = None,
|
||||
) -> None:
|
||||
if operation_id is None:
|
||||
return
|
||||
try:
|
||||
with SessionLocal() as db:
|
||||
complete_provider_operation(db, operation_id, status=status, effect_state=effect_state, summary=summary)
|
||||
db.commit()
|
||||
except Exception as exc:
|
||||
logger.explore("Receipt finalization failed", src=_SRC, payload={"operation_id": operation_id, "status": status}, error=repr(exc))
|
||||
# #endregion ScenarioExecution.BrowserReceipt.Finalize
|
||||
# #endregion ScenarioExecution.BrowserReceipt
|
||||
@@ -38,6 +38,7 @@ from src.services.dashboard_testing.execution.live_adapter import LiveAdapterRes
|
||||
from src.services.dashboard_testing.execution.live_binding import (
|
||||
binding_from_step,
|
||||
)
|
||||
from src.services.dashboard_testing.execution.mime_sniff import sniff_mime
|
||||
from src.services.dashboard_testing.execution.provider_runtime import (
|
||||
ProviderEventLoop,
|
||||
ProviderSubmissionOverflow,
|
||||
@@ -165,6 +166,7 @@ def build_screenshot_provider(
|
||||
artifact_refs: list[str] = []
|
||||
artifact_digests: dict[str, str] = {}
|
||||
artifact_byte_lengths: dict[str, int] = {}
|
||||
artifact_content_types: dict[str, str] = {}
|
||||
for index, path in enumerate(jpeg_paths):
|
||||
raw = Path(path).read_bytes()
|
||||
if len(raw) > max_screenshot_bytes:
|
||||
@@ -182,6 +184,8 @@ def build_screenshot_provider(
|
||||
artifact_refs.append(content_ref)
|
||||
artifact_digests[content_ref] = digest
|
||||
artifact_byte_lengths[content_ref] = len(raw)
|
||||
# Sniff MIME from the captured bytes (a WebP capture must not be labelled jpeg).
|
||||
artifact_content_types[content_ref] = sniff_mime(raw) or "image/jpeg"
|
||||
logger.reflect(
|
||||
"Screenshot evidence stored", src=_SRC,
|
||||
payload={"run_id": run_id, "artifact_index": index, "bytes": len(raw)},
|
||||
@@ -199,7 +203,7 @@ def build_screenshot_provider(
|
||||
# ScenarioExecution.EvaluationAdapter.Manifest admit screenshot
|
||||
# evidence (the default byte_length is absent for capture outcomes).
|
||||
"artifact_byte_lengths": artifact_byte_lengths,
|
||||
"artifact_content_types": {ref: "image/jpeg" for ref in artifact_refs},
|
||||
"artifact_content_types": artifact_content_types,
|
||||
},
|
||||
artifact_refs=artifact_refs,
|
||||
artifact_digests=artifact_digests,
|
||||
|
||||
@@ -17,6 +17,7 @@ from src.models.scenario_run import ScenarioRun, ScenarioStepRun
|
||||
|
||||
from .agent_evaluation import AgentEvaluation, persist_agent_evaluation, validate_evaluation_evidence
|
||||
from .artifacts import register_step_evidence
|
||||
from .baseline_resolver import stamp_baseline_pin
|
||||
from .decision_policy import decide_step_outcome, policy_inputs_from_outcome, verified_evidence_refs
|
||||
from .dispatch import dispatch_step
|
||||
from .executor_registry import ScenarioExecutorRegistry
|
||||
@@ -329,6 +330,7 @@ def _advance_run(db: Session, run: ScenarioRun, registry: ScenarioExecutorRegist
|
||||
evaluation_record = _evaluation_record_from_outcome(step.step_outcome)
|
||||
evaluation_publish_failed = False
|
||||
if step_meta.get("tool") == "agent_evaluation" and isinstance(evaluation_record, dict):
|
||||
evaluation_record = stamp_baseline_pin(evaluation_record, run.runner_plan or {})
|
||||
try:
|
||||
record = AgentEvaluation.model_validate(evaluation_record)
|
||||
validate_evaluation_evidence(
|
||||
|
||||
200
backend/tests/api/test_scenario_live_bindings_api.py
Normal file
200
backend/tests/api/test_scenario_live_bindings_api.py
Normal file
@@ -0,0 +1,200 @@
|
||||
# #region Test.Api.ScenarioLiveBindings [C:3] [TYPE Module] [SEMANTICS test,api,scenario,execution,live-binding,admin]
|
||||
# @BRIEF Admin CRUD for deployment live bindings via an isolated FastAPI app + fake ConfigManager.
|
||||
# @RELATION BINDS_TO -> [Api.ScenarioLiveBindings]
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.Runner.ConfiguredBinding]
|
||||
# @TEST_CONTRACT: GET/PUT/PATCH /api/scenario-live-bindings -> validated settings.scenario_live_execution_bindings
|
||||
# @TEST_EDGE: invalid_snapshot -> 422 before persistence
|
||||
# @TEST_EDGE: binding_ref_mismatch -> 422
|
||||
# @TEST_EDGE: unknown_toggle -> 404
|
||||
# @TEST_INVARIANT Api.ScenarioLiveBindings: a registered, enabled binding is resolved server-side by
|
||||
# ScenarioExecution.Runner.ConfiguredBinding; disabled/unregistered is not.
|
||||
# -> VERIFIED_BY: test_registered_binding_resolves_server_side
|
||||
# @TEST_INVARIANT Api.ScenarioLiveBindings: non-admin is denied 403 by the real has_permission checker.
|
||||
# -> VERIFIED_BY: test_non_admin_denied_403
|
||||
from __future__ import annotations
|
||||
|
||||
from fastapi import FastAPI
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from src.api.routes.dashboard_testing.scenario_live_bindings import router
|
||||
from src.core.config_models import GlobalSettings
|
||||
from src.dependencies import get_config_manager, get_current_user
|
||||
from src.models.auth import Role, User
|
||||
|
||||
_BINDING_REF = "ss-prod-d11-live-001"
|
||||
_SNAPSHOT = {
|
||||
"binding_ref": _BINDING_REF,
|
||||
"environment_id": "env-prod-01",
|
||||
"dashboard_release_id": "release-1",
|
||||
"release_fingerprint": "f" * 64,
|
||||
"dashboard_id": 11,
|
||||
"query_model_fingerprint": "q" * 64,
|
||||
"execution_principal_fingerprint": "p" * 64,
|
||||
"rls_security_fingerprint": "r" * 64,
|
||||
"browser_safe_checkpoint_ref": None,
|
||||
"browser_action_binding_ref": None,
|
||||
"evidence_owner_type": "scenario_run",
|
||||
"evidence_ref_policy": "draft_storage_raw_response",
|
||||
}
|
||||
|
||||
|
||||
_QUERY_MODEL = {
|
||||
"schema_version": 1,
|
||||
"environment_id": "env-prod-01",
|
||||
"dashboard_id": 11,
|
||||
"title": "Sales Dashboard",
|
||||
"query_model_fingerprint": "q" * 64,
|
||||
}
|
||||
|
||||
|
||||
class _FakeConfig:
|
||||
def __init__(self) -> None:
|
||||
self.settings = GlobalSettings()
|
||||
|
||||
|
||||
class _FakeConfigManager:
|
||||
def __init__(self) -> None:
|
||||
self._config = _FakeConfig()
|
||||
|
||||
def get_config(self):
|
||||
return self._config
|
||||
|
||||
def update_global_settings(self, settings):
|
||||
self._config.settings = settings
|
||||
return self._config
|
||||
|
||||
|
||||
def _admin() -> User:
|
||||
role = Role(id="role-admin", name="admin", is_admin=True)
|
||||
role.permissions = []
|
||||
user = User(id="user-admin", username="admin", email="admin@test.com")
|
||||
user.roles = [role]
|
||||
return user
|
||||
|
||||
|
||||
def _outsider() -> User:
|
||||
role = Role(id="role-viewer", name="viewer", is_admin=False)
|
||||
role.permissions = []
|
||||
user = User(id="user-viewer", username="viewer", email="viewer@test.com")
|
||||
user.roles = [role]
|
||||
return user
|
||||
|
||||
|
||||
class _Env:
|
||||
def __init__(self) -> None:
|
||||
self.config_manager = _FakeConfigManager()
|
||||
self.app = FastAPI()
|
||||
self.app.include_router(router)
|
||||
self.app.dependency_overrides[get_config_manager] = lambda: self.config_manager
|
||||
self.client = TestClient(self.app)
|
||||
|
||||
def as_user(self, user: User) -> None:
|
||||
self.app.dependency_overrides[get_current_user] = lambda: user
|
||||
|
||||
|
||||
def _put_body(**overrides) -> dict:
|
||||
body = {"enabled": True, "binding_snapshot": dict(_SNAPSHOT), "query_model_snapshot": dict(_QUERY_MODEL)}
|
||||
body.update(overrides)
|
||||
return body
|
||||
|
||||
|
||||
# #region Test.Api.ScenarioLiveBindings.Crud [C:2] [TYPE Function]
|
||||
def test_list_empty_then_register_and_toggle():
|
||||
env = _Env()
|
||||
env.as_user(_admin())
|
||||
|
||||
assert env.client.get("/api/scenario-live-bindings").json() == []
|
||||
|
||||
created = env.client.put(f"/api/scenario-live-bindings/{_BINDING_REF}", json=_put_body())
|
||||
assert created.status_code == 200
|
||||
assert created.json()["binding_ref"] == _BINDING_REF
|
||||
assert created.json()["enabled"] is True
|
||||
|
||||
listed = env.client.get("/api/scenario-live-bindings").json()
|
||||
assert [item["binding_ref"] for item in listed] == [_BINDING_REF]
|
||||
|
||||
toggled = env.client.patch(f"/api/scenario-live-bindings/{_BINDING_REF}", json={"enabled": False})
|
||||
assert toggled.status_code == 200
|
||||
assert toggled.json()["enabled"] is False
|
||||
|
||||
|
||||
def test_put_rejects_invalid_snapshot_without_persisting():
|
||||
env = _Env()
|
||||
env.as_user(_admin())
|
||||
bad = {**_SNAPSHOT, "evidence_owner_type": "draft"}
|
||||
response = env.client.put(f"/api/scenario-live-bindings/{_BINDING_REF}", json=_put_body(binding_snapshot=bad))
|
||||
assert response.status_code == 422
|
||||
assert env.config_manager.get_config().settings.scenario_live_execution_bindings == []
|
||||
|
||||
|
||||
def test_put_rejects_binding_ref_mismatch():
|
||||
env = _Env()
|
||||
env.as_user(_admin())
|
||||
response = env.client.put("/api/scenario-live-bindings/other-ref", json=_put_body())
|
||||
assert response.status_code == 422
|
||||
assert env.config_manager.get_config().settings.scenario_live_execution_bindings == []
|
||||
|
||||
|
||||
def test_put_rejects_invalid_and_mismatched_query_model():
|
||||
env = _Env()
|
||||
env.as_user(_admin())
|
||||
invalid = env.client.put(
|
||||
f"/api/scenario-live-bindings/{_BINDING_REF}",
|
||||
json=_put_body(query_model_snapshot={"totally": "bogus"}),
|
||||
)
|
||||
assert invalid.status_code == 422
|
||||
|
||||
mismatched = env.client.put(
|
||||
f"/api/scenario-live-bindings/{_BINDING_REF}",
|
||||
json=_put_body(query_model_snapshot={**_QUERY_MODEL, "environment_id": "other-env"}),
|
||||
)
|
||||
assert mismatched.status_code == 422
|
||||
assert env.config_manager.get_config().settings.scenario_live_execution_bindings == []
|
||||
|
||||
|
||||
def test_toggle_unknown_binding_is_404():
|
||||
env = _Env()
|
||||
env.as_user(_admin())
|
||||
response = env.client.patch("/api/scenario-live-bindings/missing-ref", json={"enabled": True})
|
||||
assert response.status_code == 404
|
||||
|
||||
|
||||
def test_non_admin_denied_403():
|
||||
env = _Env()
|
||||
env.as_user(_outsider())
|
||||
assert env.client.get("/api/scenario-live-bindings").status_code == 403
|
||||
assert env.client.put(f"/api/scenario-live-bindings/{_BINDING_REF}", json=_put_body()).status_code == 403
|
||||
assert env.client.patch(f"/api/scenario-live-bindings/{_BINDING_REF}", json={"enabled": True}).status_code == 403
|
||||
# #endregion Test.Api.ScenarioLiveBindings.Crud
|
||||
|
||||
|
||||
# #region Test.Api.ScenarioLiveBindings.Resolve [C:2] [TYPE Function]
|
||||
def test_registered_binding_resolves_server_side():
|
||||
from src.services.dashboard_testing.execution.start_run import _resolve_configured_live_binding
|
||||
|
||||
env = _Env()
|
||||
env.as_user(_admin())
|
||||
env.client.put(f"/api/scenario-live-bindings/{_BINDING_REF}", json=_put_body())
|
||||
plan = {"steps": [{"dashboard_id": 11}]}
|
||||
principal = _SNAPSHOT["execution_principal_fingerprint"]
|
||||
|
||||
resolved = _resolve_configured_live_binding(
|
||||
environment_id="env-prod-01", plan=plan, principal_fingerprint=principal,
|
||||
config_manager=env.config_manager,
|
||||
)
|
||||
assert resolved is not None
|
||||
assert resolved.binding_ref == _BINDING_REF
|
||||
|
||||
# A different principal or environment never adopts the binding.
|
||||
assert _resolve_configured_live_binding(
|
||||
environment_id="env-prod-01", plan=plan, principal_fingerprint="z" * 64,
|
||||
config_manager=env.config_manager,
|
||||
) is None
|
||||
|
||||
env.client.patch(f"/api/scenario-live-bindings/{_BINDING_REF}", json={"enabled": False})
|
||||
assert _resolve_configured_live_binding(
|
||||
environment_id="env-prod-01", plan=plan, principal_fingerprint=principal,
|
||||
config_manager=env.config_manager,
|
||||
) is None
|
||||
# #endregion Test.Api.ScenarioLiveBindings.Resolve
|
||||
# #endregion Test.Api.ScenarioLiveBindings
|
||||
@@ -8,6 +8,7 @@ from __future__ import annotations
|
||||
|
||||
import uuid
|
||||
from datetime import UTC, datetime
|
||||
from hashlib import sha256
|
||||
|
||||
import pytest
|
||||
|
||||
@@ -307,6 +308,29 @@ def test_omitted_criterion_id_maps_to_the_single_pinned_criterion():
|
||||
# #endregion Test.ScenarioExecution.EvaluationAdapter.Normalize
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.EvaluationAdapter.Identity [C:2] [TYPE Function]
|
||||
# @BRIEF Provider text cannot author baseline identity; the adapter zeroes any provider baseline_pin.
|
||||
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter.Normalize: a provider-supplied baseline_pin is
|
||||
# discarded; only the walker's server-resolved plan pin is persisted.
|
||||
# -> VERIFIED_BY: test_adapter_drops_provider_supplied_baseline_pin
|
||||
def test_adapter_drops_provider_supplied_baseline_pin():
|
||||
record = _run_adapter_with_response({
|
||||
"status": "succeeded",
|
||||
"verdict": "inconclusive",
|
||||
"confidence": 0.2,
|
||||
"findings": [{
|
||||
"criterion_id": "crit-visual",
|
||||
"evidence_artifact_ids": ["draft:run-norm:" + "e" * 64],
|
||||
"message": "claimed pin",
|
||||
}],
|
||||
"reason_codes": [],
|
||||
"baseline_pin": {"catalog_revision_id": "provider-claimed"},
|
||||
"usage": {"input_tokens": 1, "output_tokens": 1},
|
||||
})
|
||||
assert record["baseline_pin"] == {}
|
||||
# #endregion Test.ScenarioExecution.EvaluationAdapter.Identity
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.EvaluationAdapter.Empty [C:2] [TYPE Function]
|
||||
# @BRIEF Missing prior-step evidence is fail-closed; the adapter never synthesizes PASS.
|
||||
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter: empty completed fails closed. -> VERIFIED_BY: adapter_empty_completed_fail_closed
|
||||
@@ -343,4 +367,197 @@ def test_composition_root_evaluation_adapter_is_composed_and_fail_closed():
|
||||
assert outcome["status"] == "inconclusive"
|
||||
assert outcome["error_code"] is not None
|
||||
# #endregion Test.ScenarioExecution.EvaluationAdapter.Composition
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.EvaluationAdapter.Multimodal [C:3] [TYPE Function]
|
||||
# @BRIEF Captured images reach the judge as sniffed content-parts; text-only providers stay text-only.
|
||||
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter.Images: only manifest image items whose bytes
|
||||
# sniff to an allowlisted image MIME are attached; unreadable/non-image items are skipped.
|
||||
# -> VERIFIED_BY: adapter_attaches_sniffed_images, adapter_skips_invalid_evidence
|
||||
# @TEST_INVARIANT ScenarioExecution.AgentEvaluation.Messages: images become base64 data URLs only for a
|
||||
# multimodal provider. -> VERIFIED_BY: submit_evaluation_multimodal_gate
|
||||
_PNG_BYTES = b"\x89PNG\r\n\x1a\n" + b"evidence"
|
||||
_JPEG_BYTES = b"\xff\xd8\xff\xe0" + b"evidence"
|
||||
|
||||
|
||||
class _RetrievingStore(_EvidenceStore):
|
||||
def __init__(self, payloads: dict[str, bytes]) -> None:
|
||||
super().__init__()
|
||||
self.payloads = payloads
|
||||
|
||||
def retrieve(self, content_ref: str):
|
||||
return self.payloads.get(content_ref)
|
||||
|
||||
|
||||
def _image_outcome(ref: str, digest: str, payload: bytes, **nested_overrides) -> dict:
|
||||
nested = {
|
||||
"tool": "screenshot",
|
||||
"artifact_digests": {ref: digest},
|
||||
"artifact_content_types": {ref: "image/png"},
|
||||
"artifact_byte_lengths": {ref: len(payload)},
|
||||
}
|
||||
nested.update(nested_overrides)
|
||||
return {"status": "passed", "artifact_refs": [ref], "step_outcome": nested}
|
||||
|
||||
|
||||
def test_adapter_attaches_sniffed_images_from_manifest():
|
||||
from src.services.dashboard_testing.execution.evaluation_adapter import evaluation_adapter_from
|
||||
|
||||
png_digest = sha256(_PNG_BYTES).hexdigest()
|
||||
prior_ref = f"draft:run-mm:{png_digest}"
|
||||
run_id = str(uuid.uuid4())
|
||||
captured: dict = {}
|
||||
|
||||
def _submit(**kwargs):
|
||||
captured.update(kwargs)
|
||||
return _eval_dict()
|
||||
|
||||
store = _RetrievingStore({prior_ref: _PNG_BYTES})
|
||||
adapter = evaluation_adapter_from(storage=store, submit=_submit)
|
||||
adapter(
|
||||
{
|
||||
"logical_step_id": "step-eval-1",
|
||||
"scenario_run_id": run_id,
|
||||
"attempt": 1,
|
||||
"agent_evaluation_spec": _spec().model_dump(mode="json"),
|
||||
},
|
||||
{"s1-shot": _image_outcome(prior_ref, png_digest, _PNG_BYTES)},
|
||||
)
|
||||
|
||||
assert captured["images"] == [(_PNG_BYTES, "image/png")]
|
||||
|
||||
|
||||
def test_adapter_skips_invalid_evidence_before_attaching():
|
||||
from src.services.dashboard_testing.execution.evaluation_adapter import evaluation_adapter_from
|
||||
|
||||
json_ref = "draft:run-mm-json:" + ("a" * 64)
|
||||
bad_ref = "draft:run-mm-bad:" + ("b" * 64)
|
||||
unreadable_ref = "draft:run-mm-missing:" + ("c" * 64)
|
||||
run_id = str(uuid.uuid4())
|
||||
captured: dict = {}
|
||||
|
||||
def _submit(**kwargs):
|
||||
captured.update(kwargs)
|
||||
return _eval_dict()
|
||||
|
||||
store = _RetrievingStore({bad_ref: b"not-an-image-signature"})
|
||||
adapter = evaluation_adapter_from(storage=store, submit=_submit)
|
||||
adapter(
|
||||
{
|
||||
"logical_step_id": "step-eval-1",
|
||||
"scenario_run_id": run_id,
|
||||
"attempt": 1,
|
||||
"agent_evaluation_spec": _spec().model_dump(mode="json"),
|
||||
},
|
||||
{
|
||||
"s1-json": {
|
||||
"status": "passed",
|
||||
"artifact_refs": [json_ref],
|
||||
"step_outcome": {
|
||||
"tool": "superset_api",
|
||||
"artifact_digests": {json_ref: "a" * 64},
|
||||
"artifact_content_types": {json_ref: "application/json"},
|
||||
"artifact_byte_lengths": {json_ref: 10},
|
||||
},
|
||||
},
|
||||
"s2-bad": _image_outcome(bad_ref, "b" * 64, b"not-an-image-signature"),
|
||||
"s3-missing": _image_outcome(unreadable_ref, "c" * 64, b"\x89PNG-still"),
|
||||
},
|
||||
)
|
||||
|
||||
assert captured["images"] == []
|
||||
|
||||
|
||||
def test_evaluation_messages_text_only_and_multimodal():
|
||||
from src.services.dashboard_testing.execution.agent_evaluation import _evaluation_messages
|
||||
|
||||
assert _evaluation_messages("P", None) == [{"role": "user", "content": "P"}]
|
||||
parts = _evaluation_messages("P", [(b"abc", "image/png")])
|
||||
content = parts[0]["content"]
|
||||
assert content[0] == {"type": "text", "text": "P"}
|
||||
assert content[1]["image_url"]["url"] == "data:image/png;base64,YWJj"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("is_multimodal, expects_parts", [(True, True), (False, False)])
|
||||
def test_submit_evaluation_gates_images_on_multimodal_provider(monkeypatch, is_multimodal, expects_parts):
|
||||
import asyncio
|
||||
|
||||
import src.plugins.llm_analysis.service as llm_service_module
|
||||
import src.services.llm_provider as llm_provider_module
|
||||
from src.services.dashboard_testing.execution import agent_evaluation as ae
|
||||
|
||||
class _Provider:
|
||||
provider_type = "openai"
|
||||
base_url = "http://local-llm"
|
||||
is_multimodal = False
|
||||
|
||||
_Provider.is_multimodal = is_multimodal
|
||||
|
||||
seen: dict = {}
|
||||
|
||||
class _Client:
|
||||
def __init__(self, *args, **kwargs):
|
||||
pass
|
||||
|
||||
async def get_json_completion(self, messages):
|
||||
seen["messages"] = messages
|
||||
return {"status": "succeeded"}
|
||||
|
||||
class _ProviderService:
|
||||
def __init__(self, db):
|
||||
pass
|
||||
|
||||
def get_provider(self, provider_id):
|
||||
return _Provider()
|
||||
|
||||
def get_decrypted_api_key(self, provider_id):
|
||||
return "key"
|
||||
|
||||
monkeypatch.setattr(llm_provider_module, "LLMProviderService", _ProviderService)
|
||||
monkeypatch.setattr(llm_service_module, "LLMClient", _Client)
|
||||
monkeypatch.setattr(ae, "claim_capacity", lambda *args, **kwargs: {"lease_id": "lease-1"})
|
||||
monkeypatch.setattr(ae, "release_capacity", lambda *args, **kwargs: None)
|
||||
|
||||
result = asyncio.run(ae.submit_evaluation(
|
||||
object(), spec=_spec(), prompt="P", images=[(b"abc", "image/png")],
|
||||
environment_id="env", environment_class="DEV", run_id="run", logical_step_id="step",
|
||||
))
|
||||
|
||||
assert result == {"status": "succeeded"}
|
||||
content = seen["messages"][0]["content"]
|
||||
if expects_parts:
|
||||
assert isinstance(content, list)
|
||||
assert content[1]["image_url"]["url"] == "data:image/png;base64,YWJj"
|
||||
else:
|
||||
assert content == "P"
|
||||
|
||||
|
||||
def _image_manifest_item(ref: str, data: bytes) -> dict:
|
||||
return {
|
||||
"artifact_id": ref, "sha256": sha256(data).hexdigest(),
|
||||
"content_type": "image/png", "byte_length": len(data), "role": "actual",
|
||||
}
|
||||
|
||||
|
||||
def test_image_payloads_respect_count_and_byte_budget(monkeypatch):
|
||||
from src.services.dashboard_testing.execution import evaluation_images as ev
|
||||
|
||||
payload_a = b"\x89PNG\r\n\x1a\n" + b"a" * 92
|
||||
payload_b = b"\x89PNG\r\n\x1a\n" + b"b" * 92
|
||||
store = _RetrievingStore({"ref-a": payload_a, "ref-b": payload_b})
|
||||
manifest = [_image_manifest_item("ref-a", payload_a), _image_manifest_item("ref-b", payload_b)]
|
||||
|
||||
assert ev.image_payloads_from_manifest(manifest, store, max_images=1) == [(payload_a, "image/png")]
|
||||
|
||||
monkeypatch.setattr(ev, "MAX_EVALUATION_IMAGE_BYTES", 150)
|
||||
assert ev.image_payloads_from_manifest(manifest, store, max_images=8) == [(payload_a, "image/png")]
|
||||
|
||||
|
||||
def test_image_payloads_skip_digest_mismatch():
|
||||
from src.services.dashboard_testing.execution.evaluation_images import image_payloads_from_manifest
|
||||
|
||||
store = _RetrievingStore({"ref-a": _PNG_BYTES})
|
||||
tampered = {**_image_manifest_item("ref-a", _PNG_BYTES), "sha256": "0" * 64}
|
||||
assert image_payloads_from_manifest([tampered], store, max_images=8) == []
|
||||
# #endregion Test.ScenarioExecution.EvaluationAdapter.Multimodal
|
||||
# #endregion Test.ScenarioExecution.AgentEvaluation
|
||||
@@ -32,6 +32,7 @@ from src.services.dashboard_testing.execution.baseline_resolver import (
|
||||
graph_has_baseline_refs,
|
||||
load_published_catalog,
|
||||
resolve_baseline_pin,
|
||||
stamp_baseline_pin,
|
||||
)
|
||||
from src.services.dashboard_testing.execution.runner import start_run
|
||||
from src.services.dashboard_testing.scenario.templates import (
|
||||
@@ -295,4 +296,23 @@ def test_idempotency_rejects_different_pin():
|
||||
session.close()
|
||||
# #endregion Test.ScenarioExecution.BaselineResolver.Idempotency
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.BaselineResolver.Stamp [C:2] [TYPE Function]
|
||||
# @BRIEF baseline_pin is server-owned: the plan pin always wins; a provider-claimed pin is discarded.
|
||||
# @TEST_INVARIANT ScenarioExecution.BaselineResolver.StampEvaluation: any non-plan pin, including one
|
||||
# supplied by the provider, is replaced by the resolved plan pin (or {}). -> VERIFIED_BY: test_stamp_baseline_pin_enforces_server_authority
|
||||
def test_stamp_baseline_pin_enforces_server_authority():
|
||||
plan_pin = {"catalog_revision_id": "r-1", "baseline_set_id": "ss-prod-visual", "baseline_set_version": "1"}
|
||||
filled = stamp_baseline_pin({"baseline_pin": {}}, {"baseline_pin": plan_pin})
|
||||
assert filled["baseline_pin"] == plan_pin
|
||||
|
||||
# A provider-claimed pin must never survive: the server plan pin is authoritative.
|
||||
claimed = {"catalog_revision_id": "provider-claimed"}
|
||||
overwritten = stamp_baseline_pin({"baseline_pin": claimed}, {"baseline_pin": plan_pin})
|
||||
assert overwritten["baseline_pin"] == plan_pin
|
||||
|
||||
assert stamp_baseline_pin({"baseline_pin": claimed}, {"baseline_pin": None})["baseline_pin"] == {}
|
||||
assert stamp_baseline_pin({"baseline_pin": {}}, {}) == {"baseline_pin": {}}
|
||||
assert stamp_baseline_pin(None, {"baseline_pin": plan_pin}) is None
|
||||
# #endregion Test.ScenarioExecution.BaselineResolver.Stamp
|
||||
# #endregion Test.ScenarioExecution.BaselineResolver
|
||||
|
||||
@@ -419,4 +419,33 @@ def test_read_only_pass_reports_effect_state_none(loop_runtime, clean_env):
|
||||
assert result.details["effect_state"] == "none"
|
||||
assert "operation_id" not in result.details
|
||||
# #endregion Test.ScenarioExecution.BrowserProvider.MutationOutcome
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.BrowserProvider.MimeSniff [C:2] [TYPE Function] [SEMANTICS test,provider,browser,mime]
|
||||
# @BRIEF Content type is sniffed from stored evidence bytes, never assumed image/png.
|
||||
# @TEST_INVARIANT A WebP evidence payload must not be labelled image/png (evaluation MIME match).
|
||||
def test_browser_evidence_content_type_is_sniffed(loop_runtime, clean_env):
|
||||
binding = clean_env
|
||||
webp = b"RIFF\x00\x00\x00\x00WEBP" + b"payload"
|
||||
provider = build_browser_provider(transport=FakeTransport(evidence=webp), storage=FakeStorage(), loop=loop_runtime)
|
||||
|
||||
result = provider(LiveProviderContext(binding=binding, step=build_step(binding), completed={}))
|
||||
|
||||
assert result.status == "passed"
|
||||
ref = result.artifact_refs[0]
|
||||
assert result.details["artifact_content_types"][ref] == "image/webp"
|
||||
|
||||
|
||||
def test_browser_unrecognized_evidence_falls_back_to_png(loop_runtime, clean_env):
|
||||
binding = clean_env
|
||||
provider = build_browser_provider(
|
||||
transport=FakeTransport(evidence=b"raw-evidence-bytes"), storage=FakeStorage(), loop=loop_runtime,
|
||||
)
|
||||
|
||||
result = provider(LiveProviderContext(binding=binding, step=build_step(binding), completed={}))
|
||||
|
||||
assert result.status == "passed"
|
||||
ref = result.artifact_refs[0]
|
||||
assert result.details["artifact_content_types"][ref] == "image/png"
|
||||
# #endregion Test.ScenarioExecution.BrowserProvider.MimeSniff
|
||||
# #endregion Test.ScenarioExecution.BrowserProvider
|
||||
|
||||
@@ -270,5 +270,39 @@ def test_capture_running_on_shared_loop_is_verified(loop_runtime, clean_env):
|
||||
|
||||
assert result.status == "passed"
|
||||
assert observed["loop"] == id(loop_runtime._loop)
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.ScreenshotProvider.MimeSniff [C:2] [TYPE Function] [SEMANTICS test,provider,screenshot,mime]
|
||||
# @BRIEF Content types are sniffed per ref from the captured bytes, never assumed image/jpeg.
|
||||
# @TEST_INVARIANT WebP or PNG captures must not be labelled image/jpeg (evaluation MIME match).
|
||||
def test_capture_content_types_are_sniffed_per_ref(loop_runtime, clean_env, tmp_path):
|
||||
binding = clean_env
|
||||
webp = b"RIFF\x00\x00\x00\x00WEBP" + b"payload"
|
||||
jpeg = b"\xff\xd8\xff\xe0" + b"payload"
|
||||
first, second = tmp_path / "a.webp", tmp_path / "b.jpg"
|
||||
first.write_bytes(webp)
|
||||
second.write_bytes(jpeg)
|
||||
service = FakeCaptureService(paths=[str(first), str(second)])
|
||||
provider = build_screenshot_provider(service=service, storage=FakeStorage(), loop=loop_runtime)
|
||||
|
||||
result = provider(LiveProviderContext(binding=binding, step=build_step(binding), completed={}))
|
||||
|
||||
assert result.status == "passed"
|
||||
types = result.details["artifact_content_types"]
|
||||
refs = result.artifact_refs
|
||||
assert [types[refs[0]], types[refs[1]]] == ["image/webp", "image/jpeg"]
|
||||
|
||||
|
||||
def test_capture_unrecognized_bytes_fall_back_to_jpeg(loop_runtime, clean_env):
|
||||
binding = clean_env
|
||||
service = FakeCaptureService(payload=b"not-an-image-signature")
|
||||
provider = build_screenshot_provider(service=service, storage=FakeStorage(), loop=loop_runtime)
|
||||
|
||||
result = provider(LiveProviderContext(binding=binding, step=build_step(binding), completed={}))
|
||||
|
||||
assert result.status == "passed"
|
||||
ref = result.artifact_refs[0]
|
||||
assert result.details["artifact_content_types"][ref] == "image/jpeg"
|
||||
# #endregion Test.ScenarioExecution.ScreenshotProvider.MimeSniff
|
||||
# #endregion Test.ScenarioExecution.ScreenshotProvider.Evidence
|
||||
# #endregion Test.ScenarioExecution.ScreenshotProvider
|
||||
|
||||
@@ -120,6 +120,12 @@ def _mock_eval_adapter(step: dict, _completed: dict) -> dict:
|
||||
# #endregion Test.ScenarioExecution.Walker.MockEvalAdapter
|
||||
|
||||
|
||||
def _mock_eval_adapter_empty_pin(step: dict, completed: dict) -> dict:
|
||||
envelope = _mock_eval_adapter(step, completed)
|
||||
envelope["evaluation_record"] = {**envelope["evaluation_record"], "baseline_pin": {}}
|
||||
return envelope
|
||||
|
||||
|
||||
def test_walker_runs_fixture_to_completion(seeded_execution):
|
||||
run = start_run(
|
||||
seeded_execution,
|
||||
@@ -436,6 +442,36 @@ def test_walker_evaluation_baseline_and_semantic_pass(seeded_execution):
|
||||
# #endregion Test.ScenarioExecution.Walker.EvaluationPass
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.Walker.EvaluationPin [C:2] [TYPE Function]
|
||||
# @BRIEF The walker stamps the plan's resolved baseline pin into a record that carries none.
|
||||
# @TEST_INVARIANT ScenarioExecution.BaselineResolver.StampEvaluation: a baseline-backed plan's pin is
|
||||
# persisted into the AgentEvaluation record. -> VERIFIED_BY: walker_stamps_plan_baseline_pin
|
||||
def test_walker_stamps_plan_baseline_pin(seeded_execution):
|
||||
run = start_run(
|
||||
seeded_execution,
|
||||
"aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaa1",
|
||||
"bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbb1",
|
||||
{},
|
||||
"env-preprod-02",
|
||||
actor="test-walker",
|
||||
idempotency_key="walker-eval-pin-001",
|
||||
auto_advance=False,
|
||||
)
|
||||
resolved_pin = {"catalog_revision_id": str(uuid.uuid4()), "baseline_set_id": "ss-prod-visual", "baseline_set_version": "1"}
|
||||
plan = _required_eval_plan("eval-pin-044")
|
||||
plan["baseline_pin"] = resolved_pin
|
||||
run.runner_plan = plan
|
||||
seeded_execution.flush()
|
||||
registry = ScenarioExecutorRegistry()
|
||||
_register_default_executors(registry, agent_evaluation_adapter=_mock_eval_adapter_empty_pin)
|
||||
|
||||
_advance_run(seeded_execution, run, registry, worker_id="test-walker")
|
||||
|
||||
row = seeded_execution.query(AgentEvaluationRow).filter_by(scenario_run_id=run.id).one()
|
||||
assert row.baseline_pin == resolved_pin
|
||||
# #endregion Test.ScenarioExecution.Walker.EvaluationPin
|
||||
|
||||
|
||||
# #region Test.ScenarioExecution.Walker.EvaluationUnavailable [C:2] [TYPE Function]
|
||||
# @BRIEF Required evaluation with no adapter is inconclusive EVALUATION_UNAVAILABLE.
|
||||
# @TEST_INVARIANT ScenarioExecution.AgentEvaluation: required + missing adapter -> EVALUATION_UNAVAILABLE. -> VERIFIED_BY: walker_required_missing_adapter_unavailable
|
||||
|
||||
Reference in New Issue
Block a user