591 lines
25 KiB
Python
591 lines
25 KiB
Python
# #region Test.ScenarioExecution.AgentEvaluation [C:4] [TYPE Module] [SEMANTICS test,scenario,evaluation,parser,boundary]
|
|
# @RELATION BINDS_TO -> [ScenarioExecution.AgentEvaluation]
|
|
# @TEST_INVARIANT ScenarioExecution.AgentEvaluation: succeeded requires a bound raw response; failure
|
|
# shapes force inconclusive; the strict parser maps malformed to parser_error.
|
|
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter: production adapter builds manifest from completed,
|
|
# stores raw bytes, and never synthesizes PASS. -> VERIFIED_BY: adapter_success_from_completed, adapter_empty_completed_fail_closed, composition_root_adapter_fail_closed
|
|
from __future__ import annotations
|
|
|
|
import uuid
|
|
from datetime import UTC, datetime
|
|
from hashlib import sha256
|
|
|
|
import pytest
|
|
|
|
from src.services.dashboard_testing.execution.agent_evaluation import (
|
|
AgentEvaluation,
|
|
parse_evaluation_response,
|
|
)
|
|
from src.services.dashboard_testing.scenario.models import AgentEvaluationSpec
|
|
|
|
_NOW = datetime(2026, 9, 10, 0, 0, 0, tzinfo=UTC)
|
|
|
|
|
|
def _eval_dict(**overrides) -> dict:
|
|
data: dict = {
|
|
"schema_version": 1,
|
|
"evaluation_id": str(uuid.uuid4()),
|
|
"scenario_run_id": str(uuid.uuid4()),
|
|
"logical_step_id": "step-eval-1",
|
|
"attempt": 1,
|
|
"operation_id": str(uuid.uuid4()),
|
|
"evaluation_spec_hash": "a" * 64,
|
|
"provider_id": "llm-provider-a",
|
|
"provider_version": "1",
|
|
"model_id": "vision-model",
|
|
"model_version": "2026-08",
|
|
"prompt_template_id": "agent-evaluation-prompt",
|
|
"prompt_template_version": "1.0.0",
|
|
"prompt_template_hash": "b" * 64,
|
|
"output_schema_hash": "c" * 64,
|
|
"input_manifest_hash": "d" * 64,
|
|
"input_manifest": [{"artifact_id": "artifact-1", "sha256": "e" * 64, "content_type": "image/jpeg", "byte_length": 100, "role": "actual"}],
|
|
"baseline_pin": {"catalog_revision_id": str(uuid.uuid4())},
|
|
"comparison_ids": [str(uuid.uuid4())],
|
|
"status": "succeeded",
|
|
"verdict": "pass",
|
|
"confidence": 0.93,
|
|
"findings": [{"finding_id": "f-1", "severity": "info", "message": "matches", "evidence_artifact_ids": ["artifact-1"], "criterion_id": "crit-visual", "criterion_kind": "semantic"}],
|
|
"reason_codes": ["EVALUATION_PASS"],
|
|
"raw_response_artifact_ref": "artifact-1",
|
|
"raw_response_sha256": "e" * 64,
|
|
"trust_policy_hash": "f" * 64,
|
|
"usage": {"input_tokens": 100, "output_tokens": 10, "cost_amount": "0.01", "currency": "USD", "pricing_version": "1"},
|
|
"started_at": _NOW,
|
|
"finished_at": _NOW,
|
|
}
|
|
data.update(overrides)
|
|
return data
|
|
|
|
|
|
def _spec() -> AgentEvaluationSpec:
|
|
return AgentEvaluationSpec.model_validate({
|
|
"schema_version": 1,
|
|
"spec_id": str(uuid.uuid4()),
|
|
"provider_id": "llm-provider-a", "provider_version": "1", "model_id": "vision-model", "model_version": "2026-08",
|
|
"prompt_template_id": "agent-evaluation-prompt", "prompt_template_version": "1.0.0", "prompt_template_hash": "a" * 64,
|
|
"evidence_refs": ["artifact-1"], "comparison_refs": ["44444444-4444-4444-8444-444444444444"],
|
|
"output_schema": "agent-evaluation.schema.json",
|
|
"decision_policy": {"policy_id": "baseline-semantic", "version": "1.0.0"},
|
|
"limits": {"timeout_ms": 60000, "max_images": 4, "max_input_tokens": 32000, "max_output_tokens": 2000, "max_cost": "1.00", "currency": "USD"},
|
|
"trust_policy_hash": "b" * 64,
|
|
"criteria": [
|
|
{"criterion_id": "crit-visual", "criterion_kind": "semantic", "description": "visual", "comparison_id": None},
|
|
],
|
|
})
|
|
|
|
|
|
def test_succeeded_record_roundtrip():
|
|
record = AgentEvaluation.model_validate(_eval_dict())
|
|
assert record.status == "succeeded"
|
|
assert record.findings[0].criterion_id == "crit-visual"
|
|
|
|
|
|
def test_succeeded_requires_raw_response():
|
|
with pytest.raises(ValueError, match="EVALUATION_RAW_RESPONSE_REQUIRED"):
|
|
AgentEvaluation.model_validate(_eval_dict(raw_response_artifact_ref=None, raw_response_sha256=None))
|
|
|
|
|
|
def test_failure_shape_forces_inconclusive():
|
|
with pytest.raises(ValueError):
|
|
AgentEvaluation.model_validate(_eval_dict(status="provider_error"))
|
|
|
|
|
|
def test_logical_step_id_slug_is_allowed():
|
|
record = AgentEvaluation.model_validate(_eval_dict(logical_step_id="s4_assert_revenue"))
|
|
assert record.logical_step_id == "s4_assert_revenue"
|
|
|
|
|
|
def test_parser_accepts_valid_response():
|
|
result = parse_evaluation_response(_eval_dict(), spec=_spec())
|
|
assert result.status == "succeeded"
|
|
|
|
|
|
def test_parser_maps_malformed_to_parser_error():
|
|
result = parse_evaluation_response({"garbage": True}, spec=_spec())
|
|
assert result.status == "parser_error"
|
|
assert result.verdict == "inconclusive"
|
|
assert result.confidence == 0
|
|
assert result.findings == []
|
|
|
|
|
|
def test_parser_maps_criterion_kind_mismatch_to_parser_error():
|
|
result = parse_evaluation_response(
|
|
_eval_dict(findings=[{"finding_id": "f-1", "severity": "info", "message": "m", "evidence_artifact_ids": ["artifact-1"], "criterion_id": "crit-visual", "criterion_kind": "deterministic_comparison"}]),
|
|
spec=_spec(),
|
|
)
|
|
assert result.status == "parser_error"
|
|
|
|
|
|
# #region Test.ScenarioExecution.EvaluationAdapter.Store [C:1] [TYPE Class]
|
|
class _EvidenceStore:
|
|
def __init__(self) -> None:
|
|
self.stored: list[tuple[str, str, bytes]] = []
|
|
|
|
def store(self, run_id: str, digest: str, data: bytes) -> str:
|
|
self.stored.append((run_id, digest, data))
|
|
return f"draft:{run_id}:{digest}"
|
|
# #endregion Test.ScenarioExecution.EvaluationAdapter.Store
|
|
|
|
|
|
# #region Test.ScenarioExecution.EvaluationAdapter.Success [C:2] [TYPE Function]
|
|
# @BRIEF Injected submit + prior-step completed artifacts produce a bound evaluation_record.
|
|
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter: completed prior-step artifacts become input_manifest. -> VERIFIED_BY: adapter_success_from_completed
|
|
def test_evaluation_adapter_builds_manifest_and_returns_record():
|
|
from src.services.dashboard_testing.execution.evaluation_adapter import evaluation_adapter_from
|
|
|
|
prior_digest = "e" * 64
|
|
prior_ref = f"draft:run-eval:{prior_digest}"
|
|
run_id = str(uuid.uuid4())
|
|
storage = _EvidenceStore()
|
|
adapter = evaluation_adapter_from(storage=storage, submit=lambda **_kwargs: _eval_dict())
|
|
result = adapter(
|
|
{
|
|
"logical_step_id": "step-eval-1",
|
|
"scenario_run_id": run_id,
|
|
"attempt": 1,
|
|
"agent_evaluation_spec": _spec().model_dump(mode="json"),
|
|
},
|
|
{
|
|
"s1-shot": {
|
|
"status": "passed",
|
|
"artifact_refs": [prior_ref],
|
|
"step_outcome": {
|
|
"tool": "screenshot",
|
|
"artifact_digests": {prior_ref: prior_digest},
|
|
"content_type": "image/jpeg",
|
|
"byte_length": 100,
|
|
},
|
|
}
|
|
},
|
|
)
|
|
assert result["evaluation_input"]["status"] == "succeeded"
|
|
assert result["evaluation_input"]["verdict"] == "pass"
|
|
assert result["evaluation_input"]["confidence"] == 0.93
|
|
assert result["evaluation_record"]["input_manifest"][0]["artifact_id"] == prior_ref
|
|
assert result["content_type"] == "application/json"
|
|
assert result["byte_length"] == len(storage.stored[0][2])
|
|
assert result["artifact_refs"] == [f"draft:{run_id}:{storage.stored[0][1]}"]
|
|
# #endregion Test.ScenarioExecution.EvaluationAdapter.Success
|
|
|
|
|
|
# #region Test.ScenarioExecution.EvaluationAdapter.Normalize [C:3] [TYPE Function]
|
|
# @BRIEF Live provider shape (server-owned fields absent, finding extras present) normalizes into a
|
|
# schema-valid record — the 2026-09-10 live canary v2 finding.
|
|
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter.Normalize: provider text can never invent
|
|
# identity fields; non-succeeded statuses normalize to the pinned failure shape.
|
|
# -> VERIFIED_BY: adapter_normalizes_live_provider_response
|
|
def test_adapter_normalizes_live_provider_response():
|
|
from src.services.dashboard_testing.execution.evaluation_adapter import evaluation_adapter_from
|
|
|
|
prior_digest = "e" * 64
|
|
prior_ref = f"draft:run-norm:{prior_digest}"
|
|
run_id = str(uuid.uuid4())
|
|
spec = _spec()
|
|
live_shape = {
|
|
# Live model response shape (2026-09-10): no provider/model/identity fields at all —
|
|
# the prompt instructs the model not to emit them; the adapter stamps the pinned spec
|
|
# identity. Semantic verdict fields and finding extras are what the model does return.
|
|
"status": "inconclusive",
|
|
"verdict": "inconclusive",
|
|
"confidence": 0.0,
|
|
"findings": [{
|
|
"criterion_id": "crit-visual",
|
|
"criterion_kind": "semantic",
|
|
"evidence_artifact_ids": [prior_ref],
|
|
"rationale": "image contents not accessible",
|
|
"reason_codes": ["missing_evidence"],
|
|
"status": "inconclusive",
|
|
}],
|
|
"reason_codes": ["missing_evidence"],
|
|
"usage": {"input_tokens": 0, "output_tokens": 0},
|
|
}
|
|
storage = _EvidenceStore()
|
|
adapter = evaluation_adapter_from(storage=storage, submit=lambda **_kwargs: live_shape)
|
|
result = adapter(
|
|
{
|
|
"logical_step_id": "step-eval-1",
|
|
"scenario_run_id": run_id,
|
|
"attempt": 1,
|
|
"agent_evaluation_spec": spec.model_dump(mode="json"),
|
|
},
|
|
{
|
|
"s1-shot": {
|
|
"status": "passed",
|
|
"artifact_refs": [prior_ref],
|
|
"step_outcome": {
|
|
"tool": "screenshot",
|
|
"artifact_digests": {prior_ref: prior_digest},
|
|
"content_type": "image/jpeg",
|
|
"byte_length": 100,
|
|
},
|
|
}
|
|
},
|
|
)
|
|
record = result["evaluation_record"]
|
|
assert record["status"] == "succeeded"
|
|
assert record["verdict"] == "inconclusive"
|
|
assert record["provider_id"] == spec.provider_id
|
|
assert record["model_id"] == spec.model_id
|
|
assert len(record["findings"]) == 1
|
|
assert record["findings"][0]["message"] == "image contents not accessible"
|
|
assert record["findings"][0]["evidence_artifact_ids"] == [prior_ref]
|
|
assert record["prompt_template_id"] == spec.prompt_template_id
|
|
assert record["trust_policy_hash"] == spec.trust_policy_hash
|
|
assert record["evaluation_id"] and record["operation_id"]
|
|
assert record["input_manifest"][0]["artifact_id"] == prior_ref
|
|
# the raw response is stored verbatim for provenance
|
|
assert b"rationale" in storage.stored[0][2]
|
|
|
|
|
|
def _run_adapter_with_response(response: dict) -> dict:
|
|
from src.services.dashboard_testing.execution.evaluation_adapter import evaluation_adapter_from
|
|
|
|
prior_digest = "e" * 64
|
|
prior_ref = f"draft:run-norm:{prior_digest}"
|
|
run_id = str(uuid.uuid4())
|
|
adapter = evaluation_adapter_from(storage=_EvidenceStore(), submit=lambda **_kwargs: response)
|
|
result = adapter(
|
|
{
|
|
"logical_step_id": "step-eval-1",
|
|
"scenario_run_id": run_id,
|
|
"attempt": 1,
|
|
"agent_evaluation_spec": _spec().model_dump(mode="json"),
|
|
},
|
|
{
|
|
"s1-shot": {
|
|
"status": "passed",
|
|
"artifact_refs": [prior_ref],
|
|
"step_outcome": {
|
|
"tool": "screenshot",
|
|
"artifact_digests": {prior_ref: prior_digest},
|
|
"content_type": "image/jpeg",
|
|
"byte_length": 100,
|
|
},
|
|
}
|
|
},
|
|
)
|
|
return result["evaluation_record"]
|
|
|
|
|
|
# @TEST_EDGE: unknown_criterion_id -> finding dropped, unsupported pass/fail verdict fails closed
|
|
# @TEST_EDGE: omitted_criterion_id_single_criterion_spec -> unambiguous mapping to the pinned criterion
|
|
def test_unknown_criterion_id_is_dropped_and_verdict_fails_closed():
|
|
record = _run_adapter_with_response({
|
|
"status": "succeeded",
|
|
"verdict": "pass",
|
|
"confidence": 0.9,
|
|
"findings": [{
|
|
"criterion_id": "crit-hallucinated",
|
|
"evidence_artifact_ids": ["draft:run-norm:" + "e" * 64],
|
|
"message": "looks fine",
|
|
}],
|
|
"reason_codes": [],
|
|
"usage": {"input_tokens": 1, "output_tokens": 1},
|
|
})
|
|
assert record["findings"] == []
|
|
assert record["verdict"] == "inconclusive"
|
|
assert record["confidence"] == 0
|
|
assert "EVALUATION_FINDING_UNSUPPORTED" in record["reason_codes"]
|
|
|
|
|
|
def test_omitted_criterion_id_maps_to_the_single_pinned_criterion():
|
|
record = _run_adapter_with_response({
|
|
"status": "succeeded",
|
|
"verdict": "pass",
|
|
"confidence": 0.9,
|
|
"findings": [{
|
|
"evidence_artifact_ids": ["draft:run-norm:" + "e" * 64],
|
|
"message": "looks fine",
|
|
}],
|
|
"reason_codes": [],
|
|
"usage": {"input_tokens": 1, "output_tokens": 1},
|
|
})
|
|
assert record["verdict"] == "pass"
|
|
assert len(record["findings"]) == 1
|
|
assert record["findings"][0]["criterion_id"] == "crit-visual"
|
|
assert record["findings"][0]["criterion_kind"] == "semantic"
|
|
# #endregion Test.ScenarioExecution.EvaluationAdapter.Normalize
|
|
|
|
|
|
# #region Test.ScenarioExecution.EvaluationAdapter.Identity [C:2] [TYPE Function]
|
|
# @BRIEF Provider text cannot author baseline identity; the adapter zeroes any provider baseline_pin.
|
|
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter.Normalize: a provider-supplied baseline_pin is
|
|
# discarded; only the walker's server-resolved plan pin is persisted.
|
|
# -> VERIFIED_BY: test_adapter_drops_provider_supplied_baseline_pin
|
|
def test_adapter_drops_provider_supplied_baseline_pin():
|
|
record = _run_adapter_with_response({
|
|
"status": "succeeded",
|
|
"verdict": "inconclusive",
|
|
"confidence": 0.2,
|
|
"findings": [{
|
|
"criterion_id": "crit-visual",
|
|
"evidence_artifact_ids": ["draft:run-norm:" + "e" * 64],
|
|
"message": "claimed pin",
|
|
}],
|
|
"reason_codes": [],
|
|
"baseline_pin": {"catalog_revision_id": "provider-claimed"},
|
|
"usage": {"input_tokens": 1, "output_tokens": 1},
|
|
})
|
|
assert record["baseline_pin"] == {}
|
|
# #endregion Test.ScenarioExecution.EvaluationAdapter.Identity
|
|
|
|
|
|
# #region Test.ScenarioExecution.EvaluationAdapter.ServerIdentity [C:2] [TYPE Function]
|
|
# @BRIEF Provider-supplied evaluation_id/operation_id are replaced by server-generated ids.
|
|
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter.Normalize: provider text never authors
|
|
# evaluation/operation identity. -> VERIFIED_BY: test_adapter_replaces_provider_identity_ids
|
|
def test_adapter_replaces_provider_identity_ids():
|
|
record = _run_adapter_with_response({
|
|
"status": "succeeded",
|
|
"verdict": "inconclusive",
|
|
"confidence": 0.2,
|
|
"findings": [{
|
|
"criterion_id": "crit-visual",
|
|
"evidence_artifact_ids": ["draft:run-norm:" + "e" * 64],
|
|
"message": "identity claim",
|
|
}],
|
|
"reason_codes": [],
|
|
"evaluation_id": "provider-chosen-eval",
|
|
"operation_id": "provider-chosen-op",
|
|
"usage": {"input_tokens": 1, "output_tokens": 1},
|
|
})
|
|
assert record["evaluation_id"] != "provider-chosen-eval"
|
|
assert record["operation_id"] != "provider-chosen-op"
|
|
# #endregion Test.ScenarioExecution.EvaluationAdapter.ServerIdentity
|
|
|
|
|
|
# #region Test.ScenarioExecution.EvaluationAdapter.Empty [C:2] [TYPE Function]
|
|
# @BRIEF Missing prior-step evidence is fail-closed; the adapter never synthesizes PASS.
|
|
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter: empty completed fails closed. -> VERIFIED_BY: adapter_empty_completed_fail_closed
|
|
def test_evaluation_adapter_empty_completed_fail_closed():
|
|
from src.services.dashboard_testing.execution.evaluation_adapter import evaluation_adapter_from
|
|
|
|
adapter = evaluation_adapter_from(storage=_EvidenceStore(), submit=lambda **_kwargs: _eval_dict())
|
|
with pytest.raises(RuntimeError, match="EVALUATION_EVIDENCE_NOT_FOUND"):
|
|
adapter(
|
|
{
|
|
"logical_step_id": "step-eval-1",
|
|
"scenario_run_id": str(uuid.uuid4()),
|
|
"agent_evaluation_spec": _spec().model_dump(mode="json"),
|
|
},
|
|
{},
|
|
)
|
|
# #endregion Test.ScenarioExecution.EvaluationAdapter.Empty
|
|
|
|
|
|
# #region Test.ScenarioExecution.EvaluationAdapter.Composition [C:2] [TYPE Function]
|
|
# @BRIEF Composition root always exposes an adapter; missing spec/evidence stays non-PASS.
|
|
# @TEST_INVARIANT ScenarioExecution.LiveCompositionRoot: agent_evaluation_adapter is composed and fail-closed. -> VERIFIED_BY: composition_root_adapter_fail_closed
|
|
def test_composition_root_evaluation_adapter_is_composed_and_fail_closed():
|
|
from src.services.dashboard_testing.execution.executors import agent_evaluation
|
|
from src.services.dashboard_testing.execution.live_composition import LiveExecutionCompositionRoot
|
|
|
|
adapter = LiveExecutionCompositionRoot().agent_evaluation_adapter()
|
|
outcome = agent_evaluation(
|
|
{"logical_step_id": "step-eval-1", "scenario_run_id": str(uuid.uuid4())},
|
|
{},
|
|
adapter=adapter,
|
|
)
|
|
assert adapter is not None
|
|
assert outcome["status"] == "inconclusive"
|
|
assert outcome["error_code"] is not None
|
|
# #endregion Test.ScenarioExecution.EvaluationAdapter.Composition
|
|
|
|
|
|
# #region Test.ScenarioExecution.EvaluationAdapter.Multimodal [C:3] [TYPE Function]
|
|
# @BRIEF Captured images reach the judge as sniffed content-parts; text-only providers stay text-only.
|
|
# @TEST_INVARIANT ScenarioExecution.EvaluationAdapter.Images: only manifest image items whose bytes
|
|
# sniff to an allowlisted image MIME are attached; unreadable/non-image items are skipped.
|
|
# -> VERIFIED_BY: adapter_attaches_sniffed_images, adapter_skips_invalid_evidence
|
|
# @TEST_INVARIANT ScenarioExecution.AgentEvaluation.Messages: images become base64 data URLs only for a
|
|
# multimodal provider. -> VERIFIED_BY: submit_evaluation_multimodal_gate
|
|
_PNG_BYTES = b"\x89PNG\r\n\x1a\n" + b"evidence"
|
|
_JPEG_BYTES = b"\xff\xd8\xff\xe0" + b"evidence"
|
|
|
|
|
|
class _RetrievingStore(_EvidenceStore):
|
|
def __init__(self, payloads: dict[str, bytes]) -> None:
|
|
super().__init__()
|
|
self.payloads = payloads
|
|
|
|
def retrieve(self, content_ref: str):
|
|
return self.payloads.get(content_ref)
|
|
|
|
|
|
def _image_outcome(ref: str, digest: str, payload: bytes, **nested_overrides) -> dict:
|
|
nested = {
|
|
"tool": "screenshot",
|
|
"artifact_digests": {ref: digest},
|
|
"artifact_content_types": {ref: "image/png"},
|
|
"artifact_byte_lengths": {ref: len(payload)},
|
|
}
|
|
nested.update(nested_overrides)
|
|
return {"status": "passed", "artifact_refs": [ref], "step_outcome": nested}
|
|
|
|
|
|
def test_adapter_attaches_sniffed_images_from_manifest():
|
|
from src.services.dashboard_testing.execution.evaluation_adapter import evaluation_adapter_from
|
|
|
|
png_digest = sha256(_PNG_BYTES).hexdigest()
|
|
prior_ref = f"draft:run-mm:{png_digest}"
|
|
run_id = str(uuid.uuid4())
|
|
captured: dict = {}
|
|
|
|
def _submit(**kwargs):
|
|
captured.update(kwargs)
|
|
return _eval_dict()
|
|
|
|
store = _RetrievingStore({prior_ref: _PNG_BYTES})
|
|
adapter = evaluation_adapter_from(storage=store, submit=_submit)
|
|
adapter(
|
|
{
|
|
"logical_step_id": "step-eval-1",
|
|
"scenario_run_id": run_id,
|
|
"attempt": 1,
|
|
"agent_evaluation_spec": _spec().model_dump(mode="json"),
|
|
},
|
|
{"s1-shot": _image_outcome(prior_ref, png_digest, _PNG_BYTES)},
|
|
)
|
|
|
|
assert captured["images"] == [(_PNG_BYTES, "image/png")]
|
|
|
|
|
|
def test_adapter_skips_invalid_evidence_before_attaching():
|
|
from src.services.dashboard_testing.execution.evaluation_adapter import evaluation_adapter_from
|
|
|
|
json_ref = "draft:run-mm-json:" + ("a" * 64)
|
|
bad_ref = "draft:run-mm-bad:" + ("b" * 64)
|
|
unreadable_ref = "draft:run-mm-missing:" + ("c" * 64)
|
|
run_id = str(uuid.uuid4())
|
|
captured: dict = {}
|
|
|
|
def _submit(**kwargs):
|
|
captured.update(kwargs)
|
|
return _eval_dict()
|
|
|
|
store = _RetrievingStore({bad_ref: b"not-an-image-signature"})
|
|
adapter = evaluation_adapter_from(storage=store, submit=_submit)
|
|
adapter(
|
|
{
|
|
"logical_step_id": "step-eval-1",
|
|
"scenario_run_id": run_id,
|
|
"attempt": 1,
|
|
"agent_evaluation_spec": _spec().model_dump(mode="json"),
|
|
},
|
|
{
|
|
"s1-json": {
|
|
"status": "passed",
|
|
"artifact_refs": [json_ref],
|
|
"step_outcome": {
|
|
"tool": "superset_api",
|
|
"artifact_digests": {json_ref: "a" * 64},
|
|
"artifact_content_types": {json_ref: "application/json"},
|
|
"artifact_byte_lengths": {json_ref: 10},
|
|
},
|
|
},
|
|
"s2-bad": _image_outcome(bad_ref, "b" * 64, b"not-an-image-signature"),
|
|
"s3-missing": _image_outcome(unreadable_ref, "c" * 64, b"\x89PNG-still"),
|
|
},
|
|
)
|
|
|
|
assert captured["images"] == []
|
|
|
|
|
|
def test_evaluation_messages_text_only_and_multimodal():
|
|
from src.services.dashboard_testing.execution.agent_evaluation import _evaluation_messages
|
|
|
|
assert _evaluation_messages("P", None) == [{"role": "user", "content": "P"}]
|
|
parts = _evaluation_messages("P", [(b"abc", "image/png")])
|
|
content = parts[0]["content"]
|
|
assert content[0] == {"type": "text", "text": "P"}
|
|
assert content[1]["image_url"]["url"] == "data:image/png;base64,YWJj"
|
|
|
|
|
|
@pytest.mark.parametrize("is_multimodal, expects_parts", [(True, True), (False, False)])
|
|
def test_submit_evaluation_gates_images_on_multimodal_provider(monkeypatch, is_multimodal, expects_parts):
|
|
import asyncio
|
|
|
|
import src.plugins.llm_analysis.service as llm_service_module
|
|
import src.services.llm_provider as llm_provider_module
|
|
from src.services.dashboard_testing.execution import agent_evaluation as ae
|
|
|
|
class _Provider:
|
|
provider_type = "openai"
|
|
base_url = "http://local-llm"
|
|
is_multimodal = False
|
|
|
|
_Provider.is_multimodal = is_multimodal
|
|
|
|
seen: dict = {}
|
|
|
|
class _Client:
|
|
def __init__(self, *args, **kwargs):
|
|
pass
|
|
|
|
async def get_json_completion(self, messages):
|
|
seen["messages"] = messages
|
|
return {"status": "succeeded"}
|
|
|
|
class _ProviderService:
|
|
def __init__(self, db):
|
|
pass
|
|
|
|
def get_provider(self, provider_id):
|
|
return _Provider()
|
|
|
|
def get_decrypted_api_key(self, provider_id):
|
|
return "key"
|
|
|
|
monkeypatch.setattr(llm_provider_module, "LLMProviderService", _ProviderService)
|
|
monkeypatch.setattr(llm_service_module, "LLMClient", _Client)
|
|
monkeypatch.setattr(ae, "claim_capacity", lambda *args, **kwargs: {"lease_id": "lease-1"})
|
|
monkeypatch.setattr(ae, "release_capacity", lambda *args, **kwargs: None)
|
|
|
|
result = asyncio.run(ae.submit_evaluation(
|
|
object(), spec=_spec(), prompt="P", images=[(b"abc", "image/png")],
|
|
environment_id="env", environment_class="DEV", run_id="run", logical_step_id="step",
|
|
))
|
|
|
|
assert result == {"status": "succeeded"}
|
|
content = seen["messages"][0]["content"]
|
|
if expects_parts:
|
|
assert isinstance(content, list)
|
|
assert content[1]["image_url"]["url"] == "data:image/png;base64,YWJj"
|
|
else:
|
|
assert content == "P"
|
|
|
|
|
|
def _image_manifest_item(ref: str, data: bytes) -> dict:
|
|
return {
|
|
"artifact_id": ref, "sha256": sha256(data).hexdigest(),
|
|
"content_type": "image/png", "byte_length": len(data), "role": "actual",
|
|
}
|
|
|
|
|
|
def test_image_payloads_respect_count_and_byte_budget(monkeypatch):
|
|
from src.services.dashboard_testing.execution import evaluation_images as ev
|
|
|
|
payload_a = b"\x89PNG\r\n\x1a\n" + b"a" * 92
|
|
payload_b = b"\x89PNG\r\n\x1a\n" + b"b" * 92
|
|
store = _RetrievingStore({"ref-a": payload_a, "ref-b": payload_b})
|
|
manifest = [_image_manifest_item("ref-a", payload_a), _image_manifest_item("ref-b", payload_b)]
|
|
|
|
assert ev.image_payloads_from_manifest(manifest, store, max_images=1) == [(payload_a, "image/png")]
|
|
|
|
monkeypatch.setattr(ev, "MAX_EVALUATION_IMAGE_BYTES", 150)
|
|
assert ev.image_payloads_from_manifest(manifest, store, max_images=8) == [(payload_a, "image/png")]
|
|
|
|
|
|
def test_image_payloads_skip_digest_mismatch():
|
|
from src.services.dashboard_testing.execution.evaluation_images import image_payloads_from_manifest
|
|
|
|
store = _RetrievingStore({"ref-a": _PNG_BYTES})
|
|
tampered = {**_image_manifest_item("ref-a", _PNG_BYTES), "sha256": "0" * 64}
|
|
assert image_payloads_from_manifest([tampered], store, max_images=8) == []
|
|
|
|
missing_digest = {**_image_manifest_item("ref-a", _PNG_BYTES)}
|
|
missing_digest.pop("sha256")
|
|
assert image_payloads_from_manifest([missing_digest], store, max_images=8) == []
|
|
# #endregion Test.ScenarioExecution.EvaluationAdapter.Multimodal
|
|
# #endregion Test.ScenarioExecution.AgentEvaluation |