feat(dashboard-testing): add chart health diagnostics and per-tab evidence
Focused matrix and cleanup checks pass. Native Superset error/recovery acceptance and final semantic freeze remain open; this is not release GO.
This commit is contained in:
@@ -28,6 +28,7 @@ class ChartDataResponse:
|
||||
parsed: dict[str, Any]
|
||||
raw_bytes: bytes
|
||||
source_response_hash: str = field(repr=False)
|
||||
http_status: int = 200
|
||||
# #endregion SupersetClient.ChartData.ResponseDTO
|
||||
|
||||
# #region SupersetClient.ChartData.SupersetChartDataMixin [C:4] [TYPE Class]
|
||||
@@ -118,11 +119,15 @@ class SupersetChartDataMixin:
|
||||
)
|
||||
raw_bytes: bytes = httpx_response.content
|
||||
source_response_hash: str = hashlib.sha256(raw_bytes).hexdigest()
|
||||
parsed: dict[str, Any] = httpx_response.json()
|
||||
try:
|
||||
parsed: dict[str, Any] = httpx_response.json()
|
||||
except ValueError:
|
||||
parsed = {'_invalid_response':True}
|
||||
return ChartDataResponse(
|
||||
parsed=parsed,
|
||||
raw_bytes=raw_bytes,
|
||||
source_response_hash=source_response_hash,
|
||||
http_status=httpx_response.status_code,
|
||||
)
|
||||
except SupersetAPIError:
|
||||
raise
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
# #region DashboardTesting.ChartHealth.Contract [C:4] [TYPE Module] [SEMANTICS chart,health,diagnostic,provenance]
|
||||
# @BRIEF Closed diagnostic observations distinguish confirmed query failure from unavailable evidence.
|
||||
# @INVARIANT Sanitized payloads never claim to be original HTTP response bytes or numeric values.
|
||||
from typing import Literal
|
||||
from pydantic import BaseModel, ConfigDict, Field
|
||||
|
||||
|
||||
# #region DashboardTesting.ChartHealth.Contract.Diagnostic [C:3] [TYPE Class]
|
||||
# @POST Identity, origin, filter/generation attribution and bounded error text remain explicit.
|
||||
class ChartDiagnostic(BaseModel):
|
||||
model_config = ConfigDict(extra='forbid', strict=True)
|
||||
schema_version: Literal[1] = 1
|
||||
chart_id: int | None = Field(default=None, gt=0)
|
||||
dataset_id: int | None = Field(default=None, gt=0)
|
||||
placement_id: str = Field(min_length=1, max_length=256)
|
||||
environment_id: str | None = None
|
||||
dashboard_id: int | None = Field(default=None,gt=0)
|
||||
run_id: str | None = None
|
||||
logical_step_id: str | None = None
|
||||
attempt: int | None = Field(default=None,ge=1)
|
||||
chart_name: str = Field(max_length=512)
|
||||
tab_path: list[str] = Field(default_factory=list, max_length=32)
|
||||
status: Literal['healthy', 'failed', 'inconclusive']
|
||||
reason_code: str = Field(min_length=1, max_length=128)
|
||||
origin: Literal['dom', 'http_response', 'synthetic_exception']
|
||||
payload_kind: Literal['sanitized_diagnostic'] = 'sanitized_diagnostic'
|
||||
filter_fingerprint: str | None = None
|
||||
request_identity: str | None = None
|
||||
load_generation: int | None = Field(default=None, ge=1)
|
||||
observed_at: str
|
||||
duration_seconds: float = Field(ge=0)
|
||||
http_status: int | None = Field(default=None, ge=100, le=599)
|
||||
application_code: str | None = Field(default=None, max_length=256)
|
||||
database_code: str | None = Field(default=None, max_length=128)
|
||||
message: str = Field(default='', max_length=4096)
|
||||
terminal_state: Literal['ready', 'error', 'loading', 'unavailable']
|
||||
original_response_sha256: str | None = None
|
||||
original_byte_length: int | None = Field(default=None, ge=0)
|
||||
truncated: bool = False
|
||||
# #endregion DashboardTesting.ChartHealth.Contract.Diagnostic
|
||||
# #endregion DashboardTesting.ChartHealth.Contract
|
||||
@@ -22,6 +22,8 @@ from .artifacts import is_valid_sha256
|
||||
from .capacity import CapacityUnavailable, claim_capacity, release_capacity
|
||||
|
||||
|
||||
# #region ScenarioExecution.AgentEvaluation.ManifestItem [C:2] [TYPE Class]
|
||||
# @BRIEF Closed declared artifact metadata for evaluation input.
|
||||
class EvaluationManifestItem(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
artifact_id: str
|
||||
@@ -31,6 +33,9 @@ class EvaluationManifestItem(BaseModel):
|
||||
role: Literal["actual", "baseline", "comparison", "context"]
|
||||
|
||||
|
||||
# #endregion ScenarioExecution.AgentEvaluation.ManifestItem
|
||||
# #region ScenarioExecution.AgentEvaluation.Finding [C:2] [TYPE Class]
|
||||
# @BRIEF Findings bind to a declared criterion and concrete evidence refs.
|
||||
class EvaluationFinding(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
finding_id: str = Field(min_length=1)
|
||||
@@ -42,6 +47,9 @@ class EvaluationFinding(BaseModel):
|
||||
criterion_kind: Literal["semantic", "deterministic_comparison"]
|
||||
|
||||
|
||||
# #endregion ScenarioExecution.AgentEvaluation.Finding
|
||||
# #region ScenarioExecution.AgentEvaluation.Usage [C:2] [TYPE Class]
|
||||
# @BRIEF Usage is optional measured transport metadata, never inferred billing.
|
||||
class EvaluationUsage(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
input_tokens: int | None = Field(default=None, ge=0)
|
||||
@@ -51,6 +59,9 @@ class EvaluationUsage(BaseModel):
|
||||
pricing_version: str | None = None
|
||||
|
||||
|
||||
# #endregion ScenarioExecution.AgentEvaluation.Usage
|
||||
# #region ScenarioExecution.AgentEvaluation.Record [C:3] [TYPE Class]
|
||||
# @INVARIANT A succeeded record retains raw response provenance; semantic review may have no deterministic comparisons.
|
||||
class AgentEvaluation(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
schema_version: Literal[1]
|
||||
@@ -71,7 +82,7 @@ class AgentEvaluation(BaseModel):
|
||||
input_manifest_hash: str = Field(pattern=r"^[a-f0-9]{64}$")
|
||||
input_manifest: list[EvaluationManifestItem] = Field(min_length=1, max_length=100)
|
||||
baseline_pin: dict[str, Any]
|
||||
comparison_ids: list[str] = Field(min_length=1, max_length=100)
|
||||
comparison_ids: list[str] = Field(default_factory=list, max_length=100)
|
||||
status: Literal["succeeded", "provider_error", "parser_error", "budget_exceeded", "cancelled", "timed_out"]
|
||||
verdict: Literal["pass", "fail", "inconclusive"]
|
||||
confidence: float = Field(ge=0, le=1)
|
||||
@@ -84,6 +95,8 @@ class AgentEvaluation(BaseModel):
|
||||
started_at: datetime
|
||||
finished_at: datetime
|
||||
|
||||
# #region ScenarioExecution.AgentEvaluation.Record.UUIDs [C:2] [TYPE Function]
|
||||
# @POST Run/operation/evaluation identities are UUIDs; logical step slugs remain valid.
|
||||
@model_validator(mode="after")
|
||||
def validate_uuid_fields(self) -> "AgentEvaluation":
|
||||
# logical_step_id is deliberately excluded: runtime step identity is a registry slug
|
||||
@@ -97,6 +110,9 @@ class AgentEvaluation(BaseModel):
|
||||
raise ValueError(f"{field} must be a UUID") from exc
|
||||
return self
|
||||
|
||||
# #endregion ScenarioExecution.AgentEvaluation.Record.UUIDs
|
||||
# #region ScenarioExecution.AgentEvaluation.Record.Failure [C:3] [TYPE Function]
|
||||
# @POST Non-succeeded records cannot claim findings, confidence or PASS.
|
||||
@model_validator(mode="after")
|
||||
def validate_failure_shape(self) -> "AgentEvaluation":
|
||||
if self.status == "succeeded" and (not self.raw_response_artifact_ref or not self.raw_response_sha256):
|
||||
@@ -104,8 +120,12 @@ class AgentEvaluation(BaseModel):
|
||||
if self.status != "succeeded" and (self.verdict != "inconclusive" or self.confidence != 0 or self.findings or not self.reason_codes):
|
||||
raise ValueError("EVALUATION_FAILURE_SHAPE_INVALID")
|
||||
return self
|
||||
# #endregion ScenarioExecution.AgentEvaluation.Record.Failure
|
||||
# #endregion ScenarioExecution.AgentEvaluation.Record
|
||||
|
||||
|
||||
# #region ScenarioExecution.AgentEvaluation.Parse [C:3] [TYPE Function]
|
||||
# @POST Declared criterion identities are checked; malformed responses become typed parser errors.
|
||||
def parse_evaluation_response(raw: dict[str, Any], *, spec: AgentEvaluationSpec) -> AgentEvaluation:
|
||||
"""Parse and cross-check a provider response; malformed responses become typed parser errors."""
|
||||
try:
|
||||
@@ -132,6 +152,9 @@ def parse_evaluation_response(raw: dict[str, Any], *, spec: AgentEvaluationSpec)
|
||||
})
|
||||
|
||||
|
||||
# #endregion ScenarioExecution.AgentEvaluation.Parse
|
||||
# #region ScenarioExecution.AgentEvaluation.Persist [C:3] [TYPE Function]
|
||||
# @POST Duplicate evaluation identity cannot overwrite prior evidence.
|
||||
def persist_agent_evaluation(db: Session, record: AgentEvaluation) -> AgentEvaluationRow:
|
||||
"""Persist one immutable identity; a duplicate identity is rejected rather than updated."""
|
||||
if db.query(AgentEvaluationRow).filter_by(
|
||||
@@ -143,6 +166,7 @@ def persist_agent_evaluation(db: Session, record: AgentEvaluation) -> AgentEvalu
|
||||
db.add(row)
|
||||
db.flush()
|
||||
return row
|
||||
# #endregion ScenarioExecution.AgentEvaluation.Persist
|
||||
|
||||
|
||||
# #region ScenarioExecution.AgentEvaluation.Evidence [C:4] [TYPE Function] [SEMANTICS evaluation,evidence,binding,p0-1]
|
||||
@@ -177,6 +201,14 @@ def validate_evaluation_evidence(
|
||||
if row.logical_step_id == logical_step_id and row.attempt == attempt
|
||||
]
|
||||
step_by_ref = {row.id: row for row in step_rows} | {row.content_ref: row for row in step_rows}
|
||||
_validate_input_refs(refs,finding_refs,owned_by_ref,run_id)
|
||||
_validate_raw_ref(step_by_ref,raw_response_artifact_ref,raw_response_sha256,succeeded,logical_step_id,attempt,run_id)
|
||||
# #endregion ScenarioExecution.AgentEvaluation.Evidence
|
||||
|
||||
|
||||
# #region ScenarioExecution.AgentEvaluation.InputRefs [C:3] [TYPE Function]
|
||||
# @POST Every input/finding ref resolves to active same-run metadata; foreign evidence refuses.
|
||||
def _validate_input_refs(refs,finding_refs,owned_by_ref,run_id):
|
||||
for ref, item in refs.items():
|
||||
row = owned_by_ref.get(ref)
|
||||
if row is None:
|
||||
@@ -191,6 +223,13 @@ def validate_evaluation_evidence(
|
||||
raise ValueError("EVALUATION_EVIDENCE_NOT_FOUND")
|
||||
if row.owner_id != run_id or row.owner_type != "scenario_run":
|
||||
raise ValueError("EVALUATION_EVIDENCE_OWNER_INVALID")
|
||||
|
||||
# #endregion ScenarioExecution.AgentEvaluation.InputRefs
|
||||
|
||||
|
||||
# #region ScenarioExecution.AgentEvaluation.RawRef [C:3] [TYPE Function]
|
||||
# @POST Successful raw evidence remains bound to exact producing step/attempt/digest.
|
||||
def _validate_raw_ref(step_by_ref,raw_response_artifact_ref,raw_response_sha256,succeeded,logical_step_id,attempt,run_id):
|
||||
if succeeded:
|
||||
if not raw_response_artifact_ref or not is_valid_sha256(raw_response_sha256):
|
||||
raise ValueError("EVALUATION_EVIDENCE_RAW_REQUIRED")
|
||||
@@ -205,7 +244,7 @@ def validate_evaluation_evidence(
|
||||
raw = step_by_ref.get(raw_response_artifact_ref)
|
||||
if raw is None:
|
||||
raise ValueError("EVALUATION_EVIDENCE_NOT_FOUND")
|
||||
# #endregion ScenarioExecution.AgentEvaluation.Evidence
|
||||
# #endregion ScenarioExecution.AgentEvaluation.RawRef
|
||||
|
||||
|
||||
# #region ScenarioExecution.AgentEvaluation.Messages [C:3] [TYPE Function] [SEMANTICS evaluation,multimodal,messages,content-parts]
|
||||
@@ -251,6 +290,7 @@ async def submit_evaluation(
|
||||
from src.plugins.llm_analysis.models import LLMProviderType
|
||||
from src.plugins.llm_analysis.service import LLMClient
|
||||
from src.services.llm_provider import LLMProviderService
|
||||
from .evaluation_capacity_identity import capacity_provider_version
|
||||
|
||||
lease = None
|
||||
try:
|
||||
@@ -259,7 +299,7 @@ async def submit_evaluation(
|
||||
raise RuntimeError("EVALUATION_PROVIDER_MISSING")
|
||||
lease = claim_capacity(db, environment_id=environment_id, environment_class=environment_class,
|
||||
workload_class="agent_evaluation", provider_id=spec.provider_id,
|
||||
provider_version=spec.provider_version, run_id=run_id,
|
||||
provider_version=capacity_provider_version(spec.provider_version), run_id=run_id,
|
||||
logical_step_id=logical_step_id)
|
||||
api_key = LLMProviderService(db).get_decrypted_api_key(spec.provider_id)
|
||||
if not api_key:
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
# #region ScenarioExecution.ChartHealth.Artifact [C:4] [TYPE Module] [SEMANTICS ownership,digest,evidence]
|
||||
# @BRIEF Share exact owned artifact proof for tab frames and evaluation records.
|
||||
from hashlib import sha256
|
||||
from src.models.scenario_artifact import ScenarioArtifact
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Artifact.Verify [C:3] [TYPE Function]
|
||||
# @POST Same-run/step/attempt active identity, MIME, length and bytes are verified before model input or result projection.
|
||||
def verify_health_artifact(journal, db, receipt):
|
||||
artifact = db.get(ScenarioArtifact,receipt['artifact_id'])
|
||||
data = journal.storage.retrieve(receipt['content_ref'])
|
||||
if (artifact is None or not artifact.is_active or artifact.owner_type != 'scenario_run' or artifact.owner_id != journal.run_id
|
||||
or artifact.logical_step_id != journal.logical_step_id or artifact.attempt != journal.attempt
|
||||
or artifact.content_ref != receipt['content_ref'] or artifact.sha256 != receipt['sha256']
|
||||
or artifact.byte_length != receipt['byte_length'] or artifact.content_type != receipt['content_type']
|
||||
or data is None or len(data) != receipt['byte_length'] or sha256(data).hexdigest() != receipt['sha256']):
|
||||
raise ValueError('CHART_HEALTH_ARTIFACT_INVALID')
|
||||
return artifact,data
|
||||
# #endregion ScenarioExecution.ChartHealth.Artifact.Verify
|
||||
# #endregion ScenarioExecution.ChartHealth.Artifact
|
||||
@@ -0,0 +1,103 @@
|
||||
# #region ScenarioExecution.ChartHealth.Delivery [C:5] [TYPE Module] [SEMANTICS notification,CAS,at-most-once,summary]
|
||||
# @BRIEF Deliver a committed named-error receipt once through explicitly configured existing providers.
|
||||
# @INVARIANT Claim commits before transport; ambiguous dispatching receipts are never automatically retried.
|
||||
import asyncio
|
||||
from hashlib import sha256
|
||||
from time import monotonic
|
||||
from sqlalchemy import update
|
||||
from src.models.scenario_automation import ScenarioNotificationEvent
|
||||
from ..query_failure import safe_message
|
||||
from .chart_health_delivery_config import read_alert_config,configured_provider
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Delivery.Body [C:3] [TYPE Function]
|
||||
# @POST Bounded redacted text names each retained chart/cause and the run; omission is explicit.
|
||||
def message_body(receipt):
|
||||
summary = receipt.payload['chart_query_errors']
|
||||
lines = [f"Scenario run: {receipt.run_id}",f"Chart query errors: {summary['count']}"]
|
||||
for item in summary.get('items',[])[:100]:
|
||||
context = '/'.join(item.get('tab_path',[]))
|
||||
text = safe_message(item.get('message',''))[0][:512]
|
||||
name = safe_message(item.get('chart_name',''))[0][:512]
|
||||
lines.append(f"{context}: {name} — Code {item.get('database_code') or 'unknown'} — {text}")
|
||||
if summary.get('omitted'):
|
||||
lines.append(f"Additional omitted errors: {summary['omitted']}")
|
||||
raw = '\n'.join(lines).encode()
|
||||
suffix = b'\n[notification summary truncated]'
|
||||
return (raw[:16384-len(suffix)]+suffix).decode(errors='ignore') if len(raw) > 16384 else raw.decode()
|
||||
# #endregion ScenarioExecution.ChartHealth.Delivery.Body
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Delivery.Claim [C:4] [TYPE Function]
|
||||
# @POST One atomic pending-to-dispatching CAS owns physical sends; no target/config secrets are persisted.
|
||||
def claim_receipt(db, receipt):
|
||||
payload = dict(receipt.payload)
|
||||
payload['health_delivery'] = {'status':'dispatching','reason_code':'DELIVERY_CLAIMED'}
|
||||
statement = update(ScenarioNotificationEvent).where(
|
||||
ScenarioNotificationEvent.id == receipt.id,
|
||||
ScenarioNotificationEvent.payload['health_delivery']['status'].as_string() == 'pending',
|
||||
).values(payload=payload).execution_options(synchronize_session=False)
|
||||
claimed = db.execute(statement).rowcount == 1
|
||||
db.commit()
|
||||
if claimed:
|
||||
db.refresh(receipt)
|
||||
return claimed
|
||||
# #endregion ScenarioExecution.ChartHealth.Delivery.Claim
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Delivery.Route [C:3] [TYPE Function]
|
||||
# @POST Each explicit route returns a bounded honest result; failed transports do not escape into terminalization.
|
||||
async def send_route(route, notifications, subject, body, timeout, provider_factory):
|
||||
identifier = sha256(f'{route.type}:{route.target}'.encode()).hexdigest()
|
||||
result = {'route_hash':identifier,'type':route.type,'status':'skipped','reason_code':'PROVIDER_UNCONFIGURED'}
|
||||
try:
|
||||
provider = provider_factory(route,notifications)
|
||||
if provider is None:
|
||||
return result
|
||||
sent = await asyncio.wait_for(provider.send(route.target,subject,body),timeout=max(0.001,timeout))
|
||||
result.update(status='delivered' if sent else 'failed',reason_code='DELIVERED' if sent else 'DELIVERY_FAILED')
|
||||
except asyncio.CancelledError:
|
||||
raise
|
||||
except Exception:
|
||||
result.update(status='failed',reason_code='DELIVERY_FAILED')
|
||||
return result
|
||||
# #endregion ScenarioExecution.ChartHealth.Delivery.Route
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Delivery.Execute [C:4] [TYPE Function]
|
||||
# @PRE Receipt is committed; only an explicit enabled policy permits sends.
|
||||
# @POST Receipt records delivered/failed/skipped outcome once; model verdict and run evidence are unchanged.
|
||||
async def deliver_receipt(db, receipt_id, *, provider_factory=configured_provider):
|
||||
receipt = db.get(ScenarioNotificationEvent,receipt_id)
|
||||
if receipt is None or not (receipt.payload or {}).get('chart_query_errors'):
|
||||
return 'unavailable'
|
||||
if not claim_receipt(db,receipt):
|
||||
return 'already_claimed'
|
||||
try:
|
||||
policy,notifications = read_alert_config(db)
|
||||
if not policy.enabled or not policy.channels:
|
||||
outcome = {'status':'skipped','reason_code':'DELIVERY_DISABLED' if not policy.enabled else 'DESTINATION_UNCONFIGURED','routes':[]}
|
||||
else:
|
||||
end,results = monotonic()+policy.timeout_seconds,[]
|
||||
body = message_body(receipt)
|
||||
for route in policy.channels:
|
||||
results.append(await send_route(route,notifications,'Dashboard chart query errors',body,end-monotonic(),provider_factory))
|
||||
outcome = {'status':'delivered' if all(item['status'] == 'delivered' for item in results) else 'failed',
|
||||
'reason_code':'DELIVERY_COMPLETE' if all(item['status'] == 'delivered' for item in results) else 'DELIVERY_INCOMPLETE','routes':results}
|
||||
except asyncio.CancelledError:
|
||||
save_delivery(db,receipt,{'status':'failed','reason_code':'DELIVERY_INTERRUPTED','routes':[]})
|
||||
raise
|
||||
except Exception:
|
||||
outcome = {'status':'skipped','reason_code':'DELIVERY_CONFIGURATION_INVALID','routes':[]}
|
||||
save_delivery(db,receipt,outcome)
|
||||
return outcome['status']
|
||||
# #endregion ScenarioExecution.ChartHealth.Delivery.Execute
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Delivery.Save [C:2] [TYPE Function]
|
||||
# @POST Delivery outcome replaces only its own claimed receipt projection and commits independently.
|
||||
def save_delivery(db, receipt, outcome):
|
||||
receipt.payload = {**receipt.payload,'health_delivery':outcome}
|
||||
db.commit()
|
||||
# #endregion ScenarioExecution.ChartHealth.Delivery.Save
|
||||
# #endregion ScenarioExecution.ChartHealth.Delivery
|
||||
@@ -0,0 +1,46 @@
|
||||
# #region ScenarioExecution.ChartHealth.DeliveryConfig [C:3] [TYPE Module] [SEMANTICS notify,opt-in,configuration,budget]
|
||||
# @BRIEF Read closed opt-in routing from existing global notification settings; no credentials enter receipts.
|
||||
from typing import Literal
|
||||
from pydantic import BaseModel, ConfigDict, Field
|
||||
from src.models.config import AppConfigRecord
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.DeliveryConfig.Route [C:2] [TYPE Class]
|
||||
# @INVARIANT A destination is explicit and bounded; only existing provider types are supported.
|
||||
class AlertRoute(BaseModel):
|
||||
model_config = ConfigDict(extra='forbid',strict=True)
|
||||
type: Literal['SMTP','TELEGRAM','SLACK']
|
||||
target: str = Field(min_length=1,max_length=2048)
|
||||
# #endregion ScenarioExecution.ChartHealth.DeliveryConfig.Route
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.DeliveryConfig.Policy [C:2] [TYPE Class]
|
||||
# @POST Missing configuration disables delivery without changing durable error receipts.
|
||||
class ScenarioAlertPolicy(BaseModel):
|
||||
model_config = ConfigDict(extra='forbid',strict=True)
|
||||
enabled: bool = False
|
||||
channels: list[AlertRoute] = Field(default_factory=list,max_length=20)
|
||||
timeout_seconds: int = Field(default=30,ge=1,le=60)
|
||||
# #endregion ScenarioExecution.ChartHealth.DeliveryConfig.Policy
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.DeliveryConfig.Read [C:2] [TYPE Function]
|
||||
# @POST The committed global record supplies routes; unknown routing fields fail closed.
|
||||
def read_alert_config(db):
|
||||
record = db.get(AppConfigRecord,'global')
|
||||
notifications = ((record.payload or {}).get('notifications') or {}) if record else {}
|
||||
return ScenarioAlertPolicy.model_validate(notifications.get('scenario_alerts') or {}),notifications
|
||||
# #endregion ScenarioExecution.ChartHealth.DeliveryConfig.Read
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.DeliveryConfig.Provider [C:3] [TYPE Function]
|
||||
# @POST Missing provider connection data returns None; existing provider implementations own transport behavior.
|
||||
def configured_provider(route, notifications):
|
||||
from src.services.notifications.providers import SMTPProvider,TelegramProvider,SlackProvider
|
||||
config = notifications.get(route.type.lower()) or {}
|
||||
required = {'SMTP':['host','from_email'],'TELEGRAM':['bot_token'],'SLACK':['webhook_url']}[route.type]
|
||||
if not all(config.get(key) for key in required):
|
||||
return None
|
||||
return {'SMTP':SMTPProvider,'TELEGRAM':TelegramProvider,'SLACK':SlackProvider}[route.type](config)
|
||||
# #endregion ScenarioExecution.ChartHealth.DeliveryConfig.Provider
|
||||
# #endregion ScenarioExecution.ChartHealth.DeliveryConfig
|
||||
@@ -0,0 +1,99 @@
|
||||
# #region ScenarioExecution.ChartHealth.DeliveryHook [C:4] [TYPE Module] [SEMANTICS after-commit,notification,capacity,rollback]
|
||||
# @BRIEF Schedule opt-in notification transport only after the owning terminal transaction commits.
|
||||
# @INVARIANT One bounded background submission uses the existing provider loop; rollback cannot send.
|
||||
from threading import Semaphore, Thread
|
||||
from sqlalchemy import event
|
||||
from src.core.database import SessionLocal
|
||||
from .chart_health_delivery_config import read_alert_config
|
||||
|
||||
_slot = Semaphore(1)
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.DeliveryHook.Arm [C:3] [TYPE Function]
|
||||
# @POST Named-error receipts get honest disabled/unconfigured/pending status; no network occurs inside terminalization.
|
||||
def arm_health_delivery(db, receipt):
|
||||
if not (receipt.payload or {}).get('chart_query_errors') or receipt.payload.get('health_delivery'):
|
||||
return
|
||||
try:
|
||||
policy,_ = read_alert_config(db)
|
||||
state = {'status':'pending','reason_code':'DELIVERY_PENDING'} if policy.enabled and policy.channels else {
|
||||
'status':'skipped','reason_code':'DELIVERY_DISABLED' if not policy.enabled else 'DESTINATION_UNCONFIGURED'}
|
||||
except Exception:
|
||||
state = {'status':'skipped','reason_code':'DELIVERY_CONFIGURATION_INVALID'}
|
||||
receipt.payload = {**receipt.payload,'health_delivery':state}
|
||||
if state['status'] != 'pending':
|
||||
return
|
||||
db.info.setdefault('chart_health_deliveries',set()).add(receipt.id)
|
||||
if not db.info.get('chart_health_delivery_hooks'):
|
||||
event.listen(db,'after_commit',after_health_commit)
|
||||
event.listen(db,'after_rollback',after_health_rollback)
|
||||
db.info['chart_health_delivery_hooks'] = True
|
||||
# #endregion ScenarioExecution.ChartHealth.DeliveryHook.Arm
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.DeliveryHook.Commit [C:3] [TYPE Function]
|
||||
# @POST At most one bounded physical dispatcher exists; terminal commit cannot fail because of transport startup.
|
||||
def after_health_commit(db):
|
||||
identifiers = sorted(db.info.pop('chart_health_deliveries',set()))[:25]
|
||||
if not identifiers:
|
||||
return
|
||||
try:
|
||||
if not _slot.acquire(blocking=False):
|
||||
mark_unavailable(identifiers,'DELIVERY_CAPACITY_UNAVAILABLE')
|
||||
return
|
||||
Thread(target=dispatch_committed,args=(identifiers,),daemon=True,name='chart-health-notifications').start()
|
||||
except Exception:
|
||||
_slot.release()
|
||||
mark_unavailable(identifiers,'DELIVERY_RUNTIME_UNAVAILABLE')
|
||||
# #endregion ScenarioExecution.ChartHealth.DeliveryHook.Commit
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.DeliveryHook.Rollback [C:1] [TYPE Function]
|
||||
# @POST Rolled-back receipt IDs are discarded before any physical send can be scheduled.
|
||||
def after_health_rollback(db):
|
||||
db.info.pop('chart_health_deliveries',None)
|
||||
# #endregion ScenarioExecution.ChartHealth.DeliveryHook.Rollback
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.DeliveryHook.Unavailable [C:3] [TYPE Function]
|
||||
# @POST Only unclaimed pending deliveries become explicitly skipped; already sent/ambiguous receipts are untouched.
|
||||
def mark_unavailable(identifiers, reason):
|
||||
from sqlalchemy import update
|
||||
from src.models.scenario_automation import ScenarioNotificationEvent
|
||||
try:
|
||||
with SessionLocal() as session:
|
||||
for identifier in identifiers:
|
||||
row = session.get(ScenarioNotificationEvent,identifier)
|
||||
if row is not None:
|
||||
payload = {**row.payload,'health_delivery':{'status':'skipped','reason_code':reason}}
|
||||
session.execute(update(ScenarioNotificationEvent).where(
|
||||
ScenarioNotificationEvent.id == identifier,
|
||||
ScenarioNotificationEvent.payload['health_delivery']['status'].as_string() == 'pending',
|
||||
).values(payload=payload).execution_options(synchronize_session=False))
|
||||
session.commit()
|
||||
except Exception:
|
||||
pass
|
||||
# #endregion ScenarioExecution.ChartHealth.DeliveryHook.Unavailable
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.DeliveryHook.Dispatch [C:3] [TYPE Function]
|
||||
# @POST Existing provider-loop admission bounds each committed receipt; startup failure records skipped rather than fake sent.
|
||||
def dispatch_committed(identifiers):
|
||||
from .provider_runtime import get_provider_event_loop
|
||||
try:
|
||||
runtime = get_provider_event_loop()
|
||||
for identifier in identifiers:
|
||||
# #region ScenarioExecution.ChartHealth.DeliveryHook.Dispatch.Send [C:1] [TYPE Function]
|
||||
# @BRIEF Keep DB lifetime on the same provider loop as its existing notification clients.
|
||||
async def send(receipt_id=identifier):
|
||||
from .chart_health_delivery import deliver_receipt
|
||||
with SessionLocal() as session:
|
||||
return await deliver_receipt(session,receipt_id)
|
||||
# #endregion ScenarioExecution.ChartHealth.DeliveryHook.Dispatch.Send
|
||||
runtime.submit(send,timeout=65)
|
||||
except Exception:
|
||||
mark_unavailable(identifiers,'DELIVERY_RUNTIME_UNAVAILABLE')
|
||||
finally:
|
||||
_slot.release()
|
||||
# #endregion ScenarioExecution.ChartHealth.DeliveryHook.Dispatch
|
||||
# #endregion ScenarioExecution.ChartHealth.DeliveryHook
|
||||
@@ -0,0 +1,31 @@
|
||||
# #region ScenarioExecution.ChartHealth.Notification [C:3] [TYPE Module] [SEMANTICS notify,health,summary,deduplication]
|
||||
# @BRIEF Add bounded grouped chart causes to the existing idempotent terminal automation receipt.
|
||||
from src.models.scenario_run import ScenarioStepRun
|
||||
from ..query_failure import safe_message
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Notification.Summary [C:3] [TYPE Function]
|
||||
# @POST One terminal receipt names failed placements and causes; truncation is explicit and no delivery channel is invented.
|
||||
def notification_health_summary(db, run_id):
|
||||
errors, omitted = [],0
|
||||
seen = set()
|
||||
for step in db.query(ScenarioStepRun).filter_by(run_id=run_id).order_by(ScenarioStepRun.logical_step_id,ScenarioStepRun.attempt.desc()).all():
|
||||
if step.logical_step_id in seen:
|
||||
continue
|
||||
seen.add(step.logical_step_id)
|
||||
outer = step.step_outcome or {}
|
||||
outcome = outer.get('step_outcome',outer)
|
||||
health = outcome.get('chart_health') or {}
|
||||
for item in health.get('errors',outcome.get('chart_diagnostics',[])):
|
||||
if item.get('status') != 'failed':
|
||||
continue
|
||||
if len(errors) >= 100:
|
||||
omitted += 1
|
||||
continue
|
||||
message, truncated = safe_message(item.get('message',''))
|
||||
errors.append({'logical_step_id':step.logical_step_id,'placement_id':item['placement_id'],
|
||||
'chart_name':item['chart_name'],'tab_path':item['tab_path'],'database_code':item.get('database_code'),
|
||||
'message':message[:512],'truncated':truncated or len(message) > 512})
|
||||
return {'chart_query_errors':{'count':len(errors)+omitted,'items':errors,'omitted':omitted}} if errors or omitted else {}
|
||||
# #endregion ScenarioExecution.ChartHealth.Notification.Summary
|
||||
# #endregion ScenarioExecution.ChartHealth.Notification
|
||||
@@ -0,0 +1,105 @@
|
||||
# #region ScenarioExecution.ChartHealth.Store [C:5] [TYPE Module] [SEMANTICS health,ownership,coverage,failed,evidence]
|
||||
# @BRIEF Commit each chart observation independently and preserve confirmed failures alongside incomplete coverage.
|
||||
# @INVARIANT Failed observations are durable before the next chart/tab; completeness never follows from a model verdict.
|
||||
from copy import deepcopy
|
||||
from hashlib import sha256
|
||||
import json
|
||||
from src.core.database import SessionLocal
|
||||
from src.models.scenario_traversal import ScenarioTraversal, ScenarioTraversalPage
|
||||
from src.models.scenario_artifact import ScenarioArtifact
|
||||
from ..chart_health_contract import ChartDiagnostic
|
||||
from .traversal_store import TraversalJournal, canonical
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Store.Journal [C:5] [TYPE Class]
|
||||
class ChartHealthJournal(TraversalJournal):
|
||||
# #region ScenarioExecution.ChartHealth.Store.Journal.Freeze [C:3] [TYPE Function]
|
||||
# @POST Exact server placements are frozen before observation; changed layouts refuse resume.
|
||||
def freeze_source(self, source):
|
||||
with SessionLocal() as db:
|
||||
row = db.query(ScenarioTraversal).filter_by(id=self.id).with_for_update().one()
|
||||
if row.state['source'] is not None and row.state['source'] != source:
|
||||
raise ValueError('CHART_HEALTH_MANIFEST_CHANGED')
|
||||
row.state = {**row.state,'source':deepcopy(source)}
|
||||
db.commit()
|
||||
# #endregion ScenarioExecution.ChartHealth.Store.Journal.Freeze
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Store.Journal.Append [C:4] [TYPE Function]
|
||||
# @POST One exact server placement, typed diagnostic, artifact and frontier advance commit atomically.
|
||||
def append_chart(self, observation):
|
||||
reason = self.check_control()
|
||||
if reason:
|
||||
raise ValueError(reason)
|
||||
value = ChartDiagnostic.model_validate(observation).model_dump()
|
||||
target = self.step.get('target_snapshot') or {}
|
||||
value.update(run_id=self.run_id,logical_step_id=self.logical_step_id,attempt=self.attempt,
|
||||
environment_id=target.get('environment_id'),dashboard_id=target.get('dashboard_id'))
|
||||
with SessionLocal() as db:
|
||||
row = db.query(ScenarioTraversal).filter_by(id=self.id).with_for_update().one()
|
||||
state = row.state
|
||||
ordinal = state['next_ordinal']
|
||||
expected = state['source']['placements'][ordinal-1]
|
||||
if (value['placement_id'],value['chart_id'],value['chart_name'],value['tab_path']) != (expected['id'],expected['chart_id'],expected['name'],expected['tab_path']):
|
||||
raise ValueError('CHART_HEALTH_PLACEMENT_MISMATCH')
|
||||
data = canonical({'chart_diagnostic':value})
|
||||
if state['byte_count']+state.get('auxiliary_bytes',0)+len(data) > self.limits.max_bytes:
|
||||
raise ValueError('CHART_HEALTH_BYTES_EXCEEDED')
|
||||
artifact = self._artifact(db,data,f'chart-health-{ordinal}.json')
|
||||
receipt = {'ordinal':ordinal,'artifact_id':artifact.id,'content_ref':artifact.content_ref,
|
||||
'sha256':artifact.sha256,'byte_length':len(data),'placement_id':value['placement_id'],
|
||||
'status':value['status'],'row_count':1,'next_available':ordinal < len(state['source']['placements'])}
|
||||
chain = sha256(state['chain_digest'].encode()+canonical(receipt)).hexdigest()
|
||||
receipt['chain_digest'] = chain
|
||||
db.add(ScenarioTraversalPage(traversal_id=self.id,ordinal=ordinal,receipt=receipt))
|
||||
row.state = {**state,'next_ordinal':ordinal+1,'row_count':ordinal,'byte_count':state['byte_count']+len(data),
|
||||
'chain_digest':chain,'terminal':not receipt['next_available']}
|
||||
db.commit()
|
||||
# #endregion ScenarioExecution.ChartHealth.Store.Journal.Append
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Store.Journal.Verify [C:4] [TYPE Function]
|
||||
# @POST Ownership/hash/type/identity/ordered-chain violations reject evidence and resume.
|
||||
def verify_receipts(self):
|
||||
with SessionLocal() as db:
|
||||
row = db.get(ScenarioTraversal,self.id)
|
||||
chain, count, size = '',0,0
|
||||
for item in db.query(ScenarioTraversalPage).filter_by(traversal_id=self.id).order_by(ScenarioTraversalPage.ordinal).yield_per(1):
|
||||
receipt = item.receipt
|
||||
value, artifact = self.load_owned(db,receipt)
|
||||
count += 1
|
||||
expected = row.state['source']['placements'][count-1]
|
||||
if (item.ordinal != count or value.placement_id != expected['id'] or value.chart_id != expected['chart_id']
|
||||
or value.tab_path != expected['tab_path'] or value.status != receipt['status']):
|
||||
raise ValueError('CHART_HEALTH_RECEIPT_INVALID')
|
||||
chain = sha256(chain.encode()+canonical({key:v for key,v in receipt.items() if key != 'chain_digest'})).hexdigest()
|
||||
if chain != receipt['chain_digest']:
|
||||
raise ValueError('CHART_HEALTH_CHAIN_INVALID')
|
||||
size += artifact.byte_length
|
||||
from .chart_health_tab_store import verify_tab_evaluations
|
||||
verify_tab_evaluations(self,db,row.state)
|
||||
if row.state['next_ordinal'] != count+1 or row.state['row_count'] != count or row.state['byte_count'] != size or row.state['chain_digest'] != chain:
|
||||
raise ValueError('CHART_HEALTH_FRONTIER_INVALID')
|
||||
# #endregion ScenarioExecution.ChartHealth.Store.Journal.Verify
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Store.Journal.Load [C:4] [TYPE Function]
|
||||
# @POST Only exact active same-run/step/attempt bytes can enter summary or model evidence.
|
||||
def load_owned(self, db, receipt):
|
||||
artifact = db.get(ScenarioArtifact,receipt['artifact_id'])
|
||||
data = self.storage.retrieve(receipt['content_ref'])
|
||||
if (artifact is None or not artifact.is_active or artifact.owner_type != 'scenario_run' or artifact.owner_id != self.run_id
|
||||
or artifact.logical_step_id != self.logical_step_id or artifact.attempt != self.attempt
|
||||
or artifact.content_type != 'application/json' or artifact.content_ref != receipt['content_ref']
|
||||
or artifact.sha256 != receipt['sha256'] or artifact.byte_length != receipt['byte_length']
|
||||
or data is None or len(data) != receipt['byte_length'] or sha256(data).hexdigest() != receipt['sha256']):
|
||||
raise ValueError('CHART_HEALTH_ARTIFACT_INVALID')
|
||||
return ChartDiagnostic.model_validate(json.loads(data)['chart_diagnostic']), artifact
|
||||
# #endregion ScenarioExecution.ChartHealth.Store.Journal.Load
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Store.Journal.Finish [C:4] [TYPE Function]
|
||||
# @POST Confirmed failure dominates partial coverage; PASS requires all expected placements conclusively healthy.
|
||||
def finish(self, status, reason):
|
||||
self.verify_receipts()
|
||||
from .chart_health_summary import finish_health
|
||||
return finish_health(self,status,reason)
|
||||
# #endregion ScenarioExecution.ChartHealth.Store.Journal.Finish
|
||||
# #endregion ScenarioExecution.ChartHealth.Store.Journal
|
||||
# #endregion ScenarioExecution.ChartHealth.Store
|
||||
@@ -0,0 +1,60 @@
|
||||
# #region ScenarioExecution.ChartHealth.Summary [C:4] [TYPE Module] [SEMANTICS health,coverage,verdict,manifest]
|
||||
# @BRIEF Aggregate durable diagnostics without conflating confirmed failure with observation completeness.
|
||||
from src.core.database import SessionLocal
|
||||
from src.models.scenario_traversal import ScenarioTraversal, ScenarioTraversalPage
|
||||
from .traversal_store import canonical
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Summary.Tabs [C:3] [TYPE Function]
|
||||
# @POST Stable tab IDs group names/causes while preserving each placement and its coverage.
|
||||
def tab_groups(source, diagnostics):
|
||||
result = []
|
||||
for tab in (source or {}).get('tabs',[]):
|
||||
expected = [item for item in source['placements'] if item['tab_path'] and item['tab_path'][-1] == tab['id']]
|
||||
observed = [item for item in diagnostics if item['tab_path'] and item['tab_path'][-1] == tab['id']]
|
||||
errors = [item for item in observed if item['status'] == 'failed']
|
||||
result.append({'tab_id':tab['id'],'name':tab.get('name',tab['id']),'expected':len(expected),'observed':len(observed),
|
||||
'errored':len(errors),'unvisited':len(expected)-len(observed),'status':'failed' if errors else
|
||||
('passed' if len(observed) == len(expected) and all(item['status'] == 'healthy' for item in observed) else 'inconclusive')})
|
||||
return result
|
||||
# #endregion ScenarioExecution.ChartHealth.Summary.Tabs
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Summary.Finish [C:4] [TYPE Function]
|
||||
# @POST Every retained placement remains addressable; a confirmed error can never become PASS or merely INCONCLUSIVE.
|
||||
def finish_health(journal, requested_status, reason):
|
||||
with SessionLocal() as db:
|
||||
row = db.query(ScenarioTraversal).filter_by(id=journal.id).with_for_update().one()
|
||||
source = row.state['source']
|
||||
receipts = db.query(ScenarioTraversalPage).filter_by(traversal_id=journal.id).order_by(ScenarioTraversalPage.ordinal).all()
|
||||
diagnostics = [journal.load_owned(db,item.receipt)[0].model_dump() for item in receipts]
|
||||
errors = sum(item['status'] == 'failed' for item in diagnostics)
|
||||
unresolved = sum(item['status'] == 'inconclusive' for item in diagnostics)
|
||||
expected = len(source['placements']) if source else 0
|
||||
complete = bool(requested_status == 'passed' and source is not None and len(diagnostics) == expected and unresolved == 0)
|
||||
evaluations = row.state.get('tab_evaluations',[])
|
||||
vlm_complete = all(item['status'] == 'succeeded' and item['coverage_complete'] for item in evaluations)
|
||||
status = 'failed' if errors else ('passed' if complete and vlm_complete else 'inconclusive')
|
||||
coverage = {'expected':expected,'observed':len(diagnostics),'checked':len(diagnostics)-unresolved,
|
||||
'errored':errors,'unresolved':unresolved,'timeout':sum(item['reason_code'] == 'CHART_LOAD_TIMEOUT' for item in diagnostics),
|
||||
'unvisited':expected-len(diagnostics),'complete':complete}
|
||||
manifest = {'schema_version':1,'kind':'chart_health','run_id':journal.run_id,'logical_step_id':journal.logical_step_id,
|
||||
'attempt':journal.attempt,'status':status,'reason_code':'CHART_QUERY_FAILED' if errors else reason,
|
||||
'interruption_reason':reason if not complete else None,'complete':complete,'coverage':coverage,
|
||||
'tab_evaluations':evaluations,'visual_coverage_complete':vlm_complete,
|
||||
'source':source,'chain_digest':row.state['chain_digest'],
|
||||
'charts':[dict(item,artifact_id=receipt.receipt['artifact_id'],content_ref=receipt.receipt['content_ref'],
|
||||
sha256=receipt.receipt['sha256']) for item,receipt in zip(diagnostics,receipts)]}
|
||||
data = canonical(manifest)
|
||||
if len(data) > 262144:
|
||||
raise ValueError('CHART_HEALTH_MANIFEST_TOO_LARGE')
|
||||
artifact = journal._artifact(db,data,'chart-health-manifest.json')
|
||||
row.status = status
|
||||
db.commit()
|
||||
return {'status':status,'reason_code':manifest['reason_code'],'complete':complete,'coverage':coverage,
|
||||
'chart_health':{'coverage':coverage,'charts':manifest['charts'],'errors':[item for item in manifest['charts'] if item['status'] == 'failed'],
|
||||
'tabs':tab_groups(source,diagnostics),'tab_evaluations':evaluations,'visual_coverage_complete':vlm_complete},
|
||||
'manifest_artifact_id':artifact.id,'manifest_ref':artifact.content_ref,
|
||||
'manifest_sha256':artifact.sha256,'manifest_byte_length':artifact.byte_length}
|
||||
# #endregion ScenarioExecution.ChartHealth.Summary.Finish
|
||||
# #endregion ScenarioExecution.ChartHealth.Summary
|
||||
@@ -0,0 +1,64 @@
|
||||
# #region ScenarioExecution.ChartHealth.TabStore [C:4] [TYPE Module] [SEMANTICS tabs,evaluation,owned,commit]
|
||||
# @BRIEF Retain captures/evaluation results as owned artifacts before leaving the active tab.
|
||||
from src.core.database import SessionLocal
|
||||
from src.models.scenario_traversal import ScenarioTraversal, ScenarioTraversalPage
|
||||
from .chart_health_artifact import verify_health_artifact
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.TabStore.Store [C:3] [TYPE Function]
|
||||
# @POST MIME/length and owner metadata bind every frame/context/evaluation to the active attempt.
|
||||
def store_health_artifact(journal, data, name, content_type='application/json'):
|
||||
reason = journal.check_control()
|
||||
if reason:
|
||||
raise ValueError(reason)
|
||||
with SessionLocal() as db:
|
||||
row = db.query(ScenarioTraversal).filter_by(id=journal.id).with_for_update().one()
|
||||
size = row.state.get('auxiliary_bytes',0)+len(data)
|
||||
if size+row.state['byte_count'] > journal.limits.max_bytes:
|
||||
raise ValueError('CHART_HEALTH_BYTES_EXCEEDED')
|
||||
artifact = journal._artifact(db,data,name,content_type)
|
||||
row.state = {**row.state,'auxiliary_bytes':size}
|
||||
db.commit()
|
||||
return {'artifact_id':artifact.id,'content_ref':artifact.content_ref,'sha256':artifact.sha256,
|
||||
'byte_length':artifact.byte_length,'content_type':artifact.content_type}
|
||||
# #endregion ScenarioExecution.ChartHealth.TabStore.Store
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.TabStore.Diagnostics [C:3] [TYPE Function]
|
||||
# @POST Current-tab payloads and manifest refs come solely from committed current-attempt diagnostic artifacts.
|
||||
def tab_diagnostics(journal, tab_path):
|
||||
with SessionLocal() as db:
|
||||
result = []
|
||||
for row in db.query(ScenarioTraversalPage).filter_by(traversal_id=journal.id).order_by(ScenarioTraversalPage.ordinal).all():
|
||||
diagnostic, artifact = journal.load_owned(db,row.receipt)
|
||||
if diagnostic.tab_path == tab_path:
|
||||
result.append({'diagnostic':diagnostic.model_dump(),'receipt':{
|
||||
'artifact_id':artifact.id,'content_ref':artifact.content_ref,'sha256':artifact.sha256,
|
||||
'byte_length':artifact.byte_length,'content_type':artifact.content_type}})
|
||||
return result
|
||||
# #endregion ScenarioExecution.ChartHealth.TabStore.Diagnostics
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.TabStore.Append [C:3] [TYPE Function]
|
||||
# @POST A persisted tab evaluation summary cannot substitute model PASS for deterministic failure.
|
||||
def append_tab_evaluation(journal, summary):
|
||||
with SessionLocal() as db:
|
||||
row = db.query(ScenarioTraversal).filter_by(id=journal.id).with_for_update().one()
|
||||
for receipt in summary['artifacts']:
|
||||
verify_health_artifact(journal,db,receipt)
|
||||
results = row.state.get('tab_evaluations',[])
|
||||
if any(item['tab_path'] == summary['tab_path'] for item in results):
|
||||
raise ValueError('CHART_HEALTH_TAB_EVALUATION_DUPLICATE')
|
||||
row.state = {**row.state,'tab_evaluations':[*results,summary]}
|
||||
db.commit()
|
||||
# #endregion ScenarioExecution.ChartHealth.TabStore.Append
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.TabStore.Verify [C:3] [TYPE Function]
|
||||
# @POST Resume verifies every captured frame/evaluation receipt; model fields remain explanation-only.
|
||||
def verify_tab_evaluations(journal, db, state):
|
||||
for summary in state.get('tab_evaluations',[]):
|
||||
for receipt in summary['artifacts']:
|
||||
verify_health_artifact(journal,db,receipt)
|
||||
# #endregion ScenarioExecution.ChartHealth.TabStore.Verify
|
||||
# #endregion ScenarioExecution.ChartHealth.TabStore
|
||||
@@ -26,6 +26,7 @@ from .evaluation_manifest import _manifest_from_completed as _manifest_from_comp
|
||||
from .evaluation_images import image_payloads_from_manifest
|
||||
from .evaluation_prompt import build_evaluation_prompt
|
||||
from .evaluation_recipe_context import prepare_recipe_text_evidence
|
||||
from .evaluation_health_context import prepare_health_context
|
||||
from .live_binding import EvidenceStorage
|
||||
|
||||
|
||||
@@ -295,6 +296,8 @@ def evaluation_adapter_from(
|
||||
submit: Callable[..., dict[str, Any]] | None = None,
|
||||
db_factory: Callable[[], Any] | None = None,
|
||||
) -> Callable[[dict[str, Any], dict[str, dict[str, Any]]], dict[str, Any]]:
|
||||
# #region ScenarioExecution.EvaluationAdapter.Own.invoke [C:3] [TYPE Function]
|
||||
# @BRIEF Evaluate the supplied step using owned completed evidence and runtime dependencies.
|
||||
def invoke(step: dict[str, Any], completed: dict[str, dict[str, Any]]) -> dict[str, Any]:
|
||||
spec = _spec_from_step(step)
|
||||
manifest = _manifest_from_completed(completed)
|
||||
@@ -309,6 +312,8 @@ def evaluation_adapter_from(
|
||||
evidence_payloads = None
|
||||
if isinstance(spec.limits, TokenEvaluationLimits):
|
||||
manifest, evidence_payloads = prepare_recipe_text_evidence(step, spec, completed, evidence, db_factory=db_factory)
|
||||
else:
|
||||
evidence_payloads = prepare_health_context(step,completed,manifest,evidence,db_factory=db_factory)
|
||||
images = image_payloads_from_manifest(manifest, evidence, max_images=spec.limits.max_images)
|
||||
if images:
|
||||
# Candidate evidence loaded; actual attachment is gated on provider multimodality
|
||||
@@ -359,6 +364,7 @@ def evaluation_adapter_from(
|
||||
"content_type": "application/json",
|
||||
"byte_length": len(raw_bytes),
|
||||
}
|
||||
# #endregion ScenarioExecution.EvaluationAdapter.Own.invoke
|
||||
|
||||
return invoke
|
||||
# #endregion ScenarioExecution.EvaluationAdapter.Factory
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
# #region ScenarioExecution.EvaluationCapacityIdentity [C:3] [TYPE Module] [SEMANTICS provider,pin,capacity,width]
|
||||
# @BRIEF Bind a full authored config pin to the existing64char capacity lease without changing evaluation provenance.
|
||||
import re
|
||||
|
||||
|
||||
# #region ScenarioExecution.EvaluationCapacityIdentity.Resolve [C:3] [TYPE Function]
|
||||
# @POST Explicit config_sha256 pins use their verified digest for capacity; arbitrary long identifiers fail closed.
|
||||
def capacity_provider_version(value):
|
||||
if re.fullmatch(r'config_sha256:[a-f0-9]{64}',value):
|
||||
return value.split(':',1)[1]
|
||||
if len(value) > 64:
|
||||
raise ValueError('EVALUATION_CAPACITY_IDENTITY_INVALID')
|
||||
return value
|
||||
# #endregion ScenarioExecution.EvaluationCapacityIdentity.Resolve
|
||||
# #endregion ScenarioExecution.EvaluationCapacityIdentity
|
||||
@@ -0,0 +1,24 @@
|
||||
# #region ScenarioExecution.ChartHealth.EvaluationContext [C:4] [TYPE Module] [SEMANTICS evaluation,diagnostic,ownership,completed]
|
||||
# @BRIEF Enrich ordinary VLM evaluation with actual owned error contents when prior runtime outcomes declare chart diagnostics.
|
||||
from src.core.database import SessionLocal
|
||||
from .evaluation_health_payloads import load_health_payloads
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.EvaluationContext.Prepare [C:3] [TYPE Function]
|
||||
# @POST Ordinary evidence behavior is unchanged; diagnostic callers require durable same-run proof before contents reach the model.
|
||||
def prepare_health_context(step, completed, manifest, storage, *, db_factory=None):
|
||||
declared = False
|
||||
for outer in completed.values():
|
||||
outcome = outer.get('step_outcome',outer) if isinstance(outer,dict) else {}
|
||||
if isinstance(outcome,dict) and (outcome.get('chart_diagnostics') or outcome.get('chart_health')):
|
||||
declared = True
|
||||
if not declared:
|
||||
return None
|
||||
db = (db_factory or SessionLocal)()
|
||||
try:
|
||||
return load_health_payloads(manifest,storage,db=db,run_id=step['scenario_run_id'])
|
||||
finally:
|
||||
if db_factory is None:
|
||||
db.close()
|
||||
# #endregion ScenarioExecution.ChartHealth.EvaluationContext.Prepare
|
||||
# #endregion ScenarioExecution.ChartHealth.EvaluationContext
|
||||
@@ -0,0 +1,47 @@
|
||||
# #region ScenarioExecution.ChartHealth.EvaluationPayloads [C:5] [TYPE Module] [SEMANTICS evaluation,owned,diagnostic,redaction,budget]
|
||||
# @BRIEF Send actual owned diagnostic contents to the model, including failed observations, within explicit text limits.
|
||||
# @INVARIANT Diagnostic JSON is evidence data, not instructions or a fabricated deterministic comparison.
|
||||
from hashlib import sha256
|
||||
import json
|
||||
from src.models.scenario_artifact import ScenarioArtifact
|
||||
from ..chart_health_contract import ChartDiagnostic
|
||||
from .evaluation_text_json import redact_json_content
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.EvaluationPayloads.Load [C:4] [TYPE Function]
|
||||
# @POST Foreign/inactive/changed bytes refuse; omissions/truncation and total bytes are explicit.
|
||||
def load_health_payloads(manifest, storage, *, db, run_id, max_bytes=262144, max_items=8):
|
||||
payloads, omitted, total = [],[],0
|
||||
for item in manifest:
|
||||
if item['content_type'] != 'application/json':
|
||||
continue
|
||||
if len(payloads) >= max_items or item['byte_length'] > max_bytes-total:
|
||||
omitted.append({'artifact_id':item['artifact_id'],'reason':'TEXT_BUDGET'})
|
||||
continue
|
||||
data = storage.retrieve(item['artifact_id'])
|
||||
if data is None or len(data) != item['byte_length'] or sha256(data).hexdigest() != item['sha256']:
|
||||
raise RuntimeError('CHART_HEALTH_EVIDENCE_INVALID')
|
||||
try:
|
||||
body = json.loads(data)
|
||||
except (ValueError, UnicodeError):
|
||||
continue
|
||||
if not isinstance(body,dict):
|
||||
continue
|
||||
diagnostic = body.get('chart_diagnostic')
|
||||
if diagnostic is None and body.get('kind') != 'chart_health':
|
||||
continue
|
||||
artifact = db.query(ScenarioArtifact).filter_by(owner_type='scenario_run',owner_id=run_id,
|
||||
content_ref=item['artifact_id'],sha256=item['sha256'],is_active=True).one_or_none()
|
||||
if artifact is None or artifact.byte_length != len(data) or artifact.content_type != 'application/json':
|
||||
raise RuntimeError('CHART_HEALTH_EVIDENCE_NOT_OWNED')
|
||||
if diagnostic is not None:
|
||||
value = ChartDiagnostic.model_validate(diagnostic).model_dump()
|
||||
else:
|
||||
value = {'coverage':body['coverage'],'charts':[ChartDiagnostic.model_validate({key:entry[key] for key in ChartDiagnostic.model_fields if key in entry}).model_dump() for entry in body['charts']]}
|
||||
payloads.append({'artifact_id':item['artifact_id'],'logical_step_id':artifact.logical_step_id,
|
||||
'attempt':artifact.attempt,'content':redact_json_content(value)})
|
||||
total += len(data)
|
||||
return {'schema_version':1,'kind':'chart_diagnostic_contents','items':payloads,
|
||||
'omitted':omitted,'byte_length':total,'truncated':bool(omitted)}
|
||||
# #endregion ScenarioExecution.ChartHealth.EvaluationPayloads.Load
|
||||
# #endregion ScenarioExecution.ChartHealth.EvaluationPayloads
|
||||
@@ -73,6 +73,8 @@ class LiveExecutionBinding:
|
||||
evidence_owner_type: str
|
||||
evidence_ref_policy: str
|
||||
|
||||
# #region ScenarioExecution.LiveBinding.Own.from_snapshot [C:3] [TYPE Function]
|
||||
# @BRIEF Validate and reconstruct the persisted binding snapshot.
|
||||
@classmethod
|
||||
def from_snapshot(cls, snapshot: object) -> LiveExecutionBinding:
|
||||
if not isinstance(snapshot, dict) or set(snapshot) != _SNAPSHOT_FIELDS:
|
||||
@@ -95,7 +97,10 @@ class LiveExecutionBinding:
|
||||
):
|
||||
raise ValueError("LIVE_BINDING_SNAPSHOT_INVALID")
|
||||
return binding
|
||||
# #endregion ScenarioExecution.LiveBinding.Own.from_snapshot
|
||||
|
||||
# #region ScenarioExecution.LiveBinding.Own.snapshot [C:3] [TYPE Function]
|
||||
# @BRIEF Serialize only persisted binding identity.
|
||||
def snapshot(self) -> dict[str, Any]:
|
||||
return {
|
||||
"binding_ref": self.binding_ref,
|
||||
@@ -111,9 +116,13 @@ class LiveExecutionBinding:
|
||||
"evidence_owner_type": self.evidence_owner_type,
|
||||
"evidence_ref_policy": self.evidence_ref_policy,
|
||||
}
|
||||
# #endregion ScenarioExecution.LiveBinding.Own.snapshot
|
||||
|
||||
# #region ScenarioExecution.LiveBinding.Own.with_query_model_fingerprint [C:1] [TYPE Function]
|
||||
# @BRIEF Return the binding with the supplied query model fingerprint.
|
||||
def with_query_model_fingerprint(self, fingerprint: str) -> LiveExecutionBinding:
|
||||
return LiveExecutionBinding(**{**self.snapshot(), "query_model_fingerprint": fingerprint})
|
||||
# #endregion ScenarioExecution.LiveBinding.Own.with_query_model_fingerprint
|
||||
# #endregion ScenarioExecution.LiveBinding.Identity
|
||||
|
||||
|
||||
@@ -122,9 +131,14 @@ class LiveExecutionBinding:
|
||||
# @DATA_CONTRACT LiveExecutionBindingResolver(binding_ref) -> ResolvedLiveExecutionBinding | None.
|
||||
# @INVARIANT Resolver-owned clients, storage, and event-loop access never enter the persisted snapshot.
|
||||
class EvidenceStorage(Protocol):
|
||||
# #region ScenarioExecution.LiveBinding.Own.store [C:1] [TYPE Function]
|
||||
# @BRIEF Define the evidence-byte persistence interface.
|
||||
def store(self, run_id: str, sha256: str, data: bytes) -> str: ...
|
||||
# #endregion ScenarioExecution.LiveBinding.Own.store
|
||||
|
||||
|
||||
# #region ScenarioExecution.LiveBinding.Own.ResolvedLiveExecutionBinding [C:3] [TYPE Class]
|
||||
# @BRIEF Keep resolved live clients, approved query model and evidence storage together.
|
||||
@dataclass(frozen=True)
|
||||
class ResolvedLiveExecutionBinding:
|
||||
binding: LiveExecutionBinding
|
||||
@@ -132,12 +146,19 @@ class ResolvedLiveExecutionBinding:
|
||||
query_model: DashboardQueryModel
|
||||
evidence_storage: EvidenceStorage
|
||||
run_async: Callable[[Awaitable[Any]], Any]
|
||||
# #endregion ScenarioExecution.LiveBinding.Own.ResolvedLiveExecutionBinding
|
||||
|
||||
|
||||
# #region ScenarioExecution.LiveBinding.Own.LiveExecutionBindingResolver [C:1] [TYPE Class]
|
||||
# @BRIEF Define the composition-owned binding-reference lookup protocol.
|
||||
class LiveExecutionBindingResolver(Protocol):
|
||||
"""Composition-owned mapping from persisted identity to authorized runtime dependencies."""
|
||||
|
||||
# #region ScenarioExecution.LiveBinding.Own.resolve [C:1] [TYPE Function]
|
||||
# @BRIEF Look up the exact binding reference.
|
||||
def resolve(self, binding_ref: str) -> ResolvedLiveExecutionBinding | None: ...
|
||||
# #endregion ScenarioExecution.LiveBinding.Own.resolve
|
||||
# #endregion ScenarioExecution.LiveBinding.Own.LiveExecutionBindingResolver
|
||||
# #endregion ScenarioExecution.LiveBinding.Runtime
|
||||
|
||||
|
||||
@@ -236,18 +257,30 @@ def _evidence_result(
|
||||
raw_bytes = envelope.raw_response_content
|
||||
if not is_valid_sha256(digest) or sha256(raw_bytes).hexdigest() != digest:
|
||||
return LiveAdapterResult(status="inconclusive", reason_code="SUPERSET_EVIDENCE_INVALID")
|
||||
diagnostic = getattr(envelope, "diagnostic", None)
|
||||
if diagnostic is not None:
|
||||
from src.services.dashboard_testing.chart_health_contract import ChartDiagnostic
|
||||
try:
|
||||
diagnostic = ChartDiagnostic.model_validate(diagnostic).model_dump()
|
||||
except ValueError:
|
||||
return LiveAdapterResult(status="inconclusive", reason_code="SUPERSET_DIAGNOSTIC_INVALID")
|
||||
run_id = step["scenario_run_id"]
|
||||
content_ref = resolved.evidence_storage.store(run_id, digest, raw_bytes)
|
||||
expected_ref = f"draft:{run_id}:{digest}"
|
||||
if content_ref != expected_ref:
|
||||
return LiveAdapterResult(status="inconclusive", reason_code="SUPERSET_EVIDENCE_REF_INVALID")
|
||||
failed = envelope.normalized_value.kind == ValueKind.UNKNOWN
|
||||
return LiveAdapterResult(
|
||||
status="passed",
|
||||
reason_code="SUPERSET_QUERY_EXECUTED",
|
||||
status=("failed" if (diagnostic or {}).get("status") == "failed" else "inconclusive") if failed else "passed",
|
||||
reason_code=(diagnostic or {}).get("reason_code", "SUPERSET_QUERY_FAILED") if failed else "SUPERSET_QUERY_EXECUTED",
|
||||
details={
|
||||
"actual": envelope.normalized_value.model_dump(mode="json"),
|
||||
"source_response_hash": digest,
|
||||
"sha256": digest,
|
||||
"payload_kind": getattr(envelope, "payload_kind", "original_response"),
|
||||
"artifact_content_types": {content_ref: "application/json"},
|
||||
"artifact_byte_lengths": {content_ref: len(raw_bytes)},
|
||||
**({"chart_diagnostics": [diagnostic]} if diagnostic else {}),
|
||||
},
|
||||
output_refs=[content_ref],
|
||||
artifact_refs=[content_ref],
|
||||
@@ -325,9 +358,6 @@ def _execute_bound_superset(
|
||||
except Exception:
|
||||
_finalize_superset_receipt(operation_id, "failed", "unknown", summary={"phase": "execution_error"})
|
||||
return LiveAdapterResult(status="inconclusive", reason_code="SUPERSET_QUERY_EXECUTION_ERROR")
|
||||
if envelope.normalized_value.kind == ValueKind.UNKNOWN:
|
||||
_finalize_superset_receipt(operation_id, "failed", "none", summary={"phase": "query_failed"})
|
||||
return LiveAdapterResult(status="failed", reason_code="SUPERSET_QUERY_FAILED")
|
||||
evidence = _evidence_result(step, resolved, envelope)
|
||||
if evidence.status == "passed":
|
||||
_finalize_superset_receipt(
|
||||
@@ -335,7 +365,7 @@ def _execute_bound_superset(
|
||||
summary={"sha256": evidence.details.get("sha256"), "artifact_refs": list(evidence.artifact_refs or [])},
|
||||
)
|
||||
else:
|
||||
_finalize_superset_receipt(operation_id, "failed", "none", summary={"phase": evidence.reason_code})
|
||||
_finalize_superset_receipt(operation_id, "failed", "none", summary={"phase": evidence.reason_code,"sha256":evidence.details.get("sha256"),"artifact_refs":list(evidence.artifact_refs or [])})
|
||||
return evidence
|
||||
# #endregion ScenarioExecution.LiveBinding.Execute
|
||||
|
||||
@@ -359,10 +389,11 @@ def superset_adapter_from(resolver: LiveExecutionBindingResolver):
|
||||
# @REJECTED Executing caller-provided SQL directly was rejected — 044 evidence must use the pinned
|
||||
# query model, principal and RLS fingerprints already authorized by the 037 binding.
|
||||
def sql_evidence_adapter_from(resolver: LiveExecutionBindingResolver):
|
||||
# #region ScenarioExecution.LiveBinding.Own.execute [C:1] [TYPE Function]
|
||||
# @BRIEF Invoke the composed live binding executor.
|
||||
def execute(step: dict[str, Any], completed: dict[str, dict[str, Any]]) -> LiveAdapterResult:
|
||||
return _execute_bound_superset(resolver, step, completed)
|
||||
|
||||
# #endregion ScenarioExecution.LiveBinding.Own.execute
|
||||
return execute
|
||||
# #endregion ScenarioExecution.LiveBinding.SqlEvidenceAdapter
|
||||
|
||||
# #endregion ScenarioExecution.LiveBinding
|
||||
|
||||
@@ -38,7 +38,7 @@ _READ_ONLY_ACTIONS = frozenset({
|
||||
"navigate_tab", "inspect_filter_state", "apply_table_filter", "extract_table",
|
||||
"scroll_to", "inspect_columns", "click", "select_rows", "download",
|
||||
# Wave-2 observe drivers (038.5.0, AGSCN-FR-024):
|
||||
"assert_dom", "inspect_filter_options", "navigate_tabs", "pagination", "wait_for_selector",
|
||||
"assert_dom", "assert_chart_health", "inspect_filter_options", "navigate_tabs", "pagination", "wait_for_selector",
|
||||
})
|
||||
_MUTATION_ACTIONS = frozenset({"row_edit", "bulk_edit"})
|
||||
# Round-2 read-only actions that carry typed inputs; validated at admission before any I/O.
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
# #region ScenarioExecution.ChartHealth.Async [C:3] [TYPE Module] [SEMANTICS async,job,polling,attribution]
|
||||
# @BRIEF Associate observed Superset5 async polling events only with already inspected chart-query jobs.
|
||||
# @RATIONALE Superset5 asyncEvent.ts exposes job_id/status/errors/result_url; none are inferred from URL text.
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Async.RecordJobs [C:3] [TYPE Function]
|
||||
# @POST Only a current chart-data202 response establishes job ownership; all recorded jobs remain bounded.
|
||||
def record_jobs(collector, payload, attribution):
|
||||
result = payload.get('result',[]) if isinstance(payload,dict) else []
|
||||
records = result if isinstance(result,list) else [result]
|
||||
for record in records:
|
||||
if isinstance(record,dict) and isinstance(record.get('job_id'),str):
|
||||
collector.jobs[record['job_id']] = attribution
|
||||
while len(collector.jobs) > 128:
|
||||
collector.jobs.pop(next(iter(collector.jobs)))
|
||||
# #endregion ScenarioExecution.ChartHealth.Async.RecordJobs
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Async.Events [C:3] [TYPE Function]
|
||||
# @POST Foreign/stale jobs cannot create errors or authorize unrelated response bodies.
|
||||
def current_events(collector, payload):
|
||||
result = payload.get('result',[]) if isinstance(payload,dict) else []
|
||||
if not isinstance(result,list):
|
||||
return []
|
||||
matches = []
|
||||
for event in result:
|
||||
if not isinstance(event,dict) or event.get('job_id') not in collector.jobs:
|
||||
continue
|
||||
attribution = collector.jobs[event['job_id']]
|
||||
if collector.latest.get(attribution[0]) != attribution[1]:
|
||||
continue
|
||||
matches.append((event,attribution))
|
||||
if event.get('status') == 'done' and isinstance(event.get('result_url'),str):
|
||||
url = urlsplit(event['result_url'])
|
||||
if not url.netloc:
|
||||
collector.result_paths[url.path] = attribution
|
||||
if len(collector.result_paths) > 128:
|
||||
collector.result_paths.pop(next(iter(collector.result_paths)))
|
||||
return matches
|
||||
# #endregion ScenarioExecution.ChartHealth.Async.Events
|
||||
# #endregion ScenarioExecution.ChartHealth.Async
|
||||
@@ -0,0 +1,119 @@
|
||||
# #region ScenarioExecution.ChartHealth.Observer [C:5] [TYPE Module] [SEMANTICS chart,health,errors,readiness,scope]
|
||||
# @BRIEF Observe exact chart placements, checking current error cards before waiting for rendered data.
|
||||
# @INVARIANT A visible attributed error is FAILED; missing/ambiguous/unsettled content stays INCONCLUSIVE.
|
||||
import asyncio
|
||||
from datetime import UTC, datetime
|
||||
from time import monotonic
|
||||
from playwright.async_api import TimeoutError as PlaywrightTimeoutError
|
||||
from ...chart_health_contract import ChartDiagnostic
|
||||
from ...query_failure import safe_message
|
||||
from .browser_health_scope import resolve_health_scope
|
||||
|
||||
_STATE = '''el => ({error: Array.from(el.querySelectorAll('.alert-danger,.ant-alert-error,.chart-error')).map(x => x.textContent).filter(Boolean).join('\\n'),
|
||||
loading: !!el.querySelector('.loading,.ant-spin-spinning,[aria-busy="true"],.chart-loading'),
|
||||
ready: !!el.querySelector('table,svg,canvas,.big_number,.big_number_total,.header-line,[data-test="no-results"],.no-results')})'''
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Observer.Diagnostic [C:3] [TYPE Function]
|
||||
# @POST Runtime attribution and redacted error text form a closed diagnostic, never a metric value.
|
||||
def diagnostic(placement, status, reason, *, start, message='', origin='dom', exchange=None):
|
||||
text, truncated = safe_message(message)
|
||||
exchange = dict(exchange or {})
|
||||
truncated = truncated or exchange.pop("truncated",False)
|
||||
import re
|
||||
code = re.search(r'\bCode:\s*(\d+)\b', text)
|
||||
return ChartDiagnostic(chart_id=placement['chart_id'], placement_id=placement['id'],
|
||||
chart_name=placement['name'], tab_path=placement['tab_path'], status=status,
|
||||
reason_code=reason, origin=origin, observed_at=datetime.now(UTC).isoformat(),
|
||||
duration_seconds=max(0.0,monotonic()-start), message=text, truncated=truncated,
|
||||
database_code=code.group(1) if code else None,
|
||||
terminal_state={'healthy':'ready','failed':'error','inconclusive':'unavailable'}[status],
|
||||
**exchange).model_dump()
|
||||
# #endregion ScenarioExecution.ChartHealth.Observer.Diagnostic
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Observer.RPCBudget [C:2] [TYPE Function]
|
||||
# @POST No RPC starts after deadline;250ms of the existing per-chart budget is reserved for SDK result/task drainage.
|
||||
def chart_rpc_timeout(end):
|
||||
remaining = end-monotonic()
|
||||
if remaining <= 0.25:
|
||||
raise TimeoutError('CHART_LOAD_TIMEOUT')
|
||||
return (remaining-0.25)*1000
|
||||
# #endregion ScenarioExecution.ChartHealth.Observer.RPCBudget
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Observer.Exchange [C:3] [TYPE Function]
|
||||
# @POST Confirmed attributed API failure does not require successful ChartRenderer DOM; transport failure remains unresolved.
|
||||
def exchange_diagnostic(placement, exchange, start):
|
||||
confirmed = exchange.get('confirmed',True)
|
||||
return diagnostic(placement,'failed' if confirmed else 'inconclusive',
|
||||
'CHART_QUERY_FAILED' if confirmed else 'CHART_REQUEST_UNAVAILABLE',start=start,
|
||||
message=exchange['message'],origin='http_response' if confirmed else 'synthetic_exception',exchange=exchange['attribution'])
|
||||
# #endregion ScenarioExecution.ChartHealth.Observer.Exchange
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Observer.Wait [C:3] [TYPE Function]
|
||||
# @POST Current attributed errors are checked during mount; bounded SDK waits finish before the enclosing chart deadline.
|
||||
async def wait_chart_or_error(panel, placement, collector, journal, end):
|
||||
while True:
|
||||
reason = journal.check_control()
|
||||
if reason:
|
||||
raise ValueError(reason)
|
||||
timeout = chart_rpc_timeout(end)
|
||||
exchange = collector.current_error(placement['chart_id'])
|
||||
if exchange:
|
||||
return None,exchange
|
||||
chart,_ = await resolve_health_scope(panel,placement)
|
||||
try:
|
||||
await chart.wait_for(state='visible',timeout=min(100,timeout))
|
||||
return chart,None
|
||||
except PlaywrightTimeoutError:
|
||||
continue
|
||||
# #endregion ScenarioExecution.ChartHealth.Observer.Wait
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Observer.Observe [C:4] [TYPE Function]
|
||||
# @POST Query errors are detected before readiness timeout; zero/empty rendered charts are healthy.
|
||||
async def observe_chart(page, panel, placement, timeout_seconds, collector, journal):
|
||||
start, end = monotonic(), monotonic()+timeout_seconds
|
||||
try:
|
||||
chart,exchange = await wait_chart_or_error(panel,placement,collector,journal,end)
|
||||
if exchange:
|
||||
return exchange_diagnostic(placement,exchange,start)
|
||||
if await chart.count() != 1:
|
||||
return diagnostic(placement,'inconclusive','CHART_PLACEMENT_AMBIGUOUS',start=start)
|
||||
await chart.scroll_into_view_if_needed(timeout=chart_rpc_timeout(end))
|
||||
return await observe_mounted(chart,placement,collector,journal,start,end)
|
||||
except (TimeoutError,PlaywrightTimeoutError):
|
||||
return diagnostic(placement,'inconclusive','CHART_LOAD_TIMEOUT',start=start)
|
||||
except ValueError as exc:
|
||||
if str(exc).startswith('CHART_PLACEMENT_'):
|
||||
return diagnostic(placement,'inconclusive',str(exc),start=start)
|
||||
raise
|
||||
except asyncio.CancelledError:
|
||||
raise
|
||||
except Exception:
|
||||
return diagnostic(placement,'inconclusive','CHART_OBSERVATION_UNAVAILABLE',start=start)
|
||||
# #endregion ScenarioExecution.ChartHealth.Observer.Observe
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Observer.Mounted [C:3] [TYPE Function]
|
||||
# @POST Mounted DOM errors precede ready content; pending generations cannot use stale render success.
|
||||
async def observe_mounted(chart, placement, collector, journal, start, end):
|
||||
while monotonic() < end:
|
||||
reason = journal.check_control()
|
||||
if reason:
|
||||
raise ValueError(reason)
|
||||
state = await chart.evaluate(_STATE,timeout=chart_rpc_timeout(end))
|
||||
exchange = collector.current_error(placement['chart_id'])
|
||||
pending = collector.latest.get(placement['chart_id']) in collector.pending
|
||||
if state['error'] and not pending:
|
||||
return diagnostic(placement,'failed','CHART_QUERY_FAILED',start=start,message=state['error'])
|
||||
if exchange:
|
||||
return exchange_diagnostic(placement,exchange,start)
|
||||
if state['ready'] and not state['loading'] and not pending:
|
||||
return diagnostic(placement,'healthy','CHART_READY',start=start)
|
||||
await asyncio.sleep(0.1)
|
||||
return diagnostic(placement,'inconclusive','CHART_LOAD_TIMEOUT',start=start)
|
||||
# #endregion ScenarioExecution.ChartHealth.Observer.Mounted
|
||||
# #endregion ScenarioExecution.ChartHealth.Observer
|
||||
@@ -0,0 +1,146 @@
|
||||
# #region ScenarioExecution.ChartHealth.Responses [C:5] [TYPE Module] [SEMANTICS http,chart,generation,filters,redaction]
|
||||
# @BRIEF Bound chart-data observations to the newest exact saved-chart request; stale and foreign failures cannot decide health.
|
||||
import asyncio
|
||||
from hashlib import sha256
|
||||
import json
|
||||
from ...query_failure import response_error, safe_message
|
||||
from .browser_chart_async import record_jobs, current_events
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Responses.Collector [C:4] [TYPE Class]
|
||||
# @INVARIANT Only exact chart/data requests with saved slice identity establish attribution; buffers/listeners are bounded and owned.
|
||||
class ChartResponseCollector:
|
||||
# #region ScenarioExecution.ChartHealth.Responses.Collector.Init [C:2] [TYPE Function]
|
||||
# @POST No listeners are attached until the composite action owns their lifetime.
|
||||
def __init__(self, page, chart_ids):
|
||||
self.page, self.chart_ids = page, set(chart_ids)
|
||||
self.latest, self.requests, self.errors, self.tasks = {}, {}, {}, set()
|
||||
self.generation = 0
|
||||
self.jobs, self.result_paths, self.pending = {}, {}, set()
|
||||
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.Init
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Responses.Collector.Request [C:3] [TYPE Function]
|
||||
# @POST Newest exact saved-chart request replaces prior errors; query/filter content is fingerprinted, not retained.
|
||||
def on_request(self, request):
|
||||
path = urlsplit(request.url).path
|
||||
if request.method == 'GET' and path in self.result_paths:
|
||||
self.requests[request] = self.result_paths[path]
|
||||
return
|
||||
if path.rstrip('/') != '/api/v1/chart/data' or request.method != 'POST':
|
||||
return
|
||||
try:
|
||||
payload = request.post_data_json
|
||||
chart_id = (payload.get('form_data') or {}).get('slice_id')
|
||||
if type(chart_id) is not int or chart_id not in self.chart_ids:
|
||||
return
|
||||
self.generation += 1
|
||||
identity = str(self.generation)
|
||||
scope = {'datasource':payload.get('datasource'),'queries':payload.get('queries')}
|
||||
fingerprint = sha256(json.dumps(scope,sort_keys=True,separators=(',',':')).encode()).hexdigest()
|
||||
self.pending.discard(self.latest.get(chart_id))
|
||||
self.latest[chart_id] = identity
|
||||
self.pending.add(identity)
|
||||
self.errors.pop(chart_id,None)
|
||||
self.requests[request] = (chart_id,identity,fingerprint)
|
||||
if len(self.requests) > 128:
|
||||
self.requests.pop(next(iter(self.requests)))
|
||||
except (ValueError, TypeError, AttributeError):
|
||||
return
|
||||
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.Request
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Responses.Collector.RequestFailed [C:3] [TYPE Function]
|
||||
# @POST A current owned transport failure is unresolved evidence, never a confirmed database error.
|
||||
def on_request_failed(self, request):
|
||||
attribution = self.requests.get(request)
|
||||
if attribution is None or self.latest.get(attribution[0]) != attribution[1]:
|
||||
return
|
||||
message,truncated = safe_message(request.failure or 'Chart request transport unavailable')
|
||||
chart_id,identity,fingerprint = attribution
|
||||
self.errors[chart_id] = {'message':message,'confirmed':False,'attribution':{
|
||||
'request_identity':identity,'load_generation':int(identity),'filter_fingerprint':fingerprint,
|
||||
'truncated':truncated}}
|
||||
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.RequestFailed
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Responses.Collector.Response [C:2] [TYPE Function]
|
||||
# @POST At most128 pending body reads can exist; missing body evidence cannot create failure.
|
||||
def on_response(self, response):
|
||||
polling = urlsplit(response.url).path.rstrip('/') == '/api/v1/async_event'
|
||||
if (response.request not in self.requests and not polling) or len(self.tasks) >= 128:
|
||||
return
|
||||
task = asyncio.create_task(self.read_response(response))
|
||||
self.tasks.add(task)
|
||||
task.add_done_callback(self.tasks.discard)
|
||||
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.Response
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Responses.Collector.Read [C:4] [TYPE Function]
|
||||
# @POST HTTP200 query errors and non2xx payloads decide only the current exact request generation.
|
||||
async def read_response(self, response):
|
||||
try:
|
||||
raw = await asyncio.wait_for(response.body(),timeout=2)
|
||||
if len(raw) > 1048576:
|
||||
return
|
||||
payload = json.loads(raw)
|
||||
if response.request not in self.requests:
|
||||
for event, attribution in current_events(self,payload):
|
||||
self.record_error(attribution,response_error(event),raw,response.status)
|
||||
return
|
||||
attribution = self.requests[response.request]
|
||||
if response.status == 202:
|
||||
record_jobs(self,payload,attribution)
|
||||
return
|
||||
error = response_error(payload,response.status)
|
||||
if error is not None and error.get('_unconfirmed'):
|
||||
return
|
||||
self.pending.discard(attribution[1])
|
||||
self.record_error(attribution,error,raw,response.status)
|
||||
except Exception:
|
||||
return
|
||||
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.Read
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Responses.Collector.Record [C:3] [TYPE Function]
|
||||
# @POST Only the newest attributed response can publish a sanitized query failure.
|
||||
def record_error(self, attribution, error, raw, http_status):
|
||||
chart_id, identity, fingerprint = attribution
|
||||
if error is None or error.get('_unconfirmed') or self.latest.get(chart_id) != identity:
|
||||
return
|
||||
self.pending.discard(identity)
|
||||
message, truncated = safe_message(error.get('message') or error.get('error'))
|
||||
self.errors[chart_id] = {'message':message,'attribution':{
|
||||
'http_status':http_status,'request_identity':identity,'load_generation':int(identity),
|
||||
'filter_fingerprint':fingerprint,'original_response_sha256':sha256(raw).hexdigest(),
|
||||
'original_byte_length':len(raw),'truncated':truncated,'application_code':str(error.get('error_type') or '')[:256] or None}}
|
||||
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.Record
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Responses.Collector.Current [C:1] [TYPE Function]
|
||||
# @BRIEF Return only current owned response attribution, never caller error state.
|
||||
def current_error(self, chart_id):
|
||||
return self.errors.get(chart_id)
|
||||
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.Current
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Responses.Collector.Start [C:1] [TYPE Function]
|
||||
# @POST Request and response listeners precede tab activation/load.
|
||||
def start(self):
|
||||
self.page.on('request',self.on_request)
|
||||
self.page.on('response',self.on_response)
|
||||
self.page.on('requestfailed',self.on_request_failed)
|
||||
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.Start
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Responses.Collector.Close [C:2] [TYPE Function]
|
||||
# @POST Cancellation/completion removes listeners and drains all owned pending tasks.
|
||||
async def close(self):
|
||||
self.page.remove_listener('request',self.on_request)
|
||||
self.page.remove_listener('response',self.on_response)
|
||||
self.page.remove_listener('requestfailed',self.on_request_failed)
|
||||
tasks = list(self.tasks)
|
||||
for task in tasks:
|
||||
task.cancel()
|
||||
if tasks:
|
||||
await asyncio.gather(*tasks,return_exceptions=True)
|
||||
self.requests.clear()
|
||||
self.pending.clear()
|
||||
self.jobs.clear()
|
||||
self.result_paths.clear()
|
||||
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.Close
|
||||
# #endregion ScenarioExecution.ChartHealth.Responses.Collector
|
||||
# #endregion ScenarioExecution.ChartHealth.Responses
|
||||
@@ -0,0 +1,36 @@
|
||||
# #region ScenarioExecution.ChartHealth.Capture [C:4] [TYPE Module] [SEMANTICS tabs,frames,coverage,owned]
|
||||
# @BRIEF Capture exact visible chart placements as finite frames; long/uncaptured tabs retain explicit omissions.
|
||||
from ..chart_health_tab_store import store_health_artifact
|
||||
from ..mime_sniff import sniff_mime
|
||||
from .browser_health_scope import resolve_health_scope
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Capture.Frames [C:4] [TYPE Function]
|
||||
# @POST Only actual PNG bytes from exact current-tab chart placements are retained; omitted frames never count as visually covered.
|
||||
async def capture_frames(page, panel, placements, journal, limit, timeout_seconds):
|
||||
frames, omitted, total = [],[],0
|
||||
for placement in placements:
|
||||
reason = journal.check_control()
|
||||
if reason:
|
||||
raise ValueError(reason)
|
||||
if len(frames) >= limit:
|
||||
omitted.append({'placement_id':placement['id'],'reason':'FRAME_BUDGET'})
|
||||
continue
|
||||
try:
|
||||
_,chart = await resolve_health_scope(panel,placement)
|
||||
if await chart.count() != 1:
|
||||
raise ValueError('CHART_PLACEMENT_AMBIGUOUS')
|
||||
await chart.scroll_into_view_if_needed(timeout=timeout_seconds*1000)
|
||||
data = await chart.screenshot(type='png',timeout=timeout_seconds*1000)
|
||||
if sniff_mime(data) != 'image/png' or len(data)+total > 10485760:
|
||||
raise ValueError('FRAME_BYTE_BUDGET')
|
||||
receipt = store_health_artifact(journal,data,f'chart-health-frame-{placement["id"]}.png','image/png')
|
||||
frames.append({'placement_id':placement['id'],'receipt':receipt})
|
||||
total += len(data)
|
||||
except ValueError as exc:
|
||||
omitted.append({'placement_id':placement['id'],'reason':str(exc)})
|
||||
except Exception:
|
||||
omitted.append({'placement_id':placement['id'],'reason':'FRAME_UNAVAILABLE'})
|
||||
return frames,omitted
|
||||
# #endregion ScenarioExecution.ChartHealth.Capture.Frames
|
||||
# #endregion ScenarioExecution.ChartHealth.Capture
|
||||
@@ -0,0 +1,102 @@
|
||||
# #region ScenarioExecution.ChartHealth.TabEvaluation [C:5] [TYPE Module] [SEMANTICS evaluation,tabs,owned,diagnostics,multimodal]
|
||||
# @BRIEF Evaluate opt-in tab frames plus actual owned diagnostics before leaving the panel; model output never erases a query failure.
|
||||
import asyncio
|
||||
from datetime import UTC, datetime
|
||||
import json
|
||||
from src.core.database import SessionLocal
|
||||
from ..agent_evaluation import submit_evaluation, parse_evaluation_response
|
||||
from ..evaluation_adapter import _normalize_provider_response, _stamp_provenance, _persist_raw_response
|
||||
from ..evaluation_prompt import build_evaluation_prompt
|
||||
from ..evaluation_health_payloads import load_health_payloads
|
||||
from ..chart_health_tab_store import tab_diagnostics, store_health_artifact, append_tab_evaluation
|
||||
from ..chart_health_artifact import verify_health_artifact
|
||||
from ...query_failure import safe_message
|
||||
from .browser_health_capture import capture_frames
|
||||
from .browser_health_provider import admit_health_provider
|
||||
from .browser_health_llm_client import BoundedHealthClient
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.TabEvaluation.Selected [C:2] [TYPE Function]
|
||||
# @POST Policy selection is explicit; confirmed problems and selected stable tab IDs determine opt-in work.
|
||||
def should_evaluate(policy, tab_path, diagnostics):
|
||||
if policy.mode == 'selected':
|
||||
return bool(tab_path and tab_path[-1] in policy.tab_ids)
|
||||
return policy.mode == 'all_visited' or any(item['diagnostic']['status'] != 'healthy' for item in diagnostics)
|
||||
# #endregion ScenarioExecution.ChartHealth.TabEvaluation.Selected
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.TabEvaluation.Inputs [C:3] [TYPE Function]
|
||||
# @POST Actual diagnostic contents and exact owned image bytes enter the prompt, within finite text/frame budgets.
|
||||
def evaluation_inputs(journal, diagnostics, frames):
|
||||
artifacts = [item['receipt'] for item in [*diagnostics,*frames]]
|
||||
manifest = [{'artifact_id':item['content_ref'],'sha256':item['sha256'],'content_type':item['content_type'],
|
||||
'byte_length':item['byte_length'],'role':'context'} for item in artifacts]
|
||||
with SessionLocal() as db:
|
||||
for receipt in artifacts:
|
||||
verify_health_artifact(journal,db,receipt)
|
||||
payloads = load_health_payloads(manifest,journal.storage,db=db,run_id=journal.run_id)
|
||||
images = [(journal.storage.retrieve(item['receipt']['content_ref']),'image/png') for item in frames]
|
||||
return manifest,payloads,images,artifacts
|
||||
# #endregion ScenarioExecution.ChartHealth.TabEvaluation.Inputs
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.TabEvaluation.Submit [C:4] [TYPE Function]
|
||||
# @POST The existing capacity/credentials/provider boundary uses only the pinned opt-in specification; no paid call is made by tests.
|
||||
async def evaluate_tab(page, panel, tab_path, placements, journal, policy, timeout_seconds):
|
||||
diagnostics = tab_diagnostics(journal,tab_path)
|
||||
if not should_evaluate(policy,tab_path,diagnostics):
|
||||
return
|
||||
existing = journal.frontier().get('tab_evaluations',[])
|
||||
if any(item['tab_path'] == tab_path for item in existing):
|
||||
return
|
||||
if len(existing) >= policy.max_tabs:
|
||||
append_tab_evaluation(journal,{'tab_path':tab_path,'status':'inconclusive','reason_code':'VLM_TAB_BUDGET',
|
||||
'coverage_complete':False,'artifacts':[],'evaluated_at':datetime.now(UTC).isoformat()})
|
||||
return
|
||||
frames, omitted = await capture_frames(page,panel,placements,journal,
|
||||
min(policy.max_frames_per_tab,policy.spec.limits.max_images),timeout_seconds)
|
||||
manifest,payloads,images,artifacts = evaluation_inputs(journal,diagnostics,frames)
|
||||
summary = {'tab_path':tab_path,'status':'inconclusive','reason_code':'VLM_EVIDENCE_INCOMPLETE',
|
||||
'coverage_complete':not omitted and not payloads['truncated'],'omitted_frames':omitted,'artifacts':artifacts,
|
||||
'evaluated_at':datetime.now(UTC).isoformat()}
|
||||
payloads['frames'] = [{'placement_id':item['placement_id'],'artifact_id':item['receipt']['content_ref'],'image_index':index} for index,item in enumerate(frames)]
|
||||
prompt = build_evaluation_prompt(policy.spec,manifest,evidence_payloads=payloads)
|
||||
if not images:
|
||||
append_tab_evaluation(journal,{**summary,'reason_code':'VLM_IMAGE_EVIDENCE_UNAVAILABLE'})
|
||||
return
|
||||
try:
|
||||
with SessionLocal() as db:
|
||||
provider = admit_health_provider(db,policy)
|
||||
if len(prompt.encode())+512 > policy.spec.limits.max_input_tokens:
|
||||
raise ValueError("CHART_HEALTH_VLM_TEXT_BUDGET")
|
||||
step = journal.step
|
||||
target = step.get('target_snapshot') or {}
|
||||
binding = step.get('live_execution_binding_snapshot') or {}
|
||||
try:
|
||||
raw = await asyncio.wait_for(submit_evaluation(db,spec=policy.spec,prompt=prompt,images=images,
|
||||
environment_id=target.get('environment_id') or binding.get('environment_id'),
|
||||
environment_class=target.get('environment_class','DEV'),run_id=journal.run_id,
|
||||
logical_step_id=journal.logical_step_id,runtime_step=step,client=BoundedHealthClient(db,provider,policy.spec)),timeout=min(timeout_seconds,policy.spec.limits.timeout_ms/1000))
|
||||
finally:
|
||||
db.commit()
|
||||
raw_bytes,digest,ref = _persist_raw_response(journal.storage,journal.run_id,raw)
|
||||
record = parse_evaluation_response(_stamp_provenance(_normalize_provider_response(raw,policy.spec),
|
||||
run_id=journal.run_id,logical_step_id=journal.logical_step_id,attempt=journal.attempt,
|
||||
manifest=manifest,content_ref=ref,digest=digest,spec=policy.spec),spec=policy.spec)
|
||||
for finding in record.findings:
|
||||
finding.message = safe_message(finding.message)[0][:2000]
|
||||
raw_receipt = store_health_artifact(journal,raw_bytes,'chart-health-vlm-response.json')
|
||||
record_receipt = store_health_artifact(journal,json.dumps(record.model_dump(mode='json'),sort_keys=True).encode(),'chart-health-vlm-record.json')
|
||||
summary.update(status='succeeded' if summary['coverage_complete'] and record.status == 'succeeded' else 'inconclusive',
|
||||
reason_code='VLM_TAB_EVALUATED' if summary['coverage_complete'] else 'VLM_EVIDENCE_INCOMPLETE',
|
||||
advisory_verdict=record.verdict, confidence=record.confidence,findings=[item.model_dump(mode='json') for item in record.findings],
|
||||
artifacts=[*artifacts,raw_receipt,record_receipt])
|
||||
except asyncio.CancelledError:
|
||||
append_tab_evaluation(journal,{**summary,'reason_code':'VLM_INTERRUPTED'})
|
||||
raise
|
||||
except Exception as exc:
|
||||
code = str(exc)
|
||||
summary['reason_code'] = code[:128] if code.startswith(('CHART_HEALTH_','EVALUATION_')) else 'VLM_PROVIDER_UNAVAILABLE'
|
||||
append_tab_evaluation(journal,summary)
|
||||
# #endregion ScenarioExecution.ChartHealth.TabEvaluation.Submit
|
||||
# #endregion ScenarioExecution.ChartHealth.TabEvaluation
|
||||
@@ -0,0 +1,29 @@
|
||||
# #region ScenarioExecution.ChartHealth.EvaluationInputs [C:4] [TYPE Module] [SEMANTICS evaluation,opt-in,tabs,limits]
|
||||
# @BRIEF Explicit per-tab opt-in pins the existing provider specification and finite capture/request budgets.
|
||||
from typing import Literal
|
||||
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
||||
from ...scenario.evaluation_models import AgentEvaluationSpec, TokenEvaluationLimits
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.EvaluationInputs.Policy [C:3] [TYPE Class]
|
||||
# @INVARIANT Semantic diagnostic evaluation requires no fake metric comparison or token-only recipe authority.
|
||||
class HealthEvaluationPolicy(BaseModel):
|
||||
model_config = ConfigDict(extra='forbid',strict=True)
|
||||
mode: Literal['all_visited','selected','problem_areas']
|
||||
tab_ids: list[str] = Field(default_factory=list,max_length=1000)
|
||||
max_tabs: int = Field(default=100,ge=1,le=1000)
|
||||
max_frames_per_tab: int = Field(default=8,ge=1,le=8)
|
||||
spec: AgentEvaluationSpec
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.EvaluationInputs.Policy.Validate [C:3] [TYPE Function]
|
||||
# @POST Selection is explicit; comparison and token-only recipe authority cannot be smuggled into health evaluation.
|
||||
@model_validator(mode='after')
|
||||
def validate_policy(self):
|
||||
if (self.mode == 'selected' and not self.tab_ids) or (self.mode != 'selected' and self.tab_ids):
|
||||
raise ValueError('CHART_HEALTH_EVALUATION_TAB_SELECTION_INVALID')
|
||||
if isinstance(self.spec.limits,TokenEvaluationLimits) or self.spec.comparison_refs or any(item.criterion_kind != 'semantic' for item in self.spec.criteria):
|
||||
raise ValueError('CHART_HEALTH_EVALUATION_SEMANTIC_ONLY')
|
||||
return self
|
||||
# #endregion ScenarioExecution.ChartHealth.EvaluationInputs.Policy.Validate
|
||||
# #endregion ScenarioExecution.ChartHealth.EvaluationInputs.Policy
|
||||
# #endregion ScenarioExecution.ChartHealth.EvaluationInputs
|
||||
@@ -0,0 +1,45 @@
|
||||
# #region ScenarioExecution.ChartHealth.BoundedClient [C:4] [TYPE Module] [SEMANTICS provider,request-budget,tokens,timeout]
|
||||
# @BRIEF Bound the existing configured multimodal SDK client to one physical request and declared output/deadline limits.
|
||||
# @REJECTED Legacy get_json_completion retries/fallbacks do not preserve a finite per-tab physical request budget.
|
||||
import json
|
||||
from src.plugins.llm_analysis.service import LLMClient
|
||||
from src.plugins.llm_analysis.models import LLMProviderType
|
||||
from src.services.llm_provider import LLMProviderService
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.BoundedClient.Client [C:3] [TYPE Class]
|
||||
class BoundedHealthClient:
|
||||
# #region ScenarioExecution.ChartHealth.BoundedClient.Client.Init [C:1] [TYPE Function]
|
||||
# @POST No credentials or sockets are accessed until the existing evaluation capacity boundary invokes the client.
|
||||
def __init__(self, db, provider, spec):
|
||||
self.db,self.provider,self.spec = db,provider,spec
|
||||
# #endregion ScenarioExecution.ChartHealth.BoundedClient.Client.Init
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.BoundedClient.Client.Complete [C:4] [TYPE Function]
|
||||
# @POST One bounded SDK request; truncation/nonobject JSON refuses and transport usage replaces model billing claims.
|
||||
async def get_json_completion(self, messages):
|
||||
key = LLMProviderService(self.db).get_decrypted_api_key(self.provider.id)
|
||||
if not key:
|
||||
raise RuntimeError('EVALUATION_PROVIDER_MISSING')
|
||||
self.db.commit()
|
||||
client = LLMClient(LLMProviderType(self.provider.provider_type),key,self.provider.base_url,self.spec.model_id)
|
||||
bounded = client.client.with_options(max_retries=0,timeout=self.spec.limits.timeout_ms/1000)
|
||||
options = {'model':self.spec.model_id,'messages':messages,'max_tokens':self.spec.limits.max_output_tokens}
|
||||
if self.provider.supports_json_object is not False:
|
||||
options['response_format'] = {'type':'json_object'}
|
||||
try:
|
||||
response = await bounded.chat.completions.create(**options)
|
||||
finally:
|
||||
await bounded.close()
|
||||
if not response.choices or response.choices[0].finish_reason == 'length':
|
||||
raise RuntimeError('EVALUATION_RESPONSE_TRUNCATED')
|
||||
raw = json.loads(response.choices[0].message.content)
|
||||
if not isinstance(raw,dict):
|
||||
raise RuntimeError('EVALUATION_RESPONSE_INVALID')
|
||||
raw.pop('usage',None)
|
||||
if response.usage is not None:
|
||||
raw['usage'] = {'input_tokens':response.usage.prompt_tokens,'output_tokens':response.usage.completion_tokens}
|
||||
return raw
|
||||
# #endregion ScenarioExecution.ChartHealth.BoundedClient.Client.Complete
|
||||
# #endregion ScenarioExecution.ChartHealth.BoundedClient.Client
|
||||
# #endregion ScenarioExecution.ChartHealth.BoundedClient
|
||||
@@ -0,0 +1,64 @@
|
||||
# #region ScenarioExecution.ChartHealth.Manifest [C:4] [TYPE Module] [SEMANTICS placement,tabs,server,identity]
|
||||
# @BRIEF Inspect stable server chart placements and nested tab paths; duplicate titles never establish identity.
|
||||
from hashlib import sha256
|
||||
import json
|
||||
from urllib.parse import urlsplit
|
||||
from .browser_tabs_manifest import parse_tabs_manifest
|
||||
from ...query_failure import safe_message
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Manifest.Parse [C:4] [TYPE Function]
|
||||
# @POST Every reachable chart placement retains its own server node ID, name, chart ID and full tab path.
|
||||
def parse_health_manifest(layout, chart_ids=None):
|
||||
source = parse_tabs_manifest(layout) if any(isinstance(node,dict) and node.get("type") == "TAB" for node in layout.values()) else {"tabs":[],"source_total":0,"manifest_digest":sha256(json.dumps(layout,sort_keys=True,separators=(",",":")).encode()).hexdigest()}
|
||||
placements, seen = [], set()
|
||||
# #region ScenarioExecution.ChartHealth.Manifest.Parse.Walk [C:3] [TYPE Function]
|
||||
# @POST Duplicate/cyclic/orphan chart nodes cannot be counted as covered.
|
||||
def walk(node_id, path):
|
||||
if node_id in seen or node_id not in layout:
|
||||
raise ValueError('CHART_HEALTH_MANIFEST_INVALID')
|
||||
seen.add(node_id)
|
||||
node = layout[node_id]
|
||||
if node.get('type') == 'TAB':
|
||||
path = [*path,node_id]
|
||||
if node.get('type') == 'CHART':
|
||||
meta = node.get('meta') or {}
|
||||
chart_id = meta.get('chartId')
|
||||
if type(chart_id) is not int or chart_id <= 0:
|
||||
raise ValueError('CHART_HEALTH_CHART_ID_INVALID')
|
||||
if chart_ids is None or chart_id in chart_ids:
|
||||
placements.append({'id':node_id,'chart_id':chart_id,
|
||||
'name':safe_message(meta.get('sliceName') or f'Chart {chart_id}')[0][:512], 'tab_path':path})
|
||||
for child in node.get('children',[]):
|
||||
walk(child,path)
|
||||
# #endregion ScenarioExecution.ChartHealth.Manifest.Parse.Walk
|
||||
walk('ROOT_ID',[])
|
||||
expected = {key for key,node in layout.items() if isinstance(node,dict) and node.get('type') == 'CHART'}
|
||||
if not expected.issubset(seen) or not placements:
|
||||
raise ValueError('CHART_HEALTH_MANIFEST_INCOMPLETE')
|
||||
if chart_ids and {item['chart_id'] for item in placements} != set(chart_ids):
|
||||
raise ValueError('CHART_HEALTH_TARGET_MISSING')
|
||||
tab_order = {item['id']:index for index,item in enumerate(source['tabs'])}
|
||||
placements.sort(key=lambda item:tab_order.get(item['tab_path'][-1],-1) if item['tab_path'] else -1)
|
||||
source['tabs'] = [dict(item,name=safe_message((layout[item['id']].get('meta') or {}).get('text') or item['id'])[0][:512]) for item in source['tabs']]
|
||||
source['placements'] = placements
|
||||
source['placement_digest'] = sha256(json.dumps(placements,sort_keys=True,separators=(',',':')).encode()).hexdigest()
|
||||
return source
|
||||
# #endregion ScenarioExecution.ChartHealth.Manifest.Parse
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Manifest.Fetch [C:3] [TYPE Function]
|
||||
# @POST Authenticated dashboard layout is the sole expected-placement authority.
|
||||
async def fetch_health_manifest(page, dashboard_id, chart_ids=None):
|
||||
url = urlsplit(page.url)
|
||||
response = await page.request.get(f'{url.scheme}://{url.netloc}/api/v1/dashboard/{dashboard_id}')
|
||||
raw = await response.body()
|
||||
if response.status != 200 or len(raw) > 1048576:
|
||||
raise ValueError('CHART_HEALTH_MANIFEST_UNAVAILABLE')
|
||||
result = json.loads(raw).get('result') or {}
|
||||
if result.get('id') != dashboard_id:
|
||||
raise ValueError('CHART_HEALTH_DASHBOARD_MISMATCH')
|
||||
layout = result.get('position_json')
|
||||
return parse_health_manifest(json.loads(layout) if isinstance(layout,str) else layout, chart_ids)
|
||||
# #endregion ScenarioExecution.ChartHealth.Manifest.Fetch
|
||||
# #endregion ScenarioExecution.ChartHealth.Manifest
|
||||
@@ -0,0 +1,21 @@
|
||||
# #region ScenarioExecution.ChartHealth.Provider [C:4] [TYPE Module] [SEMANTICS provider,configuration,pin,multimodal]
|
||||
# @BRIEF Verify the public provider/model configuration before any per-tab credentials or capacity are accessed.
|
||||
from src.models.llm import LLMProvider
|
||||
from ...scenario.metric_evaluation_provider import provider_public_config_digest
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Provider.Admit [C:3] [TYPE Function]
|
||||
# @POST Changed/nonvisual/inactive provider configuration refuses before paid requests; declared budgets remain pinned.
|
||||
def admit_health_provider(db, policy):
|
||||
provider = db.get(LLMProvider,policy.spec.provider_id)
|
||||
if provider is None or provider.is_active is not True or provider.is_multimodal is not True:
|
||||
raise ValueError('CHART_HEALTH_VLM_PROVIDER_UNAVAILABLE')
|
||||
digest = provider_public_config_digest(provider)
|
||||
if (policy.spec.provider_version != f'config_sha256:{digest}' or policy.spec.model_id != provider.default_model
|
||||
or policy.spec.model_version != f'configured-model:{provider.default_model}'):
|
||||
raise ValueError('CHART_HEALTH_VLM_PROVIDER_CHANGED')
|
||||
if provider.max_images is not None and policy.spec.limits.max_images > provider.max_images:
|
||||
raise ValueError("CHART_HEALTH_VLM_IMAGE_BUDGET")
|
||||
return provider
|
||||
# #endregion ScenarioExecution.ChartHealth.Provider.Admit
|
||||
# #endregion ScenarioExecution.ChartHealth.Provider
|
||||
@@ -0,0 +1,29 @@
|
||||
# #region ScenarioExecution.ChartHealth.Scope [C:4] [TYPE Module] [SEMANTICS native,saved-chart,body,frame,identity]
|
||||
# @BRIEF Resolve a saved chart's native Superset5 card/body even when failure removes its ChartRenderer ID.
|
||||
# @RATIONALE Pinned5.0.0 gridComponents/Chart.jsx exposes chart-grid-component/data-test-chart-id outside both render branches.
|
||||
# @REJECTED Requiring ChartRenderer's chart-id on a failed native card hid confirmed errors; selecting title/header icons would create false readiness.
|
||||
# @INVARIANT Header/menu icons are excluded from readiness; duplicate/foreign native identities refuse without first/global fallback.
|
||||
from ...query_failure import safe_message
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Scope.Resolve [C:3] [TYPE Function]
|
||||
# @POST Native frames retain card context while health reads only its exact chart body; legacy protocol IDs remain supported.
|
||||
async def resolve_health_scope(panel, placement):
|
||||
grid = panel.locator(f'[data-test="chart-grid-component"][data-test-chart-id="{placement["chart_id"]}"]')
|
||||
count = await grid.count()
|
||||
if count > 1:
|
||||
raise ValueError('CHART_PLACEMENT_AMBIGUOUS')
|
||||
if count == 1:
|
||||
name = await grid.get_attribute('data-test-chart-name')
|
||||
if name is not None and safe_message(name)[0][:512] != placement['name']:
|
||||
raise ValueError('CHART_PLACEMENT_IDENTITY_MISMATCH')
|
||||
body = grid.locator('.dashboard-chart')
|
||||
if await body.count() > 1:
|
||||
raise ValueError('CHART_PLACEMENT_AMBIGUOUS')
|
||||
return body, grid
|
||||
chart = panel.locator(f'#chart-id-{placement["chart_id"]}')
|
||||
if await chart.count() > 1:
|
||||
raise ValueError('CHART_PLACEMENT_AMBIGUOUS')
|
||||
return chart, chart
|
||||
# #endregion ScenarioExecution.ChartHealth.Scope.Resolve
|
||||
# #endregion ScenarioExecution.ChartHealth.Scope
|
||||
@@ -0,0 +1,155 @@
|
||||
# #region ScenarioExecution.ChartHealth.Sweep [C:5] [TYPE Module] [SEMANTICS chart,tabs,composite,coverage,continuation]
|
||||
# @BRIEF Continue across failed charts while committing per-placement health under cancellation and whole/per-chart budgets.
|
||||
import asyncio
|
||||
from time import monotonic
|
||||
from .browser_all_tabs import activate_tab
|
||||
from .browser_chart_health import observe_chart, diagnostic
|
||||
from .browser_chart_responses import ChartResponseCollector
|
||||
from .browser_health_manifest import fetch_health_manifest
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Sweep.Panel [C:3] [TYPE Function]
|
||||
# @POST All inspected tab ancestors activate their exact panel; charts outside tabs retain the page scope.
|
||||
async def activate_path(page, source, path, timeout):
|
||||
panel = page
|
||||
for tab_id in path:
|
||||
tab = next(item for item in source['tabs'] if item['id'] == tab_id)
|
||||
panel = await current_or_activate(page,tab,timeout)
|
||||
return panel
|
||||
# #endregion ScenarioExecution.ChartHealth.Sweep.Panel
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Sweep.CurrentPanel [C:3] [TYPE Function]
|
||||
# @POST Reusing an already selected exact tab does not trigger another chart generation; identity/visibility remain checked.
|
||||
async def current_or_activate(page, tab, timeout):
|
||||
control = page.locator(f'[role="tab"][id="{tab["parent_tabs_id"]}-tab-{tab["id"]}"]')
|
||||
if await control.count() == 1 and await control.get_attribute('aria-selected') == 'true':
|
||||
panel_id = await control.get_attribute('aria-controls')
|
||||
if not panel_id:
|
||||
raise ValueError('BROWSER_TABS_PANEL_ID_MISSING')
|
||||
panel = page.locator(f'[id="{panel_id}"][role="tabpanel"]')
|
||||
await panel.wait_for(state='visible',timeout=timeout*1000)
|
||||
if await panel.count() != 1:
|
||||
raise ValueError('BROWSER_TABS_PANEL_AMBIGUOUS')
|
||||
return panel
|
||||
return await activate_tab(page,tab['id'],tab['parent_tabs_id'],timeout)
|
||||
# #endregion ScenarioExecution.ChartHealth.Sweep.CurrentPanel
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Sweep.Controlled [C:3] [TYPE Function]
|
||||
# @POST Expensive UI operations renew both leases and remain cancellable without discarding previous errors.
|
||||
async def controlled(factory, journal, timeout):
|
||||
task = asyncio.create_task(asyncio.wait_for(factory(),timeout=timeout))
|
||||
try:
|
||||
while not task.done():
|
||||
reason = journal.check_control()
|
||||
if reason:
|
||||
raise ValueError(reason)
|
||||
await asyncio.wait({task},timeout=1)
|
||||
return await task
|
||||
finally:
|
||||
if not task.done():
|
||||
task.cancel()
|
||||
await asyncio.gather(task,return_exceptions=True)
|
||||
# #endregion ScenarioExecution.ChartHealth.Sweep.Controlled
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Sweep.One [C:3] [TYPE Function]
|
||||
# @POST An unavailable chart/tab is an explicit unresolved placement, while confirmed query errors are retained.
|
||||
async def one_chart(page, source, placement, collector, journal, timeout):
|
||||
start = monotonic()
|
||||
try:
|
||||
panel = await activate_path(page,source,placement['tab_path'],timeout)
|
||||
return await observe_chart(page,panel,placement,max(0.01,timeout-(monotonic()-start)),collector,journal)
|
||||
except ValueError as exc:
|
||||
if str(exc).startswith('BROWSER_TRAVERSAL_'):
|
||||
raise
|
||||
return diagnostic(placement,'inconclusive',str(exc)[:128],start=start)
|
||||
except TimeoutError:
|
||||
return diagnostic(placement,'inconclusive','CHART_LOAD_TIMEOUT',start=start)
|
||||
except asyncio.CancelledError:
|
||||
raise
|
||||
except Exception:
|
||||
return diagnostic(placement,'inconclusive','CHART_OBSERVATION_UNAVAILABLE',start=start)
|
||||
# #endregion ScenarioExecution.ChartHealth.Sweep.One
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Sweep.Evaluate [C:3] [TYPE Function]
|
||||
# @POST Optional VLM failure remains a retained per-tab outcome and cannot suppress deterministic errors or later tabs.
|
||||
async def evaluate_current_tab(page, source, path, journal, limits, end):
|
||||
from .browser_health_evaluation import evaluate_tab
|
||||
from ..chart_health_tab_store import append_tab_evaluation
|
||||
remaining = min(end-monotonic(),journal.remaining_seconds())
|
||||
if remaining <= 0:
|
||||
raise ValueError('CHART_HEALTH_WHOLE_TIMEOUT')
|
||||
panel = await activate_path(page,source,path,min(limits.per_chart_timeout_seconds,remaining))
|
||||
placements = [item for item in source['placements'] if item['tab_path'] == path]
|
||||
# #region ScenarioExecution.ChartHealth.Sweep.Evaluate.Call [C:1] [TYPE Function]
|
||||
# @BRIEF Keep per-tab capture and configured provider work inside the original whole deadline.
|
||||
async def evaluate():
|
||||
return await evaluate_tab(page,panel,path,placements,journal,limits.per_tab_evaluation,remaining)
|
||||
# #endregion ScenarioExecution.ChartHealth.Sweep.Evaluate.Call
|
||||
try:
|
||||
await controlled(evaluate,journal,remaining)
|
||||
except TimeoutError:
|
||||
if not any(item['tab_path'] == path for item in journal.frontier().get('tab_evaluations',[])):
|
||||
append_tab_evaluation(journal,{'tab_path':path,'status':'inconclusive','reason_code':'VLM_TIMEOUT','coverage_complete':False,'artifacts':[]})
|
||||
# #endregion ScenarioExecution.ChartHealth.Sweep.Evaluate
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Sweep.BoundedChart [C:3] [TYPE Function]
|
||||
# @POST Per-chart timeout is retained and permits the next chart while remaining inside the original whole deadline.
|
||||
async def bounded_chart(page, source, placement, collector, journal, timeout):
|
||||
# #region ScenarioExecution.ChartHealth.Sweep.Run.Observe [C:1] [TYPE Function]
|
||||
# @BRIEF Bind the exact next placement to the monitored observation operation.
|
||||
async def observe():
|
||||
return await one_chart(page,source,placement,collector,journal,timeout)
|
||||
# #endregion ScenarioExecution.ChartHealth.Sweep.Run.Observe
|
||||
try:
|
||||
value = await controlled(observe,journal,timeout)
|
||||
except TimeoutError:
|
||||
value = diagnostic(placement,'inconclusive','CHART_LOAD_TIMEOUT',start=monotonic()-timeout)
|
||||
return value
|
||||
# #endregion ScenarioExecution.ChartHealth.Sweep.BoundedChart
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Sweep.Run [C:4] [TYPE Function]
|
||||
# @POST FAILED dominates interrupted coverage; otherwise only complete conclusive observations permit PASS.
|
||||
async def sweep_health(page, dashboard_id, journal, limits):
|
||||
collector = None
|
||||
end = monotonic()+min(limits.whole_timeout_seconds,journal.remaining_seconds())
|
||||
try:
|
||||
source = await fetch_health_manifest(page,dashboard_id,getattr(limits,'chart_ids',None))
|
||||
journal.freeze_source(source)
|
||||
policy = limits.per_tab_evaluation
|
||||
if policy is not None and policy.mode == 'selected' and not set(policy.tab_ids).issubset({item['id'] for item in source['tabs']}):
|
||||
raise ValueError('CHART_HEALTH_EVALUATION_TAB_MISSING')
|
||||
collector = ChartResponseCollector(page,[item['chart_id'] for item in source['placements']])
|
||||
collector.start()
|
||||
while journal.frontier()['next_ordinal'] <= len(source['placements']):
|
||||
remaining = min(end-monotonic(),journal.remaining_seconds())
|
||||
if remaining <= 0:
|
||||
raise ValueError('CHART_HEALTH_WHOLE_TIMEOUT')
|
||||
placement = source['placements'][journal.frontier()['next_ordinal']-1]
|
||||
timeout = min(limits.per_chart_timeout_seconds,remaining)
|
||||
value = await bounded_chart(page,source,placement,collector,journal,timeout)
|
||||
journal.append_chart(value)
|
||||
next_index = journal.frontier()['next_ordinal']-1
|
||||
tab_done = next_index == len(source['placements']) or source['placements'][next_index]['tab_path'] != placement['tab_path']
|
||||
if limits.per_tab_evaluation is not None and tab_done:
|
||||
await evaluate_current_tab(page,source,placement['tab_path'],journal,limits,end)
|
||||
|
||||
if await fetch_health_manifest(page,dashboard_id,getattr(limits,'chart_ids',None)) != source:
|
||||
raise ValueError('CHART_HEALTH_MANIFEST_CHANGED')
|
||||
return journal.finish('passed','CHART_HEALTH_COMPLETE')
|
||||
except asyncio.CancelledError:
|
||||
journal.finish('inconclusive','CHART_HEALTH_INTERRUPTED')
|
||||
raise
|
||||
except Exception as exc:
|
||||
code = str(exc) if str(exc).startswith(('CHART_','BROWSER_')) else 'CHART_HEALTH_UNAVAILABLE'
|
||||
return journal.finish('inconclusive',code)
|
||||
finally:
|
||||
if collector is not None:
|
||||
await collector.close()
|
||||
# #endregion ScenarioExecution.ChartHealth.Sweep.Run
|
||||
# #endregion ScenarioExecution.ChartHealth.Sweep
|
||||
@@ -5,7 +5,7 @@ from hashlib import sha256
|
||||
import json
|
||||
from .browser_traversal_inputs import parse_traversal_input
|
||||
|
||||
TRAVERSAL_ACTIONS = frozenset({'pagination','navigate_tabs'})
|
||||
TRAVERSAL_ACTIONS = frozenset({'pagination','navigate_tabs','assert_chart_health'})
|
||||
|
||||
|
||||
# #region ScenarioExecution.Traversal.PinnedInputs.Projection [C:3] [TYPE Function]
|
||||
@@ -44,9 +44,12 @@ def resolve_pinned_browser_inputs(step):
|
||||
raise ValueError('BROWSER_TRAVERSAL_AUTHORITY_MISSING')
|
||||
validate_traversal_projection(step,run)
|
||||
inputs = (step.get('step_meta') or {}).get('action_inputs') or {}
|
||||
if (step['action'] == 'pagination' and run.runner_plan.get('action_registry_version') == '038.7.0'
|
||||
if (step['action'] == 'pagination' and run.runner_plan.get('action_registry_version') in {'038.7.0','038.8.0'}
|
||||
and 'selection' not in inputs):
|
||||
raise ValueError('BROWSER_TRAVERSAL_SELECTION_REQUIRED')
|
||||
return deepcopy(parse_traversal_input(step['action'],inputs))
|
||||
if (step['action'] == 'navigate_tabs' and run.runner_plan.get('action_registry_version') == '038.8.0'
|
||||
and inputs.get('chart_health') is not True):
|
||||
raise ValueError('CHART_HEALTH_INPUT_REQUIRED')
|
||||
return deepcopy(parse_traversal_input(step['action'],inputs,registry_version=run.runner_plan.get('action_registry_version')))
|
||||
# #endregion ScenarioExecution.Traversal.PinnedInputs.Resolve
|
||||
# #endregion ScenarioExecution.Traversal.PinnedInputs
|
||||
|
||||
@@ -86,7 +86,7 @@ def _browser_action_provider(context: seam.Any, *, transport, storage, event_loo
|
||||
seam.logger.explore("Browser capacity unavailable", src=seam._SRC, payload={"run_id": run_id}, error=str(exc))
|
||||
return seam.LiveAdapterResult(status="inconclusive", reason_code="BROWSER_CAPACITY_UNAVAILABLE")
|
||||
|
||||
if action in {'pagination', 'navigate_tabs'}:
|
||||
if action in {'pagination', 'navigate_tabs', 'assert_chart_health'}:
|
||||
from .browser_traversal_runtime import execute_traversal
|
||||
return execute_traversal(step=context.step,admission=admission,storage=storage,
|
||||
capacity_lease_id=lease_id,event_loop=event_loop,transport=transport,session_manager=session_manager)
|
||||
|
||||
@@ -60,7 +60,7 @@ _READ_ONLY_ACTIONS = frozenset({
|
||||
"navigate_tab", "inspect_filter_state", "apply_table_filter", "extract_table",
|
||||
"scroll_to", "inspect_columns", "click", "select_rows", "download",
|
||||
# Wave-2 observe drivers (038.5.0, AGSCN-FR-024):
|
||||
"assert_dom", "inspect_filter_options", "navigate_tabs", "pagination", "wait_for_selector",
|
||||
"assert_dom", "assert_chart_health", "inspect_filter_options", "navigate_tabs", "pagination", "wait_for_selector",
|
||||
})
|
||||
_MUTATION_ACTIONS = frozenset({"row_edit", "bulk_edit"})
|
||||
_ALLOWED_WAIT_STATES = frozenset({"load", "domcontentloaded", "networkidle"})
|
||||
@@ -253,7 +253,7 @@ async def _execute_on_page(
|
||||
) -> BrowserTransportOutcome:
|
||||
if action not in _READ_ONLY_ACTIONS and action not in _MUTATION_ACTIONS:
|
||||
raise BrowserTransportUnsupported("BROWSER_ACTION_NOT_SUPPORTED")
|
||||
if action in {'pagination', 'navigate_tabs'}:
|
||||
if action in {'pagination', 'navigate_tabs', 'assert_chart_health'}:
|
||||
from .browser_traversal_transport import run_traversal_transport
|
||||
return await run_traversal_transport(page,action,action_input,session_handle)
|
||||
checkpoints: list[str] = ["dashboard_open"]
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
from typing import Annotated, Literal
|
||||
from pydantic import BaseModel, ConfigDict, Field
|
||||
from .browser_sampling_inputs import SelectionPolicy, FullSelection
|
||||
from .browser_health_evaluation_inputs import HealthEvaluationPolicy
|
||||
|
||||
Selector = Annotated[str, Field(min_length=1, max_length=500)]
|
||||
|
||||
@@ -41,16 +42,43 @@ class PaginationTraversalInput(TraversalBudget):
|
||||
# @BRIEF Full server-manifest sweep; expected IDs never come from callers.
|
||||
class AllTabsTraversalInput(BaseModel):
|
||||
model_config = ConfigDict(extra='forbid', strict=True)
|
||||
chart_health: bool = False
|
||||
per_tab_evaluation: HealthEvaluationPolicy | None = None
|
||||
per_chart_timeout_seconds: int = Field(default=70, ge=1, le=90)
|
||||
per_tab_timeout_seconds: int = Field(default=70, ge=1, le=90)
|
||||
whole_timeout_seconds: int = Field(default=7200, ge=1, le=21600)
|
||||
max_bytes: int = Field(default=268435456, ge=1, le=1073741824)
|
||||
# #endregion ScenarioExecution.Traversal.Inputs.Tabs
|
||||
|
||||
|
||||
# #region ScenarioExecution.Traversal.Inputs.LegacyTabs [C:1] [TYPE Class]
|
||||
# @BRIEF Frozen0386/0387 normalized tab inputs; additive chart-health fields do not enter legacy hashes or channels.
|
||||
class LegacyTabsTraversalInput(BaseModel):
|
||||
model_config = ConfigDict(extra='forbid', strict=True)
|
||||
per_tab_timeout_seconds: int = Field(default=70, ge=1, le=90)
|
||||
whole_timeout_seconds: int = Field(default=7200, ge=1, le=21600)
|
||||
max_bytes: int = Field(default=268435456, ge=1, le=1073741824)
|
||||
# #endregion ScenarioExecution.Traversal.Inputs.LegacyTabs
|
||||
|
||||
|
||||
# #region ScenarioExecution.ChartHealth.Inputs [C:2] [TYPE Class]
|
||||
# @BRIEF Closed structural health observation; source counts, verdicts and evidence are never caller inputs.
|
||||
class ChartHealthInput(BaseModel):
|
||||
model_config = ConfigDict(extra='forbid', strict=True)
|
||||
chart_ids: list[Annotated[int, Field(gt=0)]] | None = Field(default=None, min_length=1, max_length=1000)
|
||||
per_tab_evaluation: HealthEvaluationPolicy | None = None
|
||||
per_chart_timeout_seconds: int = Field(default=70, ge=1, le=90)
|
||||
whole_timeout_seconds: int = Field(default=7200, ge=1, le=21600)
|
||||
max_bytes: int = Field(default=268435456, ge=1, le=1073741824)
|
||||
# #endregion ScenarioExecution.ChartHealth.Inputs
|
||||
|
||||
|
||||
# #region ScenarioExecution.Traversal.Inputs.Parse [C:2] [TYPE Function]
|
||||
# @BRIEF Parse the closed generic traversal action contract or reject unknown actions.
|
||||
def parse_traversal_input(action: Literal['pagination', 'navigate_tabs'], inputs: dict) -> dict:
|
||||
model = {'pagination': PaginationTraversalInput, 'navigate_tabs': AllTabsTraversalInput}.get(action)
|
||||
def parse_traversal_input(action: Literal['pagination', 'navigate_tabs', 'assert_chart_health'], inputs: dict, *, registry_version=None) -> dict:
|
||||
if action == 'navigate_tabs' and registry_version in {'038.6.0','038.7.0'}:
|
||||
return LegacyTabsTraversalInput.model_validate(inputs).model_dump()
|
||||
model = {'pagination': PaginationTraversalInput, 'navigate_tabs': AllTabsTraversalInput, 'assert_chart_health':ChartHealthInput}.get(action)
|
||||
if model is None:
|
||||
raise ValueError('BROWSER_TRAVERSAL_ACTION_UNSUPPORTED')
|
||||
return model.model_validate(inputs).model_dump()
|
||||
|
||||
@@ -5,7 +5,7 @@ import asyncio
|
||||
from ..live_adapter import LiveAdapterResult
|
||||
from ..traversal_store import TraversalJournal
|
||||
from ..traversal_tabs_store import TabsJournal
|
||||
from .browser_traversal_inputs import PaginationTraversalInput, AllTabsTraversalInput
|
||||
from .browser_traversal_inputs import PaginationTraversalInput, AllTabsTraversalInput, ChartHealthInput
|
||||
|
||||
|
||||
# #region ScenarioExecution.Traversal.Runtime.Monitor [C:3] [TYPE Function]
|
||||
@@ -33,9 +33,11 @@ async def monitor_traversal(factory, journal):
|
||||
# @POST Only owned full completeness or verified selected-page completeness can yield passed; coverage scope remains explicit.
|
||||
def execute_traversal(*, step, admission, storage, capacity_lease_id, event_loop, transport, session_manager):
|
||||
action, run_id, binding = admission['action'], admission['run_id'], admission['binding']
|
||||
model = PaginationTraversalInput if action == 'pagination' else AllTabsTraversalInput
|
||||
model = {'pagination':PaginationTraversalInput,'navigate_tabs':AllTabsTraversalInput,'assert_chart_health':ChartHealthInput}[action]
|
||||
limits = model.model_validate(admission['action_inputs'])
|
||||
journal_type = TraversalJournal if action == 'pagination' else TabsJournal
|
||||
from ..chart_health_store import ChartHealthJournal
|
||||
health = action == 'assert_chart_health' or getattr(limits,'chart_health',False)
|
||||
journal_type = ChartHealthJournal if health else (TraversalJournal if action == 'pagination' else TabsJournal)
|
||||
journal = journal_type(step,storage,limits,capacity_lease_id=capacity_lease_id)
|
||||
runtime = {'journal':journal,'limits':limits,'dashboard_id':binding.dashboard_id}
|
||||
inputs = {**admission['action_inputs'],'_traversal_runtime':runtime}
|
||||
@@ -75,7 +77,7 @@ def execute_traversal(*, step, admission, storage, capacity_lease_id, event_loop
|
||||
ref = summary['manifest_ref']
|
||||
digest = summary['manifest_sha256']
|
||||
return LiveAdapterResult(status=summary['status'],reason_code=summary['reason_code'],
|
||||
details={'action':action,'sha256':digest,'traversal':summary,'checkpoints':['traversal_manifest_owned'],
|
||||
details={'action':action,'sha256':digest,'traversal':summary,**({'chart_health':summary['chart_health']} if 'chart_health' in summary else {}),'checkpoints':['traversal_manifest_owned'],
|
||||
'artifact_byte_lengths':{ref:summary['manifest_byte_length']},'artifact_content_types':{ref:'application/json'},
|
||||
**({'browser_checkpoint':checkpoint} if checkpoint else {})},
|
||||
artifact_refs=[ref],artifact_digests={ref:digest})
|
||||
|
||||
@@ -39,6 +39,9 @@ async def run_traversal_transport(page, action, inputs, session_handle=None):
|
||||
summary = await walk_pages(reader,journal,limits)
|
||||
if reader is not None:
|
||||
page = reader.page
|
||||
elif action == 'assert_chart_health' or getattr(limits,'chart_health',False):
|
||||
from .browser_health_sweep import sweep_health
|
||||
summary = await sweep_health(page,runtime['dashboard_id'],journal,limits)
|
||||
else:
|
||||
summary = await sweep_tabs(page,runtime['dashboard_id'],journal,limits)
|
||||
return BrowserTransportOutcome(checkpoints=('traversal_manifest_owned',),page_url=page.url,details={'traversal':summary})
|
||||
|
||||
@@ -18,6 +18,8 @@ from src.services.dashboard_testing.analytics.investigation import emit_terminal
|
||||
from src.services.dashboard_testing.automation.notify import persist_notification
|
||||
|
||||
from .artifacts import invalidate_step_evidence
|
||||
from .chart_health_notification import notification_health_summary
|
||||
from .chart_health_delivery_hook import arm_health_delivery
|
||||
from .baseline_resolver import BASELINE_STALE
|
||||
from .result import build_result
|
||||
|
||||
@@ -88,6 +90,22 @@ def _close_queued_dispatch_error(db: Session, run: ScenarioRun) -> dict[str, Any
|
||||
# #endregion ScenarioExecution.Runner.CloseDispatchError
|
||||
|
||||
|
||||
# #region ScenarioExecution.Runner.BaselineStaleTerminal [C:3] [TYPE Function]
|
||||
# @BRIEF Inspect persisted baseline-stale reasons without changing terminal notifications.
|
||||
def _is_baseline_stale_terminal(db: Session, run: ScenarioRun) -> bool:
|
||||
"""True when the blocked terminal run carries a BASELINE_STALE reason on the run or a step."""
|
||||
if run.error_code == BASELINE_STALE:
|
||||
return True
|
||||
for step in db.query(ScenarioStepRun).filter(ScenarioStepRun.run_id == run.id).all():
|
||||
if step.error_code == BASELINE_STALE:
|
||||
return True
|
||||
reason_codes = (step.step_outcome or {}).get("reason_codes") or []
|
||||
if BASELINE_STALE in reason_codes:
|
||||
return True
|
||||
return False
|
||||
|
||||
# #endregion ScenarioExecution.Runner.BaselineStaleTerminal
|
||||
|
||||
# #region ScenarioExecution.Runner.TerminalSignal [C:4] [TYPE Function] [SEMANTICS scenario,execution,terminal,investigation,signal,poisoned]
|
||||
# @BRIEF Project one terminal non-pass run into the canonical analyst queue without starting work.
|
||||
# @RELATION CALLS -> [ScenarioAnalytics.Investigation.TerminalSignal]
|
||||
@@ -111,31 +129,19 @@ def _close_queued_dispatch_error(db: Session, run: ScenarioRun) -> dict[str, Any
|
||||
# @REJECTED Counting the failure inside the retry dispatcher tick — several ticks observe the same
|
||||
# failed run, so tick-side counting double-accounts one failure; the terminal moment is
|
||||
# the single accounting point.
|
||||
def _is_baseline_stale_terminal(db: Session, run: ScenarioRun) -> bool:
|
||||
"""True when the blocked terminal run carries a BASELINE_STALE reason on the run or a step."""
|
||||
if run.error_code == BASELINE_STALE:
|
||||
return True
|
||||
for step in db.query(ScenarioStepRun).filter(ScenarioStepRun.run_id == run.id).all():
|
||||
if step.error_code == BASELINE_STALE:
|
||||
return True
|
||||
reason_codes = (step.step_outcome or {}).get("reason_codes") or []
|
||||
if BASELINE_STALE in reason_codes:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _record_terminal_side_effects(db: Session, run: ScenarioRun) -> None:
|
||||
if run.status in {"failed", "blocked", "inconclusive"}:
|
||||
emit_terminal_run_signal(db, run)
|
||||
baseline_stale = run.status == "blocked" and _is_baseline_stale_terminal(db, run)
|
||||
event_type = "baseline_stale" if baseline_stale else ("blocked" if run.status == "blocked" else "failed")
|
||||
persist_notification(
|
||||
receipt = persist_notification(
|
||||
db,
|
||||
event_type=event_type,
|
||||
scenario_id=run.scenario_id,
|
||||
run_id=run.id,
|
||||
severity="warning",
|
||||
payload={
|
||||
**notification_health_summary(db,run.id),
|
||||
"status": run.status,
|
||||
"environment_id": run.environment_id,
|
||||
**({"error_code": BASELINE_STALE} if baseline_stale else {}),
|
||||
@@ -143,6 +149,7 @@ def _record_terminal_side_effects(db: Session, run: ScenarioRun) -> None:
|
||||
sla_seconds=3600,
|
||||
idempotency_key=f"terminal:{event_type}:{run.id}",
|
||||
)
|
||||
arm_health_delivery(db,receipt)
|
||||
_record_poisoned_failure(db, run)
|
||||
elif run.status == "passed":
|
||||
persist_notification(
|
||||
@@ -155,6 +162,7 @@ def _record_terminal_side_effects(db: Session, run: ScenarioRun) -> None:
|
||||
)
|
||||
_reset_poisoned_streak(run)
|
||||
|
||||
# #endregion ScenarioExecution.Runner.TerminalSignal
|
||||
|
||||
# #region ScenarioExecution.Runner.PoisonedAccounting [C:3] [TYPE Function] [SEMANTICS scenario,execution,terminal,poisoned,accounting,dlq]
|
||||
# @ingroup ScenarioExecution
|
||||
@@ -223,6 +231,5 @@ def _reset_poisoned_streak(run: ScenarioRun) -> None:
|
||||
run_id=run.id,
|
||||
)
|
||||
# #endregion ScenarioExecution.Runner.PoisonedStreakReset
|
||||
# #endregion ScenarioExecution.Runner.TerminalSignal
|
||||
|
||||
# #endregion ScenarioExecution.TerminalEffects
|
||||
|
||||
@@ -66,6 +66,9 @@ class TraversalJournal:
|
||||
input_value = limits.model_dump()
|
||||
if (step.get('step_meta') or {}).get('action_inputs',{}).get('selection') is None:
|
||||
input_value.pop('selection',None)
|
||||
for field in ('chart_health','per_chart_timeout_seconds','per_tab_evaluation'):
|
||||
if field not in (step.get('step_meta') or {}).get('action_inputs',{}):
|
||||
input_value.pop(field,None)
|
||||
input_digest = sha256(canonical(input_value)).hexdigest()
|
||||
row = db.query(ScenarioTraversal).filter_by(run_id=self.run_id,logical_step_id=self.logical_step_id,attempt=self.attempt).one_or_none()
|
||||
if row is None:
|
||||
|
||||
@@ -8,27 +8,27 @@ from __future__ import annotations
|
||||
|
||||
from datetime import UTC, datetime
|
||||
from typing import Any
|
||||
import uuid
|
||||
import uuid # noqa: F401 - walker_step_control public patch seam
|
||||
|
||||
from sqlalchemy.orm import Session
|
||||
from sqlalchemy import update
|
||||
from sqlalchemy import update # noqa: F401 - walker_step_control public patch seam
|
||||
|
||||
from src.core.logger import logger
|
||||
from src.core.logger import logger # noqa: F401 - retained walker public seam
|
||||
from src.models.scenario_run import ScenarioRun, ScenarioStepRun
|
||||
|
||||
from .agent_evaluation import AgentEvaluation, persist_agent_evaluation, validate_evaluation_evidence
|
||||
from .artifacts import register_step_evidence
|
||||
from .baseline_resolver import stamp_baseline_pin
|
||||
from .agent_evaluation import AgentEvaluation, persist_agent_evaluation, validate_evaluation_evidence # noqa: F401 - walker_publication seam
|
||||
from .artifacts import register_step_evidence # noqa: F401 - walker_publication seam
|
||||
from .baseline_resolver import stamp_baseline_pin # noqa: F401 - walker_publication seam
|
||||
from .capacity_block import CAPACITY_RETRY_CODES as _CAPACITY_RETRY_CODES
|
||||
from .capacity_block import block_run_on_capacity
|
||||
from .decision_policy import decide_step_outcome, policy_inputs_from_outcome, verified_evidence_refs
|
||||
from .decision_policy import decide_step_outcome, policy_inputs_from_outcome, verified_evidence_refs # noqa: F401 - walker_publication seam
|
||||
from .dispatch import dispatch_step
|
||||
from .executor_registry import ScenarioExecutorRegistry
|
||||
from .lifecycle import suspend_for_human
|
||||
from .lifecycle import suspend_for_human # noqa: F401 - walker_step_control seam
|
||||
from .result import build_result
|
||||
from .runner_plan import resolve_pinned_policy, validate_pinned_runner_plan
|
||||
from .runner_plan import resolve_pinned_policy, validate_pinned_runner_plan # noqa: F401 - walker_publication seam
|
||||
from .terminal_effects import _record_terminal_side_effects, _reject_malformed_plan
|
||||
from .worker import claim_step
|
||||
from .worker import claim_step # noqa: F401 - walker_step_control seam
|
||||
|
||||
|
||||
# #region ScenarioExecution.Runner.EvaluationRecord [C:2] [TYPE Function] [SEMANTICS scenario,execution,evaluation,outcome]
|
||||
@@ -228,7 +228,7 @@ def _advance_step(db, run, registry, worker_id, lease_seconds, plan, order, depe
|
||||
step, claim_failed = _claim_automated_step(db, run, step_id, order, existing, descriptor, worker_id, lease_seconds, tool)
|
||||
if claim_failed:
|
||||
return build_result(run,list(existing.values()))
|
||||
if tool == 'browser' and step_meta['action'] in {'pagination', 'navigate_tabs'}:
|
||||
if tool == 'browser' and step_meta['action'] in {'pagination', 'navigate_tabs', 'assert_chart_health'}:
|
||||
# Restricted long traversal journals renew committed worker/provider leases independently.
|
||||
db.commit()
|
||||
outcome = dispatch_step(
|
||||
@@ -249,7 +249,7 @@ def _advance_step(db, run, registry, worker_id, lease_seconds, plan, order, depe
|
||||
edges=dependencies,
|
||||
)
|
||||
db.refresh(run)
|
||||
if tool == 'browser' and step_meta['action'] in {'pagination', 'navigate_tabs'} and run.phase == 'paused':
|
||||
if tool == 'browser' and step_meta['action'] in {'pagination', 'navigate_tabs', 'assert_chart_health'} and run.phase == 'paused':
|
||||
step.status = 'queued'
|
||||
db.flush()
|
||||
return build_result(run, list(existing.values()))
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# #region BaselineEngine.QueryExecutor.Envelope [C:2] [TYPE Class] [SEMANTICS baseline,execution,envelope,immutability]
|
||||
# @ingroup BaselineEngine
|
||||
# @BRIEF Trusted execution envelope: NormalizedValue + source_response_hash.
|
||||
# @INVARIANT source_response_hash from full deterministic response bytes before extraction.
|
||||
# @INVARIANT Successful source_response_hash hashes original response bytes; error payload_kind explicitly identifies sanitized diagnostics with separate original-response provenance.
|
||||
# @DATA_CONTRACT Superset API Response -> QueryExecutionEnvelope
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -16,4 +16,6 @@ class QueryExecutionEnvelope:
|
||||
normalized_value: NormalizedValue
|
||||
source_response_hash: str = field(repr=False)
|
||||
raw_response_content: bytes = field(repr=False)
|
||||
diagnostic: dict | None = None
|
||||
payload_kind: str = "original_response"
|
||||
# #endregion BaselineEngine.QueryExecutor.Envelope
|
||||
|
||||
@@ -13,7 +13,6 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
from src.core.logger import logger
|
||||
@@ -29,7 +28,7 @@ from src.schemas.dashboard_testing import (
|
||||
ValueKind,
|
||||
Warning,
|
||||
)
|
||||
from src.services.dashboard_testing.immutability import compute_source_response_hash
|
||||
from src.services.dashboard_testing.immutability import compute_source_response_hash # noqa: F401 - retained public import seam
|
||||
from src.services.dashboard_testing.query_envelope import QueryExecutionEnvelope
|
||||
|
||||
|
||||
@@ -152,22 +151,8 @@ def _verify_authoritative_model(
|
||||
if chart.dataset_id != dataset_id:
|
||||
raise ValueError("Chart dataset differs from authoritative dashboard model")
|
||||
|
||||
# 4. Verify result_key/metrics
|
||||
if chart_id is not None:
|
||||
matching_charts = [ch for ch in query_model.charts if ch.chart_id == chart_id]
|
||||
if matching_charts:
|
||||
chart_metrics = matching_charts[0].metrics
|
||||
valid_metric_names = {m.metric_name for m in chart_metrics}
|
||||
if valid_metric_names and request.result_key not in valid_metric_names:
|
||||
logger.explore("Result key not a known metric", src="BaselineEngine.QueryExecutor.VerifyAuthoritativeModel", payload={"result_key": request.result_key, "chart_id": chart_id,
|
||||
"valid_metrics": valid_metric_names}, error="result_key is not a known metric for this chart")
|
||||
warnings.append(Warning(
|
||||
source="execution",
|
||||
resource=f"chart/{chart_id}",
|
||||
code="UNKNOWN_METRIC",
|
||||
detail=f"result_key '{request.result_key}' is not a known metric "
|
||||
f"for chart {chart_id}. Known metrics: {valid_metric_names}",
|
||||
))
|
||||
from .query_model_guard import metric_warnings
|
||||
warnings.extend(metric_warnings(request,query_model,chart_id))
|
||||
|
||||
# 5. Scope filters to the target chart
|
||||
if chart_id is not None and request.normalized_filters.filters:
|
||||
@@ -216,8 +201,8 @@ def _extract_query_result(result_data: dict, result_key: str) -> tuple[Any, str]
|
||||
# @ingroup BaselineEngine
|
||||
# @BRIEF Execute query and return trusted envelope with hash from full deterministic response bytes.
|
||||
# @PRE query_model_fingerprint matches authoritative model. chart_id/dataset_id belongs to dashboard.
|
||||
# @POST Returns QueryExecutionEnvelope with NormalizedValue + source_response_hash + raw bytes.
|
||||
# @INVARIANT source_response_hash from full response bytes before extraction, never canonical scalar.
|
||||
# @POST Returns original successful response evidence or typed sanitized error evidence, preserving explicit origin and original wire digest when available.
|
||||
# @INVARIANT Successful source_response_hash hashes full wire bytes; sanitized failure payloads hash their own bytes and retain original wire digest separately.
|
||||
# @SIDE_EFFECT Async POST to Superset /api/v1/chart/data.
|
||||
# @DATA_CONTRACT ExecuteQueryRequest + DashboardQueryModel -> QueryExecutionEnvelope
|
||||
# @RATIONALE source_response_hash computed from raw httpx bytes (via raw_response=True on execute_chart_data_raw) BEFORE extracting the canonical scalar. This ensures even Superset metadata-only response changes (query_id, colnames, column types) produce a different hash, making immutability violations detectable. The envelope returns raw_response_content bytes for DraftStorage persistence, enabling offline re-verification of the exact wire response.
|
||||
@@ -256,6 +241,7 @@ async def execute_dashboard_query_envelope(
|
||||
# Use the chart's authoritative metric definition. An ad hoc metric needs
|
||||
# its aggregate and column; its label alone is not executable in Superset.
|
||||
metric_spec: str | dict = request.result_key
|
||||
chart_model = None
|
||||
if query_model is not None and chart_id is not None:
|
||||
chart_model = next((chart for chart in query_model.charts if chart.chart_id == chart_id), None)
|
||||
descriptor = next((metric for metric in chart_model.metrics if metric.metric_name == request.result_key), None) if chart_model else None
|
||||
@@ -274,26 +260,11 @@ async def execute_dashboard_query_envelope(
|
||||
row_limit=request.max_rows,
|
||||
)
|
||||
except SupersetAPIError as e:
|
||||
logger.explore("Superset API error during execution", src="BaselineEngine.QueryExecutor.ExecuteQueryEnvelope", payload={"chart_id": chart_id, "dataset_id": dataset_id}, error=str(e))
|
||||
# On API error, envelope still contains error NormalizedValue but hash from error context
|
||||
nv = NormalizedValue(
|
||||
kind=ValueKind.UNKNOWN,
|
||||
raw_value=None,
|
||||
canonical_value=None,
|
||||
source=f"{request.environment_id}/dashboard/{request.dashboard_id}/chart/{chart_id}",
|
||||
warnings=[Warning(
|
||||
source="execution",
|
||||
resource=f"chart/{chart_id}" if chart_id else f"dataset/{dataset_id}",
|
||||
code=f"SUPERSET_{getattr(e, 'status_code', 'ERROR')}",
|
||||
detail=str(e),
|
||||
)],
|
||||
)
|
||||
error_bytes = json.dumps({"error": str(e), "status_code": getattr(e, 'status_code', 0)}, sort_keys=True).encode()
|
||||
return QueryExecutionEnvelope(
|
||||
normalized_value=nv,
|
||||
source_response_hash=compute_source_response_hash(error_bytes),
|
||||
raw_response_content=error_bytes,
|
||||
)
|
||||
logger.explore("Superset API error during execution", src="BaselineEngine.QueryExecutor.ExecuteQueryEnvelope", payload={"chart_id": chart_id, "dataset_id": dataset_id}, error="Superset query unavailable")
|
||||
from .query_failure import failure_envelope
|
||||
context = getattr(e, 'context', {})
|
||||
return failure_envelope(request, {'message':str(e)},
|
||||
http_status=context.get('status_code'), origin='synthetic_exception',chart_name=chart_model.slice_name if chart_model else None)
|
||||
|
||||
# ── Compute source_response_hash from raw httpx bytes BEFORE extraction ──
|
||||
# Uses the exact bytes received from Superset (via raw_response=True on httpx).
|
||||
@@ -305,8 +276,13 @@ async def execute_dashboard_query_envelope(
|
||||
raw_response_content: bytes = raw_response.raw_bytes
|
||||
source_response_hash: str = raw_response.source_response_hash
|
||||
result_data: dict[str, Any] = raw_response.parsed
|
||||
if "result" not in result_data and ("message" in result_data or "errors" in result_data):
|
||||
raise ValueError("Superset chart-data rejected the query")
|
||||
from .query_failure import response_error, failure_envelope
|
||||
if isinstance(result_data,dict) and result_data.get('_invalid_response'):
|
||||
return failure_envelope(request, {'message':'Superset returned a non-JSON chart response'},
|
||||
raw_bytes=raw_response_content,http_status=raw_response.http_status,confirmed=False)
|
||||
error = response_error(result_data, raw_response.http_status)
|
||||
if error is not None:
|
||||
return failure_envelope(request, error, raw_bytes=raw_response_content, http_status=raw_response.http_status,chart_name=chart_model.slice_name if chart_model else None,confirmed=not error.get("_unconfirmed",False))
|
||||
|
||||
# Extract result value from Superset response (AFTER hash computation)
|
||||
raw_value, query_id = _extract_query_result(result_data, request.result_key)
|
||||
|
||||
71
backend/src/services/dashboard_testing/query_failure.py
Normal file
71
backend/src/services/dashboard_testing/query_failure.py
Normal file
@@ -0,0 +1,71 @@
|
||||
# #region DashboardTesting.ChartHealth.QueryFailure [C:4] [TYPE Module] [SEMANTICS query,errors,redaction,evidence]
|
||||
# @BRIEF Retain bounded query failure observations without pretending reconstructed bytes are wire evidence.
|
||||
from datetime import UTC, datetime
|
||||
from hashlib import sha256
|
||||
import json
|
||||
import re
|
||||
from src.schemas.dashboard_testing import NormalizedValue, ValueKind, Warning
|
||||
from .chart_health_contract import ChartDiagnostic
|
||||
from .query_envelope import QueryExecutionEnvelope
|
||||
|
||||
|
||||
# #region DashboardTesting.ChartHealth.QueryFailure.Redact [C:2] [TYPE Function]
|
||||
# @POST Credential redaction precedes bounded diagnostics and truncation is explicit.
|
||||
def safe_message(value):
|
||||
from src.plugins.llm_analysis._redaction import RedactionService
|
||||
text = RedactionService.redact_raw_response(str(value))
|
||||
return text[:4096], len(text) > 4096
|
||||
# #endregion DashboardTesting.ChartHealth.QueryFailure.Redact
|
||||
|
||||
|
||||
# #region DashboardTesting.ChartHealth.QueryFailure.Extract [C:3] [TYPE Function]
|
||||
# @POST HTTP200 application failures and result.status=failed are errors; zero/empty successful data are not.
|
||||
def response_error(payload, http_status=200):
|
||||
if not isinstance(payload,dict):
|
||||
return {'message':'Malformed Superset chart response','_unconfirmed':True}
|
||||
records = payload.get('result', [])
|
||||
if not isinstance(records,list):
|
||||
return {'message':'Malformed Superset chart result','_unconfirmed':True}
|
||||
candidates = [payload, *(records if isinstance(records, list) else [])]
|
||||
for item in candidates:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
errors = item.get('errors')
|
||||
if errors:
|
||||
entry = errors[0] if isinstance(errors, list) else errors
|
||||
return entry if isinstance(entry, dict) else {'message':str(entry)}
|
||||
if item.get('error') or item.get('status') in {'failed', 'error'}:
|
||||
return {'message':item.get('error') or item.get('message') or 'Query failed'}
|
||||
if 'result' not in payload and payload.get('message'):
|
||||
return {'message':payload['message']}
|
||||
if http_status >= 400:
|
||||
return {'message':payload.get('message', f'HTTP {http_status}') if isinstance(payload, dict) else f'HTTP {http_status}'}
|
||||
return None
|
||||
# #endregion DashboardTesting.ChartHealth.QueryFailure.Extract
|
||||
|
||||
|
||||
# #region DashboardTesting.ChartHealth.QueryFailure.Envelope [C:4] [TYPE Function]
|
||||
# @POST Failed query details retain sanitized origin and pre-redaction wire digest when available; no raw rows/SQL reach diagnostics.
|
||||
def failure_envelope(request, error, *, raw_bytes=None, http_status=None, origin='http_response', chart_name=None, confirmed=True):
|
||||
message, truncated = safe_message(error.get('message') or error.get('error') or 'Query failed')
|
||||
db_code = re.search(r'\bCode:\s*(\d+)\b', message)
|
||||
diagnostic = ChartDiagnostic(chart_id=request.chart_id, dataset_id=request.dataset_id,
|
||||
placement_id=f'chart/{request.chart_id}' if request.chart_id else f'dataset/{request.dataset_id}',
|
||||
chart_name=safe_message(chart_name or (f'Chart {request.chart_id}' if request.chart_id else f'Dataset {request.dataset_id}'))[0][:512],
|
||||
environment_id=request.environment_id,dashboard_id=request.dashboard_id,
|
||||
status='failed' if origin == 'http_response' and confirmed else 'inconclusive',
|
||||
reason_code='SUPERSET_QUERY_FAILED' if origin == 'http_response' and confirmed else 'SUPERSET_QUERY_UNAVAILABLE',
|
||||
origin=origin, filter_fingerprint=request.normalized_filters.filters_hash,
|
||||
observed_at=datetime.now(UTC).isoformat(), duration_seconds=0.0, http_status=http_status,
|
||||
application_code=str(error.get('error_type') or error.get('code') or '')[:256] or None,
|
||||
database_code=db_code.group(1) if db_code else None, message=message,
|
||||
terminal_state='error' if origin == 'http_response' and confirmed else 'unavailable',
|
||||
original_response_sha256=sha256(raw_bytes).hexdigest() if raw_bytes is not None else None,
|
||||
original_byte_length=len(raw_bytes) if raw_bytes is not None else None, truncated=truncated)
|
||||
data = json.dumps({'chart_diagnostic':diagnostic.model_dump()},sort_keys=True,separators=(',',':')).encode()
|
||||
value = NormalizedValue(kind=ValueKind.UNKNOWN,raw_value=None,canonical_value=None,
|
||||
source=f'{request.environment_id}/dashboard/{request.dashboard_id}/chart/{request.chart_id}',
|
||||
warnings=[Warning(source='execution',resource=diagnostic.placement_id,code=diagnostic.reason_code,detail=message)])
|
||||
return QueryExecutionEnvelope(value,sha256(data).hexdigest(),data,diagnostic=diagnostic.model_dump(),payload_kind='sanitized_diagnostic')
|
||||
# #endregion DashboardTesting.ChartHealth.QueryFailure.Envelope
|
||||
# #endregion DashboardTesting.ChartHealth.QueryFailure
|
||||
30
backend/src/services/dashboard_testing/query_model_guard.py
Normal file
30
backend/src/services/dashboard_testing/query_model_guard.py
Normal file
@@ -0,0 +1,30 @@
|
||||
# #region DashboardTesting.QueryModelGuard [C:3] [TYPE Module]
|
||||
# @BRIEF Preserve authoritative metric warning behavior separately from query execution and evidence retention.
|
||||
from src.schemas.dashboard_testing import Warning
|
||||
from . import query_executor as seam
|
||||
|
||||
|
||||
# #region DashboardTesting.QueryModelGuard.MetricWarnings [C:3] [TYPE Function]
|
||||
# @POST Unknown inspected metrics retain the existing warning without authorizing another dataset/chart.
|
||||
def metric_warnings(request, query_model, chart_id):
|
||||
warnings = []
|
||||
# 4. Verify result_key/metrics
|
||||
if chart_id is not None:
|
||||
matching_charts = [ch for ch in query_model.charts if ch.chart_id == chart_id]
|
||||
if matching_charts:
|
||||
chart_metrics = matching_charts[0].metrics
|
||||
valid_metric_names = {m.metric_name for m in chart_metrics}
|
||||
if valid_metric_names and request.result_key not in valid_metric_names:
|
||||
seam.logger.explore("Result key not a known metric", src="BaselineEngine.QueryExecutor.VerifyAuthoritativeModel", payload={"result_key": request.result_key, "chart_id": chart_id,
|
||||
"valid_metrics": valid_metric_names}, error="result_key is not a known metric for this chart")
|
||||
warnings.append(Warning(
|
||||
source="execution",
|
||||
resource=f"chart/{chart_id}",
|
||||
code="UNKNOWN_METRIC",
|
||||
detail=f"result_key '{request.result_key}' is not a known metric "
|
||||
f"for chart {chart_id}. Known metrics: {valid_metric_names}",
|
||||
))
|
||||
|
||||
return warnings
|
||||
# #endregion DashboardTesting.QueryModelGuard.MetricWarnings
|
||||
# #endregion DashboardTesting.QueryModelGuard
|
||||
@@ -145,6 +145,8 @@ def _selector_hint_param(missing_step: str) -> ScenarioParameter:
|
||||
# #endregion ScenarioGraph.Compiler.SelectorHintParam
|
||||
|
||||
|
||||
# #region ScenarioGraph.Compiler.Own.compile_scenario [C:3] [TYPE Function]
|
||||
# @BRIEF Compile validated scenario steps into the runner plan.
|
||||
def compile_scenario(req: CompileScenarioRequest) -> CompiledResult:
|
||||
"""Deterministically compile a dashboard goal into a stable ScenarioGraph DAG."""
|
||||
logger.reason("Compiling scenario graph", src="ScenarioGraph.Compiler.Compile", payload={"dashboard_id": req.dashboard_id, "cases": req.objective.get("selected_case_ids")})
|
||||
@@ -154,6 +156,7 @@ def compile_scenario(req: CompileScenarioRequest) -> CompiledResult:
|
||||
except Exception as e:
|
||||
logger.explore("Compile failed; no graph produced", src="ScenarioGraph.Compiler.Compile", error=str(e))
|
||||
raise
|
||||
# #endregion ScenarioGraph.Compiler.Own.compile_scenario
|
||||
|
||||
|
||||
# #region ScenarioGraph.Compiler.CompileImpl [C:4] [TYPE Function] [SEMANTICS scenario,compiler,deterministic]
|
||||
@@ -315,6 +318,7 @@ def _build_step(
|
||||
depends_on=list(depends_on or []),
|
||||
automation_status=automation,
|
||||
checklist_case_ids=[case_id],
|
||||
action_inputs={"chart_health":True} if action == "navigate_tabs" else None,
|
||||
risk=entry["risk"],
|
||||
agent_evaluation_spec=evaluation_spec if action == "evaluate_declared_spec" else None,
|
||||
)
|
||||
|
||||
@@ -72,7 +72,7 @@ class AgentEvaluationSpec(BaseModel):
|
||||
prompt_template_version: str = Field(min_length=1)
|
||||
prompt_template_hash: str = Field(pattern=_SHA256_RE)
|
||||
evidence_refs: list[str] = Field(min_length=1, max_length=100)
|
||||
comparison_refs: list[str] = Field(min_length=1, max_length=100)
|
||||
comparison_refs: list[str] = Field(default_factory=list, max_length=100)
|
||||
tool_allowlist: list[str] = Field(default_factory=list, max_length=0)
|
||||
output_schema: Literal["agent-evaluation.schema.json"]
|
||||
decision_policy: DecisionPolicy
|
||||
|
||||
@@ -315,7 +315,7 @@ def _check_field(key: str, value: Any, spec: InputField, findings: list[StepInpu
|
||||
def validate_step_inputs(action: Any, inputs: Any) -> StepInputsValidation:
|
||||
if not isinstance(inputs, dict):
|
||||
return StepInputsValidation(False, [StepInputsFinding("INVALID_STEP_INPUT_TYPE", "inputs must be a mapping")])
|
||||
if action in {'pagination', 'navigate_tabs'}:
|
||||
if action in {'pagination', 'navigate_tabs', 'assert_chart_health'}:
|
||||
from src.services.dashboard_testing.execution.providers.browser_traversal_inputs import parse_traversal_input
|
||||
try:
|
||||
parse_traversal_input(action, inputs)
|
||||
@@ -353,6 +353,8 @@ def assert_step_inputs(action: Any, inputs: Any) -> dict[str, Any]:
|
||||
first = result.findings[0]
|
||||
raise ValueError(f"{first.code}: {first.message}")
|
||||
accepted = dict(inputs)
|
||||
if action == 'navigate_tabs':
|
||||
accepted.setdefault('chart_health',True)
|
||||
if action == 'pagination':
|
||||
accepted.setdefault('selection',{'mode':'quantiles','count':5})
|
||||
return accepted
|
||||
|
||||
@@ -16,7 +16,7 @@ from typing import Any
|
||||
PHASES = ("setup", "interact", "observe", "assert", "evidence", "report")
|
||||
TOOLS = ("browser", "superset_api", "sql_evidence", "transform", "xlsx", "assertion", "screenshot", "report", "artifact", "human", "agent_evaluation")
|
||||
RISKS = ("read", "browser_interaction", "draft_write", "human")
|
||||
ACTION_REGISTRY_VERSION = "038.7.0" # explicit owned page sampling; archived0386 missing selection remains full
|
||||
ACTION_REGISTRY_VERSION = "038.8.0" # explicit chart-health ownership; archived sampling/registry snapshots remain unchanged
|
||||
|
||||
# Registry discipline (AGSCN-FR-025): every registered action MUST be implemented with a canaried
|
||||
# driver OR carry an explicit `disabled` reason. Compile and RunnerPlan derivation reject a disabled
|
||||
@@ -35,6 +35,7 @@ REGISTERED_ACTIONS: dict[str, dict[str, Any]] = {
|
||||
# Wave-2 observe drivers implemented (browser_readonly_flows_observe.py); AGSCN-FR-025.
|
||||
"assert_dom": {"tool": "browser", "phase": "observe", "risk": "read", "capability": "browser"},
|
||||
"inspect_filter_options": {"tool": "browser", "phase": "observe", "risk": "read", "capability": "browser"},
|
||||
"assert_chart_health": {"tool": "browser", "phase": "observe", "risk": "read", "capability": "browser"},
|
||||
"navigate_tabs": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "browser"},
|
||||
"wait_for_selector": {"tool": "browser", "phase": "observe", "risk": "read", "capability": "browser"},
|
||||
"extract_table": {"tool": "browser", "phase": "observe", "risk": "read", "capability": "browser"},
|
||||
@@ -124,6 +125,7 @@ _TIMEOUTS = {
|
||||
# Restricted traversal owns its shorter persisted whole/page budgets and renewable leases.
|
||||
"pagination": 21725000,
|
||||
"navigate_tabs": 21725000,
|
||||
"assert_chart_health": 21725000,
|
||||
"download": 60000,
|
||||
"download_xlsx": 60000,
|
||||
"capture_screenshot": 30000,
|
||||
@@ -211,6 +213,7 @@ STEP_TEMPLATES: dict[str, tuple[str, str, str]] = {
|
||||
"browser_text_filter_assert": ("apply_native_filter", "browser", "interact"),
|
||||
"browser_table_filter_assert": ("apply_table_filter", "browser", "interact"),
|
||||
"browser_pagination_assert": ("pagination", "browser", "interact"),
|
||||
"browser_chart_health_assert": ("assert_chart_health", "browser", "observe"),
|
||||
"browser_tabs_sweep": ("navigate_tabs", "browser", "interact"),
|
||||
"browser_edit_refresh_evidence": ("edit_row", "browser", "interact"),
|
||||
"browser_bulk_edit_evidence": ("bulk_edit", "browser", "interact"),
|
||||
|
||||
@@ -39,14 +39,22 @@ class ScenarioValidationResult:
|
||||
graph_hash: str = ""
|
||||
|
||||
|
||||
# #region ScenarioGraph.Validator.Own._err [C:1] [TYPE Function]
|
||||
# @BRIEF Construct a scenario validation error.
|
||||
def _err(code: str, msg: str, *, step_id: str | None = None, recovery: list[str] | None = None) -> Finding:
|
||||
return Finding(code=code, severity="error", message=msg, step_id=step_id, recovery_options=recovery or [])
|
||||
# #endregion ScenarioGraph.Validator.Own._err
|
||||
|
||||
|
||||
# #region ScenarioGraph.Validator.Own._warn [C:1] [TYPE Function]
|
||||
# @BRIEF Construct a scenario validation warning.
|
||||
def _warn(code: str, msg: str, *, step_id: str | None = None) -> Finding:
|
||||
return Finding(code=code, severity="warning", message=msg, step_id=step_id)
|
||||
# #endregion ScenarioGraph.Validator.Own._warn
|
||||
|
||||
|
||||
# #region ScenarioGraph.Validator.Own._find_cycles [C:3] [TYPE Function]
|
||||
# @BRIEF Find dependency cycles in the scenario graph.
|
||||
def _find_cycles(steps: list[dict[str, Any]]) -> list[str]:
|
||||
"""Return one representative cycle path per detected cycle (deterministic order)."""
|
||||
ids = {s["id"] for s in steps}
|
||||
@@ -55,6 +63,8 @@ def _find_cycles(steps: list[dict[str, Any]]) -> list[str]:
|
||||
stack: list[str] = []
|
||||
cycles: list[str] = []
|
||||
|
||||
# #region ScenarioGraph.Validator.Own.visit [C:3] [TYPE Function]
|
||||
# @BRIEF Traverse dependency links while tracking active nodes.
|
||||
def visit(node: str) -> None:
|
||||
if node in stack:
|
||||
idx = stack.index(node)
|
||||
@@ -67,14 +77,19 @@ def _find_cycles(steps: list[dict[str, Any]]) -> list[str]:
|
||||
for d in sorted(deps.get(node, [])):
|
||||
visit(d)
|
||||
stack.pop()
|
||||
# #endregion ScenarioGraph.Validator.Own.visit
|
||||
|
||||
for node in sorted(ids):
|
||||
visit(node)
|
||||
return sorted(set(cycles))
|
||||
# #endregion ScenarioGraph.Validator.Own._find_cycles
|
||||
|
||||
|
||||
# #region ScenarioGraph.Validator.Own._detect_sql [C:1] [TYPE Function]
|
||||
# @BRIEF Detect SQL-like content in nested inputs.
|
||||
def _detect_sql(text: str | None) -> bool:
|
||||
return contains_sql_statement(text)
|
||||
# #endregion ScenarioGraph.Validator.Own._detect_sql
|
||||
|
||||
|
||||
# #region ScenarioGraph.Validator.ValidateCore [C:4] [TYPE Function] [SEMANTICS scenario,validator,checks]
|
||||
@@ -211,7 +226,7 @@ def _check_step_inputs(steps: list[dict[str, Any]], result: ScenarioValidationRe
|
||||
|
||||
for s in steps:
|
||||
action_inputs = s.get("action_inputs")
|
||||
if s.get('action') in {'pagination', 'navigate_tabs'}:
|
||||
if s.get('action') in {'pagination', 'navigate_tabs', 'assert_chart_health'}:
|
||||
action_inputs = {} if action_inputs is None else action_inputs
|
||||
elif not isinstance(action_inputs, dict) or not action_inputs:
|
||||
continue
|
||||
@@ -253,6 +268,8 @@ def _check_safety(steps: list[dict[str, Any]], result: ScenarioValidationResult)
|
||||
def _check_dashboard_context(scenario: DashboardTestScenario, result: ScenarioValidationResult) -> None:
|
||||
max_findings = 20
|
||||
|
||||
# #region ScenarioGraph.Validator.Own.walk [C:3] [TYPE Function]
|
||||
# @BRIEF Inspect nested values for forbidden dashboard identity fields.
|
||||
def walk(node: Any, path: str, depth: int, server_query: bool = False) -> None:
|
||||
if len(result.errors) >= max_findings or depth > 24:
|
||||
return
|
||||
@@ -269,6 +286,7 @@ def _check_dashboard_context(scenario: DashboardTestScenario, result: ScenarioVa
|
||||
walk(item, f"{path}[{index}]", depth + 1, server_query)
|
||||
elif isinstance(node, str) and not server_query and _detect_sql(node):
|
||||
result.errors.append(_err("FORBIDDEN_SQL", f"dashboard_context contains SQL text at {path}"))
|
||||
# #endregion ScenarioGraph.Validator.Own.walk
|
||||
|
||||
walk(scenario.dashboard_context, "dashboard_context", 0)
|
||||
# #endregion ScenarioGraph.Validator.CheckDashboardContext
|
||||
@@ -359,6 +377,8 @@ def _topological_order(steps: list[dict[str, Any]]) -> list[str]:
|
||||
# #endregion ScenarioGraph.Validator.TopologicalOrder
|
||||
|
||||
|
||||
# #region ScenarioGraph.Validator.Own.validate_scenario [C:3] [TYPE Function]
|
||||
# @BRIEF Validate a scenario using the composed graph checks.
|
||||
def validate_scenario(scenario: DashboardTestScenario, *, metric_admission=None) -> ScenarioValidationResult:
|
||||
"""Validate a candidate graph and return all deterministic findings."""
|
||||
logger.reason("Validating scenario graph", src="ScenarioGraph.Validator.Validate", payload={"scenario_id": scenario.scenario_id, "steps": len(scenario.steps)})
|
||||
@@ -370,4 +390,5 @@ def validate_scenario(scenario: DashboardTestScenario, *, metric_admission=None)
|
||||
raise
|
||||
logger.reflect("Validation complete", src="ScenarioGraph.Validator.Validate", payload={"valid": result.valid, "errors": len(result.errors), "warnings": len(result.warnings)})
|
||||
return result
|
||||
# #endregion ScenarioGraph.Validator.Own.validate_scenario
|
||||
# #endregion ScenarioGraph.Validator.Validate
|
||||
|
||||
101
backend/tests/integration/test_chart_health_capacity_pin_pg.py
Normal file
101
backend/tests/integration/test_chart_health_capacity_pin_pg.py
Normal file
@@ -0,0 +1,101 @@
|
||||
# #region Test.ChartHealth.CapacityPostgres [C:2] [TYPE Module] [SEMANTICS health,postgres,capacity,provider,pin]
|
||||
# @TEST_INVARIANT An authored78character pin admits an actual64character PostgreSQL lease before SDK work and releases it after completion.
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.AgentEvaluation.Submit]
|
||||
import asyncio
|
||||
import os
|
||||
from pathlib import Path
|
||||
import subprocess
|
||||
import sys
|
||||
from types import SimpleNamespace
|
||||
import pytest
|
||||
from sqlalchemy import create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from owned_text_recipe_fixture import owned_recipe_fixture
|
||||
from src.models.provider_capacity import CapacityLease
|
||||
from src.services.llm_provider import LLMProviderService
|
||||
from src.services.dashboard_testing.scenario.models import AgentEvaluationSpec
|
||||
from src.services.dashboard_testing.scenario.metric_evaluation_provider import provider_public_config_digest
|
||||
from src.services.dashboard_testing.execution.agent_evaluation import submit_evaluation
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_provider import admit_health_provider
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_llm_client import BoundedHealthClient
|
||||
|
||||
pytestmark = pytest.mark.integration
|
||||
|
||||
# #region Test.ChartHealth.CapacityPostgres.Context [C:1] [TYPE Function]
|
||||
@pytest.fixture
|
||||
def committed_health_source(db_factory, tmp_path, monkeypatch):
|
||||
isolated = db_factory['create_db']('_health_pin')
|
||||
environment = dict(os.environ, DATABASE_URL=isolated['host_url'])
|
||||
subprocess.run([sys.executable, '-m', 'alembic', 'upgrade', 'head'],
|
||||
cwd=Path(__file__).resolve().parents[2], env=environment, check=True, capture_output=True)
|
||||
engine = create_engine(isolated['host_url'])
|
||||
sessions = sessionmaker(bind=engine, autoflush=False)
|
||||
db = sessions()
|
||||
try:
|
||||
# Shared genuine run/publication infrastructure only; the health spec below is independent.
|
||||
source = owned_recipe_fixture(db, tmp_path, monkeypatch)
|
||||
source.provider.api_key = LLMProviderService(db).encryption.encrypt('fixture-key-never-transmitted')
|
||||
source.provider.is_multimodal = True
|
||||
source.provider.max_images = 2
|
||||
db.commit()
|
||||
yield source, sessions
|
||||
finally:
|
||||
db.rollback()
|
||||
db.close()
|
||||
engine.dispose()
|
||||
db_factory['drop_db'](isolated['db_name'])
|
||||
# #endregion Test.ChartHealth.CapacityPostgres.Context
|
||||
|
||||
# #region Test.ChartHealth.CapacityPostgres.Submit [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT Independent connection sees a committed claimed bare digest before the single physical SDK call; final lease is released, authored spec unchanged.
|
||||
def test_postgres_health_authored_pin_capacity_width_and_release(committed_health_source, monkeypatch):
|
||||
source, sessions = committed_health_source
|
||||
fields = source.spec.model_dump(mode='json')
|
||||
fields.update(provider_version='config_sha256:' + provider_public_config_digest(source.provider),
|
||||
model_id=source.provider.default_model, model_version='configured-model:' + source.provider.default_model,
|
||||
comparison_refs=[], criteria=[{'criterion_id': 'health-visual', 'criterion_kind': 'semantic',
|
||||
'description': 'Describe actual chart errors', 'comparison_id': None}],
|
||||
limits={'timeout_ms': 1500, 'max_images': 2, 'max_input_tokens': 8192,
|
||||
'max_output_tokens': 128, 'max_cost': '1.00', 'currency': 'USD'})
|
||||
spec = AgentEvaluationSpec.model_validate(fields)
|
||||
assert len(spec.provider_version) == 78
|
||||
assert admit_health_provider(source.db, SimpleNamespace(spec=spec)) is source.provider
|
||||
observed = []
|
||||
# #region Test.ChartHealth.CapacityPostgres.Submit.Sdk [C:1] [TYPE Class]
|
||||
class ExternalSdk:
|
||||
# #region Test.ChartHealth.CapacityPostgres.Submit.Sdk.Init [C:1] [TYPE Function]
|
||||
def __init__(self, *_args):
|
||||
self.client = self.chat = self.completions = self
|
||||
# #endregion Test.ChartHealth.CapacityPostgres.Submit.Sdk.Init
|
||||
# #region Test.ChartHealth.CapacityPostgres.Submit.Sdk.Options [C:1] [TYPE Function]
|
||||
def with_options(self, **options):
|
||||
assert options == {'max_retries': 0, 'timeout': 1.5}
|
||||
return self
|
||||
# #endregion Test.ChartHealth.CapacityPostgres.Submit.Sdk.Options
|
||||
# #region Test.ChartHealth.CapacityPostgres.Submit.Sdk.Create [C:1] [TYPE Function]
|
||||
async def create(self, **_options):
|
||||
with sessions() as independent:
|
||||
leases = independent.query(CapacityLease).filter_by(run_id=source.run.id, workload_class='agent_evaluation').all()
|
||||
assert len(leases) == 1 and leases[0].status == 'claimed'
|
||||
assert leases[0].provider_version == spec.provider_version[14:]
|
||||
assert len(leases[0].provider_version) == 64
|
||||
observed.append('one-physical-request')
|
||||
return SimpleNamespace(choices=[SimpleNamespace(finish_reason='stop', message=SimpleNamespace(content='{"verdict":"inconclusive"}'))], usage=None)
|
||||
# #endregion Test.ChartHealth.CapacityPostgres.Submit.Sdk.Create
|
||||
# #region Test.ChartHealth.CapacityPostgres.Submit.Sdk.Close [C:1] [TYPE Function]
|
||||
async def close(self):
|
||||
observed.append('closed')
|
||||
# #endregion Test.ChartHealth.CapacityPostgres.Submit.Sdk.Close
|
||||
# #endregion Test.ChartHealth.CapacityPostgres.Submit.Sdk
|
||||
monkeypatch.setattr('src.services.dashboard_testing.execution.providers.browser_health_llm_client.LLMClient', ExternalSdk)
|
||||
result = asyncio.run(submit_evaluation(source.db, spec=spec, prompt='actual Code: 386 diagnostic', images=[],
|
||||
environment_id='full-flow-preprod', environment_class='DEV', run_id=source.run.id,
|
||||
logical_step_id='health-check', client=BoundedHealthClient(source.db, source.provider, spec)))
|
||||
source.db.commit()
|
||||
assert result == {'verdict': 'inconclusive'} and observed == ['one-physical-request', 'closed']
|
||||
with sessions() as independent:
|
||||
lease = independent.query(CapacityLease).filter_by(run_id=source.run.id, workload_class='agent_evaluation').one()
|
||||
assert lease.status == 'released' and lease.provider_version == spec.provider_version[14:]
|
||||
assert spec.provider_version.startswith('config_sha256:') and len(spec.provider_version) == 78
|
||||
# #endregion Test.ChartHealth.CapacityPostgres.Submit
|
||||
# #endregion Test.ChartHealth.CapacityPostgres
|
||||
@@ -0,0 +1,155 @@
|
||||
# #region Test.ChartHealth.ConnectedWorker [C:4] [TYPE Module] [SEMANTICS health,real-journal,browser,errors]
|
||||
# @TEST_INVARIANT Five query errors remain named FAILED while another tab is checked; coverage is independently complete.
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Sweep]
|
||||
import json
|
||||
from copy import deepcopy
|
||||
from hashlib import sha256
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
from threading import Thread
|
||||
|
||||
import pytest
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from src.models.scenario_run import ScenarioRun, ScenarioStepRun
|
||||
from src.models.scenario_artifact import ScenarioArtifact
|
||||
from src.services.agent_runs.artifacts import DraftStorage
|
||||
from src.services.dashboard_testing.execution.capacity import claim_capacity
|
||||
from src.services.dashboard_testing.execution.worker import claim_step
|
||||
from src.services.dashboard_testing.execution.chart_health_store import ChartHealthJournal
|
||||
from src.services.dashboard_testing.execution.providers.browser_traversal_inputs import ChartHealthInput
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_sweep import sweep_health
|
||||
|
||||
NAMES = ['Debt','Cash','Margin','Forecast','Reserve','Empty healthy']
|
||||
LAYOUT = {'ROOT_ID':{'type':'ROOT','children':['TABS-main']},
|
||||
'TABS-main':{'type':'TABS','children':['TAB-daily','TAB-other']},
|
||||
'TAB-daily':{'type':'TAB','meta':{'text':'Daily summary'},'children':['CHART-1','CHART-2','CHART-3','CHART-4','CHART-5']},
|
||||
'TAB-other':{'type':'TAB','meta':{'text':'Other'},'children':['CHART-6']},
|
||||
**{f'CHART-{index}':{'type':'CHART','children':[],'meta':{'chartId':index,'sliceName':name}} for index,name in enumerate(NAMES,1)}}
|
||||
HTML = '''<button role="tab" id="TABS-main-tab-TAB-daily" aria-controls="daily" aria-selected="false" onclick="activate('daily',[1,2,3,4,5],this)">Daily summary</button>
|
||||
<button role="tab" id="TABS-main-tab-TAB-other" aria-controls="other" aria-selected="false" onclick="activate('other',[6],this)">Other</button>
|
||||
<div role="tabpanel" id="daily" hidden>''' + ''.join(f'<div id="chart-id-{index}" style="min-height:120px"><div class="loading">loading</div></div>' for index in range(1,6)) + '''</div>
|
||||
<div role="tabpanel" id="other" hidden><div id="chart-id-6" style="min-height:120px"><div class="loading">loading</div></div></div>
|
||||
<script>function activate(id, charts, control) {
|
||||
document.querySelectorAll('[role=tab]').forEach(t=>t.setAttribute('aria-selected','false'));
|
||||
document.querySelectorAll('[role=tabpanel]').forEach(t=>t.hidden=true);
|
||||
control.setAttribute('aria-selected','true'); document.getElementById(id).hidden=false;
|
||||
charts.forEach(chart => fetch('/api/v1/chart/data', {method:'POST', headers:{'Content-Type':'application/json'},
|
||||
body:JSON.stringify({form_data:{slice_id:chart}, datasource:{id:2,type:'table'}, queries:[{filters:[]}]})})
|
||||
.then(r=>r.json()).then(data=>{document.getElementById('chart-id-'+chart).innerHTML=chart===6?'<table><tbody></tbody></table>':'<div role="alert" class="alert-danger">'+data.errors[0].message+'</div>'}));
|
||||
}</script>'''
|
||||
|
||||
|
||||
# #region Test.ChartHealth.ConnectedWorker.Handler [C:2] [TYPE Class]
|
||||
# @BRIEF Literal external protocol server; production observer/journal remain unmocked.
|
||||
class Handler(BaseHTTPRequestHandler):
|
||||
# #region Test.ChartHealth.ConnectedWorker.Handler.Get [C:2] [TYPE Function]
|
||||
def do_GET(self):
|
||||
body = json.dumps({'result':{'id':42,'position_json':json.dumps(LAYOUT)}}).encode() if self.path.startswith('/api/v1/dashboard/') else HTML.encode()
|
||||
self.send_response(200)
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
# #endregion Test.ChartHealth.ConnectedWorker.Handler.Get
|
||||
|
||||
# #region Test.ChartHealth.ConnectedWorker.Handler.Post [C:2] [TYPE Function]
|
||||
def do_POST(self):
|
||||
payload = json.loads(self.rfile.read(int(self.headers['Content-Length'])))
|
||||
chart = payload['form_data']['slice_id']
|
||||
body = {'result':[{'data':[]}]} if chart == 6 else {'errors':[{'error_type':'GENERIC_DB_ENGINE_ERROR','message':f'Code: 386. DB::Exception: incompatible types in chart {chart}'}]}
|
||||
self.send_response(200 if chart == 6 else 500)
|
||||
self.end_headers()
|
||||
self.wfile.write(json.dumps(body).encode())
|
||||
# #endregion Test.ChartHealth.ConnectedWorker.Handler.Post
|
||||
|
||||
# #region Test.ChartHealth.ConnectedWorker.Handler.Log [C:1] [TYPE Function]
|
||||
# @BRIEF Suppress external fixture request logs, which are not evidence.
|
||||
def log_message(self, *_args):
|
||||
pass
|
||||
# #endregion Test.ChartHealth.ConnectedWorker.Handler.Log
|
||||
# #endregion Test.ChartHealth.ConnectedWorker.Handler
|
||||
|
||||
|
||||
# #region Test.ChartHealth.ConnectedWorker.Fixture [C:3] [TYPE Function]
|
||||
@pytest.fixture
|
||||
def health_context(seeded_execution,registry_engine,monkeypatch,tmp_path):
|
||||
db = seeded_execution
|
||||
run = db.get(ScenarioRun,'run-exec-0001')
|
||||
run.status,run.phase = 'running','executing'
|
||||
inputs = {'per_chart_timeout_seconds':2,'whole_timeout_seconds':30}
|
||||
meta = {'logical_step_id':'health-check','tool':'browser','action':'assert_chart_health','action_inputs':inputs}
|
||||
plan = {'scenario_revision_id':run.scenario_revision_id,'scenario_content_hash':run.scenario_content_hash,'steps':[meta]}
|
||||
plan['plan_hash'] = sha256(json.dumps(plan,sort_keys=True,separators=(',',':')).encode()).hexdigest()
|
||||
run.runner_plan = plan
|
||||
step = {'scenario_run_id':run.id,'logical_step_id':'health-check','action':'assert_chart_health','step_meta':deepcopy(meta),
|
||||
**{key:deepcopy(getattr(run,key)) for key in ('target_snapshot','execution_principal_fingerprint','live_execution_binding_ref','live_execution_binding_snapshot')}}
|
||||
db.add(ScenarioStepRun(run_id=run.id,logical_step_id='health-check',step_position=1,attempt=1,status='running'))
|
||||
claim_step(db,run.id,'health-check',worker_id='health-test',side_effect_key=None,idempotent=True,retry_safe=True,lease_seconds=120)
|
||||
cap = claim_capacity(db,environment_id='preprod',environment_class='PREPROD',workload_class='browser',provider_id='browser',run_id=run.id,logical_step_id='health-check',ttl_seconds=120)
|
||||
db.commit()
|
||||
factory = sessionmaker(bind=registry_engine)
|
||||
for name in ('traversal_store','chart_health_store','chart_health_summary','chart_health_tab_store'):
|
||||
monkeypatch.setattr(f'src.services.dashboard_testing.execution.{name}.SessionLocal',factory)
|
||||
storage,limits = DraftStorage(tmp_path),ChartHealthInput(**inputs)
|
||||
journal = ChartHealthJournal(step,storage,limits,capacity_lease_id=cap['lease_id'])
|
||||
return db,journal,step,storage,limits,cap['lease_id']
|
||||
# #endregion Test.ChartHealth.ConnectedWorker.Fixture
|
||||
|
||||
|
||||
# #region Test.ChartHealth.ConnectedWorker.Browser [C:3] [TYPE Function]
|
||||
# @TEST_INVARIANT Real HTTP/DOM query errors do not wait for ready content and do not skip the healthy next tab.
|
||||
@pytest.mark.asyncio
|
||||
async def test_real_browser_five_errors_continue_other_tab_and_owned_complete_manifest(health_context):
|
||||
from playwright.async_api import async_playwright
|
||||
db,journal,step,storage,limits,cap = health_context
|
||||
server = ThreadingHTTPServer(('127.0.0.1',0),Handler)
|
||||
thread = Thread(target=server.serve_forever,daemon=True)
|
||||
thread.start()
|
||||
try:
|
||||
async with async_playwright() as runtime:
|
||||
browser = await runtime.chromium.launch(headless=True)
|
||||
page = await browser.new_page()
|
||||
await page.goto(f'http://127.0.0.1:{server.server_port}/dashboard')
|
||||
result = await sweep_health(page,42,journal,limits)
|
||||
await browser.close()
|
||||
finally:
|
||||
server.shutdown()
|
||||
server.server_close()
|
||||
assert result['status'] == 'failed' and result['complete'] is True, result
|
||||
assert result['coverage'] == {'expected':6,'observed':6,'checked':6,'errored':5,'unresolved':0,'timeout':0,'unvisited':0,'complete':True}
|
||||
manifest = json.loads(storage.retrieve(result['manifest_ref']))
|
||||
assert [item['chart_name'] for item in manifest['charts']] == NAMES
|
||||
assert [item['database_code'] for item in manifest['charts'][:5]] == ['386']*5
|
||||
assert manifest['charts'][5]['status'] == 'healthy'
|
||||
assert all(db.get(ScenarioArtifact,item['artifact_id']).owner_id == step['scenario_run_id'] for item in manifest['charts'])
|
||||
resumed = ChartHealthJournal(step,storage,limits,capacity_lease_id=cap)
|
||||
assert resumed.finish('passed','CHART_HEALTH_COMPLETE')['status'] == 'failed'
|
||||
# #endregion Test.ChartHealth.ConnectedWorker.Browser
|
||||
|
||||
# #region Test.ChartHealth.ConnectedWorker.Partial [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT A retained confirmed error dominates interrupted coverage without claiming the five unvisited charts were checked.
|
||||
def test_one_error_then_interrupt_remains_failed_with_exact_partial_coverage(health_context):
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_manifest import parse_health_manifest
|
||||
_,journal,_,storage,_,_ = health_context
|
||||
journal.freeze_source(parse_health_manifest(LAYOUT))
|
||||
journal.append_chart({'chart_id':1,'placement_id':'CHART-1','chart_name':'Debt','tab_path':['TAB-daily'],
|
||||
'status':'failed','reason_code':'CHART_QUERY_FAILED','origin':'dom','observed_at':'2026-10-02T00:00:00Z',
|
||||
'duration_seconds':0.0,'message':'Code: 386. DB::Exception: incompatible types','database_code':'386','terminal_state':'error'})
|
||||
result = journal.finish('inconclusive','BROWSER_TRAVERSAL_CANCELLED')
|
||||
assert result['status'] == 'failed' and result['complete'] is False
|
||||
assert result['coverage']['errored'] == 1 and result['coverage']['checked'] == 1
|
||||
assert result['coverage']['unvisited'] == 5
|
||||
body = json.loads(storage.retrieve(result['manifest_ref']))
|
||||
assert body['interruption_reason'] == 'BROWSER_TRAVERSAL_CANCELLED'
|
||||
assert body['charts'][0]['chart_name'] == 'Debt'
|
||||
# #endregion Test.ChartHealth.ConnectedWorker.Partial
|
||||
|
||||
|
||||
# #region Test.ChartHealth.ConnectedWorker.Empty [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT An unobserved source cannot PASS even when finish is called with requested_status passed.
|
||||
def test_empty_health_observations_cannot_be_promoted_to_pass(health_context):
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_manifest import parse_health_manifest
|
||||
_,journal,*_ = health_context
|
||||
journal.freeze_source(parse_health_manifest(LAYOUT))
|
||||
result = journal.finish('passed','CHART_HEALTH_COMPLETE')
|
||||
assert result['status'] == 'inconclusive' and result['complete'] is False
|
||||
assert result['coverage']['unvisited'] == 6 and result['coverage']['checked'] == 0
|
||||
# #endregion Test.ChartHealth.ConnectedWorker.Empty
|
||||
# #endregion Test.ChartHealth.ConnectedWorker
|
||||
@@ -0,0 +1,167 @@
|
||||
# #region Test.ChartHealth.Independent [C:2] [TYPE Module] [SEMANTICS health,query-errors,async,attribution]
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Responses]
|
||||
# @TEST_INVARIANT Unknown or pending query state cannot establish deterministic success or failure.
|
||||
import asyncio
|
||||
import json
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import Mock
|
||||
|
||||
import pytest
|
||||
|
||||
from src.services.dashboard_testing.query_failure import response_error
|
||||
from src.services.dashboard_testing.execution.providers.browser_chart_responses import ChartResponseCollector
|
||||
|
||||
|
||||
# #region Test.ChartHealth.Independent.Request [C:1] [TYPE Class]
|
||||
class ExternalRequest:
|
||||
# #region Test.ChartHealth.Independent.Request.Init [C:1] [TYPE Function]
|
||||
def __init__(self, chart=11, metric='sales'):
|
||||
self.url='http://fixture.invalid/api/v1/chart/data'
|
||||
self.method='POST'
|
||||
self.post_data_json={'form_data':{'slice_id':chart},'datasource':{'id':4,'type':'table'},
|
||||
'queries':[{'filters':[{'col':'metric_selector','op':'IN','val':[metric]}]}]}
|
||||
# #endregion Test.ChartHealth.Independent.Request.Init
|
||||
# #endregion Test.ChartHealth.Independent.Request
|
||||
|
||||
|
||||
# #region Test.ChartHealth.Independent.Response [C:1] [TYPE Class]
|
||||
class ExternalResponse:
|
||||
# #region Test.ChartHealth.Independent.Response.Init [C:1] [TYPE Function]
|
||||
def __init__(self, request, payload, status=200):
|
||||
self.request,self.url,self.status=request,request.url,status
|
||||
self.raw=json.dumps(payload).encode()
|
||||
# #endregion Test.ChartHealth.Independent.Response.Init
|
||||
|
||||
# #region Test.ChartHealth.Independent.Response.Body [C:1] [TYPE Function]
|
||||
async def body(self):
|
||||
return self.raw
|
||||
# #endregion Test.ChartHealth.Independent.Response.Body
|
||||
# #endregion Test.ChartHealth.Independent.Response
|
||||
|
||||
|
||||
# #region Test.ChartHealth.Independent.Errors [C:2] [TYPE Function]
|
||||
@pytest.mark.parametrize('status,payload',[(500,{'errors':[{'error_type':'GENERIC_DB_ENGINE_ERROR','message':'Code: 386. NO_COMMON_TYPE'}]}),
|
||||
(200,{'result':[{'status':'failed','error':'Code: 386. NO_COMMON_TYPE'}]})])
|
||||
@pytest.mark.asyncio
|
||||
async def test_http_and_application_query_failures_are_attributed_to_exact_chart(status,payload):
|
||||
collector=ChartResponseCollector(None,[11])
|
||||
request=ExternalRequest()
|
||||
collector.on_request(request)
|
||||
await collector.read_response(ExternalResponse(request,payload,status))
|
||||
error=collector.current_error(11)
|
||||
assert error['message']=='Code: 386. NO_COMMON_TYPE'
|
||||
assert error['attribution']['http_status']==status and error['attribution']['load_generation']==1
|
||||
assert collector.current_error(12) is None
|
||||
# #endregion Test.ChartHealth.Independent.Errors
|
||||
|
||||
|
||||
# #region Test.ChartHealth.Independent.Healthy [C:2] [TYPE Function]
|
||||
@pytest.mark.parametrize('payload',[{'result':[{'data':[]}]},{'result':[{'data':[{'value':0}]}]}])
|
||||
def test_successful_empty_and_zero_are_not_query_errors(payload):
|
||||
assert response_error(payload,200) is None
|
||||
# #endregion Test.ChartHealth.Independent.Healthy
|
||||
|
||||
|
||||
# #region Test.ChartHealth.Independent.AsyncPending [C:2] [TYPE Function]
|
||||
@pytest.mark.asyncio
|
||||
async def test_accepted_async_job_is_pending_until_terminal_owned_result():
|
||||
collector=ChartResponseCollector(None,[11])
|
||||
request=ExternalRequest()
|
||||
collector.on_request(request)
|
||||
await collector.read_response(ExternalResponse(request,{'result':[{'job_id':'job-current','status':'pending'}]},202))
|
||||
assert collector.latest[11] in collector.pending
|
||||
assert collector.current_error(11) is None
|
||||
# #endregion Test.ChartHealth.Independent.AsyncPending
|
||||
|
||||
# #region Test.ChartHealth.Independent.AsyncResult [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT A done notification is not completed data; only its exact owned result response clears waiting.
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_done_waits_for_owned_result_payload_not_foreign_result():
|
||||
collector = ChartResponseCollector(None, [11])
|
||||
request = ExternalRequest()
|
||||
collector.on_request(request)
|
||||
await collector.read_response(ExternalResponse(request, {'result': [{'job_id': 'job-current', 'status': 'pending'}]}, 202))
|
||||
event = ExternalResponse(ExternalRequest(99), {'result': [{'job_id': 'job-current', 'status': 'done', 'result_url': '/api/v1/chart/data/owned-result'}]})
|
||||
event.url = 'http://fixture.invalid/api/v1/async_event/'
|
||||
await collector.read_response(event)
|
||||
assert collector.latest[11] in collector.pending
|
||||
result_request = ExternalRequest(99)
|
||||
result_request.url = 'http://fixture.invalid/api/v1/chart/data/owned-result'
|
||||
result_request.method = 'GET'
|
||||
collector.on_request(result_request)
|
||||
await collector.read_response(ExternalResponse(result_request, {'result': [{'data': [{'value': 0}]}]}))
|
||||
assert collector.latest[11] not in collector.pending
|
||||
assert collector.current_error(11) is None
|
||||
# #endregion Test.ChartHealth.Independent.AsyncResult
|
||||
|
||||
|
||||
# #region Test.ChartHealth.Independent.Malformed [C:2] [TYPE Function]
|
||||
@pytest.mark.asyncio
|
||||
async def test_malformed_http200_is_unresolved_not_confirmed_query_failure():
|
||||
collector=ChartResponseCollector(None,[11])
|
||||
request=ExternalRequest()
|
||||
collector.on_request(request)
|
||||
await collector.read_response(ExternalResponse(request,{'result':None},200))
|
||||
assert collector.current_error(11) is None
|
||||
assert collector.latest[11] in collector.pending
|
||||
# #endregion Test.ChartHealth.Independent.Malformed
|
||||
|
||||
|
||||
# #region Test.ChartHealth.Independent.Retry [C:2] [TYPE Function]
|
||||
@pytest.mark.asyncio
|
||||
async def test_new_same_filter_retry_supersedes_late_old_error():
|
||||
collector=ChartResponseCollector(None,[11])
|
||||
first,retry=ExternalRequest(),ExternalRequest()
|
||||
collector.on_request(first)
|
||||
collector.on_request(retry)
|
||||
await collector.read_response(ExternalResponse(first,{'errors':[{'message':'stale Code: 386'}]},500))
|
||||
assert collector.current_error(11) is None and collector.latest[11] in collector.pending
|
||||
await collector.read_response(ExternalResponse(retry,{'result':[{'data':[]}]},200))
|
||||
assert collector.current_error(11) is None and collector.latest[11] not in collector.pending
|
||||
# #endregion Test.ChartHealth.Independent.Retry
|
||||
|
||||
|
||||
# #region Test.ChartHealth.Independent.Close [C:2] [TYPE Function]
|
||||
@pytest.mark.asyncio
|
||||
async def test_close_removes_listeners_and_drains_blocked_external_response_body():
|
||||
page=SimpleNamespace(on=Mock(),remove_listener=Mock())
|
||||
collector=ChartResponseCollector(page,[11])
|
||||
request=ExternalRequest()
|
||||
collector.on_request(request)
|
||||
response=ExternalResponse(request,{'result':[{'data':[]}]})
|
||||
event=asyncio.Event()
|
||||
# #region Test.ChartHealth.Independent.Close.ExternalBody [C:1] [TYPE Function]
|
||||
async def external_body():
|
||||
await event.wait()
|
||||
return response.raw
|
||||
# #endregion Test.ChartHealth.Independent.Close.ExternalBody
|
||||
response.body=external_body
|
||||
collector.start()
|
||||
collector.on_response(response)
|
||||
await asyncio.sleep(0)
|
||||
tasks=list(collector.tasks)
|
||||
await collector.close()
|
||||
assert tasks and all(task.done() for task in tasks)
|
||||
assert page.remove_listener.call_count==3 and not collector.requests and not collector.pending
|
||||
assert {call.args[0] for call in page.remove_listener.call_args_list} == {'request', 'response', 'requestfailed'}
|
||||
# #endregion Test.ChartHealth.Independent.Close
|
||||
|
||||
# #region Test.ChartHealth.Independent.TransportFailure [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT Current transport failure is unresolved evidence; stale and foreign failures cannot contaminate its chart.
|
||||
def test_transport_failure_stays_unconfirmed_and_obeys_latest_request_owner():
|
||||
collector = ChartResponseCollector(None, [11])
|
||||
old, current, foreign = ExternalRequest(), ExternalRequest(), ExternalRequest(99)
|
||||
old.failure = current.failure = foreign.failure = 'net::ERR_CONNECTION_RESET'
|
||||
collector.on_request(old)
|
||||
collector.on_request(current)
|
||||
collector.on_request(foreign)
|
||||
collector.on_request_failed(old)
|
||||
collector.on_request_failed(foreign)
|
||||
assert collector.current_error(11) is None
|
||||
collector.on_request_failed(current)
|
||||
error = collector.current_error(11)
|
||||
assert error['confirmed'] is False and error['message'] == 'net::ERR_CONNECTION_RESET'
|
||||
assert error['attribution']['load_generation'] == 2
|
||||
assert collector.current_error(99) is None
|
||||
# #endregion Test.ChartHealth.Independent.TransportFailure
|
||||
# #endregion Test.ChartHealth.Independent
|
||||
@@ -0,0 +1,70 @@
|
||||
# #region Test.ChartHealth.LegacyWorker [C:3] [TYPE Module] [SEMANTICS legacy,registry,normalized,pin,journal]
|
||||
# @TEST_INVARIANT Pinned0386/0387 channels and journal input digests retain their original pre-health canonical fields.
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.Traversal.Inputs.LegacyTabs]
|
||||
# @RATIONALE Immutable54bdbcc4 input-module SHA256=1b6a3f5875f886cebd8e0a9d9b57eb69e5d44e51c6689352709cc5f8f398a87e; literal defaults below are its legacy DTO.
|
||||
import json
|
||||
from copy import deepcopy
|
||||
from hashlib import sha256
|
||||
import pytest
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from src.models.scenario_traversal import ScenarioTraversal
|
||||
from src.models.scenario_run import ScenarioRun
|
||||
from src.services.dashboard_testing.execution.providers.browser_traversal_inputs import parse_traversal_input,AllTabsTraversalInput
|
||||
from src.services.dashboard_testing.execution.providers.browser_pinned_inputs import resolve_pinned_browser_inputs
|
||||
from src.services.dashboard_testing.execution.traversal_tabs_store import TabsJournal
|
||||
from test_chart_health_connected_worker import health_context # noqa: F401
|
||||
|
||||
LEGACY = {'per_tab_timeout_seconds':70,'whole_timeout_seconds':7200,'max_bytes':268435456}
|
||||
|
||||
|
||||
# #region Test.ChartHealth.LegacyWorker.Channel [C:2] [TYPE Function]
|
||||
@pytest.mark.parametrize('version',['038.6.0','038.7.0'])
|
||||
def test_legacy_normalized_channel_has_exact_pre_health_fields(version):
|
||||
assert parse_traversal_input('navigate_tabs',{},registry_version=version) == LEGACY
|
||||
with pytest.raises(ValueError):
|
||||
parse_traversal_input('navigate_tabs',{'chart_health':True},registry_version=version)
|
||||
# #endregion Test.ChartHealth.LegacyWorker.Channel
|
||||
|
||||
|
||||
# #region Test.ChartHealth.LegacyWorker.Pinned [C:3] [TYPE Function]
|
||||
@pytest.mark.parametrize('version',['038.6.0','038.7.0'])
|
||||
def test_real_pinned_projection_and_journal_resume_preserve_old_digest(health_context,registry_engine,monkeypatch,version):
|
||||
db,_,step,storage,_,capacity = health_context
|
||||
run = db.get(ScenarioRun,step['scenario_run_id'])
|
||||
meta = {'logical_step_id':'health-check','tool':'browser','action':'navigate_tabs','action_inputs':{}}
|
||||
body = {'scenario_revision_id':run.scenario_revision_id,'scenario_content_hash':run.scenario_content_hash,
|
||||
'action_registry_version':version,'steps':[meta]}
|
||||
body['plan_hash'] = sha256(json.dumps(body,sort_keys=True,separators=(',',':')).encode()).hexdigest()
|
||||
run.runner_plan = deepcopy(body)
|
||||
db.commit()
|
||||
before = json.dumps(body,sort_keys=True)
|
||||
step = {**step,'action':'navigate_tabs','step_meta':deepcopy(meta)}
|
||||
monkeypatch.setattr('src.core.database.SessionLocal',sessionmaker(bind=registry_engine))
|
||||
inputs = resolve_pinned_browser_inputs(step)
|
||||
assert inputs == LEGACY
|
||||
# Existing unbound fixture journal belongs to another action; this is a new owned logical step attempt.
|
||||
db.query(ScenarioTraversal).filter_by(run_id=run.id).delete()
|
||||
db.commit()
|
||||
journal = TabsJournal(step,storage,AllTabsTraversalInput(**inputs),capacity_lease_id=capacity)
|
||||
row = db.get(ScenarioTraversal,journal.id)
|
||||
assert row.input_digest == sha256(b'{"max_bytes":268435456,"per_tab_timeout_seconds":70,"whole_timeout_seconds":7200}').hexdigest()
|
||||
resumed = TabsJournal(step,storage,AllTabsTraversalInput(**inputs),capacity_lease_id=capacity)
|
||||
assert resumed.id == journal.id and row.action == 'navigate_tabs'
|
||||
db.refresh(run)
|
||||
assert json.dumps(run.runner_plan,sort_keys=True) == before
|
||||
# #endregion Test.ChartHealth.LegacyWorker.Pinned
|
||||
|
||||
|
||||
# #region Test.ChartHealth.LegacyWorker.LayoutVersion [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT Superset's actual v2 layout header is metadata, not a node; exact tab/chart identity remains authoritative.
|
||||
def test_native_superset_layout_version_header_is_not_treated_as_node():
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_manifest import parse_health_manifest
|
||||
source = parse_health_manifest({'DASHBOARD_VERSION_KEY':'v2',
|
||||
'ROOT_ID':{'type':'ROOT','children':['TABS-main']},
|
||||
'TABS-main':{'type':'TABS','children':['TAB-health']},
|
||||
'TAB-health':{'type':'TAB','children':['CHART-19'],'meta':{'text':'Actual tab'}},
|
||||
'CHART-19':{'type':'CHART','children':[],'meta':{'chartId':19,'sliceName':'Actual chart'}}})
|
||||
assert source['source_total'] == 1
|
||||
assert source['placements'] == [{'id':'CHART-19','chart_id':19,'name':'Actual chart','tab_path':['TAB-health']}]
|
||||
# #endregion Test.ChartHealth.LegacyWorker.LayoutVersion
|
||||
# #endregion Test.ChartHealth.LegacyWorker
|
||||
@@ -0,0 +1,188 @@
|
||||
# #region Test.ChartHealth.MatrixPlacements [C:2] [TYPE Module] [SEMANTICS health,placement,tabs,timeout,partial]
|
||||
# @TEST_INVARIANT Repeated chart IDs retain separate placement/tab identities; inaccessible or timed-out placements cannot erase an earlier confirmed failure.
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Sweep]
|
||||
import json
|
||||
import asyncio
|
||||
import gc
|
||||
from copy import deepcopy
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
from threading import Thread
|
||||
from unittest.mock import AsyncMock, Mock
|
||||
|
||||
import pytest
|
||||
from test_chart_health_connected_worker import health_context # noqa: F401
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_sweep import sweep_health
|
||||
|
||||
REPEATED_LAYOUT = {'ROOT_ID': {'type': 'ROOT', 'children': ['TABS-main']},
|
||||
'TABS-main': {'type': 'TABS', 'children': ['TAB-first', 'TAB-second']},
|
||||
'TAB-first': {'type': 'TAB', 'meta': {'text': 'Same title'}, 'children': ['CHART-first']},
|
||||
'TAB-second': {'type': 'TAB', 'meta': {'text': 'Same title'}, 'children': ['CHART-second']},
|
||||
'CHART-first': {'type': 'CHART', 'meta': {'chartId': 11, 'sliceName': 'Repeated chart'}},
|
||||
'CHART-second': {'type': 'CHART', 'meta': {'chartId': 11, 'sliceName': 'Repeated chart'}}}
|
||||
|
||||
# #region Test.ChartHealth.MatrixPlacements.Document [C:1] [TYPE Function]
|
||||
def external_document(mode):
|
||||
second_chart = 11 if mode == 'repeat' else 12
|
||||
disabled = 'disabled' if mode == 'unopenable' else ''
|
||||
return f'''<button role="tab" id="TABS-main-tab-TAB-first" aria-controls="first" aria-selected="false"
|
||||
onclick="activate('first',11,this)">Same title</button>
|
||||
<button {disabled} role="tab" id="TABS-main-tab-TAB-second" aria-controls="second" aria-selected="false"
|
||||
onclick="activate('second',{second_chart},this)">Same title</button>
|
||||
<div id="first" role="tabpanel" hidden style="min-height:120px"><div id="chart-id-11" style="min-height:120px"><div class="loading">loading</div></div></div>
|
||||
<div id="second" role="tabpanel" hidden style="min-height:120px"><div id="chart-id-{second_chart}" style="min-height:120px"><div class="loading">loading</div></div></div>
|
||||
<script>function activate(id, chart, control) {{
|
||||
document.querySelectorAll('[role=tab]').forEach(t=>t.setAttribute('aria-selected','false'));
|
||||
document.querySelectorAll('[role=tabpanel]').forEach(t=>t.hidden=true);
|
||||
control.setAttribute('aria-selected','true');document.getElementById(id).hidden=false;
|
||||
if (id==='second' && '{mode}'==='renderer_timeout') return;
|
||||
fetch('/api/v1/chart/data',{{method:'POST',headers:{{'Content-Type':'application/json'}},
|
||||
body:JSON.stringify({{form_data:{{slice_id:chart}},datasource:{{id:4,type:'table'}},queries:[{{filters:[{{col:'tab_scope',op:'IN',val:[id]}}]}}]}})}})
|
||||
.then(r=>r.json()).then(data=>{{const root=document.getElementById(id).querySelector('#chart-id-'+chart);
|
||||
root.innerHTML=id==='first'?'<div class="alert-danger">'+data.errors[0].message+'</div>':'<table><tbody></tbody></table>';}});
|
||||
}}</script>'''
|
||||
# #endregion Test.ChartHealth.MatrixPlacements.Document
|
||||
|
||||
# #region Test.ChartHealth.MatrixPlacements.Handler [C:1] [TYPE Class]
|
||||
class ExternalHandler(BaseHTTPRequestHandler):
|
||||
# #region Test.ChartHealth.MatrixPlacements.Handler.Get [C:1] [TYPE Function]
|
||||
def do_GET(self):
|
||||
mode = self.server.fixture_mode
|
||||
layout = deepcopy(REPEATED_LAYOUT)
|
||||
if mode != 'repeat':
|
||||
layout['CHART-second']['meta']['chartId'] = 12
|
||||
body = json.dumps({'result': {'id': 42, 'position_json': layout}}).encode() if self.path.startswith('/api/v1/dashboard/') else external_document(mode).encode()
|
||||
self.send_response(200)
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
# #endregion Test.ChartHealth.MatrixPlacements.Handler.Get
|
||||
# #region Test.ChartHealth.MatrixPlacements.Handler.Post [C:1] [TYPE Function]
|
||||
def do_POST(self):
|
||||
request = json.loads(self.rfile.read(int(self.headers['Content-Length'])))
|
||||
tab = request['queries'][0]['filters'][0]['val'][0]
|
||||
payload = {'errors': [{'error_type': 'GENERIC_DB_ENGINE_ERROR', 'message': 'Code: 386. DB::Exception: NO_COMMON_TYPE'}]} if tab == 'first' else {'result': [{'data': []}]}
|
||||
self.send_response(500 if tab == 'first' else 200)
|
||||
self.end_headers()
|
||||
self.wfile.write(json.dumps(payload).encode())
|
||||
# #endregion Test.ChartHealth.MatrixPlacements.Handler.Post
|
||||
# #region Test.ChartHealth.MatrixPlacements.Handler.Log [C:1] [TYPE Function]
|
||||
def log_message(self, *_args):
|
||||
pass
|
||||
# #endregion Test.ChartHealth.MatrixPlacements.Handler.Log
|
||||
# #endregion Test.ChartHealth.MatrixPlacements.Handler
|
||||
|
||||
# #region Test.ChartHealth.MatrixPlacements.Execute [C:1] [TYPE Function]
|
||||
async def execute_external_fixture(context, mode):
|
||||
from playwright.async_api import async_playwright
|
||||
_, journal, _, storage, limits, _ = context
|
||||
server = ThreadingHTTPServer(('127.0.0.1', 0), ExternalHandler)
|
||||
server.fixture_mode = mode
|
||||
Thread(target=server.serve_forever, daemon=True).start()
|
||||
try:
|
||||
async with async_playwright() as runtime:
|
||||
browser = await runtime.chromium.launch(headless=True)
|
||||
try:
|
||||
page = await browser.new_page()
|
||||
await page.goto(f'http://127.0.0.1:{server.server_port}/{mode}')
|
||||
result = await sweep_health(page, 42, journal, limits)
|
||||
finally:
|
||||
await browser.close()
|
||||
finally:
|
||||
server.shutdown()
|
||||
server.server_close()
|
||||
return result, json.loads(storage.retrieve(result['manifest_ref']))
|
||||
# #endregion Test.ChartHealth.MatrixPlacements.Execute
|
||||
|
||||
# #region Test.ChartHealth.MatrixPlacements.Repeated [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT Same chart11 appears twice and newest tab-specific successful query cannot rewrite the already committed first-tab failure.
|
||||
@pytest.mark.asyncio
|
||||
async def test_same_chart_id_two_tabs_has_distinct_owned_placements(health_context):
|
||||
result, manifest = await execute_external_fixture(health_context, 'repeat')
|
||||
assert result['status'] == 'failed' and result['complete'] is True
|
||||
assert result['coverage'] == {'expected': 2, 'observed': 2, 'checked': 2, 'errored': 1,
|
||||
'unresolved': 0, 'timeout': 0, 'unvisited': 0, 'complete': True}
|
||||
first, second = manifest['charts']
|
||||
assert [(item['chart_id'], item['placement_id'], item['tab_path'], item['status']) for item in [first, second]] == [
|
||||
(11, 'CHART-first', ['TAB-first'], 'failed'), (11, 'CHART-second', ['TAB-second'], 'healthy')]
|
||||
assert first['database_code'] == '386'
|
||||
assert first['artifact_id'] != second['artifact_id'] and first['sha256'] != second['sha256']
|
||||
assert first['run_id'] == second['run_id'] == health_context[2]['scenario_run_id']
|
||||
assert first['attempt'] == second['attempt'] == 1
|
||||
# #endregion Test.ChartHealth.MatrixPlacements.Repeated
|
||||
|
||||
# #region Test.ChartHealth.MatrixPlacements.Unopenable [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT A disabled second tab after a retained Code386 produces unresolved coverage and cannot downgrade confirmed FAILED.
|
||||
@pytest.mark.asyncio
|
||||
async def test_unopenable_tab_after_confirmed_error_remains_failed_partial(health_context):
|
||||
result, manifest = await execute_external_fixture(health_context, 'unopenable')
|
||||
assert result['status'] == 'failed' and result['complete'] is False
|
||||
assert result['coverage'] == {'expected': 2, 'observed': 2, 'checked': 1, 'errored': 1,
|
||||
'unresolved': 1, 'timeout': 1, 'unvisited': 0, 'complete': False}
|
||||
assert [(item['placement_id'], item['tab_path'], item['status']) for item in manifest['charts']] == [
|
||||
('CHART-first', ['TAB-first'], 'failed'), ('CHART-second', ['TAB-second'], 'inconclusive')]
|
||||
assert manifest['charts'][0]['database_code'] == '386'
|
||||
assert manifest['charts'][1]['reason_code'] == 'CHART_LOAD_TIMEOUT'
|
||||
# #endregion Test.ChartHealth.MatrixPlacements.Unopenable
|
||||
|
||||
# #region Test.ChartHealth.MatrixPlacements.RenderTimeout [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT An activated second panel that stays loading exhausts only its own budget and records the exact unchecked placement.
|
||||
@pytest.mark.asyncio
|
||||
async def test_renderer_timeout_after_confirmed_error_remains_failed_partial(health_context):
|
||||
loop = asyncio.get_running_loop()
|
||||
previous_handler = loop.get_exception_handler()
|
||||
unhandled = []
|
||||
# #region Test.ChartHealth.MatrixPlacements.RenderTimeout.Exception [C:1] [TYPE Function]
|
||||
def observe_unhandled(active_loop, context):
|
||||
unhandled.append(context)
|
||||
active_loop.default_exception_handler(context) # Record normally; never hide the actual warning.
|
||||
# #endregion Test.ChartHealth.MatrixPlacements.RenderTimeout.Exception
|
||||
loop.set_exception_handler(observe_unhandled)
|
||||
try:
|
||||
result, manifest = await execute_external_fixture(health_context, 'renderer_timeout')
|
||||
await asyncio.sleep(0.05)
|
||||
gc.collect()
|
||||
await asyncio.sleep(0.05)
|
||||
gc.collect()
|
||||
finally:
|
||||
loop.set_exception_handler(previous_handler)
|
||||
assert result['status'] == 'failed' and result['complete'] is False
|
||||
assert result['coverage'] == {'expected': 2, 'observed': 2, 'checked': 1, 'errored': 1,
|
||||
'unresolved': 1, 'timeout': 1, 'unvisited': 0, 'complete': False}
|
||||
assert manifest['charts'][0]['status'] == 'failed' and manifest['charts'][0]['database_code'] == '386'
|
||||
assert manifest['charts'][1]['chart_id'] == 12 and manifest['charts'][1]['tab_path'] == ['TAB-second']
|
||||
assert manifest['charts'][1]['status'] == 'inconclusive' and manifest['charts'][1]['reason_code'] == 'CHART_LOAD_TIMEOUT'
|
||||
assert unhandled == [], [str(item.get('exception') or item.get('message')) for item in unhandled]
|
||||
# #endregion Test.ChartHealth.MatrixPlacements.RenderTimeout
|
||||
|
||||
# #region Test.ChartHealth.MatrixPlacements.ExpiredRpc [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT If the exact monotonic budget expires during control checking, no new Playwright RPC may start beyond that deadline.
|
||||
@pytest.mark.asyncio
|
||||
async def test_expired_remaining_budget_does_not_start_one_ms_browser_rpc(health_context, monkeypatch):
|
||||
from src.services.dashboard_testing.execution.providers.browser_chart_health import observe_chart
|
||||
from src.services.dashboard_testing.execution.providers.browser_chart_responses import ChartResponseCollector
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_manifest import parse_health_manifest
|
||||
_, journal, *_ = health_context
|
||||
layout = deepcopy(REPEATED_LAYOUT)
|
||||
layout['CHART-second']['meta']['chartId'] = 12
|
||||
source = parse_health_manifest(layout)
|
||||
journal.freeze_source(source)
|
||||
journal.append_chart({'chart_id': 11, 'placement_id': 'CHART-first', 'chart_name': 'Repeated chart',
|
||||
'tab_path': ['TAB-first'], 'status': 'failed', 'reason_code': 'CHART_QUERY_FAILED', 'origin': 'dom',
|
||||
'observed_at': '2026-10-02T00:00:00Z', 'duration_seconds': 0, 'message': 'Code: 386. NO_COMMON_TYPE',
|
||||
'database_code': '386', 'terminal_state': 'error'})
|
||||
# External monotonic clock: 2s deadline; check-control crosses1.9→2.1s.
|
||||
clock = Mock(side_effect=[0.0, 0.0, 1.9, 2.1, 2.1, 2.1])
|
||||
monkeypatch.setattr('src.services.dashboard_testing.execution.providers.browser_chart_health.monotonic', clock)
|
||||
chart = Mock()
|
||||
chart.wait_for = AsyncMock()
|
||||
chart.count = AsyncMock(return_value=1)
|
||||
chart.scroll_into_view_if_needed = AsyncMock()
|
||||
chart.evaluate = AsyncMock(return_value={'error': '', 'ready': False, 'loading': True})
|
||||
panel = Mock()
|
||||
panel.locator.return_value = chart
|
||||
value = await observe_chart(None, panel, source['placements'][1], 2, ChartResponseCollector(None, [11, 12]), journal)
|
||||
assert value['status'] == 'inconclusive' and value['reason_code'] == 'CHART_LOAD_TIMEOUT'
|
||||
assert chart.evaluate.await_count == 0
|
||||
journal.append_chart(value)
|
||||
assert journal.finish('passed', 'CHART_HEALTH_COMPLETE')['status'] == 'failed'
|
||||
# #endregion Test.ChartHealth.MatrixPlacements.ExpiredRpc
|
||||
# #endregion Test.ChartHealth.MatrixPlacements
|
||||
@@ -0,0 +1,82 @@
|
||||
# #region Test.ChartHealth.NativeCanary [C:4] [TYPE Module] [SEMANTICS actual,superset,clickhouse,error,recovery]
|
||||
# @TEST_INVARIANT Actual Superset/ClickHouse query errors yield five386 diagnostics plus checked healthy other-tab; own metric recovery yields health PASS.
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Sweep]
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
from copy import deepcopy
|
||||
from hashlib import sha256
|
||||
import pytest
|
||||
from src.models.scenario_run import ScenarioRun
|
||||
from src.models.scenario_traversal import ScenarioTraversal
|
||||
from src.services.dashboard_testing.execution.chart_health_store import ChartHealthJournal
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_sweep import sweep_health
|
||||
from src.services.dashboard_testing.execution.providers.browser_traversal_inputs import ChartHealthInput
|
||||
from test_chart_health_connected_worker import health_context # noqa: F401
|
||||
|
||||
|
||||
# #region Test.ChartHealth.NativeCanary.Private [C:2] [TYPE Function]
|
||||
# @INVARIANT Local lab password is consumed only for login and never printed or included in the report.
|
||||
def lab_password():
|
||||
for line in Path(os.environ['CHART_HEALTH_NATIVE_ENV_FILE']).read_text().splitlines():
|
||||
if line.startswith('FIXTURE_PASSWORD='):
|
||||
return line.split('=',1)[1].strip().strip('\"\'')
|
||||
raise RuntimeError('CHART_HEALTH_CANARY_PASSWORD_UNAVAILABLE')
|
||||
# #endregion Test.ChartHealth.NativeCanary.Private
|
||||
|
||||
|
||||
# #region Test.ChartHealth.NativeCanary.Journal [C:3] [TYPE Function]
|
||||
# @POST Real owned fixture projection names the actual isolated PREPROD dashboard; it makes no public authoring/release claim.
|
||||
def native_journal(context, dashboard_id):
|
||||
db,_,step,storage,_,capacity = context
|
||||
run = db.get(ScenarioRun,step['scenario_run_id'])
|
||||
limits = ChartHealthInput(per_chart_timeout_seconds=70,whole_timeout_seconds=600)
|
||||
meta = {**step['step_meta'],'action_inputs':{'per_chart_timeout_seconds':70,'whole_timeout_seconds':600}}
|
||||
body = {key:deepcopy(value) for key,value in run.runner_plan.items() if key != 'plan_hash'}
|
||||
body['steps'],body['action_registry_version'] = [meta],'038.8.0'
|
||||
body['plan_hash'] = sha256(json.dumps(body,sort_keys=True,separators=(',',':')).encode()).hexdigest()
|
||||
target = {**run.target_snapshot,'dashboard_id':dashboard_id}
|
||||
run.runner_plan,run.target_snapshot = body,target
|
||||
db.query(ScenarioTraversal).filter_by(run_id=run.id).delete()
|
||||
db.commit()
|
||||
step = {**step,'step_meta':meta,'target_snapshot':deepcopy(target)}
|
||||
return ChartHealthJournal(step,storage,limits,capacity_lease_id=capacity),limits
|
||||
# #endregion Test.ChartHealth.NativeCanary.Journal
|
||||
|
||||
|
||||
# #region Test.ChartHealth.NativeCanary.Run [C:4] [TYPE Function]
|
||||
# @POST Source-backed typed health is retained for both phases without mutating any source data or finance assets.
|
||||
@pytest.mark.integration
|
||||
@pytest.mark.asyncio
|
||||
async def test_actual_clickhouse386_and_own_metric_recovery(health_context):
|
||||
if not os.getenv('CHART_HEALTH_NATIVE_DASHBOARD_ID'):
|
||||
pytest.skip('explicit isolated native canary configuration required')
|
||||
from playwright.async_api import async_playwright
|
||||
dashboard_id = int(os.environ['CHART_HEALTH_NATIVE_DASHBOARD_ID'])
|
||||
journal,limits = native_journal(health_context,dashboard_id)
|
||||
async with async_playwright() as runtime:
|
||||
browser = await runtime.chromium.launch(headless=True)
|
||||
try:
|
||||
page = await browser.new_page(viewport={'width':1600,'height':1000})
|
||||
await page.goto(os.environ['CHART_HEALTH_NATIVE_URL']+'/login/',wait_until='domcontentloaded')
|
||||
await page.locator('#username').fill('admin')
|
||||
await page.locator('#password').fill(lab_password())
|
||||
await page.locator('input[type="submit"],button[type="submit"]').click()
|
||||
await page.wait_for_url(lambda url:'/login' not in url,timeout=30000)
|
||||
await page.goto(os.environ['CHART_HEALTH_NATIVE_URL']+f'/superset/dashboard/{dashboard_id}/?force=true',wait_until='domcontentloaded')
|
||||
result = await sweep_health(page,dashboard_id,journal,limits)
|
||||
finally:
|
||||
await browser.close()
|
||||
output = Path(os.environ['CHART_HEALTH_NATIVE_OUTPUT'])
|
||||
output.mkdir(parents=True,exist_ok=False)
|
||||
(output/'report.json').write_text(json.dumps({'scope':'Actual Superset/ClickHouse observer+owned journal integration, not public release lifecycle',
|
||||
'phase':os.environ['CHART_HEALTH_NATIVE_PHASE'],'dashboard_id':dashboard_id,'result':result},indent=2))
|
||||
assert result['complete'] is True and result['coverage']['checked'] == 6, result
|
||||
if os.environ['CHART_HEALTH_NATIVE_PHASE'] == 'error':
|
||||
assert result['status'] == 'failed' and result['coverage']['errored'] == 5, result
|
||||
assert [item['database_code'] for item in result['chart_health']['errors']] == ['386']*5
|
||||
assert result['chart_health']['charts'][5]['status'] == 'healthy'
|
||||
else:
|
||||
assert result['status'] == 'passed' and result['coverage']['errored'] == 0, result
|
||||
# #endregion Test.ChartHealth.NativeCanary.Run
|
||||
# #endregion Test.ChartHealth.NativeCanary
|
||||
@@ -0,0 +1,47 @@
|
||||
# #region Test.ChartHealth.NativeDOMProbe [C:3] [TYPE Module] [SEMANTICS native,diagnostic,DOM,identity]
|
||||
# @BRIEF Inventory real isolated Superset error/healthy card DOM; this is a diagnostic, not health acceptance.
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import pytest
|
||||
from test_chart_health_native_canary import lab_password
|
||||
|
||||
SCRIPT = '''() => {
|
||||
const labels=['Canary Debt','Canary Cash','Canary Margin','Canary Forecast','Canary Reserve','Canary Healthy'];
|
||||
const attrs=e=>Object.fromEntries([...e.attributes].map(a=>[a.name,a.value]));
|
||||
return labels.map(label=>({label,matches:[...document.querySelectorAll('*')].filter(e=>e.children.length===0 && e.textContent.trim()===label)
|
||||
.map(e=>{const ancestors=[];let node=e;for(let i=0;i<7 && node;i++,node=node.parentElement){ancestors.push({tag:node.tagName,attributes:attrs(node),text:node.innerText.slice(0,10000),html:node.outerHTML.slice(0,18000)})}return ancestors})}));
|
||||
}'''
|
||||
|
||||
|
||||
# #region Test.ChartHealth.NativeDOMProbe.Run [C:3] [TYPE Function]
|
||||
# @POST Both phases retain bounded actual card ancestors without observer fallbacks or synthetic query outcomes.
|
||||
@pytest.mark.integration
|
||||
@pytest.mark.asyncio
|
||||
async def test_native_saved_card_structure_diagnostic_only():
|
||||
if not os.getenv('CHART_HEALTH_NATIVE_DASHBOARD_ID'):
|
||||
pytest.skip('explicit native diagnostic configuration required')
|
||||
from playwright.async_api import async_playwright
|
||||
dashboard_id = int(os.environ['CHART_HEALTH_NATIVE_DASHBOARD_ID'])
|
||||
async with async_playwright() as runtime:
|
||||
browser = await runtime.chromium.launch(headless=True)
|
||||
try:
|
||||
page = await browser.new_page(viewport={'width':1600,'height':1000})
|
||||
await page.goto(os.environ['CHART_HEALTH_NATIVE_URL']+'/login/',wait_until='domcontentloaded')
|
||||
await page.locator('#username').fill('admin')
|
||||
await page.locator('#password').fill(lab_password())
|
||||
await page.locator('input[type="submit"],button[type="submit"]').click()
|
||||
await page.wait_for_url(lambda url:'/login' not in url,timeout=30000)
|
||||
await page.goto(os.environ['CHART_HEALTH_NATIVE_URL']+f'/superset/dashboard/{dashboard_id}/?force=true',wait_until='domcontentloaded')
|
||||
await page.get_by_text('Canary Debt',exact=True).wait_for(state='visible',timeout=30000)
|
||||
await page.wait_for_timeout(3000)
|
||||
report = await page.evaluate(SCRIPT)
|
||||
finally:
|
||||
await browser.close()
|
||||
output = Path(os.environ['CHART_HEALTH_NATIVE_OUTPUT'])
|
||||
output.mkdir(parents=True,exist_ok=False)
|
||||
(output/'dom-diagnostic.json').write_text(json.dumps({'scope':'native card DOM inventory only, not health PASS',
|
||||
'phase':os.environ['CHART_HEALTH_NATIVE_PHASE'],'dashboard_id':dashboard_id,'cards':report},indent=2))
|
||||
assert len(report) == 6 and all(item['matches'] for item in report)
|
||||
# #endregion Test.ChartHealth.NativeDOMProbe.Run
|
||||
# #endregion Test.ChartHealth.NativeDOMProbe
|
||||
@@ -0,0 +1,76 @@
|
||||
# #region Test.ChartHealth.NativeScope [C:2] [TYPE Module] [SEMANTICS health,native,scope,error,identity]
|
||||
# @TEST_INVARIANT Native saved-card identity and chart body, never header SVG or foreign cards, establish health; attributed HTTP errors need no mounted renderer.
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Scope]
|
||||
import pytest
|
||||
from test_chart_health_connected_worker import health_context # noqa: F401
|
||||
from test_chart_health_owned_evaluation_authority import LAYOUT
|
||||
from test_chart_health_independent_authority import ExternalRequest, ExternalResponse
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_manifest import parse_health_manifest
|
||||
from src.services.dashboard_testing.execution.providers.browser_chart_health import observe_chart
|
||||
from src.services.dashboard_testing.execution.providers.browser_chart_responses import ChartResponseCollector
|
||||
|
||||
ERROR_CARD = '''<div data-test="chart-grid-component" data-test-chart-id="11" data-test-chart-name="Same title">
|
||||
<header><svg width="24" height="24"></svg></header><div class="dashboard-chart" style="height:120px">
|
||||
<div class="alert-danger">Code: 386. NO_COMMON_TYPE</div></div></div>'''
|
||||
HEADER_ONLY = '''<div data-test="chart-grid-component" data-test-chart-id="11" data-test-chart-name="Same title">
|
||||
<header><svg width="24" height="24"></svg></header><div class="dashboard-chart" style="height:120px"></div></div>'''
|
||||
FOREIGN_CARD = '''<div data-test="chart-grid-component" data-test-chart-id="99" data-test-chart-name="Same title">
|
||||
<div class="dashboard-chart" style="height:120px"><table><tbody><tr><td>0</td></tr></tbody></table></div></div>'''
|
||||
|
||||
# #region Test.ChartHealth.NativeScope.Observe [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT Literal native error cards without ChartRenderer ID fail; header-only, duplicate, foreign and wrong-name observations cannot PASS.
|
||||
@pytest.mark.parametrize('html,status,reason', [
|
||||
(ERROR_CARD, 'failed', 'CHART_QUERY_FAILED'),
|
||||
(HEADER_ONLY, 'inconclusive', 'CHART_LOAD_TIMEOUT'),
|
||||
(ERROR_CARD + ERROR_CARD, 'inconclusive', 'CHART_PLACEMENT_AMBIGUOUS'),
|
||||
(FOREIGN_CARD, 'inconclusive', 'CHART_LOAD_TIMEOUT'),
|
||||
(ERROR_CARD.replace('Same title', 'Wrong title'), 'inconclusive', 'CHART_PLACEMENT_IDENTITY_MISMATCH'),
|
||||
])
|
||||
@pytest.mark.asyncio
|
||||
async def test_native_body_scope_rejects_unowned_or_unready_visuals(health_context, html, status, reason):
|
||||
from playwright.async_api import async_playwright
|
||||
_, journal, *_ = health_context
|
||||
source = parse_health_manifest(LAYOUT)
|
||||
journal.freeze_source(source)
|
||||
async with async_playwright() as runtime:
|
||||
browser = await runtime.chromium.launch(headless=True)
|
||||
try:
|
||||
page = await browser.new_page()
|
||||
await page.set_content(html)
|
||||
value = await observe_chart(page, page, source['placements'][0], 2, ChartResponseCollector(page, [11]), journal)
|
||||
finally:
|
||||
await browser.close()
|
||||
assert value['chart_id'] == 11 and value['placement_id'] == 'CHART-a'
|
||||
assert value['status'] == status and value['reason_code'] == reason
|
||||
if status == 'failed':
|
||||
assert value['database_code'] == '386'
|
||||
journal.append_chart(value)
|
||||
assert journal.finish('inconclusive', 'CHART_HEALTH_INTERRUPTED')['complete'] is False
|
||||
# #endregion Test.ChartHealth.NativeScope.Observe
|
||||
|
||||
# #region Test.ChartHealth.NativeScope.EarlyHttp [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT A latest server-attributed Code386 HTTP failure fails before a missing ChartRenderer can consume its load deadline.
|
||||
@pytest.mark.asyncio
|
||||
async def test_owned_http_error_does_not_wait_for_missing_chart_renderer(health_context):
|
||||
from playwright.async_api import async_playwright
|
||||
_, journal, *_ = health_context
|
||||
source = parse_health_manifest(LAYOUT)
|
||||
journal.freeze_source(source)
|
||||
collector = ChartResponseCollector(None, [11])
|
||||
request = ExternalRequest()
|
||||
collector.on_request(request)
|
||||
await collector.read_response(ExternalResponse(request, {'errors': [{'message': 'Code: 386. NO_COMMON_TYPE'}]}, 500))
|
||||
async with async_playwright() as runtime:
|
||||
browser = await runtime.chromium.launch(headless=True)
|
||||
try:
|
||||
page = await browser.new_page()
|
||||
await page.set_content('<div>Renderer absent</div>')
|
||||
value = await observe_chart(page, page, source['placements'][0], 2, collector, journal)
|
||||
finally:
|
||||
await browser.close()
|
||||
assert value['status'] == 'failed' and value['reason_code'] == 'CHART_QUERY_FAILED'
|
||||
assert value['origin'] == 'http_response' and value['http_status'] == 500
|
||||
assert value['request_identity'] == '1' and value['load_generation'] == 1
|
||||
assert value['database_code'] == '386' and value['duration_seconds'] < 1
|
||||
# #endregion Test.ChartHealth.NativeScope.EarlyHttp
|
||||
# #endregion Test.ChartHealth.NativeScope
|
||||
@@ -0,0 +1,73 @@
|
||||
# #region Test.ChartHealth.NestedDom [C:2] [TYPE Module] [SEMANTICS health,nested,tabs,offscreen,placement]
|
||||
# @TEST_INVARIANT Nested paths and below-viewport charts use exact identities and remain independently retained.
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Sweep]
|
||||
import json
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
from threading import Thread
|
||||
import pytest
|
||||
from test_chart_health_connected_worker import health_context # noqa: F401
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_sweep import sweep_health
|
||||
|
||||
LAYOUT = {'ROOT_ID': {'type': 'ROOT', 'children': ['TABS-outer']},
|
||||
'TABS-outer': {'type': 'TABS', 'children': ['TAB-parent']},
|
||||
'TAB-parent': {'type': 'TAB', 'meta': {'text': 'Repeated title'}, 'children': ['CHART-top', 'TABS-inner']},
|
||||
'CHART-top': {'type': 'CHART', 'meta': {'chartId': 11, 'sliceName': 'Same title'}},
|
||||
'TABS-inner': {'type': 'TABS', 'children': ['TAB-child']},
|
||||
'TAB-child': {'type': 'TAB', 'meta': {'text': 'Repeated title'}, 'children': ['CHART-bottom']},
|
||||
'CHART-bottom': {'type': 'CHART', 'meta': {'chartId': 12, 'sliceName': 'Same title'}}}
|
||||
HTML = '''<button role="tab" id="TABS-outer-tab-TAB-parent" aria-controls="parent" aria-selected="false"
|
||||
onclick="this.setAttribute('aria-selected','true');document.getElementById('parent').hidden=false">Repeated title</button>
|
||||
<div id="parent" role="tabpanel" hidden style="min-height:120px">
|
||||
<div id="chart-id-11"><div class="alert-danger">Code: 386. NO_COMMON_TYPE</div></div>
|
||||
<button role="tab" id="TABS-inner-tab-TAB-child" aria-controls="child" aria-selected="false"
|
||||
onclick="this.setAttribute('aria-selected','true');document.getElementById('child').hidden=false">Repeated title</button>
|
||||
<div id="child" role="tabpanel" hidden style="min-height:120px">
|
||||
<div style="height:1800px"></div><div id="chart-id-12" style="height:120px"><table><tbody></tbody></table></div>
|
||||
</div></div>'''
|
||||
|
||||
# #region Test.ChartHealth.NestedDom.Handler [C:1] [TYPE Class]
|
||||
class ExternalHandler(BaseHTTPRequestHandler):
|
||||
# #region Test.ChartHealth.NestedDom.Handler.Get [C:1] [TYPE Function]
|
||||
def do_GET(self):
|
||||
body = json.dumps({'result': {'id': 42, 'position_json': LAYOUT}}).encode() if self.path.startswith('/api/v1/dashboard/') else HTML.encode()
|
||||
self.send_response(200)
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
# #endregion Test.ChartHealth.NestedDom.Handler.Get
|
||||
# #region Test.ChartHealth.NestedDom.Handler.Log [C:1] [TYPE Function]
|
||||
def log_message(self, *_args):
|
||||
pass
|
||||
# #endregion Test.ChartHealth.NestedDom.Handler.Log
|
||||
# #endregion Test.ChartHealth.NestedDom.Handler
|
||||
|
||||
# #region Test.ChartHealth.NestedDom.Sweep [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT Real Chromium opens parent/child stable IDs and scrolls the healthy empty chart into view after recording the error.
|
||||
@pytest.mark.asyncio
|
||||
async def test_nested_duplicate_names_offscreen_empty_chart_complete_owned_coverage(health_context):
|
||||
from playwright.async_api import async_playwright
|
||||
_, journal, _, storage, limits, _ = health_context
|
||||
server = ThreadingHTTPServer(('127.0.0.1', 0), ExternalHandler)
|
||||
thread = Thread(target=server.serve_forever, daemon=True)
|
||||
thread.start()
|
||||
try:
|
||||
async with async_playwright() as runtime:
|
||||
browser = await runtime.chromium.launch(headless=True)
|
||||
page = await browser.new_page(viewport={'width': 1000, 'height': 600})
|
||||
await page.goto(f'http://127.0.0.1:{server.server_port}/dashboard')
|
||||
result = await sweep_health(page, 42, journal, limits)
|
||||
position = await page.locator('#chart-id-12').bounding_box()
|
||||
assert position is not None and 0 <= position['y'] < 600
|
||||
assert await page.locator('#TABS-inner-tab-TAB-child').get_attribute('aria-selected') == 'true'
|
||||
await browser.close()
|
||||
finally:
|
||||
server.shutdown()
|
||||
server.server_close()
|
||||
assert result['status'] == 'failed' and result['complete'] is True
|
||||
assert result['coverage'] == {'expected': 2, 'observed': 2, 'checked': 2, 'errored': 1,
|
||||
'unresolved': 0, 'timeout': 0, 'unvisited': 0, 'complete': True}
|
||||
manifest = json.loads(storage.retrieve(result['manifest_ref']))
|
||||
assert [(item['chart_id'], item['tab_path'], item['status']) for item in manifest['charts']] == [
|
||||
(11, ['TAB-parent'], 'failed'), (12, ['TAB-parent', 'TAB-child'], 'healthy')]
|
||||
assert [item['chart_name'] for item in manifest['charts']] == ['Same title', 'Same title']
|
||||
# #endregion Test.ChartHealth.NestedDom.Sweep
|
||||
# #endregion Test.ChartHealth.NestedDom
|
||||
@@ -0,0 +1,122 @@
|
||||
# #region Test.ChartHealth.NotificationWorker [C:4] [TYPE Module] [SEMANTICS configured,notification,receipt,CAS]
|
||||
# @TEST_INVARIANT Explicit routing sends bounded named causes once; disabled configuration and rollback send nothing.
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Delivery]
|
||||
from copy import deepcopy
|
||||
import pytest
|
||||
from src.models.config import AppConfigRecord
|
||||
from src.services.dashboard_testing.automation.notify import persist_notification
|
||||
from src.services.dashboard_testing.execution.chart_health_delivery import deliver_receipt
|
||||
from src.services.dashboard_testing.execution.chart_health_delivery_hook import arm_health_delivery
|
||||
|
||||
SUMMARY = {'chart_query_errors':{'count':5,'omitted':0,'items':[
|
||||
{'chart_name':name,'tab_path':['TAB-daily'],'database_code':'386','message':'Code: 386. NO_COMMON_TYPE'}
|
||||
for name in ('Debt','Cash','Margin','Forecast','Reserve')]}}
|
||||
|
||||
|
||||
# #region Test.ChartHealth.NotificationWorker.Receipt [C:2] [TYPE Function]
|
||||
# @BRIEF Real durable receipt and committed global settings, independent of notification transport.
|
||||
def notification_fixture(db, *, enabled=True, channels=None):
|
||||
config = db.get(AppConfigRecord,'global')
|
||||
payload = {'notifications':{'scenario_alerts':{'enabled':enabled,'channels':channels if channels is not None else [{'type':'SLACK','target':'test-only-destination'}]}}}
|
||||
if config:
|
||||
config.payload = payload
|
||||
else:
|
||||
db.add(AppConfigRecord(id='global',payload=payload))
|
||||
row = persist_notification(db,event_type='failed',scenario_id='scn-exec-0001',run_id='run-exec-0001',
|
||||
payload=deepcopy(SUMMARY),idempotency_key='health-notification-test')
|
||||
return row
|
||||
# #endregion Test.ChartHealth.NotificationWorker.Receipt
|
||||
|
||||
|
||||
# #region Test.ChartHealth.NotificationWorker.Provider [C:2] [TYPE Class]
|
||||
# @BRIEF Configured fake boundary records messages without touching any external destination.
|
||||
class FakeProvider:
|
||||
# #region Test.ChartHealth.NotificationWorker.Provider.Init [C:1] [TYPE Function]
|
||||
def __init__(self, calls, sent=True):
|
||||
self.calls,self.sent = calls,sent
|
||||
# #endregion Test.ChartHealth.NotificationWorker.Provider.Init
|
||||
|
||||
# #region Test.ChartHealth.NotificationWorker.Provider.Send [C:1] [TYPE Function]
|
||||
async def send(self, target, subject, body):
|
||||
self.calls.append((target,subject,body))
|
||||
return self.sent
|
||||
# #endregion Test.ChartHealth.NotificationWorker.Provider.Send
|
||||
# #endregion Test.ChartHealth.NotificationWorker.Provider
|
||||
|
||||
|
||||
# #region Test.ChartHealth.NotificationWorker.Once [C:3] [TYPE Function]
|
||||
@pytest.mark.asyncio
|
||||
async def test_committed_named_errors_send_once_with_atomic_receipt_claim(seeded_execution):
|
||||
db,calls = seeded_execution,[]
|
||||
row = notification_fixture(db)
|
||||
row.payload = {**row.payload,'health_delivery':{'status':'pending'}}
|
||||
db.commit()
|
||||
provider = FakeProvider(calls)
|
||||
assert await deliver_receipt(db,row.id,provider_factory=lambda *_:provider) == 'delivered'
|
||||
assert await deliver_receipt(db,row.id,provider_factory=lambda *_:provider) == 'already_claimed'
|
||||
assert len(calls) == 1 and calls[0][0] == 'test-only-destination'
|
||||
assert all(name in calls[0][2] for name in ('Debt','Cash','Margin','Forecast','Reserve'))
|
||||
assert 'Code 386' in calls[0][2] and 'run-exec-0001' in calls[0][2]
|
||||
db.refresh(row)
|
||||
assert row.payload['health_delivery']['status'] == 'delivered'
|
||||
assert 'test-only-destination' not in str(row.payload)
|
||||
# #endregion Test.ChartHealth.NotificationWorker.Once
|
||||
|
||||
|
||||
# #region Test.ChartHealth.NotificationWorker.Failure [C:2] [TYPE Function]
|
||||
@pytest.mark.asyncio
|
||||
async def test_false_transport_retains_delivery_failure_without_retry(seeded_execution):
|
||||
db,calls = seeded_execution,[]
|
||||
row = notification_fixture(db)
|
||||
row.payload = {**row.payload,'health_delivery':{'status':'pending'}}
|
||||
db.commit()
|
||||
provider = FakeProvider(calls,False)
|
||||
assert await deliver_receipt(db,row.id,provider_factory=lambda *_:provider) == 'failed'
|
||||
assert await deliver_receipt(db,row.id,provider_factory=lambda *_:provider) == 'already_claimed'
|
||||
assert len(calls) == 1
|
||||
assert row.payload['health_delivery']['routes'][0]['reason_code'] == 'DELIVERY_FAILED'
|
||||
# #endregion Test.ChartHealth.NotificationWorker.Failure
|
||||
|
||||
|
||||
# #region Test.ChartHealth.NotificationWorker.Disabled [C:2] [TYPE Function]
|
||||
@pytest.mark.parametrize('enabled,channels,reason',[(False,[],'DELIVERY_DISABLED'),(True,[],'DESTINATION_UNCONFIGURED')])
|
||||
def test_disabled_or_missing_destination_has_explicit_skip_and_no_dispatch(seeded_execution,enabled,channels,reason):
|
||||
db = seeded_execution
|
||||
row = notification_fixture(db,enabled=enabled,channels=channels)
|
||||
arm_health_delivery(db,row)
|
||||
db.commit()
|
||||
assert row.payload['health_delivery'] == {'status':'skipped','reason_code':reason}
|
||||
assert not db.info.get('chart_health_deliveries')
|
||||
# #endregion Test.ChartHealth.NotificationWorker.Disabled
|
||||
|
||||
|
||||
# #region Test.ChartHealth.NotificationWorker.Commit [C:3] [TYPE Function]
|
||||
def test_after_commit_dispatch_is_armed_once_and_rollback_cannot_start(seeded_execution,monkeypatch):
|
||||
from src.services.dashboard_testing.execution import chart_health_delivery_hook as seam
|
||||
db,started = seeded_execution,[]
|
||||
# #region Test.ChartHealth.NotificationWorker.Commit.Thread [C:1] [TYPE Class]
|
||||
class ExternalThread:
|
||||
# #region Test.ChartHealth.NotificationWorker.Commit.Thread.Init [C:1] [TYPE Function]
|
||||
def __init__(self, **kwargs):
|
||||
self.identifiers = kwargs['args'][0]
|
||||
# #endregion Test.ChartHealth.NotificationWorker.Commit.Thread.Init
|
||||
# #region Test.ChartHealth.NotificationWorker.Commit.Thread.Start [C:1] [TYPE Function]
|
||||
def start(self):
|
||||
started.append(self.identifiers)
|
||||
seam._slot.release()
|
||||
# #endregion Test.ChartHealth.NotificationWorker.Commit.Thread.Start
|
||||
# #endregion Test.ChartHealth.NotificationWorker.Commit.Thread
|
||||
monkeypatch.setattr(seam,'Thread',ExternalThread)
|
||||
row = notification_fixture(db)
|
||||
arm_health_delivery(db,row)
|
||||
assert not started
|
||||
db.rollback()
|
||||
assert not started and not db.info.get('chart_health_deliveries')
|
||||
row = notification_fixture(db)
|
||||
arm_health_delivery(db,row)
|
||||
arm_health_delivery(db,row)
|
||||
assert not started
|
||||
db.commit()
|
||||
assert started == [[row.id]]
|
||||
# #endregion Test.ChartHealth.NotificationWorker.Commit
|
||||
# #endregion Test.ChartHealth.NotificationWorker
|
||||
@@ -0,0 +1,125 @@
|
||||
# #region Test.ChartHealth.OwnedIndependent [C:3] [TYPE Module]
|
||||
# @TEST_INVARIANT Committed chart failures dominate advisory verdicts; model inputs require exact owned bytes and attempts.
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Store]
|
||||
import json
|
||||
|
||||
import pytest
|
||||
from src.models.scenario_artifact import ScenarioArtifact
|
||||
from src.services.dashboard_testing.execution.chart_health_artifact import verify_health_artifact
|
||||
from src.services.dashboard_testing.execution.chart_health_tab_store import append_tab_evaluation, store_health_artifact
|
||||
from src.services.dashboard_testing.execution.evaluation_health_payloads import load_health_payloads
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_manifest import parse_health_manifest
|
||||
|
||||
# Infrastructure only: committed run, step lease, capacity, real DraftStorage and fresh DB sessions.
|
||||
from test_chart_health_connected_worker import health_context # noqa: F401
|
||||
|
||||
LAYOUT = {'ROOT_ID': {'type': 'ROOT', 'children': ['CHART-a', 'CHART-b']},
|
||||
'CHART-a': {'type': 'CHART', 'meta': {'chartId': 11, 'sliceName': 'Same title'}},
|
||||
'CHART-b': {'type': 'CHART', 'meta': {'chartId': 12, 'sliceName': 'Same title'}}}
|
||||
|
||||
|
||||
# #region Test.ChartHealth.OwnedIndependent.Append [C:1] [TYPE Function]
|
||||
def append(journal, chart, status='failed'):
|
||||
journal.append_chart({'chart_id': chart, 'placement_id': 'CHART-a' if chart == 11 else 'CHART-b',
|
||||
'chart_name': 'Same title', 'tab_path': [], 'status': status,
|
||||
'reason_code': 'CHART_QUERY_FAILED' if status == 'failed' else 'CHART_READY',
|
||||
'origin': 'dom', 'observed_at': '2026-10-02T00:00:00Z', 'duration_seconds': 0,
|
||||
'message': 'Code: 386. DB::Exception: incompatible types' if status == 'failed' else '',
|
||||
'database_code': '386' if status == 'failed' else None,
|
||||
'terminal_state': 'error' if status == 'failed' else 'ready'})
|
||||
# #endregion Test.ChartHealth.OwnedIndependent.Append
|
||||
|
||||
|
||||
# #region Test.ChartHealth.OwnedIndependent.Partial [C:2] [TYPE Function]
|
||||
def test_partial_error_has_named_owned_bytes_and_exact_unvisited(health_context):
|
||||
db, journal, step, storage, *_ = health_context
|
||||
journal.freeze_source(parse_health_manifest(LAYOUT))
|
||||
append(journal, 11)
|
||||
result = journal.finish('inconclusive', 'BROWSER_TRAVERSAL_CANCELLED')
|
||||
assert result['status'] == 'failed' and result['complete'] is False
|
||||
assert result['coverage'] == {'expected': 2, 'observed': 1, 'checked': 1, 'errored': 1,
|
||||
'unresolved': 0, 'timeout': 0, 'unvisited': 1, 'complete': False}
|
||||
chart = result['chart_health']['charts'][0]
|
||||
artifact = db.get(ScenarioArtifact, chart['artifact_id'])
|
||||
payload = load_health_payloads([{'artifact_id': chart['content_ref'], 'sha256': chart['sha256'],
|
||||
'byte_length': artifact.byte_length, 'content_type': 'application/json'}], storage, db=db, run_id=step['scenario_run_id'])
|
||||
assert payload['items'][0]['content']['chart_id'] == 11
|
||||
assert payload['items'][0]['content']['database_code'] == '386'
|
||||
assert 'incompatible types' in payload['items'][0]['content']['message']
|
||||
assert payload['truncated'] is False
|
||||
# #endregion Test.ChartHealth.OwnedIndependent.Partial
|
||||
|
||||
|
||||
# #region Test.ChartHealth.OwnedIndependent.Advisory [C:2] [TYPE Function]
|
||||
def test_advisory_pass_never_erases_complete_deterministic_failure(health_context):
|
||||
_, journal, _, storage, *_ = health_context
|
||||
journal.freeze_source(parse_health_manifest(LAYOUT))
|
||||
append(journal, 11)
|
||||
append(journal, 12, 'healthy')
|
||||
receipt = store_health_artifact(journal, b'{"verdict":"pass"}', 'judge.json')
|
||||
append_tab_evaluation(journal, {'tab_path': [], 'status': 'succeeded', 'coverage_complete': True,
|
||||
'verdict': 'pass', 'artifacts': [receipt]})
|
||||
result = journal.finish('passed', 'CHART_HEALTH_COMPLETE')
|
||||
assert result['status'] == 'failed' and result['complete'] is True
|
||||
assert result['coverage']['errored'] == 1
|
||||
assert [item['chart_id'] for item in result['chart_health']['charts']] == [11, 12]
|
||||
assert result['chart_health']['tab_evaluations'][0]['verdict'] == 'pass'
|
||||
assert json.loads(storage.retrieve(result['manifest_ref']))['status'] == 'failed'
|
||||
# #endregion Test.ChartHealth.OwnedIndependent.Advisory
|
||||
|
||||
|
||||
# #region Test.ChartHealth.OwnedIndependent.Ownership [C:2] [TYPE Function]
|
||||
@pytest.mark.parametrize('field,value', [('owner_id', 'foreign-run'), ('attempt', 2), ('is_active', False)])
|
||||
def test_foreign_attempt_or_inactive_artifact_refuses_before_evaluation(health_context, field, value):
|
||||
db, journal, *_ = health_context
|
||||
receipt = store_health_artifact(journal, b'{"verdict":"pass"}', 'judge.json')
|
||||
artifact = db.get(ScenarioArtifact, receipt['artifact_id'])
|
||||
setattr(artifact, field, value)
|
||||
db.commit()
|
||||
with pytest.raises(ValueError, match='CHART_HEALTH_ARTIFACT_INVALID'):
|
||||
verify_health_artifact(journal, db, receipt)
|
||||
# #endregion Test.ChartHealth.OwnedIndependent.Ownership
|
||||
|
||||
|
||||
# #region Test.ChartHealth.OwnedIndependent.Auxiliary [C:2] [TYPE Function]
|
||||
def test_tab_capture_between_chart_observations_preserves_exact_frontier(health_context):
|
||||
_, journal, *_ = health_context
|
||||
journal.freeze_source(parse_health_manifest(LAYOUT))
|
||||
append(journal, 11)
|
||||
store_health_artifact(journal, b'{"verdict":"pass"}', 'first-tab-judge.json')
|
||||
append(journal, 12, 'healthy')
|
||||
result = journal.finish('passed', 'CHART_HEALTH_COMPLETE')
|
||||
assert result['status'] == 'failed' and result['coverage']['observed'] == 2
|
||||
# #endregion Test.ChartHealth.OwnedIndependent.Auxiliary
|
||||
|
||||
# #region Test.ChartHealth.OwnedIndependent.Budget [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT Omitted diagnostic text is explicit and cannot be presented as complete model coverage.
|
||||
def test_diagnostic_byte_budget_reports_omission_without_trusting_partial_payload(health_context):
|
||||
db, journal, step, storage, *_ = health_context
|
||||
journal.freeze_source(parse_health_manifest(LAYOUT))
|
||||
append(journal, 11)
|
||||
result = journal.finish('inconclusive', 'BROWSER_TRAVERSAL_CANCELLED')
|
||||
chart = result['chart_health']['charts'][0]
|
||||
artifact = db.get(ScenarioArtifact, chart['artifact_id'])
|
||||
payload = load_health_payloads([{'artifact_id': chart['content_ref'], 'sha256': chart['sha256'],
|
||||
'byte_length': artifact.byte_length, 'content_type': 'application/json'}], storage,
|
||||
db=db, run_id=step['scenario_run_id'], max_bytes=1)
|
||||
assert payload['items'] == [] and payload['byte_length'] == 0 and payload['truncated'] is True
|
||||
assert payload['omitted'] == [{'artifact_id': chart['content_ref'], 'reason': 'TEXT_BUDGET'}]
|
||||
# #endregion Test.ChartHealth.OwnedIndependent.Budget
|
||||
|
||||
# #region Test.ChartHealth.OwnedIndependent.Tampered [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT A forged receipt hash cannot authorize actual diagnostic bytes for the model.
|
||||
def test_changed_diagnostic_digest_refuses_model_input(health_context):
|
||||
db, journal, step, storage, *_ = health_context
|
||||
journal.freeze_source(parse_health_manifest(LAYOUT))
|
||||
append(journal, 11)
|
||||
result = journal.finish('inconclusive', 'BROWSER_TRAVERSAL_CANCELLED')
|
||||
chart = result['chart_health']['charts'][0]
|
||||
artifact = db.get(ScenarioArtifact, chart['artifact_id'])
|
||||
with pytest.raises(RuntimeError, match='CHART_HEALTH_EVIDENCE_INVALID'):
|
||||
load_health_payloads([{'artifact_id': chart['content_ref'], 'sha256': '0' * 64,
|
||||
'byte_length': artifact.byte_length, 'content_type': 'application/json'}], storage,
|
||||
db=db, run_id=step['scenario_run_id'])
|
||||
# #endregion Test.ChartHealth.OwnedIndependent.Tampered
|
||||
# #endregion Test.ChartHealth.OwnedIndependent
|
||||
@@ -0,0 +1,71 @@
|
||||
# #region Test.ChartHealth.ResponseWorker [C:3] [TYPE Module] [SEMANTICS http,async,filters,stale]
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Responses]
|
||||
# @TEST_INVARIANT Late errors cannot cross the newest saved-chart filter generation, and foreign async jobs are ignored.
|
||||
import json
|
||||
from hashlib import sha256
|
||||
import pytest
|
||||
from src.services.dashboard_testing.execution.providers.browser_chart_responses import ChartResponseCollector
|
||||
|
||||
|
||||
# #region Test.ChartHealth.ResponseWorker.Request [C:1] [TYPE Class]
|
||||
class Request:
|
||||
# #region Test.ChartHealth.ResponseWorker.Request.Init [C:1] [TYPE Function]
|
||||
# @BRIEF Literal external request snapshot, independent of production attribution helpers.
|
||||
def __init__(self, chart, value):
|
||||
self.url = 'http://superset.invalid/api/v1/chart/data'
|
||||
self.method = 'POST'
|
||||
self.post_data_json = {'form_data':{'slice_id':chart},'datasource':{'id':2,'type':'table'},
|
||||
'queries':[{'filters':[{'col':'metric','op':'IN','val':[value]}]}]}
|
||||
# #endregion Test.ChartHealth.ResponseWorker.Request.Init
|
||||
# #endregion Test.ChartHealth.ResponseWorker.Request
|
||||
|
||||
|
||||
# #region Test.ChartHealth.ResponseWorker.Response [C:1] [TYPE Class]
|
||||
class Response:
|
||||
# #region Test.ChartHealth.ResponseWorker.Response.Init [C:1] [TYPE Function]
|
||||
def __init__(self, request, payload, status=200, url=None):
|
||||
self.request,self.raw,self.status = request,json.dumps(payload).encode(),status
|
||||
self.url = url or request.url
|
||||
# #endregion Test.ChartHealth.ResponseWorker.Response.Init
|
||||
# #region Test.ChartHealth.ResponseWorker.Response.Body [C:1] [TYPE Function]
|
||||
async def body(self):
|
||||
return self.raw
|
||||
# #endregion Test.ChartHealth.ResponseWorker.Response.Body
|
||||
# #endregion Test.ChartHealth.ResponseWorker.Response
|
||||
|
||||
|
||||
# #region Test.ChartHealth.ResponseWorker.Stale [C:2] [TYPE Function]
|
||||
@pytest.mark.asyncio
|
||||
async def test_late_old_filter_error_cannot_contaminate_new_success_or_other_chart():
|
||||
collector = ChartResponseCollector(None,[1])
|
||||
old,new,foreign = Request(1,'old'),Request(1,'new'),Request(99,'new')
|
||||
collector.on_request(old)
|
||||
collector.on_request(new)
|
||||
collector.on_request(foreign)
|
||||
await collector.read_response(Response(old,{'errors':[{'message':'Code: 386. old-filter failure'}]},500))
|
||||
await collector.read_response(Response(new,{'result':[{'data':[{'value':0}]}]}))
|
||||
await collector.read_response(Response(foreign,{'errors':[{'message':'foreign failure'}]},500))
|
||||
assert collector.current_error(1) is None and collector.current_error(99) is None
|
||||
assert len(collector.requests) == 2
|
||||
# #endregion Test.ChartHealth.ResponseWorker.Stale
|
||||
|
||||
|
||||
# #region Test.ChartHealth.ResponseWorker.Async [C:2] [TYPE Function]
|
||||
@pytest.mark.asyncio
|
||||
async def test_polling_errors_require_exact_current_job_and_keep_original_response_digest():
|
||||
collector = ChartResponseCollector(None,[1])
|
||||
request = Request(1,'sales')
|
||||
collector.on_request(request)
|
||||
await collector.read_response(Response(request,{'result':[{'job_id':'owned-job','status':'pending'}]},202))
|
||||
polling = Response(Request(99,'ignored'),{'result':[{'job_id':'foreign-job','status':'error','errors':[{'message':'foreign'}]}]},url='http://superset.invalid/api/v1/async_event/')
|
||||
await collector.read_response(polling)
|
||||
assert collector.current_error(1) is None
|
||||
current = Response(Request(99,'ignored'),{'result':[{'job_id':'owned-job','status':'error','errors':[{'error_type':'GENERIC_DB_ENGINE_ERROR','message':'Code: 386. DB::Exception: incompatible types'}]}]},url='http://superset.invalid/api/v1/async_event/')
|
||||
await collector.read_response(current)
|
||||
error = collector.current_error(1)
|
||||
assert error['message'] == 'Code: 386. DB::Exception: incompatible types'
|
||||
assert error['attribution']['load_generation'] == 1
|
||||
assert error['attribution']['original_response_sha256'] == sha256(current.raw).hexdigest()
|
||||
assert error['attribution']['original_byte_length'] == len(current.raw)
|
||||
# #endregion Test.ChartHealth.ResponseWorker.Async
|
||||
# #endregion Test.ChartHealth.ResponseWorker
|
||||
@@ -0,0 +1,77 @@
|
||||
# #region Test.ChartHealth.SdkWorker [C:3] [TYPE Module] [SEMANTICS physical-request,SDK,budget,credentials]
|
||||
# @TEST_INVARIANT The real encrypted credential boundary feeds one external SDK request with retries disabled and declared budgets.
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.BoundedClient]
|
||||
import json
|
||||
from types import SimpleNamespace
|
||||
import pytest
|
||||
from src.core.encryption import get_encryption_manager
|
||||
from src.models.llm import LLMProvider
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_llm_client import BoundedHealthClient
|
||||
|
||||
|
||||
# #region Test.ChartHealth.SdkWorker.Context [C:2] [TYPE Function]
|
||||
@pytest.fixture
|
||||
def sdk_context(seeded_execution,monkeypatch):
|
||||
from src.services.dashboard_testing.execution.providers import browser_health_llm_client as seam
|
||||
db,calls = seeded_execution,[]
|
||||
provider = LLMProvider(id='bounded-health-sdk',name='Fake SDK only',provider_type='openai',
|
||||
base_url='http://fixture.invalid/v1',api_key=get_encryption_manager().encrypt('fake-key-never-sent'),
|
||||
default_model='vision-fixture',is_active=True,is_multimodal=True,supports_json_object=True)
|
||||
db.add(provider)
|
||||
db.commit()
|
||||
response = SimpleNamespace(choices=[SimpleNamespace(finish_reason='stop',message=SimpleNamespace(content=json.dumps({'verdict':'pass','usage':{'input_tokens':9999}})))],
|
||||
usage=SimpleNamespace(prompt_tokens=17,completion_tokens=9))
|
||||
# #region Test.ChartHealth.SdkWorker.Context.Boundary [C:2] [TYPE Class]
|
||||
class ExternalSdk:
|
||||
# #region Test.ChartHealth.SdkWorker.Context.Boundary.Init [C:1] [TYPE Function]
|
||||
def __init__(self, provider_type,key,url,model):
|
||||
calls.append(('credential',key))
|
||||
self.client,self.chat,self.completions = self,self,self
|
||||
# #endregion Test.ChartHealth.SdkWorker.Context.Boundary.Init
|
||||
# #region Test.ChartHealth.SdkWorker.Context.Boundary.Options [C:1] [TYPE Function]
|
||||
def with_options(self, **options):
|
||||
calls.append(('options',options))
|
||||
return self
|
||||
# #endregion Test.ChartHealth.SdkWorker.Context.Boundary.Options
|
||||
# #region Test.ChartHealth.SdkWorker.Context.Boundary.Create [C:1] [TYPE Function]
|
||||
async def create(self, **options):
|
||||
calls.append(('request',options))
|
||||
return response
|
||||
# #endregion Test.ChartHealth.SdkWorker.Context.Boundary.Create
|
||||
# #region Test.ChartHealth.SdkWorker.Context.Boundary.Close [C:1] [TYPE Function]
|
||||
async def close(self):
|
||||
calls.append(('closed',True))
|
||||
# #endregion Test.ChartHealth.SdkWorker.Context.Boundary.Close
|
||||
# #endregion Test.ChartHealth.SdkWorker.Context.Boundary
|
||||
monkeypatch.setattr(seam,'LLMClient',ExternalSdk)
|
||||
spec = SimpleNamespace(model_id='vision-fixture',limits=SimpleNamespace(timeout_ms=1500,max_output_tokens=128))
|
||||
return BoundedHealthClient(db,provider,spec),calls,response
|
||||
# #endregion Test.ChartHealth.SdkWorker.Context
|
||||
|
||||
|
||||
# #region Test.ChartHealth.SdkWorker.Budgets [C:2] [TYPE Function]
|
||||
@pytest.mark.asyncio
|
||||
async def test_single_physical_request_uses_declared_limits_and_transport_usage(sdk_context):
|
||||
client,calls,_ = sdk_context
|
||||
result = await client.get_json_completion([{'role':'user','content':'Code: 386'}])
|
||||
assert calls[0] == ('credential','fake-key-never-sent')
|
||||
assert ('options',{'max_retries':0,'timeout':1.5}) in calls
|
||||
requests = [item for kind,item in calls if kind == 'request']
|
||||
assert len(requests) == 1
|
||||
assert requests[0]['max_tokens'] == 128 and requests[0]['response_format'] == {'type':'json_object'}
|
||||
assert result['usage'] == {'input_tokens':17,'output_tokens':9}
|
||||
assert calls[-1] == ('closed',True)
|
||||
# #endregion Test.ChartHealth.SdkWorker.Budgets
|
||||
|
||||
|
||||
# #region Test.ChartHealth.SdkWorker.Truncation [C:2] [TYPE Function]
|
||||
@pytest.mark.asyncio
|
||||
async def test_truncated_response_refuses_without_second_request_and_closes(sdk_context):
|
||||
client,calls,response = sdk_context
|
||||
response.choices[0].finish_reason = 'length'
|
||||
with pytest.raises(RuntimeError,match='EVALUATION_RESPONSE_TRUNCATED'):
|
||||
await client.get_json_completion([])
|
||||
assert sum(kind == 'request' for kind,_ in calls) == 1
|
||||
assert calls[-1] == ('closed',True)
|
||||
# #endregion Test.ChartHealth.SdkWorker.Truncation
|
||||
# #endregion Test.ChartHealth.SdkWorker
|
||||
@@ -0,0 +1,71 @@
|
||||
# #region Test.ChartHealth.SlowClock [C:2] [TYPE Module] [SEMANTICS health,slow,timeout,coverage]
|
||||
# @TEST_INVARIANT Slow healthy charts have independent deadlines; an immediate error remains a durable failure.
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Observer]
|
||||
from unittest.mock import AsyncMock, Mock
|
||||
import pytest
|
||||
from test_chart_health_connected_worker import health_context # noqa: F401
|
||||
from test_chart_health_owned_evaluation_authority import LAYOUT
|
||||
from src.services.dashboard_testing.execution.providers.browser_chart_health import observe_chart
|
||||
from src.services.dashboard_testing.execution.providers.browser_chart_responses import ChartResponseCollector
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_manifest import parse_health_manifest
|
||||
|
||||
# #region Test.ChartHealth.SlowClock.Clock [C:1] [TYPE Class]
|
||||
class ExternalClock:
|
||||
# #region Test.ChartHealth.SlowClock.Clock.Init [C:1] [TYPE Function]
|
||||
def __init__(self):
|
||||
self.now = 0.0
|
||||
# #endregion Test.ChartHealth.SlowClock.Clock.Init
|
||||
# #region Test.ChartHealth.SlowClock.Clock.Read [C:1] [TYPE Function]
|
||||
def __call__(self):
|
||||
return self.now
|
||||
# #endregion Test.ChartHealth.SlowClock.Clock.Read
|
||||
# #endregion Test.ChartHealth.SlowClock.Clock
|
||||
|
||||
# #region Test.ChartHealth.SlowClock.Chart [C:1] [TYPE Function]
|
||||
def external_chart(state):
|
||||
chart = Mock()
|
||||
chart.wait_for = AsyncMock()
|
||||
chart.count = AsyncMock(return_value=1)
|
||||
chart.scroll_into_view_if_needed = AsyncMock()
|
||||
chart.evaluate = AsyncMock(side_effect=state)
|
||||
return chart
|
||||
# #endregion Test.ChartHealth.SlowClock.Chart
|
||||
|
||||
# #region Test.ChartHealth.SlowClock.Success [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT Simulated 45/60-second loading does not delay immediate Code386 retention or consume another chart's deadline.
|
||||
@pytest.mark.parametrize('seconds', [45.0, 60.0])
|
||||
@pytest.mark.asyncio
|
||||
async def test_slow_chart_and_immediate_error_keep_separate_observation_budgets(health_context, monkeypatch, seconds):
|
||||
_, journal, *_ = health_context
|
||||
source = parse_health_manifest(LAYOUT)
|
||||
journal.freeze_source(source)
|
||||
clock = ExternalClock()
|
||||
monkeypatch.setattr('src.services.dashboard_testing.execution.providers.browser_chart_health.monotonic', clock)
|
||||
error = external_chart(lambda *_args, **_kwargs: {'error': 'Code: 386. NO_COMMON_TYPE', 'loading': False, 'ready': False})
|
||||
# #region Test.ChartHealth.SlowClock.Success.Render [C:1] [TYPE Function]
|
||||
def render(*_args, **_kwargs):
|
||||
clock.now += seconds / 3
|
||||
return {'error': '', 'loading': clock.now < seconds, 'ready': clock.now >= seconds}
|
||||
# #endregion Test.ChartHealth.SlowClock.Success.Render
|
||||
slow = external_chart(render)
|
||||
absent_native = Mock()
|
||||
absent_native.count = AsyncMock(return_value=0)
|
||||
panel = Mock()
|
||||
panel.locator.side_effect = {
|
||||
'[data-test="chart-grid-component"][data-test-chart-id="11"]': absent_native,
|
||||
'[data-test="chart-grid-component"][data-test-chart-id="12"]': absent_native,
|
||||
'#chart-id-11': error, '#chart-id-12': slow,
|
||||
}.__getitem__
|
||||
collector = ChartResponseCollector(None, [11, 12])
|
||||
first = await observe_chart(None, panel, source['placements'][0], 70, collector, journal)
|
||||
assert first['status'] == 'failed' and first['database_code'] == '386' and first['duration_seconds'] == 0
|
||||
journal.append_chart(first)
|
||||
second = await observe_chart(None, panel, source['placements'][1], 70, collector, journal)
|
||||
assert second['status'] == 'healthy' and second['duration_seconds'] == seconds
|
||||
assert error.evaluate.await_count == 1 and slow.evaluate.await_count == 3
|
||||
journal.append_chart(second)
|
||||
result = journal.finish('passed', 'CHART_HEALTH_COMPLETE')
|
||||
assert result['status'] == 'failed' and result['complete'] is True
|
||||
assert result['coverage']['checked'] == 2 and result['coverage']['errored'] == 1
|
||||
# #endregion Test.ChartHealth.SlowClock.Success
|
||||
# #endregion Test.ChartHealth.SlowClock
|
||||
@@ -0,0 +1,168 @@
|
||||
# #region Test.ChartHealth.VlmIndependent [C:2] [TYPE Module] [SEMANTICS health,provider,pin,owned,images]
|
||||
# @TEST_INVARIANT Fresh persisted visual provider configuration and owned diagnostic/image bytes precede model work.
|
||||
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Provider]
|
||||
from types import SimpleNamespace
|
||||
import base64
|
||||
import json
|
||||
from copy import deepcopy
|
||||
from unittest.mock import AsyncMock, Mock
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from src.models.llm import LLMProvider
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_provider import admit_health_provider
|
||||
import pytest
|
||||
from test_chart_health_connected_worker import health_context # noqa: F401
|
||||
from test_chart_health_owned_evaluation_authority import append, LAYOUT
|
||||
from src.services.dashboard_testing.execution.chart_health_tab_store import store_health_artifact, tab_diagnostics
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_manifest import parse_health_manifest
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_evaluation import evaluation_inputs
|
||||
from test_chart_health_sdk_worker import sdk_context # noqa: F401
|
||||
|
||||
# #region Test.ChartHealth.VlmIndependent.UuidJournal [C:1] [TYPE Function]
|
||||
def uuid_journal(context):
|
||||
from src.models.scenario_run import ScenarioRun, ScenarioStepRun
|
||||
from src.services.dashboard_testing.execution.worker import claim_step
|
||||
from src.services.dashboard_testing.execution.capacity import claim_capacity
|
||||
from src.services.dashboard_testing.execution.chart_health_store import ChartHealthJournal
|
||||
db, old, old_step, storage, limits, _ = context
|
||||
original = db.get(ScenarioRun, old.run_id)
|
||||
fields = {column.name: deepcopy(getattr(original, column.name)) for column in ScenarioRun.__table__.columns if column.name != 'id'}
|
||||
fields['idempotency_key'] = 'independent-whole-health-evaluation'
|
||||
run = ScenarioRun(id='126b6988-28fc-4697-8f6a-954d000443c1', **fields)
|
||||
db.add(run)
|
||||
db.flush()
|
||||
step = deepcopy(old_step)
|
||||
step['scenario_run_id'] = run.id
|
||||
run.runner_plan = deepcopy(original.runner_plan)
|
||||
db.add(ScenarioStepRun(run_id=run.id, logical_step_id='health-check', step_position=1, attempt=1, status='running'))
|
||||
claim_step(db, run.id, 'health-check', worker_id='whole-test', side_effect_key=None, idempotent=True, retry_safe=True, lease_seconds=120)
|
||||
capacity = claim_capacity(db, environment_id='preprod', environment_class='PREPROD', workload_class='browser', provider_id='browser', run_id=run.id, logical_step_id='health-check', ttl_seconds=120)
|
||||
db.commit()
|
||||
return ChartHealthJournal(step, storage, limits, capacity_lease_id=capacity['lease_id'])
|
||||
# #endregion Test.ChartHealth.VlmIndependent.UuidJournal
|
||||
|
||||
PIN = 'config_sha256:01a89824f8992e6a9b79cc577e88d48ea1e8057ad0dec367d7650a1726b76a29'
|
||||
|
||||
# #region Test.ChartHealth.VlmIndependent.Provider [C:1] [TYPE Function]
|
||||
@pytest.fixture
|
||||
def provider_context(health_context):
|
||||
db = health_context[0]
|
||||
provider = LLMProvider(id='health-provider', name='Fixture', provider_type='openai',
|
||||
base_url='http://fixture.invalid/v1', api_key='fixture-unused-no-decryption',
|
||||
default_model='vision-fixture', is_active=True, is_multimodal=True, max_images=2,
|
||||
context_window=8192, max_output_tokens=512, supports_json_object=True)
|
||||
db.add(provider)
|
||||
db.commit()
|
||||
policy = SimpleNamespace(spec=SimpleNamespace(provider_id='health-provider', provider_version=PIN,
|
||||
model_id='vision-fixture', model_version='configured-model:vision-fixture',
|
||||
limits=SimpleNamespace(max_images=2)))
|
||||
return db, provider, policy
|
||||
# #endregion Test.ChartHealth.VlmIndependent.Provider
|
||||
|
||||
# #region Test.ChartHealth.VlmIndependent.Positive [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT Literal authored public configuration pin admits the matching persisted provider.
|
||||
def test_exact_visual_provider_pin_admits_without_network_or_key_read(provider_context):
|
||||
db, provider, policy = provider_context
|
||||
assert admit_health_provider(db, policy) is provider
|
||||
# #endregion Test.ChartHealth.VlmIndependent.Positive
|
||||
|
||||
# #region Test.ChartHealth.VlmIndependent.Changed [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT Changed URL/model or unavailable vision authority refuses prior to charged work.
|
||||
@pytest.mark.parametrize('field,value,code', [
|
||||
('base_url', 'http://different.invalid/v1', 'CHART_HEALTH_VLM_PROVIDER_CHANGED'),
|
||||
('default_model', 'different-model', 'CHART_HEALTH_VLM_PROVIDER_CHANGED'),
|
||||
('is_active', False, 'CHART_HEALTH_VLM_PROVIDER_UNAVAILABLE'),
|
||||
('is_multimodal', False, 'CHART_HEALTH_VLM_PROVIDER_UNAVAILABLE')])
|
||||
def test_changed_provider_configuration_rejects_old_authored_pin(provider_context, field, value, code):
|
||||
db, provider, policy = provider_context
|
||||
setattr(provider, field, value)
|
||||
db.commit()
|
||||
with pytest.raises(ValueError, match=code):
|
||||
admit_health_provider(db, policy)
|
||||
# #endregion Test.ChartHealth.VlmIndependent.Changed
|
||||
|
||||
# #region Test.ChartHealth.VlmIndependent.ImageBudget [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT Caller cannot enlarge image budget past pinned provider capability.
|
||||
def test_image_budget_exceeding_matching_provider_is_rejected(provider_context):
|
||||
db, _, policy = provider_context
|
||||
policy.spec.limits.max_images = 3
|
||||
with pytest.raises(ValueError, match='CHART_HEALTH_VLM_IMAGE_BUDGET'):
|
||||
admit_health_provider(db, policy)
|
||||
# #endregion Test.ChartHealth.VlmIndependent.ImageBudget
|
||||
|
||||
# #region Test.ChartHealth.VlmIndependent.OwnedImages [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT Actual same-attempt PNG bytes accompany actual Code386 diagnostic contents, not references alone.
|
||||
def test_owned_image_and_error_contents_cross_real_evaluation_input_boundary(health_context, registry_engine, monkeypatch):
|
||||
_, journal, _, _, *_ = health_context
|
||||
monkeypatch.setattr('src.services.dashboard_testing.execution.providers.browser_health_evaluation.SessionLocal', sessionmaker(bind=registry_engine))
|
||||
journal.freeze_source(parse_health_manifest(LAYOUT))
|
||||
append(journal, 11)
|
||||
image = base64.b64decode('iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8/x8AAwMCAO+aRZsAAAAASUVORK5CYII=')
|
||||
receipt = store_health_artifact(journal, image, 'owned-tab.png', 'image/png')
|
||||
manifest, payload, images, artifacts = evaluation_inputs(journal, tab_diagnostics(journal, []), [{'receipt': receipt}])
|
||||
assert images == [(image, 'image/png')]
|
||||
assert len(manifest) == 2 and len(artifacts) == 2
|
||||
assert payload['items'][0]['content']['chart_id'] == 11
|
||||
assert payload['items'][0]['content']['database_code'] == '386'
|
||||
assert 'incompatible types' in payload['items'][0]['content']['message']
|
||||
# #endregion Test.ChartHealth.VlmIndependent.OwnedImages
|
||||
|
||||
# #region Test.ChartHealth.VlmIndependent.WholeTab [C:2] [TYPE Function]
|
||||
# @TEST_INVARIANT Real tab evaluation retains image/content/raw/record and permits later chart while deterministic FAIL dominates model PASS.
|
||||
@pytest.mark.asyncio
|
||||
async def test_whole_tab_evaluation_then_next_chart_preserves_failure(health_context, sdk_context, registry_engine, monkeypatch):
|
||||
from src.services.dashboard_testing.scenario.models import AgentEvaluationSpec
|
||||
from src.services.dashboard_testing.scenario.metric_evaluation_provider import provider_public_config_digest
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_evaluation import evaluate_tab
|
||||
from src.services.dashboard_testing.execution.providers.browser_health_evaluation_inputs import HealthEvaluationPolicy
|
||||
journal = uuid_journal(health_context)
|
||||
sdk, calls, response = sdk_context
|
||||
monkeypatch.setattr('src.services.dashboard_testing.execution.providers.browser_health_evaluation.SessionLocal', sessionmaker(bind=registry_engine))
|
||||
spec = AgentEvaluationSpec.model_validate({'schema_version': 1,
|
||||
'spec_id': 'e5e5e5e5-e5e5-4e5e-8e5e-e5e5e5e5e5e5', 'provider_id': sdk.provider.id,
|
||||
'provider_version': 'config_sha256:' + provider_public_config_digest(sdk.provider),
|
||||
'model_id': 'vision-fixture', 'model_version': 'configured-model:vision-fixture',
|
||||
'prompt_template_id': 'health-prompt', 'prompt_template_version': '1.0.0', 'prompt_template_hash': 'a' * 64,
|
||||
'evidence_refs': ['health-check'], 'comparison_refs': [], 'output_schema': 'agent-evaluation.schema.json',
|
||||
'decision_policy': {'policy_id': 'baseline-semantic', 'version': '1.0.0'},
|
||||
'limits': {'timeout_ms': 1500, 'max_images': 2, 'max_input_tokens': 32000, 'max_output_tokens': 128, 'max_cost': '1.00', 'currency': 'USD'},
|
||||
'trust_policy_hash': 'b' * 64,
|
||||
'criteria': [{'criterion_id': 'visual', 'criterion_kind': 'semantic', 'description': 'Explain visible chart errors', 'comparison_id': None}]})
|
||||
policy = HealthEvaluationPolicy(mode='all_visited', spec=spec)
|
||||
journal.freeze_source(parse_health_manifest(LAYOUT))
|
||||
append(journal, 11)
|
||||
diagnostic_ref = tab_diagnostics(journal, [])[0]['receipt']['content_ref']
|
||||
response.choices[0].message.content = json.dumps({'verdict': 'pass', 'confidence': 0.9,
|
||||
'findings': [{'criterion_id': 'visual', 'severity': 'info', 'message': 'Error card text is readable',
|
||||
'evidence_artifact_ids': [diagnostic_ref]}]})
|
||||
# External browser screenshot protocol only; local capture/evaluation/authority are unmocked.
|
||||
chart = Mock()
|
||||
chart.count = AsyncMock(return_value=1)
|
||||
chart.scroll_into_view_if_needed = AsyncMock()
|
||||
chart.screenshot = AsyncMock(return_value=base64.b64decode('iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8/x8AAwMCAO+aRZsAAAAASUVORK5CYII='))
|
||||
absent_native = Mock()
|
||||
absent_native.count = AsyncMock(return_value=0)
|
||||
panel = Mock()
|
||||
panel.locator.side_effect = {
|
||||
'[data-test="chart-grid-component"][data-test-chart-id="11"]': absent_native,
|
||||
'#chart-id-11': chart,
|
||||
}.__getitem__
|
||||
await evaluate_tab(None, panel, [], [{'id': 'CHART-a', 'chart_id': 11}], journal, policy, 5)
|
||||
append(journal, 12, 'healthy')
|
||||
result = journal.finish('passed', 'CHART_HEALTH_COMPLETE')
|
||||
assert result['status'] == 'failed' and result['complete'] is True
|
||||
evaluation = result['chart_health']['tab_evaluations'][0]
|
||||
assert evaluation['status'] == 'succeeded' and evaluation['advisory_verdict'] == 'pass', evaluation
|
||||
assert len(evaluation['artifacts']) == 4 # diagnostic, actual image, SDK response and typed record
|
||||
requests = [options for kind, options in calls if kind == 'request']
|
||||
assert len(requests) == 1
|
||||
assert 'Code: 386' in json.dumps(requests[0]['messages'])
|
||||
assert 'data:image/png;base64,' in json.dumps(requests[0]['messages'])
|
||||
from src.models.provider_capacity import CapacityLease
|
||||
leases = health_context[0].query(CapacityLease).filter_by(run_id=journal.run_id, workload_class='agent_evaluation').all()
|
||||
assert len(leases) == 1 and leases[0].status == 'released'
|
||||
assert leases[0].provider_version == spec.provider_version.removeprefix('config_sha256:')
|
||||
assert len(leases[0].provider_version) == 64
|
||||
record = json.loads(journal.storage.retrieve(evaluation['artifacts'][-1]['content_ref']))
|
||||
assert record['provider_version'] == spec.provider_version and record['comparison_ids'] == []
|
||||
# #endregion Test.ChartHealth.VlmIndependent.WholeTab
|
||||
# #endregion Test.ChartHealth.VlmIndependent
|
||||
@@ -30,6 +30,8 @@ from src.services.dashboard_testing.query_executor import execute_dashboard_quer
|
||||
# ── Helpers ────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
# #region Test.DashboardTesting.QueryExecutor.Own._chart_data_response [C:3] [TYPE Function]
|
||||
# @BRIEF Build fixture response bytes and their digest from the supplied parsed result.
|
||||
def _chart_data_response(result_dict: dict) -> ChartDataResponse:
|
||||
"""Build a ChartDataResponse from a result dict (simulates httpx raw_response)."""
|
||||
raw = json.dumps(result_dict, sort_keys=True, default=str).encode("utf-8")
|
||||
@@ -39,6 +41,7 @@ def _chart_data_response(result_dict: dict) -> ChartDataResponse:
|
||||
raw_bytes=raw,
|
||||
source_response_hash=hashlib.sha256(raw).hexdigest(),
|
||||
)
|
||||
# #endregion Test.DashboardTesting.QueryExecutor.Own._chart_data_response
|
||||
|
||||
|
||||
# #region Test.DashboardTesting.QueryExecutor.NoSQLRejection [C:3] [TYPE Function] [SEMANTICS testing,baseline,no-sql,security]
|
||||
@@ -123,6 +126,7 @@ async def test_superset_error_preserved():
|
||||
assert "SUPERSET" in result.warnings[0].code
|
||||
|
||||
|
||||
# #region Test.DashboardTesting.QueryExecutor.Own.test_chart_data_error_body_cannot_become_empty_baseline_value [C:3] [TYPE Function]
|
||||
@pytest.mark.asyncio
|
||||
async def test_chart_data_error_body_cannot_become_empty_baseline_value():
|
||||
client = AsyncMock()
|
||||
@@ -133,8 +137,15 @@ async def test_chart_data_error_body_cannot_become_empty_baseline_value():
|
||||
environment_id="dev", dashboard_id=42, chart_id=128, result_key="revenue",
|
||||
normalized_filters=NormalizedFilterContext(filters=[], filters_hash="sha256:empty"),
|
||||
)
|
||||
with pytest.raises(ValueError, match="chart-data rejected"):
|
||||
await execute_dashboard_query_envelope(client, request)
|
||||
result = await execute_dashboard_query_envelope(client, request)
|
||||
assert result.normalized_value.kind == ValueKind.UNKNOWN
|
||||
assert result.diagnostic["status"] == "failed"
|
||||
assert result.diagnostic["message"] == "invalid query"
|
||||
assert result.payload_kind == "sanitized_diagnostic"
|
||||
assert result.diagnostic["original_response_sha256"] != result.source_response_hash
|
||||
assert b"invalid query" in result.raw_response_content
|
||||
# #endregion Test.DashboardTesting.QueryExecutor.Own.test_chart_data_error_body_cannot_become_empty_baseline_value
|
||||
|
||||
# #endregion Test.DashboardTesting.QueryExecutor.SupersetErrorTaxonomy
|
||||
|
||||
# #region Test.DashboardTesting.QueryExecutor.TemporalFilterMapping [C:3] [TYPE Function] [SEMANTICS testing,baseline,temporal,filter]
|
||||
@@ -179,6 +190,8 @@ async def test_temporal_filter_mapped_correctly():
|
||||
|
||||
# ── Authoritative model test helpers ─────────────────────────────────
|
||||
|
||||
# #region Test.DashboardTesting.QueryExecutor.Own._make_basic_query_model [C:3] [TYPE Function]
|
||||
# @BRIEF Build a minimal approved query model with the requested charts and filter targets.
|
||||
def _make_basic_query_model(
|
||||
chart_ids: list[int] | None = None,
|
||||
filter_targets: dict[str, list[int]] | None = None,
|
||||
@@ -236,6 +249,7 @@ def _make_basic_query_model(
|
||||
native_filters=native_filters,
|
||||
query_model_fingerprint=fingerprint,
|
||||
)
|
||||
# #endregion Test.DashboardTesting.QueryExecutor.Own._make_basic_query_model
|
||||
|
||||
# ── Authoritative model tests ────────────────────────────────────────
|
||||
|
||||
@@ -431,26 +445,35 @@ from src.services.dashboard_testing.query_executor import (
|
||||
)
|
||||
|
||||
|
||||
# #region Test.DashboardTesting.QueryExecutor.Own.test_check_forbidden_fields_rejects_sql [C:1] [TYPE Function]
|
||||
def test_check_forbidden_fields_rejects_sql():
|
||||
with pytest.raises(ValueError, match="Forbidden fields"):
|
||||
_check_forbidden_fields({"sql": "SELECT 1", "chart_id": 1})
|
||||
# #endregion Test.DashboardTesting.QueryExecutor.Own.test_check_forbidden_fields_rejects_sql
|
||||
|
||||
|
||||
# #region Test.DashboardTesting.QueryExecutor.Own.test_check_forbidden_fields_clean_passes [C:1] [TYPE Function]
|
||||
def test_check_forbidden_fields_clean_passes():
|
||||
_check_forbidden_fields({"chart_id": 1, "result_key": "k"}) # no raise
|
||||
# #endregion Test.DashboardTesting.QueryExecutor.Own.test_check_forbidden_fields_clean_passes
|
||||
|
||||
|
||||
# #region Test.DashboardTesting.QueryExecutor.Own.test_extract_query_result_non_dict_first_record [C:1] [TYPE Function]
|
||||
def test_extract_query_result_non_dict_first_record():
|
||||
raw_value, query_id = _extract_query_result({"result": [[1, 2]], "query_id": "q-1"}, "k")
|
||||
assert raw_value == [1, 2]
|
||||
assert query_id == "q-1"
|
||||
# #endregion Test.DashboardTesting.QueryExecutor.Own.test_extract_query_result_non_dict_first_record
|
||||
|
||||
|
||||
# #region Test.DashboardTesting.QueryExecutor.Own.test_extract_query_result_empty [C:1] [TYPE Function]
|
||||
def test_extract_query_result_empty():
|
||||
raw_value, query_id = _extract_query_result({"result": []}, "k")
|
||||
assert raw_value is None
|
||||
# #endregion Test.DashboardTesting.QueryExecutor.Own.test_extract_query_result_empty
|
||||
|
||||
|
||||
# #region Test.DashboardTesting.QueryExecutor.Own.test_build_filters_in_operator_with_values [C:3] [TYPE Function]
|
||||
def test_build_filters_in_operator_with_values():
|
||||
from src.schemas.dashboard_testing import FilterValue, NormalizedFilter, NormalizedFilterContext
|
||||
nf = NormalizedFilterContext(
|
||||
@@ -465,8 +488,10 @@ def test_build_filters_in_operator_with_values():
|
||||
out = _build_chart_data_filters(nf)
|
||||
assert out[0]["operator"] == "IN"
|
||||
assert out[0]["comparator"] == ["a", "b"]
|
||||
# #endregion Test.DashboardTesting.QueryExecutor.Own.test_build_filters_in_operator_with_values
|
||||
|
||||
|
||||
# #region Test.DashboardTesting.QueryExecutor.Own.test_build_filters_generic_fallback [C:3] [TYPE Function]
|
||||
def test_build_filters_generic_fallback():
|
||||
from src.schemas.dashboard_testing import FilterValue, NormalizedFilter, NormalizedFilterContext
|
||||
nf = NormalizedFilterContext(
|
||||
@@ -481,8 +506,10 @@ def test_build_filters_generic_fallback():
|
||||
out = _build_chart_data_filters(nf)
|
||||
assert out[0]["operator"] == "=="
|
||||
assert out[0]["comparator"] == "x"
|
||||
# #endregion Test.DashboardTesting.QueryExecutor.Own.test_build_filters_generic_fallback
|
||||
|
||||
|
||||
# #region Test.DashboardTesting.QueryExecutor.Own.test_envelope_requires_chart_or_dataset [C:3] [TYPE Function]
|
||||
@pytest.mark.asyncio
|
||||
async def test_envelope_requires_chart_or_dataset():
|
||||
request = ExecuteQueryRequest.model_construct(
|
||||
@@ -491,4 +518,5 @@ async def test_envelope_requires_chart_or_dataset():
|
||||
)
|
||||
with pytest.raises(ValueError, match="Either chart_id or dataset_id"):
|
||||
await execute_dashboard_query_envelope(AsyncMock(), request)
|
||||
# #endregion Test.DashboardTesting.QueryExecutor.Own.test_envelope_requires_chart_or_dataset
|
||||
# #endregion Test.DashboardTesting.QueryExecutor.Branches
|
||||
|
||||
116
docker/full-flow/chart_health_canary_seed.py
Normal file
116
docker/full-flow/chart_health_canary_seed.py
Normal file
@@ -0,0 +1,116 @@
|
||||
#!/usr/bin/env python3
|
||||
# #region FullFlow.ChartHealth.CanarySeed [C:4] [TYPE Module] [SEMANTICS superset,clickhouse,fixture,type-conflict,recovery]
|
||||
# @PRE Existing isolatedDEV Superset and finance ClickHouse binding/source are available.
|
||||
# @POST Six isolated native charts expose five actual ClickHouse386 errors then recover by changing only their own metric.
|
||||
# @INVARIANT Existing finance dashboards/datasets/data/release evidence are never modified.
|
||||
import argparse
|
||||
import json
|
||||
from uuid import UUID,uuid5
|
||||
from superset.app import create_app
|
||||
|
||||
NAMESPACE = UUID('496dfb4d-c02b-4cdb-8bcc-bd7c17fe851b')
|
||||
BROKEN = "sum(if(amount_cents > 0, toInt64(1), 'chart-health-type-conflict'))"
|
||||
HEALTHY = 'sum(amount_cents)'
|
||||
NAMES = ['Canary Debt','Canary Cash','Canary Margin','Canary Forecast','Canary Reserve','Canary Healthy']
|
||||
|
||||
|
||||
# #region FullFlow.ChartHealth.CanarySeed.Identity [C:1] [TYPE Function]
|
||||
# @POST Namespace isolates every fixture asset from existing finance identities.
|
||||
def identity(key):
|
||||
return uuid5(NAMESPACE,key)
|
||||
# #endregion FullFlow.ChartHealth.CanarySeed.Identity
|
||||
|
||||
|
||||
# #region FullFlow.ChartHealth.CanarySeed.Source [C:3] [TYPE Function]
|
||||
# @POST The own virtual dataset reads one real existing source row; no ClickHouse DDL/DML is issued.
|
||||
def own_source(session, owner):
|
||||
from superset.models.core import Database
|
||||
from superset.connectors.sqla.models import SqlaTable,SqlMetric
|
||||
database = session.query(Database).filter_by(uuid=uuid5(UUID('a75d5902-c347-490c-a282-f00acc1e1230'),'database')).one()
|
||||
source = session.query(SqlaTable).filter_by(uuid=identity('dataset')).first()
|
||||
if source is None:
|
||||
source = SqlaTable(table_name='ss_tools_chart_health_canary',database=database,owners=[owner],uuid=identity('dataset'),
|
||||
sql='SELECT amount_cents FROM finance_lab.finance_positions LIMIT 1')
|
||||
session.add(source)
|
||||
source.fetch_metadata()
|
||||
source.metrics = [SqlMetric(metric_name='canary_value',expression=BROKEN),SqlMetric(metric_name='healthy_value',expression=HEALTHY)]
|
||||
session.flush()
|
||||
if source.sql != 'SELECT amount_cents FROM finance_lab.finance_positions LIMIT 1':
|
||||
raise ValueError('CHART_HEALTH_CANARY_SOURCE_CONFLICT')
|
||||
return source
|
||||
# #endregion FullFlow.ChartHealth.CanarySeed.Source
|
||||
|
||||
|
||||
# #region FullFlow.ChartHealth.CanarySeed.Charts [C:3] [TYPE Function]
|
||||
# @POST Only namespaced chart metadata is created; saved query errors come from actual engine metric typing.
|
||||
def own_charts(session, owner, source):
|
||||
from superset.models.slice import Slice
|
||||
charts = []
|
||||
for index,name in enumerate(NAMES):
|
||||
chart = session.query(Slice).filter_by(uuid=identity(f'chart:{index}')).first()
|
||||
if chart is None:
|
||||
params = {'viz_type':'big_number_total','datasource':f'{source.id}__table','metric':'healthy_value' if index == 5 else 'canary_value',
|
||||
'time_range':'No filter','adhoc_filters':[],'row_limit':1,'number_format':',.0f'}
|
||||
chart = Slice(slice_name=name,viz_type='big_number_total',datasource_id=source.id,datasource_type='table',
|
||||
params=json.dumps(params),owners=[owner],uuid=identity(f'chart:{index}'))
|
||||
session.add(chart)
|
||||
session.flush()
|
||||
if chart.datasource_id != source.id:
|
||||
raise ValueError('CHART_HEALTH_CANARY_CHART_CONFLICT')
|
||||
charts.append(chart)
|
||||
return charts
|
||||
# #endregion FullFlow.ChartHealth.CanarySeed.Charts
|
||||
|
||||
|
||||
# #region FullFlow.ChartHealth.CanarySeed.Layout [C:3] [TYPE Function]
|
||||
# @POST Exact saved placement IDs and tab ancestry match real Superset layout semantics.
|
||||
def own_layout(charts):
|
||||
tabs = ['TAB-health-errors','TAB-health-healthy']
|
||||
layout = {'DASHBOARD_VERSION_KEY':'v2','ROOT_ID':{'id':'ROOT_ID','type':'ROOT','children':['GRID_ID']},
|
||||
'GRID_ID':{'id':'GRID_ID','type':'GRID','children':['TABS-health'],'parents':['ROOT_ID']},
|
||||
'TABS-health':{'id':'TABS-health','type':'TABS','children':tabs,'parents':['ROOT_ID','GRID_ID'],'meta':{}}}
|
||||
for tab_id,group in zip(tabs,(charts[:5],charts[5:])):
|
||||
parents = ['ROOT_ID','GRID_ID','TABS-health']
|
||||
rows = [f'{tab_id}-row-{index}' for index in range((len(group)+1)//2)]
|
||||
layout[tab_id] = {'id':tab_id,'type':'TAB','children':rows,'parents':parents,'meta':{'text':'Daily summary' if len(group) == 5 else 'Healthy other tab'}}
|
||||
for index,row_id in enumerate(rows):
|
||||
pair = group[index*2:index*2+2]
|
||||
layout[row_id] = {'id':row_id,'type':'ROW','children':[f'CHART-{item.id}' for item in pair],'parents':parents+[tab_id],'meta':{'background':'BACKGROUND_TRANSPARENT'}}
|
||||
for chart in pair:
|
||||
layout[f'CHART-{chart.id}'] = {'id':f'CHART-{chart.id}','type':'CHART','children':[],'parents':parents+[tab_id,row_id],
|
||||
'meta':{'chartId':chart.id,'uuid':str(chart.uuid),'sliceName':chart.slice_name,'width':6 if len(pair) == 2 else 12,'height':50}}
|
||||
return layout
|
||||
# #endregion FullFlow.ChartHealth.CanarySeed.Layout
|
||||
|
||||
|
||||
# #region FullFlow.ChartHealth.CanarySeed.Main [C:4] [TYPE Function]
|
||||
# @SIDE_EFFECT Changes only namespaced native fixture metric/dashboard metadata, never existing data or execution receipts.
|
||||
def main():
|
||||
phase = argparse.ArgumentParser()
|
||||
phase.add_argument('--phase',choices=['error','recovered'],required=True)
|
||||
args = phase.parse_args()
|
||||
with create_app().app_context():
|
||||
from superset import db,security_manager
|
||||
from superset.models.dashboard import Dashboard
|
||||
owner = security_manager.find_user(username='admin')
|
||||
if owner is None:
|
||||
raise ValueError('CHART_HEALTH_CANARY_ADMIN_UNAVAILABLE')
|
||||
source = own_source(db.session,owner)
|
||||
metric = next(item for item in source.metrics if item.metric_name == 'canary_value')
|
||||
if metric.expression not in {BROKEN,HEALTHY}:
|
||||
raise ValueError('CHART_HEALTH_CANARY_METRIC_CONFLICT')
|
||||
metric.expression = BROKEN if args.phase == 'error' else HEALTHY
|
||||
charts = own_charts(db.session,owner,source)
|
||||
dashboard = db.session.query(Dashboard).filter_by(uuid=identity('dashboard')).first()
|
||||
if dashboard is None:
|
||||
dashboard = Dashboard(dashboard_title='Isolated chart health canary',slug='chart-health-canary',published=True,owners=[owner],slices=charts,
|
||||
uuid=identity('dashboard'),position_json=json.dumps(own_layout(charts)),json_metadata=json.dumps({'native_filter_configuration':[],
|
||||
'timed_refresh_immune_slices':[],'expanded_slices':{},'refresh_frequency':0,'color_scheme':'supersetColors','label_colors':{},'shared_label_colors':{}}))
|
||||
db.session.add(dashboard)
|
||||
db.session.commit()
|
||||
print(json.dumps({'phase':args.phase,'dashboard_id':dashboard.id,'slug':dashboard.slug,'chart_ids':[item.id for item in charts],'dataset_id':source.id}))
|
||||
# #endregion FullFlow.ChartHealth.CanarySeed.Main
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
# #endregion FullFlow.ChartHealth.CanarySeed
|
||||
@@ -0,0 +1,70 @@
|
||||
<!-- #region ScenarioRunMonitor.Component.ChartHealth [C:4] [TYPE Component] [SEMANTICS chart,health,errors,coverage,diagnostic] -->
|
||||
<!-- @BRIEF Show named failed charts, tab context and partial coverage before technical evidence disclosures. -->
|
||||
<!-- @UX_STATE absent -> hidden by parent; observed -> counts and stable tab groups; partial -> explicit unchecked/unresolved counts alongside FAILED; expanded -> per-placement cause, filters/time/digest and owned evidence. -->
|
||||
<!-- @UX_STATE TabFilter(all|stableTabId): local display filter never mutates verdict, coverage or stored observations. -->
|
||||
<!-- @UX_REACTIVITY groups/runId derive only from the parent-selected run; keyed run selection remounts local filters/disclosures. -->
|
||||
<!-- @INVARIANT Model explanation never overwrites confirmed query failure and query failures never render numeric delta. -->
|
||||
<script lang="ts">
|
||||
import { chartErrorCauses } from "./chart-health";
|
||||
import type { ChartHealthGroup } from "./chart-health";
|
||||
import EvidenceViewer from "./EvidenceViewer.svelte";
|
||||
import { Button } from "$lib/ui";
|
||||
let { groups, runId, onselectstep }: { groups: ChartHealthGroup[]; runId: string; onselectstep?: (_id: string) => void } = $props();
|
||||
let tab = $state("");
|
||||
let chart = $state("");
|
||||
let showAll = $state(false);
|
||||
</script>
|
||||
<section class="mt-4 rounded-lg border border-border bg-surface p-3" aria-label="Chart query health">
|
||||
<h3 class="font-semibold text-text">Chart query health</h3>
|
||||
{#each groups as group (group.stepId)}
|
||||
<div class="mt-3 rounded border border-border p-3">
|
||||
<p class="font-medium text-destructive">{group.errors.length} charts failed <span class="text-text-muted">· {group.status}</span></p>
|
||||
{#each chartErrorCauses(group.errors) as cause (cause.cause)}
|
||||
<p class="mt-1 text-sm">{cause.cause} · {cause.charts.length} charts: {cause.charts.map(item => item.chart_name).join(", ")}</p>
|
||||
{/each}
|
||||
{#if Object.keys(group.coverage).length}
|
||||
<p class="mt-1 text-sm text-text-muted">Checked {String(group.coverage.checked ?? 0)} / {String(group.coverage.expected ?? 0)} · unresolved {String(group.coverage.unresolved ?? 0)} · unvisited {String(group.coverage.unvisited ?? 0)} · {group.coverage.complete === true ? "Complete coverage" : "Partial coverage"}</p>
|
||||
{/if}
|
||||
{#if group.tabs.length}
|
||||
<label class="mt-3 block text-sm">Tab
|
||||
<select class="ml-2 rounded border border-border bg-surface p-1" bind:value={tab}>
|
||||
<option value="">All tabs</option>
|
||||
{#each group.tabs as item (String(item.tab_id))}<option value={String(item.tab_id)}>{String(item.name)} · {String(item.errored)} errors</option>{/each}
|
||||
</select>
|
||||
</label>
|
||||
{/if}
|
||||
<label class="mt-2 block text-sm">Chart
|
||||
<select class="ml-2 rounded border border-border bg-surface p-1" bind:value={chart}>
|
||||
<option value="">All charts</option>{#each group.charts as item (item.placement_id)}<option value={item.placement_id}>{item.chart_name} · {item.status ?? "unavailable"}</option>{/each}
|
||||
</select>
|
||||
</label>
|
||||
<label class="mt-2 block text-sm"><input type="checkbox" bind:checked={showAll} /> Show healthy and unresolved observations</label>
|
||||
<ul class="mt-2 space-y-2">
|
||||
{#each group.charts.filter(error => (showAll || error.status === "failed" || chart === error.placement_id) && (!tab || error.tab_path.includes(tab)) && (!chart || error.placement_id === chart)) as error (error.placement_id)}
|
||||
<li class="rounded border border-border p-2">
|
||||
<strong>{error.chart_name}</strong><span class="ml-2 text-xs">{error.status ?? "failed"}</span><span class="ml-2 text-xs text-text-muted">{group.tabs.find(item => item.tab_id === error.tab_path.at(-1))?.name ?? error.tab_path.join(" / ")}</span>
|
||||
<p class="mt-1 whitespace-pre-wrap break-words text-sm">{error.message}</p>
|
||||
{#if error.database_code}<p class="text-xs text-text-muted">Database code {error.database_code}</p>{/if}
|
||||
<details class="mt-2 text-xs text-text-muted"><summary>Diagnostic evidence</summary>
|
||||
<p class="mt-1 break-all">{error.placement_id} · chart {error.chart_id ?? "unavailable"} · {error.origin ?? "unavailable"}</p>
|
||||
<p class="break-all">{error.observed_at ?? "unavailable"} · filters {error.filter_fingerprint ?? "unobserved"}</p>
|
||||
{#if error.truncated}<p>Message truncated; retained coverage remains explicit.</p>{/if}
|
||||
<p class="break-all">{error.sha256 ?? "See step evidence"}</p>
|
||||
{#if error.artifact_id && runId}<EvidenceViewer {runId} artifactId={error.artifact_id} contentType="application/json" sha256={error.sha256} label={error.chart_name} />{/if}
|
||||
</details>
|
||||
</li>
|
||||
{/each}
|
||||
</ul>
|
||||
{#if group.evaluations.length}
|
||||
<details class="mt-3 text-sm"><summary>Model review by tab (separate from query health)</summary>
|
||||
{#each group.evaluations as evaluation, index (index)}
|
||||
<p class="mt-2">{Array.isArray(evaluation.tab_path) ? evaluation.tab_path.join(" / ") : "Tab"} · {String(evaluation.status)} · {String(evaluation.advisory_verdict ?? "inconclusive")} · {evaluation.coverage_complete === true ? "Captured requested frames" : "Incomplete visual evidence"}</p>
|
||||
{#if Array.isArray(evaluation.findings)}{#each evaluation.findings as finding, index (index)}<p class="text-text-muted">{String(finding.message ?? "")}</p>{/each}{/if}
|
||||
{/each}
|
||||
</details>
|
||||
{/if}
|
||||
<Button variant="ghost" class="mt-2" onclick={() => onselectstep?.(group.stepId)}>Open step evidence</Button>
|
||||
</div>
|
||||
{/each}
|
||||
</section>
|
||||
<!-- #endregion ScenarioRunMonitor.Component.ChartHealth -->
|
||||
@@ -12,9 +12,12 @@
|
||||
<!-- @UX_STATE Selection(none|selected): failure/deviation controls emit onselectstep; the parent-owned selectedStepId drives the run-bound inspector and suppresses the matching duplicate deviation card. -->
|
||||
<!-- @UX_STATE Disclosures(collapsed|expanded): provenance and failure technical fields expand locally; failed checks and unavailable evidence retain authoritative statuses. -->
|
||||
<!-- @UX_REACTIVITY result, steps, plan, runId and selectedStepId come from the same parent-selected run; selection callbacks do not fetch or alter runtime results. -->
|
||||
<!-- @UX_STATE ChartHealth(absent|observed|partial): named persisted chart query failures render separately from numeric comparisons; partial coverage remains visible beside FAILED. -->
|
||||
<script lang="ts">
|
||||
import { t } from "$lib/i18n/index.svelte.js";
|
||||
import type { ScenarioExecutionResult, ScenarioStepRun } from "$lib/types/scenario-run";
|
||||
import ChartHealthCard from "./ChartHealthCard.svelte";
|
||||
import { chartHealthGroups } from "./chart-health";
|
||||
import RunStepInspector from "./RunStepInspector.svelte";
|
||||
import MetricDeviationCard from "./MetricDeviationCard.svelte";
|
||||
import { Button } from "$lib/ui";
|
||||
@@ -40,6 +43,7 @@
|
||||
} = $props();
|
||||
|
||||
const dt = $derived($t.dashboard_testing ?? {});
|
||||
const healthGroups = $derived(chartHealthGroups(steps));
|
||||
const differences = $derived(metricDifferences(steps));
|
||||
const hasPinnedBaselines = $derived(hasPinnedMetricBaselines(plan));
|
||||
const failedComparisons = $derived(steps.filter((step) => persistedComparison(step)?.status === "fail").length);
|
||||
@@ -75,6 +79,7 @@
|
||||
<p class="mt-1 text-sm text-text-muted" role="status">{dt.result_metric_unproven}</p>
|
||||
{:else}<p class="mt-1 text-sm text-text-muted">{dt.result_metric_no_mismatch}</p>{/if}
|
||||
</section>
|
||||
{#if healthGroups.length}{#key runId}<ChartHealthCard groups={healthGroups} {runId} {onselectstep} />{/key}{/if}
|
||||
{#if result.failures && result.failures.length > 0}
|
||||
<section class="mt-4" aria-label={dt.result_failed_aria}>
|
||||
<h3 class="text-sm font-medium text-text">{dt.result_failed_summary} ({result.failures.length})</h3>
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
// #region Test.ScenarioRunMonitor.ChartHealth [C:3] [TYPE Module] [SEMANTICS named-errors,coverage,model,context]
|
||||
// @TEST_INVARIANT Named query failures and partial coverage survive model PASS and never become numeric deltas.
|
||||
// @RELATION BINDS_TO -> [ScenarioRunMonitor.Component.ChartHealth]
|
||||
import { fireEvent, render, screen, within } from "@testing-library/svelte";
|
||||
import { expect, it } from "vitest";
|
||||
import ChartHealthCard from "../ChartHealthCard.svelte";
|
||||
import type { ChartHealthGroup } from "../chart-health";
|
||||
|
||||
const names = ["Debt", "Cash", "Margin", "Forecast", "Reserve"];
|
||||
const failed = names.map((chart_name, index) => ({ status: "failed", placement_id: `CHART-${index+1}`,
|
||||
chart_id: index+1, chart_name, tab_path: ["TAB-daily"], message: "Code: 386. NO_COMMON_TYPE", database_code: "386" }));
|
||||
const healthy = { status: "healthy", placement_id: "CHART-6", chart_id: 6, chart_name: "Healthy zero", tab_path: ["TAB-other"], message: "" };
|
||||
const group: ChartHealthGroup = { stepId: "chart-health", status: "failed", errors: failed, charts: [...failed, healthy],
|
||||
coverage: { expected: 7, checked: 6, unresolved: 0, unvisited: 1, complete: false },
|
||||
tabs: [{ tab_id: "TAB-daily", name: "Daily summary", errored: 5 }, { tab_id: "TAB-other", name: "Other", errored: 0 }],
|
||||
evaluations: [{ tab_path: ["TAB-daily"], status: "succeeded", advisory_verdict: "pass", coverage_complete: false,
|
||||
findings: [{ message: "All visible cards look normal" }] }] };
|
||||
|
||||
// #region Test.ScenarioRunMonitor.ChartHealth.Causes [C:2] [TYPE Function]
|
||||
it("shows five named Code386 placements and explicit unchecked coverage despite advisory PASS", () => {
|
||||
render(ChartHealthCard, { props: { groups: [group], runId: "run-a" } });
|
||||
const region = screen.getByRole("region", { name: "Chart query health" });
|
||||
const items = within(region).getAllByRole("listitem");
|
||||
expect(items).toHaveLength(5);
|
||||
names.forEach((name, index) => expect(within(items[index]).getByText(name, { exact: true })).toBeTruthy());
|
||||
expect(within(region).getByText(/Checked 6 \/ 7.*unvisited 1.*Partial coverage/)).toBeTruthy();
|
||||
expect(within(region).getByText(/5 charts failed/)).toBeTruthy();
|
||||
expect(within(region).getByText(/pass.*Incomplete visual evidence/)).toBeTruthy();
|
||||
expect(within(region).queryByText(/delta|expected value|actual value/i)).toBeNull();
|
||||
});
|
||||
// #endregion Test.ScenarioRunMonitor.ChartHealth.Causes
|
||||
|
||||
// #region Test.ScenarioRunMonitor.ChartHealth.Selection [C:2] [TYPE Function]
|
||||
it("selects the actual healthy placement and its tab without changing failed coverage", async () => {
|
||||
render(ChartHealthCard, { props: { groups: [group], runId: "run-a" } });
|
||||
await fireEvent.change(screen.getByLabelText("Tab"), { target: { value: "TAB-other" } });
|
||||
await fireEvent.change(screen.getByLabelText("Chart"), { target: { value: "CHART-6" } });
|
||||
const items = screen.getAllByRole("listitem");
|
||||
expect(items).toHaveLength(1);
|
||||
expect(within(items[0]).getByText("Healthy zero", { exact: true })).toBeTruthy();
|
||||
expect(within(items[0]).getByText("healthy", { exact: true })).toBeTruthy();
|
||||
expect(screen.getByText(/5 charts failed/)).toBeTruthy();
|
||||
expect(screen.getByText(/Partial coverage/)).toBeTruthy();
|
||||
});
|
||||
// #endregion Test.ScenarioRunMonitor.ChartHealth.Selection
|
||||
// #endregion Test.ScenarioRunMonitor.ChartHealth
|
||||
92
frontend/src/lib/components/scenario-run/chart-health.ts
Normal file
92
frontend/src/lib/components/scenario-run/chart-health.ts
Normal file
@@ -0,0 +1,92 @@
|
||||
// #region ScenarioRunMonitor.ChartHealth [C:3] [TYPE Module] [SEMANTICS health,errors,coverage,projection]
|
||||
// @BRIEF Project persisted chart errors and coverage without inferring numeric deviations or changing verdicts.
|
||||
import type { ScenarioStepRun } from "$lib/types/scenario-run";
|
||||
|
||||
export interface ChartError {
|
||||
status?: string; chart_id?: number; placement_id: string; chart_name: string; tab_path: string[];
|
||||
message: string; database_code?: string; application_code?: string; observed_at?: string;
|
||||
filter_fingerprint?: string; origin?: string; truncated?: boolean; artifact_id?: string; sha256?: string;
|
||||
}
|
||||
export interface ChartHealthGroup {
|
||||
stepId: string; status: string; errors: ChartError[]; charts: ChartError[]; coverage: Record<string, unknown>;
|
||||
tabs: Record<string, unknown>[]; evaluations: Record<string, unknown>[];
|
||||
}
|
||||
|
||||
// #region ScenarioRunMonitor.ChartHealth.Record [C:1] [TYPE Function]
|
||||
// @BRIEF Accept object envelopes only.
|
||||
function record(value: unknown): Record<string, unknown> | null {
|
||||
return value !== null && typeof value === "object" && !Array.isArray(value) ? value as Record<string, unknown> : null;
|
||||
}
|
||||
// #endregion ScenarioRunMonitor.ChartHealth.Record
|
||||
|
||||
// #region ScenarioRunMonitor.ChartHealth.Groups [C:3] [TYPE Function]
|
||||
// @POST Only persisted explicit failed diagnostics establish query errors; healthy/uncertain observations remain separate.
|
||||
export function chartHealthGroups(steps: ScenarioStepRun[]): ChartHealthGroup[] {
|
||||
const groups: ChartHealthGroup[] = [];
|
||||
for (const step of steps) {
|
||||
const group = healthGroup(step);
|
||||
if (group) groups.push(group);
|
||||
}
|
||||
return groups;
|
||||
}
|
||||
// #endregion ScenarioRunMonitor.ChartHealth.Groups
|
||||
|
||||
// #region ScenarioRunMonitor.ChartHealth.Array [C:1] [TYPE Function]
|
||||
// @BRIEF Persisted non-array fields project as absent collections.
|
||||
function array(value: unknown): unknown[] {
|
||||
return Array.isArray(value) ? value : [];
|
||||
}
|
||||
// #endregion ScenarioRunMonitor.ChartHealth.Array
|
||||
|
||||
// #region ScenarioRunMonitor.ChartHealth.Failed [C:2] [TYPE Function]
|
||||
// @POST A query error requires explicit persisted failed status and placement/name identity.
|
||||
function failedDiagnostic(value: unknown): boolean {
|
||||
const item = record(value);
|
||||
return item?.status === "failed" && typeof item.placement_id === "string" && typeof item.chart_name === "string";
|
||||
}
|
||||
// #endregion ScenarioRunMonitor.ChartHealth.Failed
|
||||
|
||||
// #region ScenarioRunMonitor.ChartHealth.Group [C:3] [TYPE Function]
|
||||
// @POST The selected step keeps its own verdict, chart observations and independent visual/checked coverage.
|
||||
function healthGroup(step: ScenarioStepRun): ChartHealthGroup | null {
|
||||
const source = healthSource(step);
|
||||
if (!source) return null;
|
||||
const errors = array(source.diagnostics).filter(failedDiagnostic) as ChartError[];
|
||||
return { stepId: step.logical_step_id, status: step.status, errors, ...healthFields(source.health, errors) };
|
||||
}
|
||||
// #endregion ScenarioRunMonitor.ChartHealth.Group
|
||||
|
||||
// #region ScenarioRunMonitor.ChartHealth.Source [C:2] [TYPE Function]
|
||||
// @POST Nested persisted outcomes take precedence; absent health evidence remains hidden.
|
||||
function healthSource(step: ScenarioStepRun): { health: Record<string, unknown> | null; diagnostics: unknown } | null {
|
||||
const outer = record(step.step_outcome);
|
||||
const outcome = record(outer?.step_outcome) ?? outer;
|
||||
const health = record(outcome?.chart_health);
|
||||
const diagnostics = health?.errors ?? outcome?.chart_diagnostics;
|
||||
if (!health && !Array.isArray(diagnostics)) return null;
|
||||
return { health, diagnostics };
|
||||
}
|
||||
// #endregion ScenarioRunMonitor.ChartHealth.Source
|
||||
|
||||
// #region ScenarioRunMonitor.ChartHealth.Fields [C:2] [TYPE Function]
|
||||
// @POST Additive observed collections never reinterpret the selected step verdict.
|
||||
function healthFields(health: Record<string, unknown> | null, errors: ChartError[]): Omit<ChartHealthGroup, "stepId" | "status" | "errors"> {
|
||||
return {
|
||||
charts: (Array.isArray(health?.charts) ? health.charts : errors) as ChartError[],
|
||||
coverage: record(health?.coverage) ?? {}, tabs: array(health?.tabs) as Record<string, unknown>[],
|
||||
evaluations: array(health?.tab_evaluations) as Record<string, unknown>[] };
|
||||
}
|
||||
// #endregion ScenarioRunMonitor.ChartHealth.Fields
|
||||
|
||||
// #region ScenarioRunMonitor.ChartHealth.Causes [C:2] [TYPE Function]
|
||||
// @POST Shared error codes group cause summaries without losing any affected placement or its individual message.
|
||||
export function chartErrorCauses(errors: ChartError[]): { cause: string; charts: ChartError[] }[] {
|
||||
const groups = new Map<string, ChartError[]>();
|
||||
for (const error of errors) {
|
||||
const cause = error.database_code ? `Database code ${error.database_code}` : error.application_code ?? error.message;
|
||||
groups.set(cause, [...(groups.get(cause) ?? []), error]);
|
||||
}
|
||||
return [...groups.entries()].map(([cause, charts]) => ({ cause, charts }));
|
||||
}
|
||||
// #endregion ScenarioRunMonitor.ChartHealth.Causes
|
||||
// #endregion ScenarioRunMonitor.ChartHealth
|
||||
70
scripts/chart-health-native-canary.py
Normal file
70
scripts/chart-health-native-canary.py
Normal file
@@ -0,0 +1,70 @@
|
||||
#!/usr/bin/env python3
|
||||
# #region Tooling.ChartHealth.NativeCanary [C:4] [TYPE Module] [SEMANTICS local,isolated,recovery,privileged]
|
||||
# @PRE Parent authorizes access to the existing isolated laboratory; no build/deploy/data mutation is performed.
|
||||
# @POST Error and recovered phases run real production observer checks; recovery is attempted in finally even if error check fails.
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
|
||||
|
||||
# #region Tooling.ChartHealth.NativeCanary.Seed [C:3] [TYPE Function]
|
||||
# @POST Only native namespaced metric metadata changes; captured Superset startup logs/credentials are never emitted.
|
||||
def seed(env_file, phase):
|
||||
command = ['docker','compose','-p','ss-tools-full-flow','--env-file',str(env_file),
|
||||
'-f','docker-compose.full-flow.yml','-f','docker-compose.full-flow.finance.yml',
|
||||
'exec','-T','superset-preprod','/app/.venv/bin/python','/fixture/chart_health_canary_seed.py','--phase',phase]
|
||||
result = subprocess.run(command,cwd=ROOT,capture_output=True,text=True,timeout=120)
|
||||
if result.returncode:
|
||||
raise RuntimeError(f'CHART_HEALTH_NATIVE_SEED_FAILED_{result.returncode}')
|
||||
records = [json.loads(line) for line in result.stdout.splitlines() if line.startswith('{"phase":')]
|
||||
if len(records) != 1:
|
||||
raise RuntimeError('CHART_HEALTH_NATIVE_SEED_IDENTITY_UNAVAILABLE')
|
||||
return records[0]
|
||||
# #endregion Tooling.ChartHealth.NativeCanary.Seed
|
||||
|
||||
|
||||
# #region Tooling.ChartHealth.NativeCanary.Verify [C:3] [TYPE Function]
|
||||
# @POST Test output and phase result are retained separately; a failure is not relabeled PASS.
|
||||
def verify(args, record):
|
||||
environment = {**os.environ,'CHART_HEALTH_NATIVE_ENV_FILE':str(args.env_file),
|
||||
'CHART_HEALTH_NATIVE_URL':'http://127.0.0.1:18112','CHART_HEALTH_NATIVE_DASHBOARD_ID':str(record['dashboard_id']),
|
||||
'CHART_HEALTH_NATIVE_PHASE':record['phase'],'CHART_HEALTH_NATIVE_OUTPUT':str(args.output/record['phase'])}
|
||||
leaf = 'test_chart_health_native_dom_worker.py' if args.diagnose else 'test_chart_health_native_canary.py'
|
||||
command = [str(ROOT/'backend/.venv/bin/python'),'-m','pytest','-q','--run-integration',
|
||||
f'tests/services/dashboard_testing/registry/{leaf}']
|
||||
result = subprocess.run(command,cwd=ROOT/'backend',env=environment,capture_output=True,text=True,timeout=700)
|
||||
(args.output/f'{record["phase"]}-pytest.log').write_text(result.stdout+result.stderr)
|
||||
return result.returncode
|
||||
# #endregion Tooling.ChartHealth.NativeCanary.Verify
|
||||
|
||||
|
||||
# #region Tooling.ChartHealth.NativeCanary.Main [C:4] [TYPE Function]
|
||||
# @POST A fresh evidence directory preserves both phases; only both actual checks yield exit0.
|
||||
def main():
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument('--env-file',type=Path,required=True)
|
||||
parser.add_argument('--output',type=Path,required=True)
|
||||
parser.add_argument('--diagnose',action='store_true')
|
||||
args = parser.parse_args()
|
||||
args.output = args.output.resolve()
|
||||
args.output.mkdir(parents=True,exist_ok=False)
|
||||
results = {'error':None,'recovered':None}
|
||||
try:
|
||||
record = seed(args.env_file,'error')
|
||||
results['error'] = verify(args,record)
|
||||
finally:
|
||||
record = seed(args.env_file,'recovered')
|
||||
results['recovered'] = verify(args,record)
|
||||
(args.output/'phase-results.json').write_text(json.dumps(results,indent=2))
|
||||
print(json.dumps({'scope':'native DOM diagnostic only' if args.diagnose else 'isolated native canary','phase_exit_codes':results}))
|
||||
return 0 if results == {'error':0,'recovered':0} else 1
|
||||
# #endregion Tooling.ChartHealth.NativeCanary.Main
|
||||
|
||||
if __name__ == '__main__':
|
||||
sys.exit(main())
|
||||
# #endregion Tooling.ChartHealth.NativeCanary
|
||||
@@ -41,7 +41,7 @@
|
||||
# @INVARIANT Acquire load semaphore before request reaches shared client semaphore; never manually re-acquire shared semaphore.
|
||||
# @INVARIANT Breaker/stop prevents new queue intake; in-flight requests drain within deadline.
|
||||
# @DATA_CONTRACT ValidatedLoadProfile -> LoadRun + LoadExecution[] + LoadRunAggregate
|
||||
# @RELATION CALLS -> [SupersetClient.ChartData.Execute]
|
||||
# @REJECTED Historical design-stub CALLS -> [SupersetClient.ChartData.Execute] is retired: the canonical RunnerPool invokes an injected executor, so a direct chart-data call is not established by this module.
|
||||
# @RELATION DEPENDS_ON -> [Core.Manager.CreateTask]
|
||||
# @RELATION DEPENDS_ON -> [Core.ClientRegistry.GetSemaphore]
|
||||
# @RATIONALE Async workers directly model executions; HTTP pool alone hides queue pressure.
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,650 @@
|
||||
{
|
||||
"baseline": "54bdbcc4",
|
||||
"historical_source_triples": 123,
|
||||
"valid_required_triples": 122,
|
||||
"valid_indexed_triples": 122,
|
||||
"missing_valid_indexed_triples": [],
|
||||
"required_triples": [
|
||||
[
|
||||
"Api.DashboardTesting",
|
||||
"DEPENDS_ON",
|
||||
"BaselineEngine.QueryExecutor.ExecuteQuery"
|
||||
],
|
||||
[
|
||||
"Api.DashboardTesting.ExecuteQuery",
|
||||
"CALLS",
|
||||
"BaselineEngine.QueryExecutor.ExecuteQuery"
|
||||
],
|
||||
[
|
||||
"Api.DashboardTesting.Scenario",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioGraph.Compiler.Compile"
|
||||
],
|
||||
[
|
||||
"Api.DashboardTesting.Scenario",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioGraph.Validator.Validate"
|
||||
],
|
||||
[
|
||||
"Api.ScenarioLiveBindings",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioExecution.LiveBinding.Identity"
|
||||
],
|
||||
[
|
||||
"BaselineEngine.Api",
|
||||
"DEPENDS_ON",
|
||||
"BaselineEngine.QueryExecutor.Execute"
|
||||
],
|
||||
[
|
||||
"BaselineEngine.Candidates.Capture",
|
||||
"DEPENDS_ON",
|
||||
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
|
||||
],
|
||||
[
|
||||
"BaselineEngine.Inheritance",
|
||||
"DEPENDS_ON",
|
||||
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
|
||||
],
|
||||
[
|
||||
"BaselineEngine.Inheritance.Execute",
|
||||
"DEPENDS_ON",
|
||||
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
|
||||
],
|
||||
[
|
||||
"BaselineEngine.QueryExecutor.Execute",
|
||||
"DEPENDS_ON",
|
||||
"SupersetClient.ChartData.Execute"
|
||||
],
|
||||
[
|
||||
"BaselineEngine.Verification.ExecutorMetric.Async",
|
||||
"DEPENDS_ON",
|
||||
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
|
||||
],
|
||||
[
|
||||
"Core.ConfigModels.ScenarioLiveExecutionBinding",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioExecution.LiveBinding.Identity"
|
||||
],
|
||||
[
|
||||
"LoadTesting.Executor.ExecuteSupersetChart",
|
||||
"CALLS",
|
||||
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
|
||||
],
|
||||
[
|
||||
"McpInterface.ScenarioPipeline",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioGraph.Compiler.Compile"
|
||||
],
|
||||
[
|
||||
"McpServer.ScenarioTools",
|
||||
"CALLS",
|
||||
"ScenarioGraph.Compiler.Compile"
|
||||
],
|
||||
[
|
||||
"McpServer.ScenarioTools",
|
||||
"CALLS",
|
||||
"ScenarioGraph.Validator.Validate"
|
||||
],
|
||||
[
|
||||
"McpServer.ToolsScenario",
|
||||
"CALLS",
|
||||
"ScenarioGraph.Compiler.Compile"
|
||||
],
|
||||
[
|
||||
"McpServer.TraversalGuidance.Build",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioGraph.Templates"
|
||||
],
|
||||
[
|
||||
"ScenarioEditor.Apply",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioGraph.Validator.Validate"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.BrowserProvider",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioExecution.BrowserProvider.Admission"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.BrowserProvider",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioExecution.BrowserProvider.Transport"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.BrowserProvider.Admission",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioExecution.LiveBinding.Identity"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.BrowserProvider.NativeFilter",
|
||||
"IMPLEMENTS",
|
||||
"ScenarioExecution.BrowserProvider.Transport"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.BrowserProvider.ReadOnlyActions",
|
||||
"IMPLEMENTS",
|
||||
"ScenarioExecution.BrowserProvider.Transport"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.BrowserProvider.ReadOnlyActions.ObserveFlows",
|
||||
"IMPLEMENTS",
|
||||
"ScenarioExecution.BrowserProvider.Transport"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.BrowserProvider.Reconciler",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioExecution.BrowserProvider.Transport"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.BrowserProvider.Session",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioExecution.BrowserProvider.Transport"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.BrowserProvider.Session.Registry",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioExecution.BrowserProvider.Transport"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.BrowserProvider.TableEvidence",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioExecution.BrowserProvider.Admission.Evidence"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.BrowserProvider.TableEvidence.Observation",
|
||||
"CALLS",
|
||||
"ScenarioExecution.BrowserProvider.Admission.Evidence"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.BrowserProvider.Transport.Factory",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioExecution.BrowserProvider.Transport"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.BrowserProvider.Transport.MutationFlow",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioExecution.BrowserProvider.Transport"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.DecisionPolicy",
|
||||
"CALLED_BY",
|
||||
"ScenarioExecution.Runner.Walker"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.EvaluationAdapter",
|
||||
"CALLS",
|
||||
"ScenarioExecution.AgentEvaluation"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.EvaluationBinding",
|
||||
"CALLED_BY",
|
||||
"ScenarioExecution.Runner.Walker"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.EvaluationImages",
|
||||
"CALLED_BY",
|
||||
"ScenarioExecution.EvaluationAdapter"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.EvaluationPrompt",
|
||||
"CALLED_BY",
|
||||
"ScenarioExecution.EvaluationAdapter"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.Executors.SqlEvidence",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioExecution.LiveBinding.SqlEvidenceAdapter"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.LiveBinding",
|
||||
"CALLS",
|
||||
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.LiveBinding.SqlEvidenceAdapter",
|
||||
"CALLS",
|
||||
"ScenarioExecution.LiveBinding.Execute"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.LiveCanaryV5",
|
||||
"VERIFIES",
|
||||
"ScenarioExecution.Runner.Walker"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.LiveCompositionRoot",
|
||||
"CALLS",
|
||||
"ScenarioExecution.EvaluationAdapter.Factory"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.LiveCompositionRoot",
|
||||
"CALLS",
|
||||
"ScenarioExecution.LiveBinding.SupersetAdapter"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.LiveCompositionRoot",
|
||||
"DEPENDS_ON",
|
||||
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.LiveCompositionRoot.EvaluationAdapter",
|
||||
"CALLS",
|
||||
"ScenarioExecution.EvaluationAdapter.Factory"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.ProviderPreflight",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioGraph.Templates.ActionRegistry"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.Runner",
|
||||
"CALLS",
|
||||
"ScenarioExecution.LiveBinding.SupersetAdapter"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.Runner.ConfiguredBinding",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioExecution.LiveBinding.Identity"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.Runner.CrashRecovery",
|
||||
"CALLS",
|
||||
"ScenarioExecution.Runner.TerminalSignal"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.Runner.CrashRecovery",
|
||||
"CALLS",
|
||||
"ScenarioExecution.Runner.Walker"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.Runner.CrashRecovery",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioExecution.LiveBinding.Identity"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.Runner.DefaultRegistry",
|
||||
"CALLS",
|
||||
"ScenarioExecution.LiveBinding.SupersetAdapter"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.Runner.QueuedDispatch",
|
||||
"CALLS",
|
||||
"ScenarioExecution.Runner.TerminalSignal"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.Runner.QueuedDispatch",
|
||||
"CALLS",
|
||||
"ScenarioExecution.Runner.Walker"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.Traversal.PinnedInputs.Resolve",
|
||||
"CALLS",
|
||||
"ScenarioExecution.Traversal.Inputs.Parse"
|
||||
],
|
||||
[
|
||||
"ScenarioExecution.Traversal.PinnedInputs.Resolve",
|
||||
"CALLS",
|
||||
"ScenarioExecution.Traversal.PinnedInputs.Projection"
|
||||
],
|
||||
[
|
||||
"ScenarioGraph.Api",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioGraph.Compiler.Compile"
|
||||
],
|
||||
[
|
||||
"ScenarioGraph.Api",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioGraph.Validator.Validate"
|
||||
],
|
||||
[
|
||||
"ScenarioGraph.Compiler.ChainEmission",
|
||||
"CALLED_BY",
|
||||
"ScenarioGraph.Compiler.CompileGraph"
|
||||
],
|
||||
[
|
||||
"ScenarioGraph.Compiler.ChainEmission",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioGraph.Compiler.BuildStep"
|
||||
],
|
||||
[
|
||||
"ScenarioGraph.Compiler.ChainEmission",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioGraph.Templates"
|
||||
],
|
||||
[
|
||||
"ScenarioGraph.MetricEvaluationProvider",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioGraph.Models.MetricTextEvaluationRecipe"
|
||||
],
|
||||
[
|
||||
"ScenarioGraph.MetricEvaluationRecipe",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioGraph.Models.MetricTextEvaluationRecipe"
|
||||
],
|
||||
[
|
||||
"ScenarioGraph.MetricRecipeStepInputs.Validate",
|
||||
"CALLS",
|
||||
"ScenarioGraph.StepInputs.EmbeddedLiterals"
|
||||
],
|
||||
[
|
||||
"ScenarioGraph.MetricResultSchema",
|
||||
"DEPENDS_ON",
|
||||
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
|
||||
],
|
||||
[
|
||||
"ScenarioGraph.ServerOwnedPipeline",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioGraph.Compiler.Compile"
|
||||
],
|
||||
[
|
||||
"ScenarioGraph.ServerOwnedPipeline",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioGraph.Validator.Validate"
|
||||
],
|
||||
[
|
||||
"ScenarioGraph.StepInputs",
|
||||
"CALLED_BY",
|
||||
"ScenarioGraph.Validator.CheckStepInputs"
|
||||
],
|
||||
[
|
||||
"Stage6.TableTextRecipeSlice",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioGraph.Models.MetricTextEvaluationRecipe"
|
||||
],
|
||||
[
|
||||
"SupersetBaselineEngine.QA.Audit",
|
||||
"VERIFIES",
|
||||
"BaselineEngine.QueryExecutor.ExecuteQuery"
|
||||
],
|
||||
[
|
||||
"Test.DashboardTesting.CandidateCapture",
|
||||
"BINDS_TO",
|
||||
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
|
||||
],
|
||||
[
|
||||
"Test.DashboardTesting.ChartDataRaw",
|
||||
"BINDS_TO",
|
||||
"SupersetClient.ChartData.Execute"
|
||||
],
|
||||
[
|
||||
"Test.DashboardTesting.MetricExecutorCatalog",
|
||||
"VERIFIES",
|
||||
"BaselineEngine.QueryExecutor.Envelope"
|
||||
],
|
||||
[
|
||||
"Test.DashboardTesting.MetricExecutorCatalog",
|
||||
"VERIFIES",
|
||||
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
|
||||
],
|
||||
[
|
||||
"Test.DashboardTesting.QueryExecutor",
|
||||
"VERIFIES",
|
||||
"BaselineEngine.QueryExecutor.ExecuteQuery"
|
||||
],
|
||||
[
|
||||
"Test.EvaluationText.BrowserCommitFrontier",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.Runner.Walker"
|
||||
],
|
||||
[
|
||||
"Test.EvaluationText.SubmitRuntime",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.AgentEvaluation"
|
||||
],
|
||||
[
|
||||
"Test.MetricBrowser.InputAuthority",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.BrowserProvider.Admission.Gate"
|
||||
],
|
||||
[
|
||||
"Test.MetricTextRecipe.TokenContext",
|
||||
"BINDS_TO",
|
||||
"ScenarioGraph.Validator.Validate"
|
||||
],
|
||||
[
|
||||
"Test.MetricTextRecipe.ValidatorAuthority",
|
||||
"BINDS_TO",
|
||||
"ScenarioGraph.Validator.Validate"
|
||||
],
|
||||
[
|
||||
"Test.Scenario.Compiler",
|
||||
"BINDS_TO",
|
||||
"ScenarioGraph.Compiler.Compile"
|
||||
],
|
||||
[
|
||||
"Test.Scenario.MetricGraphV2Authority",
|
||||
"BINDS_TO",
|
||||
"ScenarioGraph.Compiler.Compile"
|
||||
],
|
||||
[
|
||||
"Test.Scenario.Validator",
|
||||
"BINDS_TO",
|
||||
"ScenarioGraph.Validator.Validate"
|
||||
],
|
||||
[
|
||||
"Test.Scenario.ValidatorBelief",
|
||||
"BINDS_TO",
|
||||
"ScenarioGraph.Validator.Validate"
|
||||
],
|
||||
[
|
||||
"Test.Scenario.ValidatorProperties",
|
||||
"BINDS_TO",
|
||||
"ScenarioGraph.Validator.Validate"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioAutomation.NotificationsWiring",
|
||||
"VERIFIES",
|
||||
"ScenarioExecution.Runner.TerminalSignal"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioEditor.StepInputs",
|
||||
"BINDS_TO",
|
||||
"ScenarioGraph.StepInputs"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.AgentEvaluation",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.AgentEvaluation"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.AgentEvaluationStore",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.AgentEvaluation"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.BrowserLimitsCleanup",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.BrowserProvider.Transport"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.BrowserReadOnlyActions",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.BrowserProvider.Admission"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.BrowserReadOnlyActions",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.BrowserProvider.Transport"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.BrowserTableRetainedAuthority",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.BrowserProvider.Transport"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.CancelTimeout",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.Runner.Walker"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.CrashRecovery",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.Runner.TerminalSignal"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.DueAdmissionAtomicity",
|
||||
"VERIFIES",
|
||||
"ScenarioExecution.Runner.TerminalSignal"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.EvaluationBinding",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.Runner.Walker"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.LiveBinding",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.LiveBinding"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.LiveBinding",
|
||||
"VERIFIES",
|
||||
"ScenarioExecution.LiveBinding.Evidence"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.LiveBinding",
|
||||
"VERIFIES",
|
||||
"ScenarioExecution.LiveBinding.Execute"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.LlmInjectionOffline",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.EvaluationAdapter"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.MetricDispatchContinuation",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.Runner.Walker"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.QueuedDispatch",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.Runner.Walker"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.RetryClosure",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.Runner.Walker"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.RetryDispatch",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.Runner.TerminalSignal"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.SupersetProviderContract",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.LiveBinding"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.SupersetProviderContract",
|
||||
"VERIFIES",
|
||||
"ScenarioExecution.LiveBinding.Evidence"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.SupersetProviderContract",
|
||||
"VERIFIES",
|
||||
"ScenarioExecution.LiveBinding.Execute"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.SupersetProviderContract",
|
||||
"VERIFIES",
|
||||
"ScenarioExecution.LiveBinding.Payload"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.TerminalSignals",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.Runner.Walker"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.TerminalSignals",
|
||||
"VERIFIES",
|
||||
"ScenarioExecution.Runner.TerminalSignal"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.TraversalInputs",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.Traversal.Inputs"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.TraversalJournal",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.Traversal.Store"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.Walker",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.Runner.Walker"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioExecution.Walker",
|
||||
"VERIFIES",
|
||||
"ScenarioExecution.Runner.Walker"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioGraph.AgentEvaluationModels",
|
||||
"BINDS_TO",
|
||||
"ScenarioGraph.Models.AgentEvaluationSpec"
|
||||
],
|
||||
[
|
||||
"Test.ScenarioRunMonitor.InspectionResult",
|
||||
"BINDS_TO",
|
||||
"ScenarioRunMonitor.Component.Result"
|
||||
],
|
||||
[
|
||||
"Test.SecurityOpsOffline",
|
||||
"VERIFIES",
|
||||
"ScenarioExecution.EvaluationAdapter.Raw"
|
||||
],
|
||||
[
|
||||
"Test.Traversal.SampleJournal",
|
||||
"BINDS_TO",
|
||||
"ScenarioExecution.Traversal.Store"
|
||||
],
|
||||
[
|
||||
"Tooling.FinanceSemanticCuration.Report",
|
||||
"VERIFIES",
|
||||
"ScenarioExecution.Traversal.Runtime.Provider.Factory"
|
||||
],
|
||||
[
|
||||
"Tooling.FinanceSemanticCuration.Report",
|
||||
"VERIFIES",
|
||||
"ScenarioExecution.Traversal.Runtime.Provider.Submit"
|
||||
],
|
||||
[
|
||||
"VerificationProgram.Reconciliation",
|
||||
"DEPENDS_ON",
|
||||
"ScenarioGraph.Templates.ActionRegistry"
|
||||
]
|
||||
],
|
||||
"retired_documentation_assertion": {
|
||||
"triple": [
|
||||
"LoadTesting.RunnerPool",
|
||||
"CALLS",
|
||||
"SupersetClient.ChartData.Execute"
|
||||
],
|
||||
"authorization": "Root explicitly accepted122trueindexed +1 retired invaliddocumentation assertion",
|
||||
"reason": "Duplicate fenced design contract says CALLS; canonical production pool invokes an injected executor. Direct chart-data call is not established. No production CALLS relation manufactured.",
|
||||
"canonical_ID_preserved": "LoadTesting.RunnerPool",
|
||||
"canonical_source_untouched": "backend/src/services/load_testing/runner_pool.py",
|
||||
"incoming_references_unchanged": true
|
||||
},
|
||||
"metadata_scope_extension": [
|
||||
{
|
||||
"path": "specs/050-mcp-interface/plans/T029a-table-text-recipe-slice-2026-10-01.md",
|
||||
"before_sha256": "8d53d368d499f0899338d191b5696537d5091b84d0e6da8c3ec13b779e86c5e1",
|
||||
"after_sha256": "b7c27b3d8014227c7a470b50d9f27297ab58c901d1f4cffd02242706857e6ead",
|
||||
"diff": "--- specs/050-mcp-interface/plans/T029a-table-text-recipe-slice-2026-10-01.md\n+++ specs/050-mcp-interface/plans/T029a-table-text-recipe-slice-2026-10-01.md\n@@ -6,7 +6,6 @@\n and heavy 40\u201360-second dashboard acceptance remain open.\n \n ## @{ Stage6.TableTextRecipeSlice [C:5] [TYPE ADR] [SEMANTICS recipe,text,context,provider,token-budget]\n-\n @BRIEF Closed owned metric/table recipe with exact filter provenance, honest provider aliases and nonmonetary limits.\n @RELATION DEPENDS_ON -> [Stage6.OwnedTextEvidenceSlice]\n @RELATION DEPENDS_ON -> [ScenarioGraph.MetricEvaluationRecipe.Compile]\n@@ -137,7 +136,6 @@\n ## @} Stage6.TableTextRecipeSlice\n \n ## @{ Stage6.TableTextCapacityNativeDecision [C:5] [TYPE ADR] [SEMANTICS recipe,capacity,native-filter,transactions]\n-\n @BRIEF Pending native scope and provider-wide capacity authority for the owned text recipe.\n @RELATION DEPENDS_ON -> [Stage6.TableTextRecipeSlice]\n @PRE Recipe scope and public provider identity are verified before a committed capacity claim or HTTP request.\n@@ -265,7 +263,6 @@\n ## @} Stage6.TableTextCapacityNativeDecision\n \n ## @{ Stage6.TableTextRuntimeInputDecision [C:5] [TYPE ADR] [SEMANTICS recipe,runtime,inputs,authority,snapshot]\n-\n @BRIEF Preserve registry snapshot identity while conveying admitted recipe inputs to the real browser provider.\n @RELATION DEPENDS_ON -> [Stage6.TableTextRecipeSlice]\n @PRE Runtime graph/step identity is proved against the persisted immutable plan before recipe input projection.\n"
|
||||
},
|
||||
{
|
||||
"path": "specs/040-dashboard-load-testing/contracts/modules.md",
|
||||
"before_sha256": "eb376a4eeddefdf4982daef439cc0661bdab70f890f0219bc0055067741dfa1f",
|
||||
"after_sha256": "d33c10b84e7a00af4cdb14fb9dd03bd66838a77f13c2393edfe22b0884131bf9",
|
||||
"diff": "--- specs/040-dashboard-load-testing/contracts/modules.md\n+++ specs/040-dashboard-load-testing/contracts/modules.md\n@@ -41,7 +41,7 @@\n # @INVARIANT Acquire load semaphore before request reaches shared client semaphore; never manually re-acquire shared semaphore.\n # @INVARIANT Breaker/stop prevents new queue intake; in-flight requests drain within deadline.\n # @DATA_CONTRACT ValidatedLoadProfile -> LoadRun + LoadExecution[] + LoadRunAggregate\n-# @RELATION CALLS -> [SupersetClient.ChartData.Execute]\n+# @REJECTED Historical design-stub CALLS -> [SupersetClient.ChartData.Execute] is retired: the canonical RunnerPool invokes an injected executor, so a direct chart-data call is not established by this module.\n # @RELATION DEPENDS_ON -> [Core.Manager.CreateTask]\n # @RELATION DEPENDS_ON -> [Core.ClientRegistry.GetSemaphore]\n # @RATIONALE Async workers directly model executions; HTTP pool alone hides queue pressure.\n"
|
||||
}
|
||||
],
|
||||
"config_changes": null,
|
||||
"limitations": [
|
||||
"Snapshot after metadata-only retirement/header normalization; final combined owner source freeze, legacy parity and navigation receipts follow separately.",
|
||||
"This is not123indexedPASS and not globalGO."
|
||||
]
|
||||
}
|
||||
@@ -6,7 +6,6 @@ Intermediate BLOCKED/pending entries remain chronological evidence. Global GO
|
||||
and heavy 40–60-second dashboard acceptance remain open.
|
||||
|
||||
## @{ Stage6.TableTextRecipeSlice [C:5] [TYPE ADR] [SEMANTICS recipe,text,context,provider,token-budget]
|
||||
|
||||
@BRIEF Closed owned metric/table recipe with exact filter provenance, honest provider aliases and nonmonetary limits.
|
||||
@RELATION DEPENDS_ON -> [Stage6.OwnedTextEvidenceSlice]
|
||||
@RELATION DEPENDS_ON -> [ScenarioGraph.MetricEvaluationRecipe.Compile]
|
||||
@@ -137,7 +136,6 @@ isolated Docker contour.
|
||||
## @} Stage6.TableTextRecipeSlice
|
||||
|
||||
## @{ Stage6.TableTextCapacityNativeDecision [C:5] [TYPE ADR] [SEMANTICS recipe,capacity,native-filter,transactions]
|
||||
|
||||
@BRIEF Pending native scope and provider-wide capacity authority for the owned text recipe.
|
||||
@RELATION DEPENDS_ON -> [Stage6.TableTextRecipeSlice]
|
||||
@PRE Recipe scope and public provider identity are verified before a committed capacity claim or HTTP request.
|
||||
@@ -265,7 +263,6 @@ inconclusive; its diagnosed runtime-input correction is described below.
|
||||
## @} Stage6.TableTextCapacityNativeDecision
|
||||
|
||||
## @{ Stage6.TableTextRuntimeInputDecision [C:5] [TYPE ADR] [SEMANTICS recipe,runtime,inputs,authority,snapshot]
|
||||
|
||||
@BRIEF Preserve registry snapshot identity while conveying admitted recipe inputs to the real browser provider.
|
||||
@RELATION DEPENDS_ON -> [Stage6.TableTextRecipeSlice]
|
||||
@PRE Runtime graph/step identity is proved against the persisted immutable plan before recipe input projection.
|
||||
|
||||
@@ -0,0 +1,285 @@
|
||||
# Chart health, query errors and per-tab VLM — implementation handoff
|
||||
|
||||
## @{ ScenarioGraph.ChartHealth.Plan [C:4] [TYPE ADR] [SEMANTICS chart-health,query-errors,tabs,vlm,evidence]
|
||||
@BRIEF Preserve Superset query failures as owned evidence, check every visited tab and expose actionable chart diagnostics to analysts and VLM.
|
||||
@PRE Existing sampling, ownership, cancellation, registry pinning and baseline bindings remain authoritative.
|
||||
@POST Confirmed chart failures are durable and attributable; partial coverage is explicit; VLM cannot erase a deterministic failure.
|
||||
@INVARIANT Evidence identifies run, attempt, tab path, chart placement, filter state and load generation; stale or unrelated responses cannot establish a verdict.
|
||||
@RATIONALE SQL-error cards can currently become generic timeouts, and the API adapter returns before retaining error evidence. Shared observation logic supports both a graph action and composite tab traversal without introducing a general nested-graph engine.
|
||||
@REJECTED Screenshot-only detection, HTTP-status-only success, global unscoped error text searches and arbitrary nested graph execution were rejected for this slice.
|
||||
|
||||
## Authority and scope
|
||||
|
||||
### Implementation checkpoint — 2026-10-02
|
||||
|
||||
S1–S5 are connected in source and the focused runtime gates pass; acceptance is
|
||||
still pending the independent semantic audit. The new authoring registry is `038.8.0`; historical
|
||||
`038.6` full traversal and `038.7` sampling inputs/journals keep their meanings.
|
||||
New `navigate_tabs` authoring explicitly enables chart health. Per-tab visual
|
||||
evaluation is opt-in and requires a pinned active visual provider configuration.
|
||||
|
||||
Retained evidence so far:
|
||||
|
||||
- Independent literal HTTP/application/async/stale/transport and real owned
|
||||
artifact/provider/whole-tab evaluation checks: **26 passed**. Three causal
|
||||
failures were fixed: async202 must remain pending; malformed200 cannot confirm
|
||||
query failure; auxiliary frame bytes must not be added twice to diagnostic
|
||||
frontier bytes. Whole evaluation now captures actual PNG+Code386 contents,
|
||||
retains raw/record artifacts and continues to a later chart. Advisory PASS
|
||||
cannot erase a confirmed chart FAIL.
|
||||
- Existing evaluation/models compatibility: **28 passed**. Semantic review has
|
||||
no invented comparison IDs. The full authored provider pin stays in evidence;
|
||||
only its exact validated64hex digest enters the existing64char capacity lease.
|
||||
The actual retained lease uses the64char digest while the owned record keeps
|
||||
the full78char pin. These checks use SQLite; actual PostgreSQL admission is
|
||||
subsequently checked against an isolated real PostgreSQL Testcontainers instance:
|
||||
**1 passed in4.86s** with authored78→committed64→released lease identity.
|
||||
- Fake external SDK boundary: **2 passed**, with actual encrypted DB credential,
|
||||
one physical request, retries disabled, declared timeout/output bounds,
|
||||
transport usage and closure. No paid provider call occurred.
|
||||
- Configured fake notifications: **5 passed**; analyst DOM: **2 passed**.
|
||||
Durable summary is always retained. Optional delivery uses the existing global
|
||||
`notifications.scenario_alerts` namespace and existing SMTP/TELEGRAM/SLACK
|
||||
providers, after terminal commit. It is disabled by default; destinations
|
||||
are explicit. Atomic pending-to-dispatching CAS gives at-most-once attempts,
|
||||
with explicit skipped/failed results and no automatic ambiguous-send retry.
|
||||
- Final combined focused backend boundary suite: **81 passed in3.07s**;
|
||||
root privileged connected Chromium suite: **3 passed in3.66s**, including
|
||||
all five named Code386 cards plus the healthy empty chart on another tab.
|
||||
Checked coverage is6/6, errors5, complete=true, overall FAILED. There were
|
||||
no late TargetClosed/unretrieved Future warnings in the returned output.
|
||||
- All current touched Python surfaces pass explicit F/C901≤10 and production
|
||||
module length<400. Final anchor,
|
||||
relation and source-hash closure remains required before readiness claims.
|
||||
|
||||
Bounded post-review fixes and proofs:
|
||||
|
||||
- Frontend projection was split into cohesive source/field helpers. The explicit
|
||||
TypeScript parser ESLint gate enforces complexity≤10 with zero errors/warnings;
|
||||
analyst DOM checks remain **2 passed**. The repository's default ESLint config
|
||||
does not match ordinary `.ts` files, so an ignored-file run is not accepted as
|
||||
this proof.
|
||||
- `038.6.0`/`038.7.0` pinned tab inputs now use their frozen original three-field
|
||||
DTO instead of receiving additive health fields. **4 passed** cover exact
|
||||
normalized old defaults, rejection of new flags, real committed projection,
|
||||
unchanged persisted plan bytes, old input digest and journal resume. Literal
|
||||
baseline derives from immutable `54bdbcc4`, sourceSHA
|
||||
`1b6a3f5875f886cebd8e0a9d9b57eb69e5d44e51c6689352709cc5f8f398a87e`.
|
||||
- Independent nested/offscreen Chromium: **1 passed in2.80s**. Simulated45/60s
|
||||
budget checks: **2 passed**. These supplement rather than replace the complete
|
||||
five-error connected Chromium proof.
|
||||
- Native Superset/ClickHouse canary is prepared in
|
||||
`docker/full-flow/chart_health_canary_seed.py` and
|
||||
`scripts/chart-health-native-canary.py`. It creates only namespaced PREPROD
|
||||
metadata, reads one existing real finance source row, provokes an actual
|
||||
Int64/String metric type conflict, and recovers only its own metric in finally.
|
||||
Both error/recovered phase checks retain actual observer+owned journal results;
|
||||
they make no public release/authoring claim. Actual execution remains pending.
|
||||
|
||||
The first connected Chromium fixture retained five confirmed errors but reported
|
||||
incomplete coverage: its empty table collapsed the next panel to zero height.
|
||||
The fixture now gives cards their real Superset-style minimum height. The complete
|
||||
coverage assertion is unchanged; the fresh privileged browser gate now passes.
|
||||
Earlier failed receipts/results are retained and must not be counted as PASS.
|
||||
|
||||
Declared operational limits: chart health does not fix the separate expensive
|
||||
mid-page navigation limitation of default pagination quantiles. Server logs and
|
||||
unobserved async WebSocket bodies are not available evidence. Unresolved response
|
||||
bodies/loading/transport stay INCONCLUSIVE; confirmed errors stay FAILED even
|
||||
when chart or visual coverage is partial. No real notification or deployment has
|
||||
been performed during this implementation packet.
|
||||
|
||||
User request: create an agent plan and delegate implementation of the discussed
|
||||
query-error/chart-health/VLM flow. Work from current HEAD and preserve unrelated
|
||||
maintenance, Git UI, login and translation changes. Follow INV1–7 and canonical
|
||||
skills: self-implementation, semantics-core/python/testing; semantics-svelte for
|
||||
UI changes. Required gates: real contract boundaries, preserved original IDs and
|
||||
incoming relations, strict C901 <=10, production Python modules <400 lines.
|
||||
|
||||
## Current gaps to reproduce first
|
||||
|
||||
- `query_executor.py`: API errors become UNKNOWN values with warnings and an
|
||||
error-context envelope; this is not necessarily the original HTTP body.
|
||||
- `execution/live_binding.py`: UNKNOWN returns `SUPERSET_QUERY_FAILED` before
|
||||
`_evidence_result`; other exceptions become generic execution errors.
|
||||
- `execution/providers/browser_all_tabs.py`: waits for chart content before
|
||||
checking an alert; catches the failure as INCONCLUSIVE and stops the sweep.
|
||||
- `execution/evaluation_adapter.py`, `evaluation_manifest.py`,
|
||||
`evaluation_prompt.py`: VLM gets images/manifest; ordinary JSON references
|
||||
do not automatically provide the actual diagnostic contents to the model.
|
||||
- No implemented `assert_chart_health` action or general nested graph runtime
|
||||
has been established. Verify exact action/schema/registry seams before edits.
|
||||
|
||||
## Ordered implementation slices
|
||||
|
||||
### S1 — Typed diagnostic evidence and failure retention
|
||||
|
||||
Define a versioned diagnostic schema before implementation. Include run/step/
|
||||
attempt, environment/dashboard, stable chart ID and placement/tab path, effective
|
||||
filter fingerprint, load generation/request identity, timestamps/duration,
|
||||
HTTP status, application/DB error code, bounded message, observation origin,
|
||||
terminal loading state, evidence references/digests and truncation indicators.
|
||||
Distinguish original response, sanitized retained payload and synthetic exception
|
||||
context; never label a reconstructed JSON object as original wire evidence.
|
||||
|
||||
Retain evidence before returning FAILED/INCONCLUSIVE. Propagate references and
|
||||
typed diagnostics through provider receipts, step outcomes and persistence.
|
||||
Preserve real failure details on API exceptions; missing/unreadable evidence
|
||||
must be explicit. Detect errors inside HTTP-200 payloads as well as non-2xx.
|
||||
|
||||
### S2 — Shared chart-health observation and graph action
|
||||
|
||||
Implement one common observer, used by a new admitted `assert_chart_health`
|
||||
action and tab traversal. Add descriptor, input/output schemas, registry version,
|
||||
executor routing and MCP explanations through existing authoritative mechanisms.
|
||||
Never reinterpret an already pinned registry snapshot or existing saved plan.
|
||||
|
||||
Observe scoped DOM error cards and matched chart-data requests. Install network
|
||||
listeners before the corresponding load/filter action; bound buffers and remove
|
||||
listeners on completion/cancellation. A request must match the expected chart,
|
||||
filter state and current load generation. Handle asynchronous query polling.
|
||||
Check errors while waiting for readiness, not after requiring rendered content.
|
||||
Exclude unrelated HTTP failures and stale responses from chart verdicts.
|
||||
|
||||
Successful empty results and zeros are health successes; data-presence/business
|
||||
assertions belong to their own steps. Expected-error scenarios require explicit
|
||||
typed expectations. Confirmed query failure is FAILED; unresolved loading or
|
||||
unattributed/infrastructure-only observations are INCONCLUSIVE with evidence.
|
||||
|
||||
### S3 — Composite `navigate_tabs` integration and coverage
|
||||
|
||||
Keep `navigate_tabs` as a composite driver for this slice. For each stable tab:
|
||||
activate ancestors/panel -> observe its chart placements -> capture evidence ->
|
||||
optional VLM evaluation while the tab is active -> advance. Reuse S2; do not
|
||||
duplicate detectors or add arbitrary graph callbacks/recursive graph execution.
|
||||
|
||||
Visit charts below the viewport and nested panels using bounded owned scrolling.
|
||||
Resolve placements by stable IDs and paths, not titles. Continue after individual
|
||||
confirmed chart failures to collect other findings. Persist per-chart/per-tab
|
||||
results incrementally; cancellation, leases and ownership remain enforced.
|
||||
|
||||
Separate verdict from coverage: observed FAILED dominates even when coverage is
|
||||
partial; otherwise incomplete coverage is INCONCLUSIVE; PASS requires complete
|
||||
requested coverage. Show expected/checked/error/timeout/unvisited counts. Keep
|
||||
pagination sampling independent from chart/tab health coverage. Document that
|
||||
many tabs and 40–60 second queries can make this step take a long time.
|
||||
|
||||
### S4 — Per-tab VLM diagnostic context
|
||||
|
||||
When enabled, support explicit all-visited/selected/problem-area scope. Capture
|
||||
and evaluate the chosen tab before leaving it; long tabs need multiple frames
|
||||
with explicit coverage. Attach bounded *contents* of owned diagnostic artifacts
|
||||
alongside screenshots and criteria, using the existing evidence-as-data boundary.
|
||||
Artifact references alone are insufficient. Validate ownership/hashes and budget;
|
||||
report omitted/truncated diagnostics or images and insufficient coverage.
|
||||
|
||||
The evaluation path must accept diagnostic evidence from failed observations
|
||||
without allowing comparison of fabricated metric values. VLM may explain the
|
||||
failure and assess appearance; it cannot change a confirmed deterministic FAIL
|
||||
to PASS. Preserve deterministic and VLM verdicts separately. Do not silently
|
||||
convert every existing visual evaluation into a paid per-tab VLM call.
|
||||
|
||||
Server-side Superset logs are optional future enrichment requiring a separate
|
||||
correlated connector. HTTP responses, scoped DOM and browser request failures
|
||||
are this slice's sources; do not claim access to server logs. Exclude credentials,
|
||||
cookies and unnecessary sensitive response/query content from model payloads.
|
||||
|
||||
### S5 — Analyst results, graph and configured notifications
|
||||
|
||||
Show a summary such as “5 charts failed — Daily summary”, with tab/chart names,
|
||||
cause, filters, time and proof links. Group equal causes while retaining every
|
||||
affected placement. Expand/filter the composite graph node by tab/chart; keep
|
||||
partial coverage visible alongside FAILED. Error charts have no numeric baseline
|
||||
delta; preserve baseline comparisons for successful producers.
|
||||
|
||||
Use the existing automation notification machinery for run failures, with a
|
||||
bounded chart-error summary and deduplication. Delivery depends on configured
|
||||
channels; tests use fake delivery. Do not send real external messages during
|
||||
implementation. Add UX_STATE before component code and keep technical details
|
||||
collapsible beneath actionable explanations.
|
||||
|
||||
## Acceptance matrix
|
||||
|
||||
### Current acceptance ledger — 2026-10-02
|
||||
|
||||
All ten functional rows now have focused executable assertions passing. Final
|
||||
acceptance remains open: timeout cleanup emits an unhandled Playwright future,
|
||||
and native error detection fails. This ledger does not establish release GO.
|
||||
|
||||
| Row | Status | Actual proof / remaining subcase |
|
||||
|---|---|---|
|
||||
| 1 | PASS, protocol fixture | Real Chromium: five Code386 cards retained, healthy next tab checked, 6/6 coverage and deterministic FAILED. |
|
||||
| 2 | PASS | Independent HTTP500, HTTP200 application error and owned asynchronous terminal-result fixtures. |
|
||||
| 3 | PASS | Two independent 45/60-second simulated-clock cases; immediate error recorded at duration zero while slow chart uses its own budget. |
|
||||
| 4 | PASS | Stale/retried/foreign request ownership and request-failure attribution; listener/task cleanup. |
|
||||
| 5 | PASS | Successful zero and empty results remain healthy, including real browser empty table. |
|
||||
| 6 | PASS | Real Chromium proves duplicate titles, nested paths, offscreen scrolling and same chart ID in distinct placements with failed/healthy owned observations. |
|
||||
| 7 | PASS assertions; cleanup OPEN | Cancellation, unopenable-tab and renderer-timeout after confirmed FAIL retain exact partial coverage. The three placement cases pass, but renderer timeout emits an unhandled Playwright future; a strict cleanup oracle is pending. |
|
||||
| 8 | PASS, fake external SDK | Owned diagnostic contents plus actual PNG cross the real evaluation boundary; digest/ownership/budget guards and advisory-PASS isolation proven. |
|
||||
| 9 | PASS, focused compatibility | Version-pinned legacy DTO fixtures prove old plan bytes, normalized fields, input digest and journal resume; existing focused compatibility/ownership gates pass. |
|
||||
| 10 | PASS, fake delivery | Two analyst DOM cases and five configured fake notification cases; no real external messages. |
|
||||
|
||||
Additional required native gate: existing Superset/ClickHouse stand is available.
|
||||
The first actual error/recovery canary failed both phases before observations:
|
||||
Superset layout contains a scalar `DASHBOARD_VERSION_KEY: v2`, which the health
|
||||
manifest parser incorrectly treated as a node. The owner added a dictionary-node
|
||||
guard and a literal regression case (legacy/layout combined gate: five PASS).
|
||||
The second native canary completed: recovery PASS, error phase FAIL. Six charts
|
||||
were observed, only one checked; five error-card locators timed out and produced
|
||||
INCONCLUSIVE instead of confirmed failure. The owner is repairing native error
|
||||
DOM handling and timeout cleanup. Output paths are retained separately per attempt.
|
||||
|
||||
The bounded repair checks current attributed API errors before waiting for
|
||||
successful chart-renderer DOM. Native Superset5 saved-card identity is resolved
|
||||
through its exact chart-grid-component/chart-id attributes; readiness is scoped
|
||||
to the chart body, excluding header icons, while screenshots retain the whole
|
||||
card. The existing per-chart deadline reserves 250ms for SDK drainage and never
|
||||
starts a new RPC after the remaining budget expires. The independent expired-RPC
|
||||
regression passes after its retained causal failure; response/owned-storage and
|
||||
legacy/layout gates pass 23 cases. Actual native DOM inventory, strict native
|
||||
error/recovery rerun and real Chromium timeout-cleanup verification remain open.
|
||||
|
||||
Additional completed gates: real isolated PostgreSQL capacity admission/release
|
||||
one PASS, nested/offscreen Chromium one PASS, strict TypeScript complexity <=10
|
||||
and Python C901 <=10. Final source hash/index closure follows the necessary native
|
||||
manifest fix. Historical documentation edge retirement is explicit in the
|
||||
semantic edge-policy receipt; it must not be counted as a preserved runtime CALLS.
|
||||
|
||||
Use meaningful fixtures with production paths and @TEST_INVARIANT traceability:
|
||||
|
||||
1. Five error cards like the user's screenshot: all five attributable findings
|
||||
retained; other charts checked; overall FAILED, never generic PASS.
|
||||
2. HTTP 500; HTTP 200 with application error; asynchronous job failure.
|
||||
3. Slow success (40–60s simulated clock) beside an immediate failure; independent
|
||||
budgets, prompt failure observation and cancellation responsiveness.
|
||||
4. Late old-filter response, duplicate/retried request, unrelated failed endpoint.
|
||||
5. Successful zero and empty data; no automatic query-error classification.
|
||||
6. Duplicate titles, repeated chart placement, nested tabs, offscreen chart.
|
||||
7. Unopenable tab/renderer timeout/cancellation after one confirmed failure:
|
||||
durable FAILED plus partial coverage and exact unchecked counts.
|
||||
8. VLM payload contains actual diagnostic text and scoped screenshots; missing
|
||||
artifacts/budget truncation explicit; model PASS cannot erase query FAIL.
|
||||
9. Existing pinned registry plans, metric/baseline flows and ownership/cleanup
|
||||
gates retain behavior; no weakening old fixtures to make unrelated failures pass.
|
||||
10. UI makes each affected chart/cause discoverable; fake notification receives
|
||||
one grouped summary, with no live external delivery.
|
||||
|
||||
Run focused backend/frontend checks. Add a local Superset/ClickHouse fixture with
|
||||
an intentional type-conflict query and recovery if the existing test stand is
|
||||
available; preserve datasets and existing release evidence. Do not run a new
|
||||
625-page sweep or real LLM billing just to validate these diagnostics. Record
|
||||
actual commands/results and distinguish inherited failures from regressions.
|
||||
|
||||
## Agent result and review gates
|
||||
|
||||
Return <RESULT> with exact changed paths, completed S1–S5 status, evidence/test
|
||||
commands, API/registry compatibility, semantic index/ID/edge receipts and remaining
|
||||
limitations. Update this plan and main handoff with actual statuses. Delegate
|
||||
independent verification and semantic curation at integration; retain review
|
||||
evidence. No automatic global GO, commit, deployment or external notification.
|
||||
Root handles privilege-only checks when sandbox blocks TestClient/Docker.
|
||||
## @} ScenarioGraph.ChartHealth.Plan
|
||||
|
||||
## Commit checkpoint — 2026-10-02
|
||||
|
||||
User explicitly authorized commit and push. Post-fix real Chromium matrix: four PASS in 10.54s, including cleanup and expired-deadline guards, with no unhandled future warning. Affected independent checks: 28 PASS. Native Superset error/recovery acceptance and final semantic freeze remain open; this commit is not release GO.
|
||||
Reference in New Issue
Block a user