feat(dashboard-testing): add chart health diagnostics and per-tab evidence

Focused matrix and cleanup checks pass. Native Superset error/recovery acceptance and final semantic freeze remain open; this is not release GO.
This commit is contained in:
2026-10-02 16:23:22 +03:00
parent 54bdbcc403
commit 3c99e3ca77
73 changed files with 7027 additions and 105 deletions

View File

@@ -28,6 +28,7 @@ class ChartDataResponse:
parsed: dict[str, Any]
raw_bytes: bytes
source_response_hash: str = field(repr=False)
http_status: int = 200
# #endregion SupersetClient.ChartData.ResponseDTO
# #region SupersetClient.ChartData.SupersetChartDataMixin [C:4] [TYPE Class]
@@ -118,11 +119,15 @@ class SupersetChartDataMixin:
)
raw_bytes: bytes = httpx_response.content
source_response_hash: str = hashlib.sha256(raw_bytes).hexdigest()
parsed: dict[str, Any] = httpx_response.json()
try:
parsed: dict[str, Any] = httpx_response.json()
except ValueError:
parsed = {'_invalid_response':True}
return ChartDataResponse(
parsed=parsed,
raw_bytes=raw_bytes,
source_response_hash=source_response_hash,
http_status=httpx_response.status_code,
)
except SupersetAPIError:
raise

View File

@@ -0,0 +1,41 @@
# #region DashboardTesting.ChartHealth.Contract [C:4] [TYPE Module] [SEMANTICS chart,health,diagnostic,provenance]
# @BRIEF Closed diagnostic observations distinguish confirmed query failure from unavailable evidence.
# @INVARIANT Sanitized payloads never claim to be original HTTP response bytes or numeric values.
from typing import Literal
from pydantic import BaseModel, ConfigDict, Field
# #region DashboardTesting.ChartHealth.Contract.Diagnostic [C:3] [TYPE Class]
# @POST Identity, origin, filter/generation attribution and bounded error text remain explicit.
class ChartDiagnostic(BaseModel):
model_config = ConfigDict(extra='forbid', strict=True)
schema_version: Literal[1] = 1
chart_id: int | None = Field(default=None, gt=0)
dataset_id: int | None = Field(default=None, gt=0)
placement_id: str = Field(min_length=1, max_length=256)
environment_id: str | None = None
dashboard_id: int | None = Field(default=None,gt=0)
run_id: str | None = None
logical_step_id: str | None = None
attempt: int | None = Field(default=None,ge=1)
chart_name: str = Field(max_length=512)
tab_path: list[str] = Field(default_factory=list, max_length=32)
status: Literal['healthy', 'failed', 'inconclusive']
reason_code: str = Field(min_length=1, max_length=128)
origin: Literal['dom', 'http_response', 'synthetic_exception']
payload_kind: Literal['sanitized_diagnostic'] = 'sanitized_diagnostic'
filter_fingerprint: str | None = None
request_identity: str | None = None
load_generation: int | None = Field(default=None, ge=1)
observed_at: str
duration_seconds: float = Field(ge=0)
http_status: int | None = Field(default=None, ge=100, le=599)
application_code: str | None = Field(default=None, max_length=256)
database_code: str | None = Field(default=None, max_length=128)
message: str = Field(default='', max_length=4096)
terminal_state: Literal['ready', 'error', 'loading', 'unavailable']
original_response_sha256: str | None = None
original_byte_length: int | None = Field(default=None, ge=0)
truncated: bool = False
# #endregion DashboardTesting.ChartHealth.Contract.Diagnostic
# #endregion DashboardTesting.ChartHealth.Contract

View File

@@ -22,6 +22,8 @@ from .artifacts import is_valid_sha256
from .capacity import CapacityUnavailable, claim_capacity, release_capacity
# #region ScenarioExecution.AgentEvaluation.ManifestItem [C:2] [TYPE Class]
# @BRIEF Closed declared artifact metadata for evaluation input.
class EvaluationManifestItem(BaseModel):
model_config = ConfigDict(extra="forbid")
artifact_id: str
@@ -31,6 +33,9 @@ class EvaluationManifestItem(BaseModel):
role: Literal["actual", "baseline", "comparison", "context"]
# #endregion ScenarioExecution.AgentEvaluation.ManifestItem
# #region ScenarioExecution.AgentEvaluation.Finding [C:2] [TYPE Class]
# @BRIEF Findings bind to a declared criterion and concrete evidence refs.
class EvaluationFinding(BaseModel):
model_config = ConfigDict(extra="forbid")
finding_id: str = Field(min_length=1)
@@ -42,6 +47,9 @@ class EvaluationFinding(BaseModel):
criterion_kind: Literal["semantic", "deterministic_comparison"]
# #endregion ScenarioExecution.AgentEvaluation.Finding
# #region ScenarioExecution.AgentEvaluation.Usage [C:2] [TYPE Class]
# @BRIEF Usage is optional measured transport metadata, never inferred billing.
class EvaluationUsage(BaseModel):
model_config = ConfigDict(extra="forbid")
input_tokens: int | None = Field(default=None, ge=0)
@@ -51,6 +59,9 @@ class EvaluationUsage(BaseModel):
pricing_version: str | None = None
# #endregion ScenarioExecution.AgentEvaluation.Usage
# #region ScenarioExecution.AgentEvaluation.Record [C:3] [TYPE Class]
# @INVARIANT A succeeded record retains raw response provenance; semantic review may have no deterministic comparisons.
class AgentEvaluation(BaseModel):
model_config = ConfigDict(extra="forbid")
schema_version: Literal[1]
@@ -71,7 +82,7 @@ class AgentEvaluation(BaseModel):
input_manifest_hash: str = Field(pattern=r"^[a-f0-9]{64}$")
input_manifest: list[EvaluationManifestItem] = Field(min_length=1, max_length=100)
baseline_pin: dict[str, Any]
comparison_ids: list[str] = Field(min_length=1, max_length=100)
comparison_ids: list[str] = Field(default_factory=list, max_length=100)
status: Literal["succeeded", "provider_error", "parser_error", "budget_exceeded", "cancelled", "timed_out"]
verdict: Literal["pass", "fail", "inconclusive"]
confidence: float = Field(ge=0, le=1)
@@ -84,6 +95,8 @@ class AgentEvaluation(BaseModel):
started_at: datetime
finished_at: datetime
# #region ScenarioExecution.AgentEvaluation.Record.UUIDs [C:2] [TYPE Function]
# @POST Run/operation/evaluation identities are UUIDs; logical step slugs remain valid.
@model_validator(mode="after")
def validate_uuid_fields(self) -> "AgentEvaluation":
# logical_step_id is deliberately excluded: runtime step identity is a registry slug
@@ -97,6 +110,9 @@ class AgentEvaluation(BaseModel):
raise ValueError(f"{field} must be a UUID") from exc
return self
# #endregion ScenarioExecution.AgentEvaluation.Record.UUIDs
# #region ScenarioExecution.AgentEvaluation.Record.Failure [C:3] [TYPE Function]
# @POST Non-succeeded records cannot claim findings, confidence or PASS.
@model_validator(mode="after")
def validate_failure_shape(self) -> "AgentEvaluation":
if self.status == "succeeded" and (not self.raw_response_artifact_ref or not self.raw_response_sha256):
@@ -104,8 +120,12 @@ class AgentEvaluation(BaseModel):
if self.status != "succeeded" and (self.verdict != "inconclusive" or self.confidence != 0 or self.findings or not self.reason_codes):
raise ValueError("EVALUATION_FAILURE_SHAPE_INVALID")
return self
# #endregion ScenarioExecution.AgentEvaluation.Record.Failure
# #endregion ScenarioExecution.AgentEvaluation.Record
# #region ScenarioExecution.AgentEvaluation.Parse [C:3] [TYPE Function]
# @POST Declared criterion identities are checked; malformed responses become typed parser errors.
def parse_evaluation_response(raw: dict[str, Any], *, spec: AgentEvaluationSpec) -> AgentEvaluation:
"""Parse and cross-check a provider response; malformed responses become typed parser errors."""
try:
@@ -132,6 +152,9 @@ def parse_evaluation_response(raw: dict[str, Any], *, spec: AgentEvaluationSpec)
})
# #endregion ScenarioExecution.AgentEvaluation.Parse
# #region ScenarioExecution.AgentEvaluation.Persist [C:3] [TYPE Function]
# @POST Duplicate evaluation identity cannot overwrite prior evidence.
def persist_agent_evaluation(db: Session, record: AgentEvaluation) -> AgentEvaluationRow:
"""Persist one immutable identity; a duplicate identity is rejected rather than updated."""
if db.query(AgentEvaluationRow).filter_by(
@@ -143,6 +166,7 @@ def persist_agent_evaluation(db: Session, record: AgentEvaluation) -> AgentEvalu
db.add(row)
db.flush()
return row
# #endregion ScenarioExecution.AgentEvaluation.Persist
# #region ScenarioExecution.AgentEvaluation.Evidence [C:4] [TYPE Function] [SEMANTICS evaluation,evidence,binding,p0-1]
@@ -177,6 +201,14 @@ def validate_evaluation_evidence(
if row.logical_step_id == logical_step_id and row.attempt == attempt
]
step_by_ref = {row.id: row for row in step_rows} | {row.content_ref: row for row in step_rows}
_validate_input_refs(refs,finding_refs,owned_by_ref,run_id)
_validate_raw_ref(step_by_ref,raw_response_artifact_ref,raw_response_sha256,succeeded,logical_step_id,attempt,run_id)
# #endregion ScenarioExecution.AgentEvaluation.Evidence
# #region ScenarioExecution.AgentEvaluation.InputRefs [C:3] [TYPE Function]
# @POST Every input/finding ref resolves to active same-run metadata; foreign evidence refuses.
def _validate_input_refs(refs,finding_refs,owned_by_ref,run_id):
for ref, item in refs.items():
row = owned_by_ref.get(ref)
if row is None:
@@ -191,6 +223,13 @@ def validate_evaluation_evidence(
raise ValueError("EVALUATION_EVIDENCE_NOT_FOUND")
if row.owner_id != run_id or row.owner_type != "scenario_run":
raise ValueError("EVALUATION_EVIDENCE_OWNER_INVALID")
# #endregion ScenarioExecution.AgentEvaluation.InputRefs
# #region ScenarioExecution.AgentEvaluation.RawRef [C:3] [TYPE Function]
# @POST Successful raw evidence remains bound to exact producing step/attempt/digest.
def _validate_raw_ref(step_by_ref,raw_response_artifact_ref,raw_response_sha256,succeeded,logical_step_id,attempt,run_id):
if succeeded:
if not raw_response_artifact_ref or not is_valid_sha256(raw_response_sha256):
raise ValueError("EVALUATION_EVIDENCE_RAW_REQUIRED")
@@ -205,7 +244,7 @@ def validate_evaluation_evidence(
raw = step_by_ref.get(raw_response_artifact_ref)
if raw is None:
raise ValueError("EVALUATION_EVIDENCE_NOT_FOUND")
# #endregion ScenarioExecution.AgentEvaluation.Evidence
# #endregion ScenarioExecution.AgentEvaluation.RawRef
# #region ScenarioExecution.AgentEvaluation.Messages [C:3] [TYPE Function] [SEMANTICS evaluation,multimodal,messages,content-parts]
@@ -251,6 +290,7 @@ async def submit_evaluation(
from src.plugins.llm_analysis.models import LLMProviderType
from src.plugins.llm_analysis.service import LLMClient
from src.services.llm_provider import LLMProviderService
from .evaluation_capacity_identity import capacity_provider_version
lease = None
try:
@@ -259,7 +299,7 @@ async def submit_evaluation(
raise RuntimeError("EVALUATION_PROVIDER_MISSING")
lease = claim_capacity(db, environment_id=environment_id, environment_class=environment_class,
workload_class="agent_evaluation", provider_id=spec.provider_id,
provider_version=spec.provider_version, run_id=run_id,
provider_version=capacity_provider_version(spec.provider_version), run_id=run_id,
logical_step_id=logical_step_id)
api_key = LLMProviderService(db).get_decrypted_api_key(spec.provider_id)
if not api_key:

View File

@@ -0,0 +1,20 @@
# #region ScenarioExecution.ChartHealth.Artifact [C:4] [TYPE Module] [SEMANTICS ownership,digest,evidence]
# @BRIEF Share exact owned artifact proof for tab frames and evaluation records.
from hashlib import sha256
from src.models.scenario_artifact import ScenarioArtifact
# #region ScenarioExecution.ChartHealth.Artifact.Verify [C:3] [TYPE Function]
# @POST Same-run/step/attempt active identity, MIME, length and bytes are verified before model input or result projection.
def verify_health_artifact(journal, db, receipt):
artifact = db.get(ScenarioArtifact,receipt['artifact_id'])
data = journal.storage.retrieve(receipt['content_ref'])
if (artifact is None or not artifact.is_active or artifact.owner_type != 'scenario_run' or artifact.owner_id != journal.run_id
or artifact.logical_step_id != journal.logical_step_id or artifact.attempt != journal.attempt
or artifact.content_ref != receipt['content_ref'] or artifact.sha256 != receipt['sha256']
or artifact.byte_length != receipt['byte_length'] or artifact.content_type != receipt['content_type']
or data is None or len(data) != receipt['byte_length'] or sha256(data).hexdigest() != receipt['sha256']):
raise ValueError('CHART_HEALTH_ARTIFACT_INVALID')
return artifact,data
# #endregion ScenarioExecution.ChartHealth.Artifact.Verify
# #endregion ScenarioExecution.ChartHealth.Artifact

View File

@@ -0,0 +1,103 @@
# #region ScenarioExecution.ChartHealth.Delivery [C:5] [TYPE Module] [SEMANTICS notification,CAS,at-most-once,summary]
# @BRIEF Deliver a committed named-error receipt once through explicitly configured existing providers.
# @INVARIANT Claim commits before transport; ambiguous dispatching receipts are never automatically retried.
import asyncio
from hashlib import sha256
from time import monotonic
from sqlalchemy import update
from src.models.scenario_automation import ScenarioNotificationEvent
from ..query_failure import safe_message
from .chart_health_delivery_config import read_alert_config,configured_provider
# #region ScenarioExecution.ChartHealth.Delivery.Body [C:3] [TYPE Function]
# @POST Bounded redacted text names each retained chart/cause and the run; omission is explicit.
def message_body(receipt):
summary = receipt.payload['chart_query_errors']
lines = [f"Scenario run: {receipt.run_id}",f"Chart query errors: {summary['count']}"]
for item in summary.get('items',[])[:100]:
context = '/'.join(item.get('tab_path',[]))
text = safe_message(item.get('message',''))[0][:512]
name = safe_message(item.get('chart_name',''))[0][:512]
lines.append(f"{context}: {name} — Code {item.get('database_code') or 'unknown'} — {text}")
if summary.get('omitted'):
lines.append(f"Additional omitted errors: {summary['omitted']}")
raw = '\n'.join(lines).encode()
suffix = b'\n[notification summary truncated]'
return (raw[:16384-len(suffix)]+suffix).decode(errors='ignore') if len(raw) > 16384 else raw.decode()
# #endregion ScenarioExecution.ChartHealth.Delivery.Body
# #region ScenarioExecution.ChartHealth.Delivery.Claim [C:4] [TYPE Function]
# @POST One atomic pending-to-dispatching CAS owns physical sends; no target/config secrets are persisted.
def claim_receipt(db, receipt):
payload = dict(receipt.payload)
payload['health_delivery'] = {'status':'dispatching','reason_code':'DELIVERY_CLAIMED'}
statement = update(ScenarioNotificationEvent).where(
ScenarioNotificationEvent.id == receipt.id,
ScenarioNotificationEvent.payload['health_delivery']['status'].as_string() == 'pending',
).values(payload=payload).execution_options(synchronize_session=False)
claimed = db.execute(statement).rowcount == 1
db.commit()
if claimed:
db.refresh(receipt)
return claimed
# #endregion ScenarioExecution.ChartHealth.Delivery.Claim
# #region ScenarioExecution.ChartHealth.Delivery.Route [C:3] [TYPE Function]
# @POST Each explicit route returns a bounded honest result; failed transports do not escape into terminalization.
async def send_route(route, notifications, subject, body, timeout, provider_factory):
identifier = sha256(f'{route.type}:{route.target}'.encode()).hexdigest()
result = {'route_hash':identifier,'type':route.type,'status':'skipped','reason_code':'PROVIDER_UNCONFIGURED'}
try:
provider = provider_factory(route,notifications)
if provider is None:
return result
sent = await asyncio.wait_for(provider.send(route.target,subject,body),timeout=max(0.001,timeout))
result.update(status='delivered' if sent else 'failed',reason_code='DELIVERED' if sent else 'DELIVERY_FAILED')
except asyncio.CancelledError:
raise
except Exception:
result.update(status='failed',reason_code='DELIVERY_FAILED')
return result
# #endregion ScenarioExecution.ChartHealth.Delivery.Route
# #region ScenarioExecution.ChartHealth.Delivery.Execute [C:4] [TYPE Function]
# @PRE Receipt is committed; only an explicit enabled policy permits sends.
# @POST Receipt records delivered/failed/skipped outcome once; model verdict and run evidence are unchanged.
async def deliver_receipt(db, receipt_id, *, provider_factory=configured_provider):
receipt = db.get(ScenarioNotificationEvent,receipt_id)
if receipt is None or not (receipt.payload or {}).get('chart_query_errors'):
return 'unavailable'
if not claim_receipt(db,receipt):
return 'already_claimed'
try:
policy,notifications = read_alert_config(db)
if not policy.enabled or not policy.channels:
outcome = {'status':'skipped','reason_code':'DELIVERY_DISABLED' if not policy.enabled else 'DESTINATION_UNCONFIGURED','routes':[]}
else:
end,results = monotonic()+policy.timeout_seconds,[]
body = message_body(receipt)
for route in policy.channels:
results.append(await send_route(route,notifications,'Dashboard chart query errors',body,end-monotonic(),provider_factory))
outcome = {'status':'delivered' if all(item['status'] == 'delivered' for item in results) else 'failed',
'reason_code':'DELIVERY_COMPLETE' if all(item['status'] == 'delivered' for item in results) else 'DELIVERY_INCOMPLETE','routes':results}
except asyncio.CancelledError:
save_delivery(db,receipt,{'status':'failed','reason_code':'DELIVERY_INTERRUPTED','routes':[]})
raise
except Exception:
outcome = {'status':'skipped','reason_code':'DELIVERY_CONFIGURATION_INVALID','routes':[]}
save_delivery(db,receipt,outcome)
return outcome['status']
# #endregion ScenarioExecution.ChartHealth.Delivery.Execute
# #region ScenarioExecution.ChartHealth.Delivery.Save [C:2] [TYPE Function]
# @POST Delivery outcome replaces only its own claimed receipt projection and commits independently.
def save_delivery(db, receipt, outcome):
receipt.payload = {**receipt.payload,'health_delivery':outcome}
db.commit()
# #endregion ScenarioExecution.ChartHealth.Delivery.Save
# #endregion ScenarioExecution.ChartHealth.Delivery

View File

@@ -0,0 +1,46 @@
# #region ScenarioExecution.ChartHealth.DeliveryConfig [C:3] [TYPE Module] [SEMANTICS notify,opt-in,configuration,budget]
# @BRIEF Read closed opt-in routing from existing global notification settings; no credentials enter receipts.
from typing import Literal
from pydantic import BaseModel, ConfigDict, Field
from src.models.config import AppConfigRecord
# #region ScenarioExecution.ChartHealth.DeliveryConfig.Route [C:2] [TYPE Class]
# @INVARIANT A destination is explicit and bounded; only existing provider types are supported.
class AlertRoute(BaseModel):
model_config = ConfigDict(extra='forbid',strict=True)
type: Literal['SMTP','TELEGRAM','SLACK']
target: str = Field(min_length=1,max_length=2048)
# #endregion ScenarioExecution.ChartHealth.DeliveryConfig.Route
# #region ScenarioExecution.ChartHealth.DeliveryConfig.Policy [C:2] [TYPE Class]
# @POST Missing configuration disables delivery without changing durable error receipts.
class ScenarioAlertPolicy(BaseModel):
model_config = ConfigDict(extra='forbid',strict=True)
enabled: bool = False
channels: list[AlertRoute] = Field(default_factory=list,max_length=20)
timeout_seconds: int = Field(default=30,ge=1,le=60)
# #endregion ScenarioExecution.ChartHealth.DeliveryConfig.Policy
# #region ScenarioExecution.ChartHealth.DeliveryConfig.Read [C:2] [TYPE Function]
# @POST The committed global record supplies routes; unknown routing fields fail closed.
def read_alert_config(db):
record = db.get(AppConfigRecord,'global')
notifications = ((record.payload or {}).get('notifications') or {}) if record else {}
return ScenarioAlertPolicy.model_validate(notifications.get('scenario_alerts') or {}),notifications
# #endregion ScenarioExecution.ChartHealth.DeliveryConfig.Read
# #region ScenarioExecution.ChartHealth.DeliveryConfig.Provider [C:3] [TYPE Function]
# @POST Missing provider connection data returns None; existing provider implementations own transport behavior.
def configured_provider(route, notifications):
from src.services.notifications.providers import SMTPProvider,TelegramProvider,SlackProvider
config = notifications.get(route.type.lower()) or {}
required = {'SMTP':['host','from_email'],'TELEGRAM':['bot_token'],'SLACK':['webhook_url']}[route.type]
if not all(config.get(key) for key in required):
return None
return {'SMTP':SMTPProvider,'TELEGRAM':TelegramProvider,'SLACK':SlackProvider}[route.type](config)
# #endregion ScenarioExecution.ChartHealth.DeliveryConfig.Provider
# #endregion ScenarioExecution.ChartHealth.DeliveryConfig

View File

@@ -0,0 +1,99 @@
# #region ScenarioExecution.ChartHealth.DeliveryHook [C:4] [TYPE Module] [SEMANTICS after-commit,notification,capacity,rollback]
# @BRIEF Schedule opt-in notification transport only after the owning terminal transaction commits.
# @INVARIANT One bounded background submission uses the existing provider loop; rollback cannot send.
from threading import Semaphore, Thread
from sqlalchemy import event
from src.core.database import SessionLocal
from .chart_health_delivery_config import read_alert_config
_slot = Semaphore(1)
# #region ScenarioExecution.ChartHealth.DeliveryHook.Arm [C:3] [TYPE Function]
# @POST Named-error receipts get honest disabled/unconfigured/pending status; no network occurs inside terminalization.
def arm_health_delivery(db, receipt):
if not (receipt.payload or {}).get('chart_query_errors') or receipt.payload.get('health_delivery'):
return
try:
policy,_ = read_alert_config(db)
state = {'status':'pending','reason_code':'DELIVERY_PENDING'} if policy.enabled and policy.channels else {
'status':'skipped','reason_code':'DELIVERY_DISABLED' if not policy.enabled else 'DESTINATION_UNCONFIGURED'}
except Exception:
state = {'status':'skipped','reason_code':'DELIVERY_CONFIGURATION_INVALID'}
receipt.payload = {**receipt.payload,'health_delivery':state}
if state['status'] != 'pending':
return
db.info.setdefault('chart_health_deliveries',set()).add(receipt.id)
if not db.info.get('chart_health_delivery_hooks'):
event.listen(db,'after_commit',after_health_commit)
event.listen(db,'after_rollback',after_health_rollback)
db.info['chart_health_delivery_hooks'] = True
# #endregion ScenarioExecution.ChartHealth.DeliveryHook.Arm
# #region ScenarioExecution.ChartHealth.DeliveryHook.Commit [C:3] [TYPE Function]
# @POST At most one bounded physical dispatcher exists; terminal commit cannot fail because of transport startup.
def after_health_commit(db):
identifiers = sorted(db.info.pop('chart_health_deliveries',set()))[:25]
if not identifiers:
return
try:
if not _slot.acquire(blocking=False):
mark_unavailable(identifiers,'DELIVERY_CAPACITY_UNAVAILABLE')
return
Thread(target=dispatch_committed,args=(identifiers,),daemon=True,name='chart-health-notifications').start()
except Exception:
_slot.release()
mark_unavailable(identifiers,'DELIVERY_RUNTIME_UNAVAILABLE')
# #endregion ScenarioExecution.ChartHealth.DeliveryHook.Commit
# #region ScenarioExecution.ChartHealth.DeliveryHook.Rollback [C:1] [TYPE Function]
# @POST Rolled-back receipt IDs are discarded before any physical send can be scheduled.
def after_health_rollback(db):
db.info.pop('chart_health_deliveries',None)
# #endregion ScenarioExecution.ChartHealth.DeliveryHook.Rollback
# #region ScenarioExecution.ChartHealth.DeliveryHook.Unavailable [C:3] [TYPE Function]
# @POST Only unclaimed pending deliveries become explicitly skipped; already sent/ambiguous receipts are untouched.
def mark_unavailable(identifiers, reason):
from sqlalchemy import update
from src.models.scenario_automation import ScenarioNotificationEvent
try:
with SessionLocal() as session:
for identifier in identifiers:
row = session.get(ScenarioNotificationEvent,identifier)
if row is not None:
payload = {**row.payload,'health_delivery':{'status':'skipped','reason_code':reason}}
session.execute(update(ScenarioNotificationEvent).where(
ScenarioNotificationEvent.id == identifier,
ScenarioNotificationEvent.payload['health_delivery']['status'].as_string() == 'pending',
).values(payload=payload).execution_options(synchronize_session=False))
session.commit()
except Exception:
pass
# #endregion ScenarioExecution.ChartHealth.DeliveryHook.Unavailable
# #region ScenarioExecution.ChartHealth.DeliveryHook.Dispatch [C:3] [TYPE Function]
# @POST Existing provider-loop admission bounds each committed receipt; startup failure records skipped rather than fake sent.
def dispatch_committed(identifiers):
from .provider_runtime import get_provider_event_loop
try:
runtime = get_provider_event_loop()
for identifier in identifiers:
# #region ScenarioExecution.ChartHealth.DeliveryHook.Dispatch.Send [C:1] [TYPE Function]
# @BRIEF Keep DB lifetime on the same provider loop as its existing notification clients.
async def send(receipt_id=identifier):
from .chart_health_delivery import deliver_receipt
with SessionLocal() as session:
return await deliver_receipt(session,receipt_id)
# #endregion ScenarioExecution.ChartHealth.DeliveryHook.Dispatch.Send
runtime.submit(send,timeout=65)
except Exception:
mark_unavailable(identifiers,'DELIVERY_RUNTIME_UNAVAILABLE')
finally:
_slot.release()
# #endregion ScenarioExecution.ChartHealth.DeliveryHook.Dispatch
# #endregion ScenarioExecution.ChartHealth.DeliveryHook

View File

@@ -0,0 +1,31 @@
# #region ScenarioExecution.ChartHealth.Notification [C:3] [TYPE Module] [SEMANTICS notify,health,summary,deduplication]
# @BRIEF Add bounded grouped chart causes to the existing idempotent terminal automation receipt.
from src.models.scenario_run import ScenarioStepRun
from ..query_failure import safe_message
# #region ScenarioExecution.ChartHealth.Notification.Summary [C:3] [TYPE Function]
# @POST One terminal receipt names failed placements and causes; truncation is explicit and no delivery channel is invented.
def notification_health_summary(db, run_id):
errors, omitted = [],0
seen = set()
for step in db.query(ScenarioStepRun).filter_by(run_id=run_id).order_by(ScenarioStepRun.logical_step_id,ScenarioStepRun.attempt.desc()).all():
if step.logical_step_id in seen:
continue
seen.add(step.logical_step_id)
outer = step.step_outcome or {}
outcome = outer.get('step_outcome',outer)
health = outcome.get('chart_health') or {}
for item in health.get('errors',outcome.get('chart_diagnostics',[])):
if item.get('status') != 'failed':
continue
if len(errors) >= 100:
omitted += 1
continue
message, truncated = safe_message(item.get('message',''))
errors.append({'logical_step_id':step.logical_step_id,'placement_id':item['placement_id'],
'chart_name':item['chart_name'],'tab_path':item['tab_path'],'database_code':item.get('database_code'),
'message':message[:512],'truncated':truncated or len(message) > 512})
return {'chart_query_errors':{'count':len(errors)+omitted,'items':errors,'omitted':omitted}} if errors or omitted else {}
# #endregion ScenarioExecution.ChartHealth.Notification.Summary
# #endregion ScenarioExecution.ChartHealth.Notification

View File

@@ -0,0 +1,105 @@
# #region ScenarioExecution.ChartHealth.Store [C:5] [TYPE Module] [SEMANTICS health,ownership,coverage,failed,evidence]
# @BRIEF Commit each chart observation independently and preserve confirmed failures alongside incomplete coverage.
# @INVARIANT Failed observations are durable before the next chart/tab; completeness never follows from a model verdict.
from copy import deepcopy
from hashlib import sha256
import json
from src.core.database import SessionLocal
from src.models.scenario_traversal import ScenarioTraversal, ScenarioTraversalPage
from src.models.scenario_artifact import ScenarioArtifact
from ..chart_health_contract import ChartDiagnostic
from .traversal_store import TraversalJournal, canonical
# #region ScenarioExecution.ChartHealth.Store.Journal [C:5] [TYPE Class]
class ChartHealthJournal(TraversalJournal):
# #region ScenarioExecution.ChartHealth.Store.Journal.Freeze [C:3] [TYPE Function]
# @POST Exact server placements are frozen before observation; changed layouts refuse resume.
def freeze_source(self, source):
with SessionLocal() as db:
row = db.query(ScenarioTraversal).filter_by(id=self.id).with_for_update().one()
if row.state['source'] is not None and row.state['source'] != source:
raise ValueError('CHART_HEALTH_MANIFEST_CHANGED')
row.state = {**row.state,'source':deepcopy(source)}
db.commit()
# #endregion ScenarioExecution.ChartHealth.Store.Journal.Freeze
# #region ScenarioExecution.ChartHealth.Store.Journal.Append [C:4] [TYPE Function]
# @POST One exact server placement, typed diagnostic, artifact and frontier advance commit atomically.
def append_chart(self, observation):
reason = self.check_control()
if reason:
raise ValueError(reason)
value = ChartDiagnostic.model_validate(observation).model_dump()
target = self.step.get('target_snapshot') or {}
value.update(run_id=self.run_id,logical_step_id=self.logical_step_id,attempt=self.attempt,
environment_id=target.get('environment_id'),dashboard_id=target.get('dashboard_id'))
with SessionLocal() as db:
row = db.query(ScenarioTraversal).filter_by(id=self.id).with_for_update().one()
state = row.state
ordinal = state['next_ordinal']
expected = state['source']['placements'][ordinal-1]
if (value['placement_id'],value['chart_id'],value['chart_name'],value['tab_path']) != (expected['id'],expected['chart_id'],expected['name'],expected['tab_path']):
raise ValueError('CHART_HEALTH_PLACEMENT_MISMATCH')
data = canonical({'chart_diagnostic':value})
if state['byte_count']+state.get('auxiliary_bytes',0)+len(data) > self.limits.max_bytes:
raise ValueError('CHART_HEALTH_BYTES_EXCEEDED')
artifact = self._artifact(db,data,f'chart-health-{ordinal}.json')
receipt = {'ordinal':ordinal,'artifact_id':artifact.id,'content_ref':artifact.content_ref,
'sha256':artifact.sha256,'byte_length':len(data),'placement_id':value['placement_id'],
'status':value['status'],'row_count':1,'next_available':ordinal < len(state['source']['placements'])}
chain = sha256(state['chain_digest'].encode()+canonical(receipt)).hexdigest()
receipt['chain_digest'] = chain
db.add(ScenarioTraversalPage(traversal_id=self.id,ordinal=ordinal,receipt=receipt))
row.state = {**state,'next_ordinal':ordinal+1,'row_count':ordinal,'byte_count':state['byte_count']+len(data),
'chain_digest':chain,'terminal':not receipt['next_available']}
db.commit()
# #endregion ScenarioExecution.ChartHealth.Store.Journal.Append
# #region ScenarioExecution.ChartHealth.Store.Journal.Verify [C:4] [TYPE Function]
# @POST Ownership/hash/type/identity/ordered-chain violations reject evidence and resume.
def verify_receipts(self):
with SessionLocal() as db:
row = db.get(ScenarioTraversal,self.id)
chain, count, size = '',0,0
for item in db.query(ScenarioTraversalPage).filter_by(traversal_id=self.id).order_by(ScenarioTraversalPage.ordinal).yield_per(1):
receipt = item.receipt
value, artifact = self.load_owned(db,receipt)
count += 1
expected = row.state['source']['placements'][count-1]
if (item.ordinal != count or value.placement_id != expected['id'] or value.chart_id != expected['chart_id']
or value.tab_path != expected['tab_path'] or value.status != receipt['status']):
raise ValueError('CHART_HEALTH_RECEIPT_INVALID')
chain = sha256(chain.encode()+canonical({key:v for key,v in receipt.items() if key != 'chain_digest'})).hexdigest()
if chain != receipt['chain_digest']:
raise ValueError('CHART_HEALTH_CHAIN_INVALID')
size += artifact.byte_length
from .chart_health_tab_store import verify_tab_evaluations
verify_tab_evaluations(self,db,row.state)
if row.state['next_ordinal'] != count+1 or row.state['row_count'] != count or row.state['byte_count'] != size or row.state['chain_digest'] != chain:
raise ValueError('CHART_HEALTH_FRONTIER_INVALID')
# #endregion ScenarioExecution.ChartHealth.Store.Journal.Verify
# #region ScenarioExecution.ChartHealth.Store.Journal.Load [C:4] [TYPE Function]
# @POST Only exact active same-run/step/attempt bytes can enter summary or model evidence.
def load_owned(self, db, receipt):
artifact = db.get(ScenarioArtifact,receipt['artifact_id'])
data = self.storage.retrieve(receipt['content_ref'])
if (artifact is None or not artifact.is_active or artifact.owner_type != 'scenario_run' or artifact.owner_id != self.run_id
or artifact.logical_step_id != self.logical_step_id or artifact.attempt != self.attempt
or artifact.content_type != 'application/json' or artifact.content_ref != receipt['content_ref']
or artifact.sha256 != receipt['sha256'] or artifact.byte_length != receipt['byte_length']
or data is None or len(data) != receipt['byte_length'] or sha256(data).hexdigest() != receipt['sha256']):
raise ValueError('CHART_HEALTH_ARTIFACT_INVALID')
return ChartDiagnostic.model_validate(json.loads(data)['chart_diagnostic']), artifact
# #endregion ScenarioExecution.ChartHealth.Store.Journal.Load
# #region ScenarioExecution.ChartHealth.Store.Journal.Finish [C:4] [TYPE Function]
# @POST Confirmed failure dominates partial coverage; PASS requires all expected placements conclusively healthy.
def finish(self, status, reason):
self.verify_receipts()
from .chart_health_summary import finish_health
return finish_health(self,status,reason)
# #endregion ScenarioExecution.ChartHealth.Store.Journal.Finish
# #endregion ScenarioExecution.ChartHealth.Store.Journal
# #endregion ScenarioExecution.ChartHealth.Store

View File

@@ -0,0 +1,60 @@
# #region ScenarioExecution.ChartHealth.Summary [C:4] [TYPE Module] [SEMANTICS health,coverage,verdict,manifest]
# @BRIEF Aggregate durable diagnostics without conflating confirmed failure with observation completeness.
from src.core.database import SessionLocal
from src.models.scenario_traversal import ScenarioTraversal, ScenarioTraversalPage
from .traversal_store import canonical
# #region ScenarioExecution.ChartHealth.Summary.Tabs [C:3] [TYPE Function]
# @POST Stable tab IDs group names/causes while preserving each placement and its coverage.
def tab_groups(source, diagnostics):
result = []
for tab in (source or {}).get('tabs',[]):
expected = [item for item in source['placements'] if item['tab_path'] and item['tab_path'][-1] == tab['id']]
observed = [item for item in diagnostics if item['tab_path'] and item['tab_path'][-1] == tab['id']]
errors = [item for item in observed if item['status'] == 'failed']
result.append({'tab_id':tab['id'],'name':tab.get('name',tab['id']),'expected':len(expected),'observed':len(observed),
'errored':len(errors),'unvisited':len(expected)-len(observed),'status':'failed' if errors else
('passed' if len(observed) == len(expected) and all(item['status'] == 'healthy' for item in observed) else 'inconclusive')})
return result
# #endregion ScenarioExecution.ChartHealth.Summary.Tabs
# #region ScenarioExecution.ChartHealth.Summary.Finish [C:4] [TYPE Function]
# @POST Every retained placement remains addressable; a confirmed error can never become PASS or merely INCONCLUSIVE.
def finish_health(journal, requested_status, reason):
with SessionLocal() as db:
row = db.query(ScenarioTraversal).filter_by(id=journal.id).with_for_update().one()
source = row.state['source']
receipts = db.query(ScenarioTraversalPage).filter_by(traversal_id=journal.id).order_by(ScenarioTraversalPage.ordinal).all()
diagnostics = [journal.load_owned(db,item.receipt)[0].model_dump() for item in receipts]
errors = sum(item['status'] == 'failed' for item in diagnostics)
unresolved = sum(item['status'] == 'inconclusive' for item in diagnostics)
expected = len(source['placements']) if source else 0
complete = bool(requested_status == 'passed' and source is not None and len(diagnostics) == expected and unresolved == 0)
evaluations = row.state.get('tab_evaluations',[])
vlm_complete = all(item['status'] == 'succeeded' and item['coverage_complete'] for item in evaluations)
status = 'failed' if errors else ('passed' if complete and vlm_complete else 'inconclusive')
coverage = {'expected':expected,'observed':len(diagnostics),'checked':len(diagnostics)-unresolved,
'errored':errors,'unresolved':unresolved,'timeout':sum(item['reason_code'] == 'CHART_LOAD_TIMEOUT' for item in diagnostics),
'unvisited':expected-len(diagnostics),'complete':complete}
manifest = {'schema_version':1,'kind':'chart_health','run_id':journal.run_id,'logical_step_id':journal.logical_step_id,
'attempt':journal.attempt,'status':status,'reason_code':'CHART_QUERY_FAILED' if errors else reason,
'interruption_reason':reason if not complete else None,'complete':complete,'coverage':coverage,
'tab_evaluations':evaluations,'visual_coverage_complete':vlm_complete,
'source':source,'chain_digest':row.state['chain_digest'],
'charts':[dict(item,artifact_id=receipt.receipt['artifact_id'],content_ref=receipt.receipt['content_ref'],
sha256=receipt.receipt['sha256']) for item,receipt in zip(diagnostics,receipts)]}
data = canonical(manifest)
if len(data) > 262144:
raise ValueError('CHART_HEALTH_MANIFEST_TOO_LARGE')
artifact = journal._artifact(db,data,'chart-health-manifest.json')
row.status = status
db.commit()
return {'status':status,'reason_code':manifest['reason_code'],'complete':complete,'coverage':coverage,
'chart_health':{'coverage':coverage,'charts':manifest['charts'],'errors':[item for item in manifest['charts'] if item['status'] == 'failed'],
'tabs':tab_groups(source,diagnostics),'tab_evaluations':evaluations,'visual_coverage_complete':vlm_complete},
'manifest_artifact_id':artifact.id,'manifest_ref':artifact.content_ref,
'manifest_sha256':artifact.sha256,'manifest_byte_length':artifact.byte_length}
# #endregion ScenarioExecution.ChartHealth.Summary.Finish
# #endregion ScenarioExecution.ChartHealth.Summary

View File

@@ -0,0 +1,64 @@
# #region ScenarioExecution.ChartHealth.TabStore [C:4] [TYPE Module] [SEMANTICS tabs,evaluation,owned,commit]
# @BRIEF Retain captures/evaluation results as owned artifacts before leaving the active tab.
from src.core.database import SessionLocal
from src.models.scenario_traversal import ScenarioTraversal, ScenarioTraversalPage
from .chart_health_artifact import verify_health_artifact
# #region ScenarioExecution.ChartHealth.TabStore.Store [C:3] [TYPE Function]
# @POST MIME/length and owner metadata bind every frame/context/evaluation to the active attempt.
def store_health_artifact(journal, data, name, content_type='application/json'):
reason = journal.check_control()
if reason:
raise ValueError(reason)
with SessionLocal() as db:
row = db.query(ScenarioTraversal).filter_by(id=journal.id).with_for_update().one()
size = row.state.get('auxiliary_bytes',0)+len(data)
if size+row.state['byte_count'] > journal.limits.max_bytes:
raise ValueError('CHART_HEALTH_BYTES_EXCEEDED')
artifact = journal._artifact(db,data,name,content_type)
row.state = {**row.state,'auxiliary_bytes':size}
db.commit()
return {'artifact_id':artifact.id,'content_ref':artifact.content_ref,'sha256':artifact.sha256,
'byte_length':artifact.byte_length,'content_type':artifact.content_type}
# #endregion ScenarioExecution.ChartHealth.TabStore.Store
# #region ScenarioExecution.ChartHealth.TabStore.Diagnostics [C:3] [TYPE Function]
# @POST Current-tab payloads and manifest refs come solely from committed current-attempt diagnostic artifacts.
def tab_diagnostics(journal, tab_path):
with SessionLocal() as db:
result = []
for row in db.query(ScenarioTraversalPage).filter_by(traversal_id=journal.id).order_by(ScenarioTraversalPage.ordinal).all():
diagnostic, artifact = journal.load_owned(db,row.receipt)
if diagnostic.tab_path == tab_path:
result.append({'diagnostic':diagnostic.model_dump(),'receipt':{
'artifact_id':artifact.id,'content_ref':artifact.content_ref,'sha256':artifact.sha256,
'byte_length':artifact.byte_length,'content_type':artifact.content_type}})
return result
# #endregion ScenarioExecution.ChartHealth.TabStore.Diagnostics
# #region ScenarioExecution.ChartHealth.TabStore.Append [C:3] [TYPE Function]
# @POST A persisted tab evaluation summary cannot substitute model PASS for deterministic failure.
def append_tab_evaluation(journal, summary):
with SessionLocal() as db:
row = db.query(ScenarioTraversal).filter_by(id=journal.id).with_for_update().one()
for receipt in summary['artifacts']:
verify_health_artifact(journal,db,receipt)
results = row.state.get('tab_evaluations',[])
if any(item['tab_path'] == summary['tab_path'] for item in results):
raise ValueError('CHART_HEALTH_TAB_EVALUATION_DUPLICATE')
row.state = {**row.state,'tab_evaluations':[*results,summary]}
db.commit()
# #endregion ScenarioExecution.ChartHealth.TabStore.Append
# #region ScenarioExecution.ChartHealth.TabStore.Verify [C:3] [TYPE Function]
# @POST Resume verifies every captured frame/evaluation receipt; model fields remain explanation-only.
def verify_tab_evaluations(journal, db, state):
for summary in state.get('tab_evaluations',[]):
for receipt in summary['artifacts']:
verify_health_artifact(journal,db,receipt)
# #endregion ScenarioExecution.ChartHealth.TabStore.Verify
# #endregion ScenarioExecution.ChartHealth.TabStore

View File

@@ -26,6 +26,7 @@ from .evaluation_manifest import _manifest_from_completed as _manifest_from_comp
from .evaluation_images import image_payloads_from_manifest
from .evaluation_prompt import build_evaluation_prompt
from .evaluation_recipe_context import prepare_recipe_text_evidence
from .evaluation_health_context import prepare_health_context
from .live_binding import EvidenceStorage
@@ -295,6 +296,8 @@ def evaluation_adapter_from(
submit: Callable[..., dict[str, Any]] | None = None,
db_factory: Callable[[], Any] | None = None,
) -> Callable[[dict[str, Any], dict[str, dict[str, Any]]], dict[str, Any]]:
# #region ScenarioExecution.EvaluationAdapter.Own.invoke [C:3] [TYPE Function]
# @BRIEF Evaluate the supplied step using owned completed evidence and runtime dependencies.
def invoke(step: dict[str, Any], completed: dict[str, dict[str, Any]]) -> dict[str, Any]:
spec = _spec_from_step(step)
manifest = _manifest_from_completed(completed)
@@ -309,6 +312,8 @@ def evaluation_adapter_from(
evidence_payloads = None
if isinstance(spec.limits, TokenEvaluationLimits):
manifest, evidence_payloads = prepare_recipe_text_evidence(step, spec, completed, evidence, db_factory=db_factory)
else:
evidence_payloads = prepare_health_context(step,completed,manifest,evidence,db_factory=db_factory)
images = image_payloads_from_manifest(manifest, evidence, max_images=spec.limits.max_images)
if images:
# Candidate evidence loaded; actual attachment is gated on provider multimodality
@@ -359,6 +364,7 @@ def evaluation_adapter_from(
"content_type": "application/json",
"byte_length": len(raw_bytes),
}
# #endregion ScenarioExecution.EvaluationAdapter.Own.invoke
return invoke
# #endregion ScenarioExecution.EvaluationAdapter.Factory

View File

@@ -0,0 +1,15 @@
# #region ScenarioExecution.EvaluationCapacityIdentity [C:3] [TYPE Module] [SEMANTICS provider,pin,capacity,width]
# @BRIEF Bind a full authored config pin to the existing64char capacity lease without changing evaluation provenance.
import re
# #region ScenarioExecution.EvaluationCapacityIdentity.Resolve [C:3] [TYPE Function]
# @POST Explicit config_sha256 pins use their verified digest for capacity; arbitrary long identifiers fail closed.
def capacity_provider_version(value):
if re.fullmatch(r'config_sha256:[a-f0-9]{64}',value):
return value.split(':',1)[1]
if len(value) > 64:
raise ValueError('EVALUATION_CAPACITY_IDENTITY_INVALID')
return value
# #endregion ScenarioExecution.EvaluationCapacityIdentity.Resolve
# #endregion ScenarioExecution.EvaluationCapacityIdentity

View File

@@ -0,0 +1,24 @@
# #region ScenarioExecution.ChartHealth.EvaluationContext [C:4] [TYPE Module] [SEMANTICS evaluation,diagnostic,ownership,completed]
# @BRIEF Enrich ordinary VLM evaluation with actual owned error contents when prior runtime outcomes declare chart diagnostics.
from src.core.database import SessionLocal
from .evaluation_health_payloads import load_health_payloads
# #region ScenarioExecution.ChartHealth.EvaluationContext.Prepare [C:3] [TYPE Function]
# @POST Ordinary evidence behavior is unchanged; diagnostic callers require durable same-run proof before contents reach the model.
def prepare_health_context(step, completed, manifest, storage, *, db_factory=None):
declared = False
for outer in completed.values():
outcome = outer.get('step_outcome',outer) if isinstance(outer,dict) else {}
if isinstance(outcome,dict) and (outcome.get('chart_diagnostics') or outcome.get('chart_health')):
declared = True
if not declared:
return None
db = (db_factory or SessionLocal)()
try:
return load_health_payloads(manifest,storage,db=db,run_id=step['scenario_run_id'])
finally:
if db_factory is None:
db.close()
# #endregion ScenarioExecution.ChartHealth.EvaluationContext.Prepare
# #endregion ScenarioExecution.ChartHealth.EvaluationContext

View File

@@ -0,0 +1,47 @@
# #region ScenarioExecution.ChartHealth.EvaluationPayloads [C:5] [TYPE Module] [SEMANTICS evaluation,owned,diagnostic,redaction,budget]
# @BRIEF Send actual owned diagnostic contents to the model, including failed observations, within explicit text limits.
# @INVARIANT Diagnostic JSON is evidence data, not instructions or a fabricated deterministic comparison.
from hashlib import sha256
import json
from src.models.scenario_artifact import ScenarioArtifact
from ..chart_health_contract import ChartDiagnostic
from .evaluation_text_json import redact_json_content
# #region ScenarioExecution.ChartHealth.EvaluationPayloads.Load [C:4] [TYPE Function]
# @POST Foreign/inactive/changed bytes refuse; omissions/truncation and total bytes are explicit.
def load_health_payloads(manifest, storage, *, db, run_id, max_bytes=262144, max_items=8):
payloads, omitted, total = [],[],0
for item in manifest:
if item['content_type'] != 'application/json':
continue
if len(payloads) >= max_items or item['byte_length'] > max_bytes-total:
omitted.append({'artifact_id':item['artifact_id'],'reason':'TEXT_BUDGET'})
continue
data = storage.retrieve(item['artifact_id'])
if data is None or len(data) != item['byte_length'] or sha256(data).hexdigest() != item['sha256']:
raise RuntimeError('CHART_HEALTH_EVIDENCE_INVALID')
try:
body = json.loads(data)
except (ValueError, UnicodeError):
continue
if not isinstance(body,dict):
continue
diagnostic = body.get('chart_diagnostic')
if diagnostic is None and body.get('kind') != 'chart_health':
continue
artifact = db.query(ScenarioArtifact).filter_by(owner_type='scenario_run',owner_id=run_id,
content_ref=item['artifact_id'],sha256=item['sha256'],is_active=True).one_or_none()
if artifact is None or artifact.byte_length != len(data) or artifact.content_type != 'application/json':
raise RuntimeError('CHART_HEALTH_EVIDENCE_NOT_OWNED')
if diagnostic is not None:
value = ChartDiagnostic.model_validate(diagnostic).model_dump()
else:
value = {'coverage':body['coverage'],'charts':[ChartDiagnostic.model_validate({key:entry[key] for key in ChartDiagnostic.model_fields if key in entry}).model_dump() for entry in body['charts']]}
payloads.append({'artifact_id':item['artifact_id'],'logical_step_id':artifact.logical_step_id,
'attempt':artifact.attempt,'content':redact_json_content(value)})
total += len(data)
return {'schema_version':1,'kind':'chart_diagnostic_contents','items':payloads,
'omitted':omitted,'byte_length':total,'truncated':bool(omitted)}
# #endregion ScenarioExecution.ChartHealth.EvaluationPayloads.Load
# #endregion ScenarioExecution.ChartHealth.EvaluationPayloads

View File

@@ -73,6 +73,8 @@ class LiveExecutionBinding:
evidence_owner_type: str
evidence_ref_policy: str
# #region ScenarioExecution.LiveBinding.Own.from_snapshot [C:3] [TYPE Function]
# @BRIEF Validate and reconstruct the persisted binding snapshot.
@classmethod
def from_snapshot(cls, snapshot: object) -> LiveExecutionBinding:
if not isinstance(snapshot, dict) or set(snapshot) != _SNAPSHOT_FIELDS:
@@ -95,7 +97,10 @@ class LiveExecutionBinding:
):
raise ValueError("LIVE_BINDING_SNAPSHOT_INVALID")
return binding
# #endregion ScenarioExecution.LiveBinding.Own.from_snapshot
# #region ScenarioExecution.LiveBinding.Own.snapshot [C:3] [TYPE Function]
# @BRIEF Serialize only persisted binding identity.
def snapshot(self) -> dict[str, Any]:
return {
"binding_ref": self.binding_ref,
@@ -111,9 +116,13 @@ class LiveExecutionBinding:
"evidence_owner_type": self.evidence_owner_type,
"evidence_ref_policy": self.evidence_ref_policy,
}
# #endregion ScenarioExecution.LiveBinding.Own.snapshot
# #region ScenarioExecution.LiveBinding.Own.with_query_model_fingerprint [C:1] [TYPE Function]
# @BRIEF Return the binding with the supplied query model fingerprint.
def with_query_model_fingerprint(self, fingerprint: str) -> LiveExecutionBinding:
return LiveExecutionBinding(**{**self.snapshot(), "query_model_fingerprint": fingerprint})
# #endregion ScenarioExecution.LiveBinding.Own.with_query_model_fingerprint
# #endregion ScenarioExecution.LiveBinding.Identity
@@ -122,9 +131,14 @@ class LiveExecutionBinding:
# @DATA_CONTRACT LiveExecutionBindingResolver(binding_ref) -> ResolvedLiveExecutionBinding | None.
# @INVARIANT Resolver-owned clients, storage, and event-loop access never enter the persisted snapshot.
class EvidenceStorage(Protocol):
# #region ScenarioExecution.LiveBinding.Own.store [C:1] [TYPE Function]
# @BRIEF Define the evidence-byte persistence interface.
def store(self, run_id: str, sha256: str, data: bytes) -> str: ...
# #endregion ScenarioExecution.LiveBinding.Own.store
# #region ScenarioExecution.LiveBinding.Own.ResolvedLiveExecutionBinding [C:3] [TYPE Class]
# @BRIEF Keep resolved live clients, approved query model and evidence storage together.
@dataclass(frozen=True)
class ResolvedLiveExecutionBinding:
binding: LiveExecutionBinding
@@ -132,12 +146,19 @@ class ResolvedLiveExecutionBinding:
query_model: DashboardQueryModel
evidence_storage: EvidenceStorage
run_async: Callable[[Awaitable[Any]], Any]
# #endregion ScenarioExecution.LiveBinding.Own.ResolvedLiveExecutionBinding
# #region ScenarioExecution.LiveBinding.Own.LiveExecutionBindingResolver [C:1] [TYPE Class]
# @BRIEF Define the composition-owned binding-reference lookup protocol.
class LiveExecutionBindingResolver(Protocol):
"""Composition-owned mapping from persisted identity to authorized runtime dependencies."""
# #region ScenarioExecution.LiveBinding.Own.resolve [C:1] [TYPE Function]
# @BRIEF Look up the exact binding reference.
def resolve(self, binding_ref: str) -> ResolvedLiveExecutionBinding | None: ...
# #endregion ScenarioExecution.LiveBinding.Own.resolve
# #endregion ScenarioExecution.LiveBinding.Own.LiveExecutionBindingResolver
# #endregion ScenarioExecution.LiveBinding.Runtime
@@ -236,18 +257,30 @@ def _evidence_result(
raw_bytes = envelope.raw_response_content
if not is_valid_sha256(digest) or sha256(raw_bytes).hexdigest() != digest:
return LiveAdapterResult(status="inconclusive", reason_code="SUPERSET_EVIDENCE_INVALID")
diagnostic = getattr(envelope, "diagnostic", None)
if diagnostic is not None:
from src.services.dashboard_testing.chart_health_contract import ChartDiagnostic
try:
diagnostic = ChartDiagnostic.model_validate(diagnostic).model_dump()
except ValueError:
return LiveAdapterResult(status="inconclusive", reason_code="SUPERSET_DIAGNOSTIC_INVALID")
run_id = step["scenario_run_id"]
content_ref = resolved.evidence_storage.store(run_id, digest, raw_bytes)
expected_ref = f"draft:{run_id}:{digest}"
if content_ref != expected_ref:
return LiveAdapterResult(status="inconclusive", reason_code="SUPERSET_EVIDENCE_REF_INVALID")
failed = envelope.normalized_value.kind == ValueKind.UNKNOWN
return LiveAdapterResult(
status="passed",
reason_code="SUPERSET_QUERY_EXECUTED",
status=("failed" if (diagnostic or {}).get("status") == "failed" else "inconclusive") if failed else "passed",
reason_code=(diagnostic or {}).get("reason_code", "SUPERSET_QUERY_FAILED") if failed else "SUPERSET_QUERY_EXECUTED",
details={
"actual": envelope.normalized_value.model_dump(mode="json"),
"source_response_hash": digest,
"sha256": digest,
"payload_kind": getattr(envelope, "payload_kind", "original_response"),
"artifact_content_types": {content_ref: "application/json"},
"artifact_byte_lengths": {content_ref: len(raw_bytes)},
**({"chart_diagnostics": [diagnostic]} if diagnostic else {}),
},
output_refs=[content_ref],
artifact_refs=[content_ref],
@@ -325,9 +358,6 @@ def _execute_bound_superset(
except Exception:
_finalize_superset_receipt(operation_id, "failed", "unknown", summary={"phase": "execution_error"})
return LiveAdapterResult(status="inconclusive", reason_code="SUPERSET_QUERY_EXECUTION_ERROR")
if envelope.normalized_value.kind == ValueKind.UNKNOWN:
_finalize_superset_receipt(operation_id, "failed", "none", summary={"phase": "query_failed"})
return LiveAdapterResult(status="failed", reason_code="SUPERSET_QUERY_FAILED")
evidence = _evidence_result(step, resolved, envelope)
if evidence.status == "passed":
_finalize_superset_receipt(
@@ -335,7 +365,7 @@ def _execute_bound_superset(
summary={"sha256": evidence.details.get("sha256"), "artifact_refs": list(evidence.artifact_refs or [])},
)
else:
_finalize_superset_receipt(operation_id, "failed", "none", summary={"phase": evidence.reason_code})
_finalize_superset_receipt(operation_id, "failed", "none", summary={"phase": evidence.reason_code,"sha256":evidence.details.get("sha256"),"artifact_refs":list(evidence.artifact_refs or [])})
return evidence
# #endregion ScenarioExecution.LiveBinding.Execute
@@ -359,10 +389,11 @@ def superset_adapter_from(resolver: LiveExecutionBindingResolver):
# @REJECTED Executing caller-provided SQL directly was rejected — 044 evidence must use the pinned
# query model, principal and RLS fingerprints already authorized by the 037 binding.
def sql_evidence_adapter_from(resolver: LiveExecutionBindingResolver):
# #region ScenarioExecution.LiveBinding.Own.execute [C:1] [TYPE Function]
# @BRIEF Invoke the composed live binding executor.
def execute(step: dict[str, Any], completed: dict[str, dict[str, Any]]) -> LiveAdapterResult:
return _execute_bound_superset(resolver, step, completed)
# #endregion ScenarioExecution.LiveBinding.Own.execute
return execute
# #endregion ScenarioExecution.LiveBinding.SqlEvidenceAdapter
# #endregion ScenarioExecution.LiveBinding

View File

@@ -38,7 +38,7 @@ _READ_ONLY_ACTIONS = frozenset({
"navigate_tab", "inspect_filter_state", "apply_table_filter", "extract_table",
"scroll_to", "inspect_columns", "click", "select_rows", "download",
# Wave-2 observe drivers (038.5.0, AGSCN-FR-024):
"assert_dom", "inspect_filter_options", "navigate_tabs", "pagination", "wait_for_selector",
"assert_dom", "assert_chart_health", "inspect_filter_options", "navigate_tabs", "pagination", "wait_for_selector",
})
_MUTATION_ACTIONS = frozenset({"row_edit", "bulk_edit"})
# Round-2 read-only actions that carry typed inputs; validated at admission before any I/O.

View File

@@ -0,0 +1,42 @@
# #region ScenarioExecution.ChartHealth.Async [C:3] [TYPE Module] [SEMANTICS async,job,polling,attribution]
# @BRIEF Associate observed Superset5 async polling events only with already inspected chart-query jobs.
# @RATIONALE Superset5 asyncEvent.ts exposes job_id/status/errors/result_url; none are inferred from URL text.
from urllib.parse import urlsplit
# #region ScenarioExecution.ChartHealth.Async.RecordJobs [C:3] [TYPE Function]
# @POST Only a current chart-data202 response establishes job ownership; all recorded jobs remain bounded.
def record_jobs(collector, payload, attribution):
result = payload.get('result',[]) if isinstance(payload,dict) else []
records = result if isinstance(result,list) else [result]
for record in records:
if isinstance(record,dict) and isinstance(record.get('job_id'),str):
collector.jobs[record['job_id']] = attribution
while len(collector.jobs) > 128:
collector.jobs.pop(next(iter(collector.jobs)))
# #endregion ScenarioExecution.ChartHealth.Async.RecordJobs
# #region ScenarioExecution.ChartHealth.Async.Events [C:3] [TYPE Function]
# @POST Foreign/stale jobs cannot create errors or authorize unrelated response bodies.
def current_events(collector, payload):
result = payload.get('result',[]) if isinstance(payload,dict) else []
if not isinstance(result,list):
return []
matches = []
for event in result:
if not isinstance(event,dict) or event.get('job_id') not in collector.jobs:
continue
attribution = collector.jobs[event['job_id']]
if collector.latest.get(attribution[0]) != attribution[1]:
continue
matches.append((event,attribution))
if event.get('status') == 'done' and isinstance(event.get('result_url'),str):
url = urlsplit(event['result_url'])
if not url.netloc:
collector.result_paths[url.path] = attribution
if len(collector.result_paths) > 128:
collector.result_paths.pop(next(iter(collector.result_paths)))
return matches
# #endregion ScenarioExecution.ChartHealth.Async.Events
# #endregion ScenarioExecution.ChartHealth.Async

View File

@@ -0,0 +1,119 @@
# #region ScenarioExecution.ChartHealth.Observer [C:5] [TYPE Module] [SEMANTICS chart,health,errors,readiness,scope]
# @BRIEF Observe exact chart placements, checking current error cards before waiting for rendered data.
# @INVARIANT A visible attributed error is FAILED; missing/ambiguous/unsettled content stays INCONCLUSIVE.
import asyncio
from datetime import UTC, datetime
from time import monotonic
from playwright.async_api import TimeoutError as PlaywrightTimeoutError
from ...chart_health_contract import ChartDiagnostic
from ...query_failure import safe_message
from .browser_health_scope import resolve_health_scope
_STATE = '''el => ({error: Array.from(el.querySelectorAll('.alert-danger,.ant-alert-error,.chart-error')).map(x => x.textContent).filter(Boolean).join('\\n'),
loading: !!el.querySelector('.loading,.ant-spin-spinning,[aria-busy="true"],.chart-loading'),
ready: !!el.querySelector('table,svg,canvas,.big_number,.big_number_total,.header-line,[data-test="no-results"],.no-results')})'''
# #region ScenarioExecution.ChartHealth.Observer.Diagnostic [C:3] [TYPE Function]
# @POST Runtime attribution and redacted error text form a closed diagnostic, never a metric value.
def diagnostic(placement, status, reason, *, start, message='', origin='dom', exchange=None):
text, truncated = safe_message(message)
exchange = dict(exchange or {})
truncated = truncated or exchange.pop("truncated",False)
import re
code = re.search(r'\bCode:\s*(\d+)\b', text)
return ChartDiagnostic(chart_id=placement['chart_id'], placement_id=placement['id'],
chart_name=placement['name'], tab_path=placement['tab_path'], status=status,
reason_code=reason, origin=origin, observed_at=datetime.now(UTC).isoformat(),
duration_seconds=max(0.0,monotonic()-start), message=text, truncated=truncated,
database_code=code.group(1) if code else None,
terminal_state={'healthy':'ready','failed':'error','inconclusive':'unavailable'}[status],
**exchange).model_dump()
# #endregion ScenarioExecution.ChartHealth.Observer.Diagnostic
# #region ScenarioExecution.ChartHealth.Observer.RPCBudget [C:2] [TYPE Function]
# @POST No RPC starts after deadline;250ms of the existing per-chart budget is reserved for SDK result/task drainage.
def chart_rpc_timeout(end):
remaining = end-monotonic()
if remaining <= 0.25:
raise TimeoutError('CHART_LOAD_TIMEOUT')
return (remaining-0.25)*1000
# #endregion ScenarioExecution.ChartHealth.Observer.RPCBudget
# #region ScenarioExecution.ChartHealth.Observer.Exchange [C:3] [TYPE Function]
# @POST Confirmed attributed API failure does not require successful ChartRenderer DOM; transport failure remains unresolved.
def exchange_diagnostic(placement, exchange, start):
confirmed = exchange.get('confirmed',True)
return diagnostic(placement,'failed' if confirmed else 'inconclusive',
'CHART_QUERY_FAILED' if confirmed else 'CHART_REQUEST_UNAVAILABLE',start=start,
message=exchange['message'],origin='http_response' if confirmed else 'synthetic_exception',exchange=exchange['attribution'])
# #endregion ScenarioExecution.ChartHealth.Observer.Exchange
# #region ScenarioExecution.ChartHealth.Observer.Wait [C:3] [TYPE Function]
# @POST Current attributed errors are checked during mount; bounded SDK waits finish before the enclosing chart deadline.
async def wait_chart_or_error(panel, placement, collector, journal, end):
while True:
reason = journal.check_control()
if reason:
raise ValueError(reason)
timeout = chart_rpc_timeout(end)
exchange = collector.current_error(placement['chart_id'])
if exchange:
return None,exchange
chart,_ = await resolve_health_scope(panel,placement)
try:
await chart.wait_for(state='visible',timeout=min(100,timeout))
return chart,None
except PlaywrightTimeoutError:
continue
# #endregion ScenarioExecution.ChartHealth.Observer.Wait
# #region ScenarioExecution.ChartHealth.Observer.Observe [C:4] [TYPE Function]
# @POST Query errors are detected before readiness timeout; zero/empty rendered charts are healthy.
async def observe_chart(page, panel, placement, timeout_seconds, collector, journal):
start, end = monotonic(), monotonic()+timeout_seconds
try:
chart,exchange = await wait_chart_or_error(panel,placement,collector,journal,end)
if exchange:
return exchange_diagnostic(placement,exchange,start)
if await chart.count() != 1:
return diagnostic(placement,'inconclusive','CHART_PLACEMENT_AMBIGUOUS',start=start)
await chart.scroll_into_view_if_needed(timeout=chart_rpc_timeout(end))
return await observe_mounted(chart,placement,collector,journal,start,end)
except (TimeoutError,PlaywrightTimeoutError):
return diagnostic(placement,'inconclusive','CHART_LOAD_TIMEOUT',start=start)
except ValueError as exc:
if str(exc).startswith('CHART_PLACEMENT_'):
return diagnostic(placement,'inconclusive',str(exc),start=start)
raise
except asyncio.CancelledError:
raise
except Exception:
return diagnostic(placement,'inconclusive','CHART_OBSERVATION_UNAVAILABLE',start=start)
# #endregion ScenarioExecution.ChartHealth.Observer.Observe
# #region ScenarioExecution.ChartHealth.Observer.Mounted [C:3] [TYPE Function]
# @POST Mounted DOM errors precede ready content; pending generations cannot use stale render success.
async def observe_mounted(chart, placement, collector, journal, start, end):
while monotonic() < end:
reason = journal.check_control()
if reason:
raise ValueError(reason)
state = await chart.evaluate(_STATE,timeout=chart_rpc_timeout(end))
exchange = collector.current_error(placement['chart_id'])
pending = collector.latest.get(placement['chart_id']) in collector.pending
if state['error'] and not pending:
return diagnostic(placement,'failed','CHART_QUERY_FAILED',start=start,message=state['error'])
if exchange:
return exchange_diagnostic(placement,exchange,start)
if state['ready'] and not state['loading'] and not pending:
return diagnostic(placement,'healthy','CHART_READY',start=start)
await asyncio.sleep(0.1)
return diagnostic(placement,'inconclusive','CHART_LOAD_TIMEOUT',start=start)
# #endregion ScenarioExecution.ChartHealth.Observer.Mounted
# #endregion ScenarioExecution.ChartHealth.Observer

View File

@@ -0,0 +1,146 @@
# #region ScenarioExecution.ChartHealth.Responses [C:5] [TYPE Module] [SEMANTICS http,chart,generation,filters,redaction]
# @BRIEF Bound chart-data observations to the newest exact saved-chart request; stale and foreign failures cannot decide health.
import asyncio
from hashlib import sha256
import json
from ...query_failure import response_error, safe_message
from .browser_chart_async import record_jobs, current_events
from urllib.parse import urlsplit
# #region ScenarioExecution.ChartHealth.Responses.Collector [C:4] [TYPE Class]
# @INVARIANT Only exact chart/data requests with saved slice identity establish attribution; buffers/listeners are bounded and owned.
class ChartResponseCollector:
# #region ScenarioExecution.ChartHealth.Responses.Collector.Init [C:2] [TYPE Function]
# @POST No listeners are attached until the composite action owns their lifetime.
def __init__(self, page, chart_ids):
self.page, self.chart_ids = page, set(chart_ids)
self.latest, self.requests, self.errors, self.tasks = {}, {}, {}, set()
self.generation = 0
self.jobs, self.result_paths, self.pending = {}, {}, set()
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.Init
# #region ScenarioExecution.ChartHealth.Responses.Collector.Request [C:3] [TYPE Function]
# @POST Newest exact saved-chart request replaces prior errors; query/filter content is fingerprinted, not retained.
def on_request(self, request):
path = urlsplit(request.url).path
if request.method == 'GET' and path in self.result_paths:
self.requests[request] = self.result_paths[path]
return
if path.rstrip('/') != '/api/v1/chart/data' or request.method != 'POST':
return
try:
payload = request.post_data_json
chart_id = (payload.get('form_data') or {}).get('slice_id')
if type(chart_id) is not int or chart_id not in self.chart_ids:
return
self.generation += 1
identity = str(self.generation)
scope = {'datasource':payload.get('datasource'),'queries':payload.get('queries')}
fingerprint = sha256(json.dumps(scope,sort_keys=True,separators=(',',':')).encode()).hexdigest()
self.pending.discard(self.latest.get(chart_id))
self.latest[chart_id] = identity
self.pending.add(identity)
self.errors.pop(chart_id,None)
self.requests[request] = (chart_id,identity,fingerprint)
if len(self.requests) > 128:
self.requests.pop(next(iter(self.requests)))
except (ValueError, TypeError, AttributeError):
return
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.Request
# #region ScenarioExecution.ChartHealth.Responses.Collector.RequestFailed [C:3] [TYPE Function]
# @POST A current owned transport failure is unresolved evidence, never a confirmed database error.
def on_request_failed(self, request):
attribution = self.requests.get(request)
if attribution is None or self.latest.get(attribution[0]) != attribution[1]:
return
message,truncated = safe_message(request.failure or 'Chart request transport unavailable')
chart_id,identity,fingerprint = attribution
self.errors[chart_id] = {'message':message,'confirmed':False,'attribution':{
'request_identity':identity,'load_generation':int(identity),'filter_fingerprint':fingerprint,
'truncated':truncated}}
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.RequestFailed
# #region ScenarioExecution.ChartHealth.Responses.Collector.Response [C:2] [TYPE Function]
# @POST At most128 pending body reads can exist; missing body evidence cannot create failure.
def on_response(self, response):
polling = urlsplit(response.url).path.rstrip('/') == '/api/v1/async_event'
if (response.request not in self.requests and not polling) or len(self.tasks) >= 128:
return
task = asyncio.create_task(self.read_response(response))
self.tasks.add(task)
task.add_done_callback(self.tasks.discard)
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.Response
# #region ScenarioExecution.ChartHealth.Responses.Collector.Read [C:4] [TYPE Function]
# @POST HTTP200 query errors and non2xx payloads decide only the current exact request generation.
async def read_response(self, response):
try:
raw = await asyncio.wait_for(response.body(),timeout=2)
if len(raw) > 1048576:
return
payload = json.loads(raw)
if response.request not in self.requests:
for event, attribution in current_events(self,payload):
self.record_error(attribution,response_error(event),raw,response.status)
return
attribution = self.requests[response.request]
if response.status == 202:
record_jobs(self,payload,attribution)
return
error = response_error(payload,response.status)
if error is not None and error.get('_unconfirmed'):
return
self.pending.discard(attribution[1])
self.record_error(attribution,error,raw,response.status)
except Exception:
return
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.Read
# #region ScenarioExecution.ChartHealth.Responses.Collector.Record [C:3] [TYPE Function]
# @POST Only the newest attributed response can publish a sanitized query failure.
def record_error(self, attribution, error, raw, http_status):
chart_id, identity, fingerprint = attribution
if error is None or error.get('_unconfirmed') or self.latest.get(chart_id) != identity:
return
self.pending.discard(identity)
message, truncated = safe_message(error.get('message') or error.get('error'))
self.errors[chart_id] = {'message':message,'attribution':{
'http_status':http_status,'request_identity':identity,'load_generation':int(identity),
'filter_fingerprint':fingerprint,'original_response_sha256':sha256(raw).hexdigest(),
'original_byte_length':len(raw),'truncated':truncated,'application_code':str(error.get('error_type') or '')[:256] or None}}
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.Record
# #region ScenarioExecution.ChartHealth.Responses.Collector.Current [C:1] [TYPE Function]
# @BRIEF Return only current owned response attribution, never caller error state.
def current_error(self, chart_id):
return self.errors.get(chart_id)
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.Current
# #region ScenarioExecution.ChartHealth.Responses.Collector.Start [C:1] [TYPE Function]
# @POST Request and response listeners precede tab activation/load.
def start(self):
self.page.on('request',self.on_request)
self.page.on('response',self.on_response)
self.page.on('requestfailed',self.on_request_failed)
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.Start
# #region ScenarioExecution.ChartHealth.Responses.Collector.Close [C:2] [TYPE Function]
# @POST Cancellation/completion removes listeners and drains all owned pending tasks.
async def close(self):
self.page.remove_listener('request',self.on_request)
self.page.remove_listener('response',self.on_response)
self.page.remove_listener('requestfailed',self.on_request_failed)
tasks = list(self.tasks)
for task in tasks:
task.cancel()
if tasks:
await asyncio.gather(*tasks,return_exceptions=True)
self.requests.clear()
self.pending.clear()
self.jobs.clear()
self.result_paths.clear()
# #endregion ScenarioExecution.ChartHealth.Responses.Collector.Close
# #endregion ScenarioExecution.ChartHealth.Responses.Collector
# #endregion ScenarioExecution.ChartHealth.Responses

View File

@@ -0,0 +1,36 @@
# #region ScenarioExecution.ChartHealth.Capture [C:4] [TYPE Module] [SEMANTICS tabs,frames,coverage,owned]
# @BRIEF Capture exact visible chart placements as finite frames; long/uncaptured tabs retain explicit omissions.
from ..chart_health_tab_store import store_health_artifact
from ..mime_sniff import sniff_mime
from .browser_health_scope import resolve_health_scope
# #region ScenarioExecution.ChartHealth.Capture.Frames [C:4] [TYPE Function]
# @POST Only actual PNG bytes from exact current-tab chart placements are retained; omitted frames never count as visually covered.
async def capture_frames(page, panel, placements, journal, limit, timeout_seconds):
frames, omitted, total = [],[],0
for placement in placements:
reason = journal.check_control()
if reason:
raise ValueError(reason)
if len(frames) >= limit:
omitted.append({'placement_id':placement['id'],'reason':'FRAME_BUDGET'})
continue
try:
_,chart = await resolve_health_scope(panel,placement)
if await chart.count() != 1:
raise ValueError('CHART_PLACEMENT_AMBIGUOUS')
await chart.scroll_into_view_if_needed(timeout=timeout_seconds*1000)
data = await chart.screenshot(type='png',timeout=timeout_seconds*1000)
if sniff_mime(data) != 'image/png' or len(data)+total > 10485760:
raise ValueError('FRAME_BYTE_BUDGET')
receipt = store_health_artifact(journal,data,f'chart-health-frame-{placement["id"]}.png','image/png')
frames.append({'placement_id':placement['id'],'receipt':receipt})
total += len(data)
except ValueError as exc:
omitted.append({'placement_id':placement['id'],'reason':str(exc)})
except Exception:
omitted.append({'placement_id':placement['id'],'reason':'FRAME_UNAVAILABLE'})
return frames,omitted
# #endregion ScenarioExecution.ChartHealth.Capture.Frames
# #endregion ScenarioExecution.ChartHealth.Capture

View File

@@ -0,0 +1,102 @@
# #region ScenarioExecution.ChartHealth.TabEvaluation [C:5] [TYPE Module] [SEMANTICS evaluation,tabs,owned,diagnostics,multimodal]
# @BRIEF Evaluate opt-in tab frames plus actual owned diagnostics before leaving the panel; model output never erases a query failure.
import asyncio
from datetime import UTC, datetime
import json
from src.core.database import SessionLocal
from ..agent_evaluation import submit_evaluation, parse_evaluation_response
from ..evaluation_adapter import _normalize_provider_response, _stamp_provenance, _persist_raw_response
from ..evaluation_prompt import build_evaluation_prompt
from ..evaluation_health_payloads import load_health_payloads
from ..chart_health_tab_store import tab_diagnostics, store_health_artifact, append_tab_evaluation
from ..chart_health_artifact import verify_health_artifact
from ...query_failure import safe_message
from .browser_health_capture import capture_frames
from .browser_health_provider import admit_health_provider
from .browser_health_llm_client import BoundedHealthClient
# #region ScenarioExecution.ChartHealth.TabEvaluation.Selected [C:2] [TYPE Function]
# @POST Policy selection is explicit; confirmed problems and selected stable tab IDs determine opt-in work.
def should_evaluate(policy, tab_path, diagnostics):
if policy.mode == 'selected':
return bool(tab_path and tab_path[-1] in policy.tab_ids)
return policy.mode == 'all_visited' or any(item['diagnostic']['status'] != 'healthy' for item in diagnostics)
# #endregion ScenarioExecution.ChartHealth.TabEvaluation.Selected
# #region ScenarioExecution.ChartHealth.TabEvaluation.Inputs [C:3] [TYPE Function]
# @POST Actual diagnostic contents and exact owned image bytes enter the prompt, within finite text/frame budgets.
def evaluation_inputs(journal, diagnostics, frames):
artifacts = [item['receipt'] for item in [*diagnostics,*frames]]
manifest = [{'artifact_id':item['content_ref'],'sha256':item['sha256'],'content_type':item['content_type'],
'byte_length':item['byte_length'],'role':'context'} for item in artifacts]
with SessionLocal() as db:
for receipt in artifacts:
verify_health_artifact(journal,db,receipt)
payloads = load_health_payloads(manifest,journal.storage,db=db,run_id=journal.run_id)
images = [(journal.storage.retrieve(item['receipt']['content_ref']),'image/png') for item in frames]
return manifest,payloads,images,artifacts
# #endregion ScenarioExecution.ChartHealth.TabEvaluation.Inputs
# #region ScenarioExecution.ChartHealth.TabEvaluation.Submit [C:4] [TYPE Function]
# @POST The existing capacity/credentials/provider boundary uses only the pinned opt-in specification; no paid call is made by tests.
async def evaluate_tab(page, panel, tab_path, placements, journal, policy, timeout_seconds):
diagnostics = tab_diagnostics(journal,tab_path)
if not should_evaluate(policy,tab_path,diagnostics):
return
existing = journal.frontier().get('tab_evaluations',[])
if any(item['tab_path'] == tab_path for item in existing):
return
if len(existing) >= policy.max_tabs:
append_tab_evaluation(journal,{'tab_path':tab_path,'status':'inconclusive','reason_code':'VLM_TAB_BUDGET',
'coverage_complete':False,'artifacts':[],'evaluated_at':datetime.now(UTC).isoformat()})
return
frames, omitted = await capture_frames(page,panel,placements,journal,
min(policy.max_frames_per_tab,policy.spec.limits.max_images),timeout_seconds)
manifest,payloads,images,artifacts = evaluation_inputs(journal,diagnostics,frames)
summary = {'tab_path':tab_path,'status':'inconclusive','reason_code':'VLM_EVIDENCE_INCOMPLETE',
'coverage_complete':not omitted and not payloads['truncated'],'omitted_frames':omitted,'artifacts':artifacts,
'evaluated_at':datetime.now(UTC).isoformat()}
payloads['frames'] = [{'placement_id':item['placement_id'],'artifact_id':item['receipt']['content_ref'],'image_index':index} for index,item in enumerate(frames)]
prompt = build_evaluation_prompt(policy.spec,manifest,evidence_payloads=payloads)
if not images:
append_tab_evaluation(journal,{**summary,'reason_code':'VLM_IMAGE_EVIDENCE_UNAVAILABLE'})
return
try:
with SessionLocal() as db:
provider = admit_health_provider(db,policy)
if len(prompt.encode())+512 > policy.spec.limits.max_input_tokens:
raise ValueError("CHART_HEALTH_VLM_TEXT_BUDGET")
step = journal.step
target = step.get('target_snapshot') or {}
binding = step.get('live_execution_binding_snapshot') or {}
try:
raw = await asyncio.wait_for(submit_evaluation(db,spec=policy.spec,prompt=prompt,images=images,
environment_id=target.get('environment_id') or binding.get('environment_id'),
environment_class=target.get('environment_class','DEV'),run_id=journal.run_id,
logical_step_id=journal.logical_step_id,runtime_step=step,client=BoundedHealthClient(db,provider,policy.spec)),timeout=min(timeout_seconds,policy.spec.limits.timeout_ms/1000))
finally:
db.commit()
raw_bytes,digest,ref = _persist_raw_response(journal.storage,journal.run_id,raw)
record = parse_evaluation_response(_stamp_provenance(_normalize_provider_response(raw,policy.spec),
run_id=journal.run_id,logical_step_id=journal.logical_step_id,attempt=journal.attempt,
manifest=manifest,content_ref=ref,digest=digest,spec=policy.spec),spec=policy.spec)
for finding in record.findings:
finding.message = safe_message(finding.message)[0][:2000]
raw_receipt = store_health_artifact(journal,raw_bytes,'chart-health-vlm-response.json')
record_receipt = store_health_artifact(journal,json.dumps(record.model_dump(mode='json'),sort_keys=True).encode(),'chart-health-vlm-record.json')
summary.update(status='succeeded' if summary['coverage_complete'] and record.status == 'succeeded' else 'inconclusive',
reason_code='VLM_TAB_EVALUATED' if summary['coverage_complete'] else 'VLM_EVIDENCE_INCOMPLETE',
advisory_verdict=record.verdict, confidence=record.confidence,findings=[item.model_dump(mode='json') for item in record.findings],
artifacts=[*artifacts,raw_receipt,record_receipt])
except asyncio.CancelledError:
append_tab_evaluation(journal,{**summary,'reason_code':'VLM_INTERRUPTED'})
raise
except Exception as exc:
code = str(exc)
summary['reason_code'] = code[:128] if code.startswith(('CHART_HEALTH_','EVALUATION_')) else 'VLM_PROVIDER_UNAVAILABLE'
append_tab_evaluation(journal,summary)
# #endregion ScenarioExecution.ChartHealth.TabEvaluation.Submit
# #endregion ScenarioExecution.ChartHealth.TabEvaluation

View File

@@ -0,0 +1,29 @@
# #region ScenarioExecution.ChartHealth.EvaluationInputs [C:4] [TYPE Module] [SEMANTICS evaluation,opt-in,tabs,limits]
# @BRIEF Explicit per-tab opt-in pins the existing provider specification and finite capture/request budgets.
from typing import Literal
from pydantic import BaseModel, ConfigDict, Field, model_validator
from ...scenario.evaluation_models import AgentEvaluationSpec, TokenEvaluationLimits
# #region ScenarioExecution.ChartHealth.EvaluationInputs.Policy [C:3] [TYPE Class]
# @INVARIANT Semantic diagnostic evaluation requires no fake metric comparison or token-only recipe authority.
class HealthEvaluationPolicy(BaseModel):
model_config = ConfigDict(extra='forbid',strict=True)
mode: Literal['all_visited','selected','problem_areas']
tab_ids: list[str] = Field(default_factory=list,max_length=1000)
max_tabs: int = Field(default=100,ge=1,le=1000)
max_frames_per_tab: int = Field(default=8,ge=1,le=8)
spec: AgentEvaluationSpec
# #region ScenarioExecution.ChartHealth.EvaluationInputs.Policy.Validate [C:3] [TYPE Function]
# @POST Selection is explicit; comparison and token-only recipe authority cannot be smuggled into health evaluation.
@model_validator(mode='after')
def validate_policy(self):
if (self.mode == 'selected' and not self.tab_ids) or (self.mode != 'selected' and self.tab_ids):
raise ValueError('CHART_HEALTH_EVALUATION_TAB_SELECTION_INVALID')
if isinstance(self.spec.limits,TokenEvaluationLimits) or self.spec.comparison_refs or any(item.criterion_kind != 'semantic' for item in self.spec.criteria):
raise ValueError('CHART_HEALTH_EVALUATION_SEMANTIC_ONLY')
return self
# #endregion ScenarioExecution.ChartHealth.EvaluationInputs.Policy.Validate
# #endregion ScenarioExecution.ChartHealth.EvaluationInputs.Policy
# #endregion ScenarioExecution.ChartHealth.EvaluationInputs

View File

@@ -0,0 +1,45 @@
# #region ScenarioExecution.ChartHealth.BoundedClient [C:4] [TYPE Module] [SEMANTICS provider,request-budget,tokens,timeout]
# @BRIEF Bound the existing configured multimodal SDK client to one physical request and declared output/deadline limits.
# @REJECTED Legacy get_json_completion retries/fallbacks do not preserve a finite per-tab physical request budget.
import json
from src.plugins.llm_analysis.service import LLMClient
from src.plugins.llm_analysis.models import LLMProviderType
from src.services.llm_provider import LLMProviderService
# #region ScenarioExecution.ChartHealth.BoundedClient.Client [C:3] [TYPE Class]
class BoundedHealthClient:
# #region ScenarioExecution.ChartHealth.BoundedClient.Client.Init [C:1] [TYPE Function]
# @POST No credentials or sockets are accessed until the existing evaluation capacity boundary invokes the client.
def __init__(self, db, provider, spec):
self.db,self.provider,self.spec = db,provider,spec
# #endregion ScenarioExecution.ChartHealth.BoundedClient.Client.Init
# #region ScenarioExecution.ChartHealth.BoundedClient.Client.Complete [C:4] [TYPE Function]
# @POST One bounded SDK request; truncation/nonobject JSON refuses and transport usage replaces model billing claims.
async def get_json_completion(self, messages):
key = LLMProviderService(self.db).get_decrypted_api_key(self.provider.id)
if not key:
raise RuntimeError('EVALUATION_PROVIDER_MISSING')
self.db.commit()
client = LLMClient(LLMProviderType(self.provider.provider_type),key,self.provider.base_url,self.spec.model_id)
bounded = client.client.with_options(max_retries=0,timeout=self.spec.limits.timeout_ms/1000)
options = {'model':self.spec.model_id,'messages':messages,'max_tokens':self.spec.limits.max_output_tokens}
if self.provider.supports_json_object is not False:
options['response_format'] = {'type':'json_object'}
try:
response = await bounded.chat.completions.create(**options)
finally:
await bounded.close()
if not response.choices or response.choices[0].finish_reason == 'length':
raise RuntimeError('EVALUATION_RESPONSE_TRUNCATED')
raw = json.loads(response.choices[0].message.content)
if not isinstance(raw,dict):
raise RuntimeError('EVALUATION_RESPONSE_INVALID')
raw.pop('usage',None)
if response.usage is not None:
raw['usage'] = {'input_tokens':response.usage.prompt_tokens,'output_tokens':response.usage.completion_tokens}
return raw
# #endregion ScenarioExecution.ChartHealth.BoundedClient.Client.Complete
# #endregion ScenarioExecution.ChartHealth.BoundedClient.Client
# #endregion ScenarioExecution.ChartHealth.BoundedClient

View File

@@ -0,0 +1,64 @@
# #region ScenarioExecution.ChartHealth.Manifest [C:4] [TYPE Module] [SEMANTICS placement,tabs,server,identity]
# @BRIEF Inspect stable server chart placements and nested tab paths; duplicate titles never establish identity.
from hashlib import sha256
import json
from urllib.parse import urlsplit
from .browser_tabs_manifest import parse_tabs_manifest
from ...query_failure import safe_message
# #region ScenarioExecution.ChartHealth.Manifest.Parse [C:4] [TYPE Function]
# @POST Every reachable chart placement retains its own server node ID, name, chart ID and full tab path.
def parse_health_manifest(layout, chart_ids=None):
source = parse_tabs_manifest(layout) if any(isinstance(node,dict) and node.get("type") == "TAB" for node in layout.values()) else {"tabs":[],"source_total":0,"manifest_digest":sha256(json.dumps(layout,sort_keys=True,separators=(",",":")).encode()).hexdigest()}
placements, seen = [], set()
# #region ScenarioExecution.ChartHealth.Manifest.Parse.Walk [C:3] [TYPE Function]
# @POST Duplicate/cyclic/orphan chart nodes cannot be counted as covered.
def walk(node_id, path):
if node_id in seen or node_id not in layout:
raise ValueError('CHART_HEALTH_MANIFEST_INVALID')
seen.add(node_id)
node = layout[node_id]
if node.get('type') == 'TAB':
path = [*path,node_id]
if node.get('type') == 'CHART':
meta = node.get('meta') or {}
chart_id = meta.get('chartId')
if type(chart_id) is not int or chart_id <= 0:
raise ValueError('CHART_HEALTH_CHART_ID_INVALID')
if chart_ids is None or chart_id in chart_ids:
placements.append({'id':node_id,'chart_id':chart_id,
'name':safe_message(meta.get('sliceName') or f'Chart {chart_id}')[0][:512], 'tab_path':path})
for child in node.get('children',[]):
walk(child,path)
# #endregion ScenarioExecution.ChartHealth.Manifest.Parse.Walk
walk('ROOT_ID',[])
expected = {key for key,node in layout.items() if isinstance(node,dict) and node.get('type') == 'CHART'}
if not expected.issubset(seen) or not placements:
raise ValueError('CHART_HEALTH_MANIFEST_INCOMPLETE')
if chart_ids and {item['chart_id'] for item in placements} != set(chart_ids):
raise ValueError('CHART_HEALTH_TARGET_MISSING')
tab_order = {item['id']:index for index,item in enumerate(source['tabs'])}
placements.sort(key=lambda item:tab_order.get(item['tab_path'][-1],-1) if item['tab_path'] else -1)
source['tabs'] = [dict(item,name=safe_message((layout[item['id']].get('meta') or {}).get('text') or item['id'])[0][:512]) for item in source['tabs']]
source['placements'] = placements
source['placement_digest'] = sha256(json.dumps(placements,sort_keys=True,separators=(',',':')).encode()).hexdigest()
return source
# #endregion ScenarioExecution.ChartHealth.Manifest.Parse
# #region ScenarioExecution.ChartHealth.Manifest.Fetch [C:3] [TYPE Function]
# @POST Authenticated dashboard layout is the sole expected-placement authority.
async def fetch_health_manifest(page, dashboard_id, chart_ids=None):
url = urlsplit(page.url)
response = await page.request.get(f'{url.scheme}://{url.netloc}/api/v1/dashboard/{dashboard_id}')
raw = await response.body()
if response.status != 200 or len(raw) > 1048576:
raise ValueError('CHART_HEALTH_MANIFEST_UNAVAILABLE')
result = json.loads(raw).get('result') or {}
if result.get('id') != dashboard_id:
raise ValueError('CHART_HEALTH_DASHBOARD_MISMATCH')
layout = result.get('position_json')
return parse_health_manifest(json.loads(layout) if isinstance(layout,str) else layout, chart_ids)
# #endregion ScenarioExecution.ChartHealth.Manifest.Fetch
# #endregion ScenarioExecution.ChartHealth.Manifest

View File

@@ -0,0 +1,21 @@
# #region ScenarioExecution.ChartHealth.Provider [C:4] [TYPE Module] [SEMANTICS provider,configuration,pin,multimodal]
# @BRIEF Verify the public provider/model configuration before any per-tab credentials or capacity are accessed.
from src.models.llm import LLMProvider
from ...scenario.metric_evaluation_provider import provider_public_config_digest
# #region ScenarioExecution.ChartHealth.Provider.Admit [C:3] [TYPE Function]
# @POST Changed/nonvisual/inactive provider configuration refuses before paid requests; declared budgets remain pinned.
def admit_health_provider(db, policy):
provider = db.get(LLMProvider,policy.spec.provider_id)
if provider is None or provider.is_active is not True or provider.is_multimodal is not True:
raise ValueError('CHART_HEALTH_VLM_PROVIDER_UNAVAILABLE')
digest = provider_public_config_digest(provider)
if (policy.spec.provider_version != f'config_sha256:{digest}' or policy.spec.model_id != provider.default_model
or policy.spec.model_version != f'configured-model:{provider.default_model}'):
raise ValueError('CHART_HEALTH_VLM_PROVIDER_CHANGED')
if provider.max_images is not None and policy.spec.limits.max_images > provider.max_images:
raise ValueError("CHART_HEALTH_VLM_IMAGE_BUDGET")
return provider
# #endregion ScenarioExecution.ChartHealth.Provider.Admit
# #endregion ScenarioExecution.ChartHealth.Provider

View File

@@ -0,0 +1,29 @@
# #region ScenarioExecution.ChartHealth.Scope [C:4] [TYPE Module] [SEMANTICS native,saved-chart,body,frame,identity]
# @BRIEF Resolve a saved chart's native Superset5 card/body even when failure removes its ChartRenderer ID.
# @RATIONALE Pinned5.0.0 gridComponents/Chart.jsx exposes chart-grid-component/data-test-chart-id outside both render branches.
# @REJECTED Requiring ChartRenderer's chart-id on a failed native card hid confirmed errors; selecting title/header icons would create false readiness.
# @INVARIANT Header/menu icons are excluded from readiness; duplicate/foreign native identities refuse without first/global fallback.
from ...query_failure import safe_message
# #region ScenarioExecution.ChartHealth.Scope.Resolve [C:3] [TYPE Function]
# @POST Native frames retain card context while health reads only its exact chart body; legacy protocol IDs remain supported.
async def resolve_health_scope(panel, placement):
grid = panel.locator(f'[data-test="chart-grid-component"][data-test-chart-id="{placement["chart_id"]}"]')
count = await grid.count()
if count > 1:
raise ValueError('CHART_PLACEMENT_AMBIGUOUS')
if count == 1:
name = await grid.get_attribute('data-test-chart-name')
if name is not None and safe_message(name)[0][:512] != placement['name']:
raise ValueError('CHART_PLACEMENT_IDENTITY_MISMATCH')
body = grid.locator('.dashboard-chart')
if await body.count() > 1:
raise ValueError('CHART_PLACEMENT_AMBIGUOUS')
return body, grid
chart = panel.locator(f'#chart-id-{placement["chart_id"]}')
if await chart.count() > 1:
raise ValueError('CHART_PLACEMENT_AMBIGUOUS')
return chart, chart
# #endregion ScenarioExecution.ChartHealth.Scope.Resolve
# #endregion ScenarioExecution.ChartHealth.Scope

View File

@@ -0,0 +1,155 @@
# #region ScenarioExecution.ChartHealth.Sweep [C:5] [TYPE Module] [SEMANTICS chart,tabs,composite,coverage,continuation]
# @BRIEF Continue across failed charts while committing per-placement health under cancellation and whole/per-chart budgets.
import asyncio
from time import monotonic
from .browser_all_tabs import activate_tab
from .browser_chart_health import observe_chart, diagnostic
from .browser_chart_responses import ChartResponseCollector
from .browser_health_manifest import fetch_health_manifest
# #region ScenarioExecution.ChartHealth.Sweep.Panel [C:3] [TYPE Function]
# @POST All inspected tab ancestors activate their exact panel; charts outside tabs retain the page scope.
async def activate_path(page, source, path, timeout):
panel = page
for tab_id in path:
tab = next(item for item in source['tabs'] if item['id'] == tab_id)
panel = await current_or_activate(page,tab,timeout)
return panel
# #endregion ScenarioExecution.ChartHealth.Sweep.Panel
# #region ScenarioExecution.ChartHealth.Sweep.CurrentPanel [C:3] [TYPE Function]
# @POST Reusing an already selected exact tab does not trigger another chart generation; identity/visibility remain checked.
async def current_or_activate(page, tab, timeout):
control = page.locator(f'[role="tab"][id="{tab["parent_tabs_id"]}-tab-{tab["id"]}"]')
if await control.count() == 1 and await control.get_attribute('aria-selected') == 'true':
panel_id = await control.get_attribute('aria-controls')
if not panel_id:
raise ValueError('BROWSER_TABS_PANEL_ID_MISSING')
panel = page.locator(f'[id="{panel_id}"][role="tabpanel"]')
await panel.wait_for(state='visible',timeout=timeout*1000)
if await panel.count() != 1:
raise ValueError('BROWSER_TABS_PANEL_AMBIGUOUS')
return panel
return await activate_tab(page,tab['id'],tab['parent_tabs_id'],timeout)
# #endregion ScenarioExecution.ChartHealth.Sweep.CurrentPanel
# #region ScenarioExecution.ChartHealth.Sweep.Controlled [C:3] [TYPE Function]
# @POST Expensive UI operations renew both leases and remain cancellable without discarding previous errors.
async def controlled(factory, journal, timeout):
task = asyncio.create_task(asyncio.wait_for(factory(),timeout=timeout))
try:
while not task.done():
reason = journal.check_control()
if reason:
raise ValueError(reason)
await asyncio.wait({task},timeout=1)
return await task
finally:
if not task.done():
task.cancel()
await asyncio.gather(task,return_exceptions=True)
# #endregion ScenarioExecution.ChartHealth.Sweep.Controlled
# #region ScenarioExecution.ChartHealth.Sweep.One [C:3] [TYPE Function]
# @POST An unavailable chart/tab is an explicit unresolved placement, while confirmed query errors are retained.
async def one_chart(page, source, placement, collector, journal, timeout):
start = monotonic()
try:
panel = await activate_path(page,source,placement['tab_path'],timeout)
return await observe_chart(page,panel,placement,max(0.01,timeout-(monotonic()-start)),collector,journal)
except ValueError as exc:
if str(exc).startswith('BROWSER_TRAVERSAL_'):
raise
return diagnostic(placement,'inconclusive',str(exc)[:128],start=start)
except TimeoutError:
return diagnostic(placement,'inconclusive','CHART_LOAD_TIMEOUT',start=start)
except asyncio.CancelledError:
raise
except Exception:
return diagnostic(placement,'inconclusive','CHART_OBSERVATION_UNAVAILABLE',start=start)
# #endregion ScenarioExecution.ChartHealth.Sweep.One
# #region ScenarioExecution.ChartHealth.Sweep.Evaluate [C:3] [TYPE Function]
# @POST Optional VLM failure remains a retained per-tab outcome and cannot suppress deterministic errors or later tabs.
async def evaluate_current_tab(page, source, path, journal, limits, end):
from .browser_health_evaluation import evaluate_tab
from ..chart_health_tab_store import append_tab_evaluation
remaining = min(end-monotonic(),journal.remaining_seconds())
if remaining <= 0:
raise ValueError('CHART_HEALTH_WHOLE_TIMEOUT')
panel = await activate_path(page,source,path,min(limits.per_chart_timeout_seconds,remaining))
placements = [item for item in source['placements'] if item['tab_path'] == path]
# #region ScenarioExecution.ChartHealth.Sweep.Evaluate.Call [C:1] [TYPE Function]
# @BRIEF Keep per-tab capture and configured provider work inside the original whole deadline.
async def evaluate():
return await evaluate_tab(page,panel,path,placements,journal,limits.per_tab_evaluation,remaining)
# #endregion ScenarioExecution.ChartHealth.Sweep.Evaluate.Call
try:
await controlled(evaluate,journal,remaining)
except TimeoutError:
if not any(item['tab_path'] == path for item in journal.frontier().get('tab_evaluations',[])):
append_tab_evaluation(journal,{'tab_path':path,'status':'inconclusive','reason_code':'VLM_TIMEOUT','coverage_complete':False,'artifacts':[]})
# #endregion ScenarioExecution.ChartHealth.Sweep.Evaluate
# #region ScenarioExecution.ChartHealth.Sweep.BoundedChart [C:3] [TYPE Function]
# @POST Per-chart timeout is retained and permits the next chart while remaining inside the original whole deadline.
async def bounded_chart(page, source, placement, collector, journal, timeout):
# #region ScenarioExecution.ChartHealth.Sweep.Run.Observe [C:1] [TYPE Function]
# @BRIEF Bind the exact next placement to the monitored observation operation.
async def observe():
return await one_chart(page,source,placement,collector,journal,timeout)
# #endregion ScenarioExecution.ChartHealth.Sweep.Run.Observe
try:
value = await controlled(observe,journal,timeout)
except TimeoutError:
value = diagnostic(placement,'inconclusive','CHART_LOAD_TIMEOUT',start=monotonic()-timeout)
return value
# #endregion ScenarioExecution.ChartHealth.Sweep.BoundedChart
# #region ScenarioExecution.ChartHealth.Sweep.Run [C:4] [TYPE Function]
# @POST FAILED dominates interrupted coverage; otherwise only complete conclusive observations permit PASS.
async def sweep_health(page, dashboard_id, journal, limits):
collector = None
end = monotonic()+min(limits.whole_timeout_seconds,journal.remaining_seconds())
try:
source = await fetch_health_manifest(page,dashboard_id,getattr(limits,'chart_ids',None))
journal.freeze_source(source)
policy = limits.per_tab_evaluation
if policy is not None and policy.mode == 'selected' and not set(policy.tab_ids).issubset({item['id'] for item in source['tabs']}):
raise ValueError('CHART_HEALTH_EVALUATION_TAB_MISSING')
collector = ChartResponseCollector(page,[item['chart_id'] for item in source['placements']])
collector.start()
while journal.frontier()['next_ordinal'] <= len(source['placements']):
remaining = min(end-monotonic(),journal.remaining_seconds())
if remaining <= 0:
raise ValueError('CHART_HEALTH_WHOLE_TIMEOUT')
placement = source['placements'][journal.frontier()['next_ordinal']-1]
timeout = min(limits.per_chart_timeout_seconds,remaining)
value = await bounded_chart(page,source,placement,collector,journal,timeout)
journal.append_chart(value)
next_index = journal.frontier()['next_ordinal']-1
tab_done = next_index == len(source['placements']) or source['placements'][next_index]['tab_path'] != placement['tab_path']
if limits.per_tab_evaluation is not None and tab_done:
await evaluate_current_tab(page,source,placement['tab_path'],journal,limits,end)
if await fetch_health_manifest(page,dashboard_id,getattr(limits,'chart_ids',None)) != source:
raise ValueError('CHART_HEALTH_MANIFEST_CHANGED')
return journal.finish('passed','CHART_HEALTH_COMPLETE')
except asyncio.CancelledError:
journal.finish('inconclusive','CHART_HEALTH_INTERRUPTED')
raise
except Exception as exc:
code = str(exc) if str(exc).startswith(('CHART_','BROWSER_')) else 'CHART_HEALTH_UNAVAILABLE'
return journal.finish('inconclusive',code)
finally:
if collector is not None:
await collector.close()
# #endregion ScenarioExecution.ChartHealth.Sweep.Run
# #endregion ScenarioExecution.ChartHealth.Sweep

View File

@@ -5,7 +5,7 @@ from hashlib import sha256
import json
from .browser_traversal_inputs import parse_traversal_input
TRAVERSAL_ACTIONS = frozenset({'pagination','navigate_tabs'})
TRAVERSAL_ACTIONS = frozenset({'pagination','navigate_tabs','assert_chart_health'})
# #region ScenarioExecution.Traversal.PinnedInputs.Projection [C:3] [TYPE Function]
@@ -44,9 +44,12 @@ def resolve_pinned_browser_inputs(step):
raise ValueError('BROWSER_TRAVERSAL_AUTHORITY_MISSING')
validate_traversal_projection(step,run)
inputs = (step.get('step_meta') or {}).get('action_inputs') or {}
if (step['action'] == 'pagination' and run.runner_plan.get('action_registry_version') == '038.7.0'
if (step['action'] == 'pagination' and run.runner_plan.get('action_registry_version') in {'038.7.0','038.8.0'}
and 'selection' not in inputs):
raise ValueError('BROWSER_TRAVERSAL_SELECTION_REQUIRED')
return deepcopy(parse_traversal_input(step['action'],inputs))
if (step['action'] == 'navigate_tabs' and run.runner_plan.get('action_registry_version') == '038.8.0'
and inputs.get('chart_health') is not True):
raise ValueError('CHART_HEALTH_INPUT_REQUIRED')
return deepcopy(parse_traversal_input(step['action'],inputs,registry_version=run.runner_plan.get('action_registry_version')))
# #endregion ScenarioExecution.Traversal.PinnedInputs.Resolve
# #endregion ScenarioExecution.Traversal.PinnedInputs

View File

@@ -86,7 +86,7 @@ def _browser_action_provider(context: seam.Any, *, transport, storage, event_loo
seam.logger.explore("Browser capacity unavailable", src=seam._SRC, payload={"run_id": run_id}, error=str(exc))
return seam.LiveAdapterResult(status="inconclusive", reason_code="BROWSER_CAPACITY_UNAVAILABLE")
if action in {'pagination', 'navigate_tabs'}:
if action in {'pagination', 'navigate_tabs', 'assert_chart_health'}:
from .browser_traversal_runtime import execute_traversal
return execute_traversal(step=context.step,admission=admission,storage=storage,
capacity_lease_id=lease_id,event_loop=event_loop,transport=transport,session_manager=session_manager)

View File

@@ -60,7 +60,7 @@ _READ_ONLY_ACTIONS = frozenset({
"navigate_tab", "inspect_filter_state", "apply_table_filter", "extract_table",
"scroll_to", "inspect_columns", "click", "select_rows", "download",
# Wave-2 observe drivers (038.5.0, AGSCN-FR-024):
"assert_dom", "inspect_filter_options", "navigate_tabs", "pagination", "wait_for_selector",
"assert_dom", "assert_chart_health", "inspect_filter_options", "navigate_tabs", "pagination", "wait_for_selector",
})
_MUTATION_ACTIONS = frozenset({"row_edit", "bulk_edit"})
_ALLOWED_WAIT_STATES = frozenset({"load", "domcontentloaded", "networkidle"})
@@ -253,7 +253,7 @@ async def _execute_on_page(
) -> BrowserTransportOutcome:
if action not in _READ_ONLY_ACTIONS and action not in _MUTATION_ACTIONS:
raise BrowserTransportUnsupported("BROWSER_ACTION_NOT_SUPPORTED")
if action in {'pagination', 'navigate_tabs'}:
if action in {'pagination', 'navigate_tabs', 'assert_chart_health'}:
from .browser_traversal_transport import run_traversal_transport
return await run_traversal_transport(page,action,action_input,session_handle)
checkpoints: list[str] = ["dashboard_open"]

View File

@@ -4,6 +4,7 @@
from typing import Annotated, Literal
from pydantic import BaseModel, ConfigDict, Field
from .browser_sampling_inputs import SelectionPolicy, FullSelection
from .browser_health_evaluation_inputs import HealthEvaluationPolicy
Selector = Annotated[str, Field(min_length=1, max_length=500)]
@@ -41,16 +42,43 @@ class PaginationTraversalInput(TraversalBudget):
# @BRIEF Full server-manifest sweep; expected IDs never come from callers.
class AllTabsTraversalInput(BaseModel):
model_config = ConfigDict(extra='forbid', strict=True)
chart_health: bool = False
per_tab_evaluation: HealthEvaluationPolicy | None = None
per_chart_timeout_seconds: int = Field(default=70, ge=1, le=90)
per_tab_timeout_seconds: int = Field(default=70, ge=1, le=90)
whole_timeout_seconds: int = Field(default=7200, ge=1, le=21600)
max_bytes: int = Field(default=268435456, ge=1, le=1073741824)
# #endregion ScenarioExecution.Traversal.Inputs.Tabs
# #region ScenarioExecution.Traversal.Inputs.LegacyTabs [C:1] [TYPE Class]
# @BRIEF Frozen0386/0387 normalized tab inputs; additive chart-health fields do not enter legacy hashes or channels.
class LegacyTabsTraversalInput(BaseModel):
model_config = ConfigDict(extra='forbid', strict=True)
per_tab_timeout_seconds: int = Field(default=70, ge=1, le=90)
whole_timeout_seconds: int = Field(default=7200, ge=1, le=21600)
max_bytes: int = Field(default=268435456, ge=1, le=1073741824)
# #endregion ScenarioExecution.Traversal.Inputs.LegacyTabs
# #region ScenarioExecution.ChartHealth.Inputs [C:2] [TYPE Class]
# @BRIEF Closed structural health observation; source counts, verdicts and evidence are never caller inputs.
class ChartHealthInput(BaseModel):
model_config = ConfigDict(extra='forbid', strict=True)
chart_ids: list[Annotated[int, Field(gt=0)]] | None = Field(default=None, min_length=1, max_length=1000)
per_tab_evaluation: HealthEvaluationPolicy | None = None
per_chart_timeout_seconds: int = Field(default=70, ge=1, le=90)
whole_timeout_seconds: int = Field(default=7200, ge=1, le=21600)
max_bytes: int = Field(default=268435456, ge=1, le=1073741824)
# #endregion ScenarioExecution.ChartHealth.Inputs
# #region ScenarioExecution.Traversal.Inputs.Parse [C:2] [TYPE Function]
# @BRIEF Parse the closed generic traversal action contract or reject unknown actions.
def parse_traversal_input(action: Literal['pagination', 'navigate_tabs'], inputs: dict) -> dict:
model = {'pagination': PaginationTraversalInput, 'navigate_tabs': AllTabsTraversalInput}.get(action)
def parse_traversal_input(action: Literal['pagination', 'navigate_tabs', 'assert_chart_health'], inputs: dict, *, registry_version=None) -> dict:
if action == 'navigate_tabs' and registry_version in {'038.6.0','038.7.0'}:
return LegacyTabsTraversalInput.model_validate(inputs).model_dump()
model = {'pagination': PaginationTraversalInput, 'navigate_tabs': AllTabsTraversalInput, 'assert_chart_health':ChartHealthInput}.get(action)
if model is None:
raise ValueError('BROWSER_TRAVERSAL_ACTION_UNSUPPORTED')
return model.model_validate(inputs).model_dump()

View File

@@ -5,7 +5,7 @@ import asyncio
from ..live_adapter import LiveAdapterResult
from ..traversal_store import TraversalJournal
from ..traversal_tabs_store import TabsJournal
from .browser_traversal_inputs import PaginationTraversalInput, AllTabsTraversalInput
from .browser_traversal_inputs import PaginationTraversalInput, AllTabsTraversalInput, ChartHealthInput
# #region ScenarioExecution.Traversal.Runtime.Monitor [C:3] [TYPE Function]
@@ -33,9 +33,11 @@ async def monitor_traversal(factory, journal):
# @POST Only owned full completeness or verified selected-page completeness can yield passed; coverage scope remains explicit.
def execute_traversal(*, step, admission, storage, capacity_lease_id, event_loop, transport, session_manager):
action, run_id, binding = admission['action'], admission['run_id'], admission['binding']
model = PaginationTraversalInput if action == 'pagination' else AllTabsTraversalInput
model = {'pagination':PaginationTraversalInput,'navigate_tabs':AllTabsTraversalInput,'assert_chart_health':ChartHealthInput}[action]
limits = model.model_validate(admission['action_inputs'])
journal_type = TraversalJournal if action == 'pagination' else TabsJournal
from ..chart_health_store import ChartHealthJournal
health = action == 'assert_chart_health' or getattr(limits,'chart_health',False)
journal_type = ChartHealthJournal if health else (TraversalJournal if action == 'pagination' else TabsJournal)
journal = journal_type(step,storage,limits,capacity_lease_id=capacity_lease_id)
runtime = {'journal':journal,'limits':limits,'dashboard_id':binding.dashboard_id}
inputs = {**admission['action_inputs'],'_traversal_runtime':runtime}
@@ -75,7 +77,7 @@ def execute_traversal(*, step, admission, storage, capacity_lease_id, event_loop
ref = summary['manifest_ref']
digest = summary['manifest_sha256']
return LiveAdapterResult(status=summary['status'],reason_code=summary['reason_code'],
details={'action':action,'sha256':digest,'traversal':summary,'checkpoints':['traversal_manifest_owned'],
details={'action':action,'sha256':digest,'traversal':summary,**({'chart_health':summary['chart_health']} if 'chart_health' in summary else {}),'checkpoints':['traversal_manifest_owned'],
'artifact_byte_lengths':{ref:summary['manifest_byte_length']},'artifact_content_types':{ref:'application/json'},
**({'browser_checkpoint':checkpoint} if checkpoint else {})},
artifact_refs=[ref],artifact_digests={ref:digest})

View File

@@ -39,6 +39,9 @@ async def run_traversal_transport(page, action, inputs, session_handle=None):
summary = await walk_pages(reader,journal,limits)
if reader is not None:
page = reader.page
elif action == 'assert_chart_health' or getattr(limits,'chart_health',False):
from .browser_health_sweep import sweep_health
summary = await sweep_health(page,runtime['dashboard_id'],journal,limits)
else:
summary = await sweep_tabs(page,runtime['dashboard_id'],journal,limits)
return BrowserTransportOutcome(checkpoints=('traversal_manifest_owned',),page_url=page.url,details={'traversal':summary})

View File

@@ -18,6 +18,8 @@ from src.services.dashboard_testing.analytics.investigation import emit_terminal
from src.services.dashboard_testing.automation.notify import persist_notification
from .artifacts import invalidate_step_evidence
from .chart_health_notification import notification_health_summary
from .chart_health_delivery_hook import arm_health_delivery
from .baseline_resolver import BASELINE_STALE
from .result import build_result
@@ -88,6 +90,22 @@ def _close_queued_dispatch_error(db: Session, run: ScenarioRun) -> dict[str, Any
# #endregion ScenarioExecution.Runner.CloseDispatchError
# #region ScenarioExecution.Runner.BaselineStaleTerminal [C:3] [TYPE Function]
# @BRIEF Inspect persisted baseline-stale reasons without changing terminal notifications.
def _is_baseline_stale_terminal(db: Session, run: ScenarioRun) -> bool:
"""True when the blocked terminal run carries a BASELINE_STALE reason on the run or a step."""
if run.error_code == BASELINE_STALE:
return True
for step in db.query(ScenarioStepRun).filter(ScenarioStepRun.run_id == run.id).all():
if step.error_code == BASELINE_STALE:
return True
reason_codes = (step.step_outcome or {}).get("reason_codes") or []
if BASELINE_STALE in reason_codes:
return True
return False
# #endregion ScenarioExecution.Runner.BaselineStaleTerminal
# #region ScenarioExecution.Runner.TerminalSignal [C:4] [TYPE Function] [SEMANTICS scenario,execution,terminal,investigation,signal,poisoned]
# @BRIEF Project one terminal non-pass run into the canonical analyst queue without starting work.
# @RELATION CALLS -> [ScenarioAnalytics.Investigation.TerminalSignal]
@@ -111,31 +129,19 @@ def _close_queued_dispatch_error(db: Session, run: ScenarioRun) -> dict[str, Any
# @REJECTED Counting the failure inside the retry dispatcher tick — several ticks observe the same
# failed run, so tick-side counting double-accounts one failure; the terminal moment is
# the single accounting point.
def _is_baseline_stale_terminal(db: Session, run: ScenarioRun) -> bool:
"""True when the blocked terminal run carries a BASELINE_STALE reason on the run or a step."""
if run.error_code == BASELINE_STALE:
return True
for step in db.query(ScenarioStepRun).filter(ScenarioStepRun.run_id == run.id).all():
if step.error_code == BASELINE_STALE:
return True
reason_codes = (step.step_outcome or {}).get("reason_codes") or []
if BASELINE_STALE in reason_codes:
return True
return False
def _record_terminal_side_effects(db: Session, run: ScenarioRun) -> None:
if run.status in {"failed", "blocked", "inconclusive"}:
emit_terminal_run_signal(db, run)
baseline_stale = run.status == "blocked" and _is_baseline_stale_terminal(db, run)
event_type = "baseline_stale" if baseline_stale else ("blocked" if run.status == "blocked" else "failed")
persist_notification(
receipt = persist_notification(
db,
event_type=event_type,
scenario_id=run.scenario_id,
run_id=run.id,
severity="warning",
payload={
**notification_health_summary(db,run.id),
"status": run.status,
"environment_id": run.environment_id,
**({"error_code": BASELINE_STALE} if baseline_stale else {}),
@@ -143,6 +149,7 @@ def _record_terminal_side_effects(db: Session, run: ScenarioRun) -> None:
sla_seconds=3600,
idempotency_key=f"terminal:{event_type}:{run.id}",
)
arm_health_delivery(db,receipt)
_record_poisoned_failure(db, run)
elif run.status == "passed":
persist_notification(
@@ -155,6 +162,7 @@ def _record_terminal_side_effects(db: Session, run: ScenarioRun) -> None:
)
_reset_poisoned_streak(run)
# #endregion ScenarioExecution.Runner.TerminalSignal
# #region ScenarioExecution.Runner.PoisonedAccounting [C:3] [TYPE Function] [SEMANTICS scenario,execution,terminal,poisoned,accounting,dlq]
# @ingroup ScenarioExecution
@@ -223,6 +231,5 @@ def _reset_poisoned_streak(run: ScenarioRun) -> None:
run_id=run.id,
)
# #endregion ScenarioExecution.Runner.PoisonedStreakReset
# #endregion ScenarioExecution.Runner.TerminalSignal
# #endregion ScenarioExecution.TerminalEffects

View File

@@ -66,6 +66,9 @@ class TraversalJournal:
input_value = limits.model_dump()
if (step.get('step_meta') or {}).get('action_inputs',{}).get('selection') is None:
input_value.pop('selection',None)
for field in ('chart_health','per_chart_timeout_seconds','per_tab_evaluation'):
if field not in (step.get('step_meta') or {}).get('action_inputs',{}):
input_value.pop(field,None)
input_digest = sha256(canonical(input_value)).hexdigest()
row = db.query(ScenarioTraversal).filter_by(run_id=self.run_id,logical_step_id=self.logical_step_id,attempt=self.attempt).one_or_none()
if row is None:

View File

@@ -8,27 +8,27 @@ from __future__ import annotations
from datetime import UTC, datetime
from typing import Any
import uuid
import uuid # noqa: F401 - walker_step_control public patch seam
from sqlalchemy.orm import Session
from sqlalchemy import update
from sqlalchemy import update # noqa: F401 - walker_step_control public patch seam
from src.core.logger import logger
from src.core.logger import logger # noqa: F401 - retained walker public seam
from src.models.scenario_run import ScenarioRun, ScenarioStepRun
from .agent_evaluation import AgentEvaluation, persist_agent_evaluation, validate_evaluation_evidence
from .artifacts import register_step_evidence
from .baseline_resolver import stamp_baseline_pin
from .agent_evaluation import AgentEvaluation, persist_agent_evaluation, validate_evaluation_evidence # noqa: F401 - walker_publication seam
from .artifacts import register_step_evidence # noqa: F401 - walker_publication seam
from .baseline_resolver import stamp_baseline_pin # noqa: F401 - walker_publication seam
from .capacity_block import CAPACITY_RETRY_CODES as _CAPACITY_RETRY_CODES
from .capacity_block import block_run_on_capacity
from .decision_policy import decide_step_outcome, policy_inputs_from_outcome, verified_evidence_refs
from .decision_policy import decide_step_outcome, policy_inputs_from_outcome, verified_evidence_refs # noqa: F401 - walker_publication seam
from .dispatch import dispatch_step
from .executor_registry import ScenarioExecutorRegistry
from .lifecycle import suspend_for_human
from .lifecycle import suspend_for_human # noqa: F401 - walker_step_control seam
from .result import build_result
from .runner_plan import resolve_pinned_policy, validate_pinned_runner_plan
from .runner_plan import resolve_pinned_policy, validate_pinned_runner_plan # noqa: F401 - walker_publication seam
from .terminal_effects import _record_terminal_side_effects, _reject_malformed_plan
from .worker import claim_step
from .worker import claim_step # noqa: F401 - walker_step_control seam
# #region ScenarioExecution.Runner.EvaluationRecord [C:2] [TYPE Function] [SEMANTICS scenario,execution,evaluation,outcome]
@@ -228,7 +228,7 @@ def _advance_step(db, run, registry, worker_id, lease_seconds, plan, order, depe
step, claim_failed = _claim_automated_step(db, run, step_id, order, existing, descriptor, worker_id, lease_seconds, tool)
if claim_failed:
return build_result(run,list(existing.values()))
if tool == 'browser' and step_meta['action'] in {'pagination', 'navigate_tabs'}:
if tool == 'browser' and step_meta['action'] in {'pagination', 'navigate_tabs', 'assert_chart_health'}:
# Restricted long traversal journals renew committed worker/provider leases independently.
db.commit()
outcome = dispatch_step(
@@ -249,7 +249,7 @@ def _advance_step(db, run, registry, worker_id, lease_seconds, plan, order, depe
edges=dependencies,
)
db.refresh(run)
if tool == 'browser' and step_meta['action'] in {'pagination', 'navigate_tabs'} and run.phase == 'paused':
if tool == 'browser' and step_meta['action'] in {'pagination', 'navigate_tabs', 'assert_chart_health'} and run.phase == 'paused':
step.status = 'queued'
db.flush()
return build_result(run, list(existing.values()))

View File

@@ -1,7 +1,7 @@
# #region BaselineEngine.QueryExecutor.Envelope [C:2] [TYPE Class] [SEMANTICS baseline,execution,envelope,immutability]
# @ingroup BaselineEngine
# @BRIEF Trusted execution envelope: NormalizedValue + source_response_hash.
# @INVARIANT source_response_hash from full deterministic response bytes before extraction.
# @INVARIANT Successful source_response_hash hashes original response bytes; error payload_kind explicitly identifies sanitized diagnostics with separate original-response provenance.
# @DATA_CONTRACT Superset API Response -> QueryExecutionEnvelope
from __future__ import annotations
@@ -16,4 +16,6 @@ class QueryExecutionEnvelope:
normalized_value: NormalizedValue
source_response_hash: str = field(repr=False)
raw_response_content: bytes = field(repr=False)
diagnostic: dict | None = None
payload_kind: str = "original_response"
# #endregion BaselineEngine.QueryExecutor.Envelope

View File

@@ -13,7 +13,6 @@
from __future__ import annotations
import json
from typing import Any
from src.core.logger import logger
@@ -29,7 +28,7 @@ from src.schemas.dashboard_testing import (
ValueKind,
Warning,
)
from src.services.dashboard_testing.immutability import compute_source_response_hash
from src.services.dashboard_testing.immutability import compute_source_response_hash # noqa: F401 - retained public import seam
from src.services.dashboard_testing.query_envelope import QueryExecutionEnvelope
@@ -152,22 +151,8 @@ def _verify_authoritative_model(
if chart.dataset_id != dataset_id:
raise ValueError("Chart dataset differs from authoritative dashboard model")
# 4. Verify result_key/metrics
if chart_id is not None:
matching_charts = [ch for ch in query_model.charts if ch.chart_id == chart_id]
if matching_charts:
chart_metrics = matching_charts[0].metrics
valid_metric_names = {m.metric_name for m in chart_metrics}
if valid_metric_names and request.result_key not in valid_metric_names:
logger.explore("Result key not a known metric", src="BaselineEngine.QueryExecutor.VerifyAuthoritativeModel", payload={"result_key": request.result_key, "chart_id": chart_id,
"valid_metrics": valid_metric_names}, error="result_key is not a known metric for this chart")
warnings.append(Warning(
source="execution",
resource=f"chart/{chart_id}",
code="UNKNOWN_METRIC",
detail=f"result_key '{request.result_key}' is not a known metric "
f"for chart {chart_id}. Known metrics: {valid_metric_names}",
))
from .query_model_guard import metric_warnings
warnings.extend(metric_warnings(request,query_model,chart_id))
# 5. Scope filters to the target chart
if chart_id is not None and request.normalized_filters.filters:
@@ -216,8 +201,8 @@ def _extract_query_result(result_data: dict, result_key: str) -> tuple[Any, str]
# @ingroup BaselineEngine
# @BRIEF Execute query and return trusted envelope with hash from full deterministic response bytes.
# @PRE query_model_fingerprint matches authoritative model. chart_id/dataset_id belongs to dashboard.
# @POST Returns QueryExecutionEnvelope with NormalizedValue + source_response_hash + raw bytes.
# @INVARIANT source_response_hash from full response bytes before extraction, never canonical scalar.
# @POST Returns original successful response evidence or typed sanitized error evidence, preserving explicit origin and original wire digest when available.
# @INVARIANT Successful source_response_hash hashes full wire bytes; sanitized failure payloads hash their own bytes and retain original wire digest separately.
# @SIDE_EFFECT Async POST to Superset /api/v1/chart/data.
# @DATA_CONTRACT ExecuteQueryRequest + DashboardQueryModel -> QueryExecutionEnvelope
# @RATIONALE source_response_hash computed from raw httpx bytes (via raw_response=True on execute_chart_data_raw) BEFORE extracting the canonical scalar. This ensures even Superset metadata-only response changes (query_id, colnames, column types) produce a different hash, making immutability violations detectable. The envelope returns raw_response_content bytes for DraftStorage persistence, enabling offline re-verification of the exact wire response.
@@ -256,6 +241,7 @@ async def execute_dashboard_query_envelope(
# Use the chart's authoritative metric definition. An ad hoc metric needs
# its aggregate and column; its label alone is not executable in Superset.
metric_spec: str | dict = request.result_key
chart_model = None
if query_model is not None and chart_id is not None:
chart_model = next((chart for chart in query_model.charts if chart.chart_id == chart_id), None)
descriptor = next((metric for metric in chart_model.metrics if metric.metric_name == request.result_key), None) if chart_model else None
@@ -274,26 +260,11 @@ async def execute_dashboard_query_envelope(
row_limit=request.max_rows,
)
except SupersetAPIError as e:
logger.explore("Superset API error during execution", src="BaselineEngine.QueryExecutor.ExecuteQueryEnvelope", payload={"chart_id": chart_id, "dataset_id": dataset_id}, error=str(e))
# On API error, envelope still contains error NormalizedValue but hash from error context
nv = NormalizedValue(
kind=ValueKind.UNKNOWN,
raw_value=None,
canonical_value=None,
source=f"{request.environment_id}/dashboard/{request.dashboard_id}/chart/{chart_id}",
warnings=[Warning(
source="execution",
resource=f"chart/{chart_id}" if chart_id else f"dataset/{dataset_id}",
code=f"SUPERSET_{getattr(e, 'status_code', 'ERROR')}",
detail=str(e),
)],
)
error_bytes = json.dumps({"error": str(e), "status_code": getattr(e, 'status_code', 0)}, sort_keys=True).encode()
return QueryExecutionEnvelope(
normalized_value=nv,
source_response_hash=compute_source_response_hash(error_bytes),
raw_response_content=error_bytes,
)
logger.explore("Superset API error during execution", src="BaselineEngine.QueryExecutor.ExecuteQueryEnvelope", payload={"chart_id": chart_id, "dataset_id": dataset_id}, error="Superset query unavailable")
from .query_failure import failure_envelope
context = getattr(e, 'context', {})
return failure_envelope(request, {'message':str(e)},
http_status=context.get('status_code'), origin='synthetic_exception',chart_name=chart_model.slice_name if chart_model else None)
# ── Compute source_response_hash from raw httpx bytes BEFORE extraction ──
# Uses the exact bytes received from Superset (via raw_response=True on httpx).
@@ -305,8 +276,13 @@ async def execute_dashboard_query_envelope(
raw_response_content: bytes = raw_response.raw_bytes
source_response_hash: str = raw_response.source_response_hash
result_data: dict[str, Any] = raw_response.parsed
if "result" not in result_data and ("message" in result_data or "errors" in result_data):
raise ValueError("Superset chart-data rejected the query")
from .query_failure import response_error, failure_envelope
if isinstance(result_data,dict) and result_data.get('_invalid_response'):
return failure_envelope(request, {'message':'Superset returned a non-JSON chart response'},
raw_bytes=raw_response_content,http_status=raw_response.http_status,confirmed=False)
error = response_error(result_data, raw_response.http_status)
if error is not None:
return failure_envelope(request, error, raw_bytes=raw_response_content, http_status=raw_response.http_status,chart_name=chart_model.slice_name if chart_model else None,confirmed=not error.get("_unconfirmed",False))
# Extract result value from Superset response (AFTER hash computation)
raw_value, query_id = _extract_query_result(result_data, request.result_key)

View File

@@ -0,0 +1,71 @@
# #region DashboardTesting.ChartHealth.QueryFailure [C:4] [TYPE Module] [SEMANTICS query,errors,redaction,evidence]
# @BRIEF Retain bounded query failure observations without pretending reconstructed bytes are wire evidence.
from datetime import UTC, datetime
from hashlib import sha256
import json
import re
from src.schemas.dashboard_testing import NormalizedValue, ValueKind, Warning
from .chart_health_contract import ChartDiagnostic
from .query_envelope import QueryExecutionEnvelope
# #region DashboardTesting.ChartHealth.QueryFailure.Redact [C:2] [TYPE Function]
# @POST Credential redaction precedes bounded diagnostics and truncation is explicit.
def safe_message(value):
from src.plugins.llm_analysis._redaction import RedactionService
text = RedactionService.redact_raw_response(str(value))
return text[:4096], len(text) > 4096
# #endregion DashboardTesting.ChartHealth.QueryFailure.Redact
# #region DashboardTesting.ChartHealth.QueryFailure.Extract [C:3] [TYPE Function]
# @POST HTTP200 application failures and result.status=failed are errors; zero/empty successful data are not.
def response_error(payload, http_status=200):
if not isinstance(payload,dict):
return {'message':'Malformed Superset chart response','_unconfirmed':True}
records = payload.get('result', [])
if not isinstance(records,list):
return {'message':'Malformed Superset chart result','_unconfirmed':True}
candidates = [payload, *(records if isinstance(records, list) else [])]
for item in candidates:
if not isinstance(item, dict):
continue
errors = item.get('errors')
if errors:
entry = errors[0] if isinstance(errors, list) else errors
return entry if isinstance(entry, dict) else {'message':str(entry)}
if item.get('error') or item.get('status') in {'failed', 'error'}:
return {'message':item.get('error') or item.get('message') or 'Query failed'}
if 'result' not in payload and payload.get('message'):
return {'message':payload['message']}
if http_status >= 400:
return {'message':payload.get('message', f'HTTP {http_status}') if isinstance(payload, dict) else f'HTTP {http_status}'}
return None
# #endregion DashboardTesting.ChartHealth.QueryFailure.Extract
# #region DashboardTesting.ChartHealth.QueryFailure.Envelope [C:4] [TYPE Function]
# @POST Failed query details retain sanitized origin and pre-redaction wire digest when available; no raw rows/SQL reach diagnostics.
def failure_envelope(request, error, *, raw_bytes=None, http_status=None, origin='http_response', chart_name=None, confirmed=True):
message, truncated = safe_message(error.get('message') or error.get('error') or 'Query failed')
db_code = re.search(r'\bCode:\s*(\d+)\b', message)
diagnostic = ChartDiagnostic(chart_id=request.chart_id, dataset_id=request.dataset_id,
placement_id=f'chart/{request.chart_id}' if request.chart_id else f'dataset/{request.dataset_id}',
chart_name=safe_message(chart_name or (f'Chart {request.chart_id}' if request.chart_id else f'Dataset {request.dataset_id}'))[0][:512],
environment_id=request.environment_id,dashboard_id=request.dashboard_id,
status='failed' if origin == 'http_response' and confirmed else 'inconclusive',
reason_code='SUPERSET_QUERY_FAILED' if origin == 'http_response' and confirmed else 'SUPERSET_QUERY_UNAVAILABLE',
origin=origin, filter_fingerprint=request.normalized_filters.filters_hash,
observed_at=datetime.now(UTC).isoformat(), duration_seconds=0.0, http_status=http_status,
application_code=str(error.get('error_type') or error.get('code') or '')[:256] or None,
database_code=db_code.group(1) if db_code else None, message=message,
terminal_state='error' if origin == 'http_response' and confirmed else 'unavailable',
original_response_sha256=sha256(raw_bytes).hexdigest() if raw_bytes is not None else None,
original_byte_length=len(raw_bytes) if raw_bytes is not None else None, truncated=truncated)
data = json.dumps({'chart_diagnostic':diagnostic.model_dump()},sort_keys=True,separators=(',',':')).encode()
value = NormalizedValue(kind=ValueKind.UNKNOWN,raw_value=None,canonical_value=None,
source=f'{request.environment_id}/dashboard/{request.dashboard_id}/chart/{request.chart_id}',
warnings=[Warning(source='execution',resource=diagnostic.placement_id,code=diagnostic.reason_code,detail=message)])
return QueryExecutionEnvelope(value,sha256(data).hexdigest(),data,diagnostic=diagnostic.model_dump(),payload_kind='sanitized_diagnostic')
# #endregion DashboardTesting.ChartHealth.QueryFailure.Envelope
# #endregion DashboardTesting.ChartHealth.QueryFailure

View File

@@ -0,0 +1,30 @@
# #region DashboardTesting.QueryModelGuard [C:3] [TYPE Module]
# @BRIEF Preserve authoritative metric warning behavior separately from query execution and evidence retention.
from src.schemas.dashboard_testing import Warning
from . import query_executor as seam
# #region DashboardTesting.QueryModelGuard.MetricWarnings [C:3] [TYPE Function]
# @POST Unknown inspected metrics retain the existing warning without authorizing another dataset/chart.
def metric_warnings(request, query_model, chart_id):
warnings = []
# 4. Verify result_key/metrics
if chart_id is not None:
matching_charts = [ch for ch in query_model.charts if ch.chart_id == chart_id]
if matching_charts:
chart_metrics = matching_charts[0].metrics
valid_metric_names = {m.metric_name for m in chart_metrics}
if valid_metric_names and request.result_key not in valid_metric_names:
seam.logger.explore("Result key not a known metric", src="BaselineEngine.QueryExecutor.VerifyAuthoritativeModel", payload={"result_key": request.result_key, "chart_id": chart_id,
"valid_metrics": valid_metric_names}, error="result_key is not a known metric for this chart")
warnings.append(Warning(
source="execution",
resource=f"chart/{chart_id}",
code="UNKNOWN_METRIC",
detail=f"result_key '{request.result_key}' is not a known metric "
f"for chart {chart_id}. Known metrics: {valid_metric_names}",
))
return warnings
# #endregion DashboardTesting.QueryModelGuard.MetricWarnings
# #endregion DashboardTesting.QueryModelGuard

View File

@@ -145,6 +145,8 @@ def _selector_hint_param(missing_step: str) -> ScenarioParameter:
# #endregion ScenarioGraph.Compiler.SelectorHintParam
# #region ScenarioGraph.Compiler.Own.compile_scenario [C:3] [TYPE Function]
# @BRIEF Compile validated scenario steps into the runner plan.
def compile_scenario(req: CompileScenarioRequest) -> CompiledResult:
"""Deterministically compile a dashboard goal into a stable ScenarioGraph DAG."""
logger.reason("Compiling scenario graph", src="ScenarioGraph.Compiler.Compile", payload={"dashboard_id": req.dashboard_id, "cases": req.objective.get("selected_case_ids")})
@@ -154,6 +156,7 @@ def compile_scenario(req: CompileScenarioRequest) -> CompiledResult:
except Exception as e:
logger.explore("Compile failed; no graph produced", src="ScenarioGraph.Compiler.Compile", error=str(e))
raise
# #endregion ScenarioGraph.Compiler.Own.compile_scenario
# #region ScenarioGraph.Compiler.CompileImpl [C:4] [TYPE Function] [SEMANTICS scenario,compiler,deterministic]
@@ -315,6 +318,7 @@ def _build_step(
depends_on=list(depends_on or []),
automation_status=automation,
checklist_case_ids=[case_id],
action_inputs={"chart_health":True} if action == "navigate_tabs" else None,
risk=entry["risk"],
agent_evaluation_spec=evaluation_spec if action == "evaluate_declared_spec" else None,
)

View File

@@ -72,7 +72,7 @@ class AgentEvaluationSpec(BaseModel):
prompt_template_version: str = Field(min_length=1)
prompt_template_hash: str = Field(pattern=_SHA256_RE)
evidence_refs: list[str] = Field(min_length=1, max_length=100)
comparison_refs: list[str] = Field(min_length=1, max_length=100)
comparison_refs: list[str] = Field(default_factory=list, max_length=100)
tool_allowlist: list[str] = Field(default_factory=list, max_length=0)
output_schema: Literal["agent-evaluation.schema.json"]
decision_policy: DecisionPolicy

View File

@@ -315,7 +315,7 @@ def _check_field(key: str, value: Any, spec: InputField, findings: list[StepInpu
def validate_step_inputs(action: Any, inputs: Any) -> StepInputsValidation:
if not isinstance(inputs, dict):
return StepInputsValidation(False, [StepInputsFinding("INVALID_STEP_INPUT_TYPE", "inputs must be a mapping")])
if action in {'pagination', 'navigate_tabs'}:
if action in {'pagination', 'navigate_tabs', 'assert_chart_health'}:
from src.services.dashboard_testing.execution.providers.browser_traversal_inputs import parse_traversal_input
try:
parse_traversal_input(action, inputs)
@@ -353,6 +353,8 @@ def assert_step_inputs(action: Any, inputs: Any) -> dict[str, Any]:
first = result.findings[0]
raise ValueError(f"{first.code}: {first.message}")
accepted = dict(inputs)
if action == 'navigate_tabs':
accepted.setdefault('chart_health',True)
if action == 'pagination':
accepted.setdefault('selection',{'mode':'quantiles','count':5})
return accepted

View File

@@ -16,7 +16,7 @@ from typing import Any
PHASES = ("setup", "interact", "observe", "assert", "evidence", "report")
TOOLS = ("browser", "superset_api", "sql_evidence", "transform", "xlsx", "assertion", "screenshot", "report", "artifact", "human", "agent_evaluation")
RISKS = ("read", "browser_interaction", "draft_write", "human")
ACTION_REGISTRY_VERSION = "038.7.0" # explicit owned page sampling; archived0386 missing selection remains full
ACTION_REGISTRY_VERSION = "038.8.0" # explicit chart-health ownership; archived sampling/registry snapshots remain unchanged
# Registry discipline (AGSCN-FR-025): every registered action MUST be implemented with a canaried
# driver OR carry an explicit `disabled` reason. Compile and RunnerPlan derivation reject a disabled
@@ -35,6 +35,7 @@ REGISTERED_ACTIONS: dict[str, dict[str, Any]] = {
# Wave-2 observe drivers implemented (browser_readonly_flows_observe.py); AGSCN-FR-025.
"assert_dom": {"tool": "browser", "phase": "observe", "risk": "read", "capability": "browser"},
"inspect_filter_options": {"tool": "browser", "phase": "observe", "risk": "read", "capability": "browser"},
"assert_chart_health": {"tool": "browser", "phase": "observe", "risk": "read", "capability": "browser"},
"navigate_tabs": {"tool": "browser", "phase": "interact", "risk": "browser_interaction", "capability": "browser"},
"wait_for_selector": {"tool": "browser", "phase": "observe", "risk": "read", "capability": "browser"},
"extract_table": {"tool": "browser", "phase": "observe", "risk": "read", "capability": "browser"},
@@ -124,6 +125,7 @@ _TIMEOUTS = {
# Restricted traversal owns its shorter persisted whole/page budgets and renewable leases.
"pagination": 21725000,
"navigate_tabs": 21725000,
"assert_chart_health": 21725000,
"download": 60000,
"download_xlsx": 60000,
"capture_screenshot": 30000,
@@ -211,6 +213,7 @@ STEP_TEMPLATES: dict[str, tuple[str, str, str]] = {
"browser_text_filter_assert": ("apply_native_filter", "browser", "interact"),
"browser_table_filter_assert": ("apply_table_filter", "browser", "interact"),
"browser_pagination_assert": ("pagination", "browser", "interact"),
"browser_chart_health_assert": ("assert_chart_health", "browser", "observe"),
"browser_tabs_sweep": ("navigate_tabs", "browser", "interact"),
"browser_edit_refresh_evidence": ("edit_row", "browser", "interact"),
"browser_bulk_edit_evidence": ("bulk_edit", "browser", "interact"),

View File

@@ -39,14 +39,22 @@ class ScenarioValidationResult:
graph_hash: str = ""
# #region ScenarioGraph.Validator.Own._err [C:1] [TYPE Function]
# @BRIEF Construct a scenario validation error.
def _err(code: str, msg: str, *, step_id: str | None = None, recovery: list[str] | None = None) -> Finding:
return Finding(code=code, severity="error", message=msg, step_id=step_id, recovery_options=recovery or [])
# #endregion ScenarioGraph.Validator.Own._err
# #region ScenarioGraph.Validator.Own._warn [C:1] [TYPE Function]
# @BRIEF Construct a scenario validation warning.
def _warn(code: str, msg: str, *, step_id: str | None = None) -> Finding:
return Finding(code=code, severity="warning", message=msg, step_id=step_id)
# #endregion ScenarioGraph.Validator.Own._warn
# #region ScenarioGraph.Validator.Own._find_cycles [C:3] [TYPE Function]
# @BRIEF Find dependency cycles in the scenario graph.
def _find_cycles(steps: list[dict[str, Any]]) -> list[str]:
"""Return one representative cycle path per detected cycle (deterministic order)."""
ids = {s["id"] for s in steps}
@@ -55,6 +63,8 @@ def _find_cycles(steps: list[dict[str, Any]]) -> list[str]:
stack: list[str] = []
cycles: list[str] = []
# #region ScenarioGraph.Validator.Own.visit [C:3] [TYPE Function]
# @BRIEF Traverse dependency links while tracking active nodes.
def visit(node: str) -> None:
if node in stack:
idx = stack.index(node)
@@ -67,14 +77,19 @@ def _find_cycles(steps: list[dict[str, Any]]) -> list[str]:
for d in sorted(deps.get(node, [])):
visit(d)
stack.pop()
# #endregion ScenarioGraph.Validator.Own.visit
for node in sorted(ids):
visit(node)
return sorted(set(cycles))
# #endregion ScenarioGraph.Validator.Own._find_cycles
# #region ScenarioGraph.Validator.Own._detect_sql [C:1] [TYPE Function]
# @BRIEF Detect SQL-like content in nested inputs.
def _detect_sql(text: str | None) -> bool:
return contains_sql_statement(text)
# #endregion ScenarioGraph.Validator.Own._detect_sql
# #region ScenarioGraph.Validator.ValidateCore [C:4] [TYPE Function] [SEMANTICS scenario,validator,checks]
@@ -211,7 +226,7 @@ def _check_step_inputs(steps: list[dict[str, Any]], result: ScenarioValidationRe
for s in steps:
action_inputs = s.get("action_inputs")
if s.get('action') in {'pagination', 'navigate_tabs'}:
if s.get('action') in {'pagination', 'navigate_tabs', 'assert_chart_health'}:
action_inputs = {} if action_inputs is None else action_inputs
elif not isinstance(action_inputs, dict) or not action_inputs:
continue
@@ -253,6 +268,8 @@ def _check_safety(steps: list[dict[str, Any]], result: ScenarioValidationResult)
def _check_dashboard_context(scenario: DashboardTestScenario, result: ScenarioValidationResult) -> None:
max_findings = 20
# #region ScenarioGraph.Validator.Own.walk [C:3] [TYPE Function]
# @BRIEF Inspect nested values for forbidden dashboard identity fields.
def walk(node: Any, path: str, depth: int, server_query: bool = False) -> None:
if len(result.errors) >= max_findings or depth > 24:
return
@@ -269,6 +286,7 @@ def _check_dashboard_context(scenario: DashboardTestScenario, result: ScenarioVa
walk(item, f"{path}[{index}]", depth + 1, server_query)
elif isinstance(node, str) and not server_query and _detect_sql(node):
result.errors.append(_err("FORBIDDEN_SQL", f"dashboard_context contains SQL text at {path}"))
# #endregion ScenarioGraph.Validator.Own.walk
walk(scenario.dashboard_context, "dashboard_context", 0)
# #endregion ScenarioGraph.Validator.CheckDashboardContext
@@ -359,6 +377,8 @@ def _topological_order(steps: list[dict[str, Any]]) -> list[str]:
# #endregion ScenarioGraph.Validator.TopologicalOrder
# #region ScenarioGraph.Validator.Own.validate_scenario [C:3] [TYPE Function]
# @BRIEF Validate a scenario using the composed graph checks.
def validate_scenario(scenario: DashboardTestScenario, *, metric_admission=None) -> ScenarioValidationResult:
"""Validate a candidate graph and return all deterministic findings."""
logger.reason("Validating scenario graph", src="ScenarioGraph.Validator.Validate", payload={"scenario_id": scenario.scenario_id, "steps": len(scenario.steps)})
@@ -370,4 +390,5 @@ def validate_scenario(scenario: DashboardTestScenario, *, metric_admission=None)
raise
logger.reflect("Validation complete", src="ScenarioGraph.Validator.Validate", payload={"valid": result.valid, "errors": len(result.errors), "warnings": len(result.warnings)})
return result
# #endregion ScenarioGraph.Validator.Own.validate_scenario
# #endregion ScenarioGraph.Validator.Validate

View File

@@ -0,0 +1,101 @@
# #region Test.ChartHealth.CapacityPostgres [C:2] [TYPE Module] [SEMANTICS health,postgres,capacity,provider,pin]
# @TEST_INVARIANT An authored78character pin admits an actual64character PostgreSQL lease before SDK work and releases it after completion.
# @RELATION BINDS_TO -> [ScenarioExecution.AgentEvaluation.Submit]
import asyncio
import os
from pathlib import Path
import subprocess
import sys
from types import SimpleNamespace
import pytest
from sqlalchemy import create_engine
from sqlalchemy.orm import sessionmaker
from owned_text_recipe_fixture import owned_recipe_fixture
from src.models.provider_capacity import CapacityLease
from src.services.llm_provider import LLMProviderService
from src.services.dashboard_testing.scenario.models import AgentEvaluationSpec
from src.services.dashboard_testing.scenario.metric_evaluation_provider import provider_public_config_digest
from src.services.dashboard_testing.execution.agent_evaluation import submit_evaluation
from src.services.dashboard_testing.execution.providers.browser_health_provider import admit_health_provider
from src.services.dashboard_testing.execution.providers.browser_health_llm_client import BoundedHealthClient
pytestmark = pytest.mark.integration
# #region Test.ChartHealth.CapacityPostgres.Context [C:1] [TYPE Function]
@pytest.fixture
def committed_health_source(db_factory, tmp_path, monkeypatch):
isolated = db_factory['create_db']('_health_pin')
environment = dict(os.environ, DATABASE_URL=isolated['host_url'])
subprocess.run([sys.executable, '-m', 'alembic', 'upgrade', 'head'],
cwd=Path(__file__).resolve().parents[2], env=environment, check=True, capture_output=True)
engine = create_engine(isolated['host_url'])
sessions = sessionmaker(bind=engine, autoflush=False)
db = sessions()
try:
# Shared genuine run/publication infrastructure only; the health spec below is independent.
source = owned_recipe_fixture(db, tmp_path, monkeypatch)
source.provider.api_key = LLMProviderService(db).encryption.encrypt('fixture-key-never-transmitted')
source.provider.is_multimodal = True
source.provider.max_images = 2
db.commit()
yield source, sessions
finally:
db.rollback()
db.close()
engine.dispose()
db_factory['drop_db'](isolated['db_name'])
# #endregion Test.ChartHealth.CapacityPostgres.Context
# #region Test.ChartHealth.CapacityPostgres.Submit [C:2] [TYPE Function]
# @TEST_INVARIANT Independent connection sees a committed claimed bare digest before the single physical SDK call; final lease is released, authored spec unchanged.
def test_postgres_health_authored_pin_capacity_width_and_release(committed_health_source, monkeypatch):
source, sessions = committed_health_source
fields = source.spec.model_dump(mode='json')
fields.update(provider_version='config_sha256:' + provider_public_config_digest(source.provider),
model_id=source.provider.default_model, model_version='configured-model:' + source.provider.default_model,
comparison_refs=[], criteria=[{'criterion_id': 'health-visual', 'criterion_kind': 'semantic',
'description': 'Describe actual chart errors', 'comparison_id': None}],
limits={'timeout_ms': 1500, 'max_images': 2, 'max_input_tokens': 8192,
'max_output_tokens': 128, 'max_cost': '1.00', 'currency': 'USD'})
spec = AgentEvaluationSpec.model_validate(fields)
assert len(spec.provider_version) == 78
assert admit_health_provider(source.db, SimpleNamespace(spec=spec)) is source.provider
observed = []
# #region Test.ChartHealth.CapacityPostgres.Submit.Sdk [C:1] [TYPE Class]
class ExternalSdk:
# #region Test.ChartHealth.CapacityPostgres.Submit.Sdk.Init [C:1] [TYPE Function]
def __init__(self, *_args):
self.client = self.chat = self.completions = self
# #endregion Test.ChartHealth.CapacityPostgres.Submit.Sdk.Init
# #region Test.ChartHealth.CapacityPostgres.Submit.Sdk.Options [C:1] [TYPE Function]
def with_options(self, **options):
assert options == {'max_retries': 0, 'timeout': 1.5}
return self
# #endregion Test.ChartHealth.CapacityPostgres.Submit.Sdk.Options
# #region Test.ChartHealth.CapacityPostgres.Submit.Sdk.Create [C:1] [TYPE Function]
async def create(self, **_options):
with sessions() as independent:
leases = independent.query(CapacityLease).filter_by(run_id=source.run.id, workload_class='agent_evaluation').all()
assert len(leases) == 1 and leases[0].status == 'claimed'
assert leases[0].provider_version == spec.provider_version[14:]
assert len(leases[0].provider_version) == 64
observed.append('one-physical-request')
return SimpleNamespace(choices=[SimpleNamespace(finish_reason='stop', message=SimpleNamespace(content='{"verdict":"inconclusive"}'))], usage=None)
# #endregion Test.ChartHealth.CapacityPostgres.Submit.Sdk.Create
# #region Test.ChartHealth.CapacityPostgres.Submit.Sdk.Close [C:1] [TYPE Function]
async def close(self):
observed.append('closed')
# #endregion Test.ChartHealth.CapacityPostgres.Submit.Sdk.Close
# #endregion Test.ChartHealth.CapacityPostgres.Submit.Sdk
monkeypatch.setattr('src.services.dashboard_testing.execution.providers.browser_health_llm_client.LLMClient', ExternalSdk)
result = asyncio.run(submit_evaluation(source.db, spec=spec, prompt='actual Code: 386 diagnostic', images=[],
environment_id='full-flow-preprod', environment_class='DEV', run_id=source.run.id,
logical_step_id='health-check', client=BoundedHealthClient(source.db, source.provider, spec)))
source.db.commit()
assert result == {'verdict': 'inconclusive'} and observed == ['one-physical-request', 'closed']
with sessions() as independent:
lease = independent.query(CapacityLease).filter_by(run_id=source.run.id, workload_class='agent_evaluation').one()
assert lease.status == 'released' and lease.provider_version == spec.provider_version[14:]
assert spec.provider_version.startswith('config_sha256:') and len(spec.provider_version) == 78
# #endregion Test.ChartHealth.CapacityPostgres.Submit
# #endregion Test.ChartHealth.CapacityPostgres

View File

@@ -0,0 +1,155 @@
# #region Test.ChartHealth.ConnectedWorker [C:4] [TYPE Module] [SEMANTICS health,real-journal,browser,errors]
# @TEST_INVARIANT Five query errors remain named FAILED while another tab is checked; coverage is independently complete.
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Sweep]
import json
from copy import deepcopy
from hashlib import sha256
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from threading import Thread
import pytest
from sqlalchemy.orm import sessionmaker
from src.models.scenario_run import ScenarioRun, ScenarioStepRun
from src.models.scenario_artifact import ScenarioArtifact
from src.services.agent_runs.artifacts import DraftStorage
from src.services.dashboard_testing.execution.capacity import claim_capacity
from src.services.dashboard_testing.execution.worker import claim_step
from src.services.dashboard_testing.execution.chart_health_store import ChartHealthJournal
from src.services.dashboard_testing.execution.providers.browser_traversal_inputs import ChartHealthInput
from src.services.dashboard_testing.execution.providers.browser_health_sweep import sweep_health
NAMES = ['Debt','Cash','Margin','Forecast','Reserve','Empty healthy']
LAYOUT = {'ROOT_ID':{'type':'ROOT','children':['TABS-main']},
'TABS-main':{'type':'TABS','children':['TAB-daily','TAB-other']},
'TAB-daily':{'type':'TAB','meta':{'text':'Daily summary'},'children':['CHART-1','CHART-2','CHART-3','CHART-4','CHART-5']},
'TAB-other':{'type':'TAB','meta':{'text':'Other'},'children':['CHART-6']},
**{f'CHART-{index}':{'type':'CHART','children':[],'meta':{'chartId':index,'sliceName':name}} for index,name in enumerate(NAMES,1)}}
HTML = '''<button role="tab" id="TABS-main-tab-TAB-daily" aria-controls="daily" aria-selected="false" onclick="activate('daily',[1,2,3,4,5],this)">Daily summary</button>
<button role="tab" id="TABS-main-tab-TAB-other" aria-controls="other" aria-selected="false" onclick="activate('other',[6],this)">Other</button>
<div role="tabpanel" id="daily" hidden>''' + ''.join(f'<div id="chart-id-{index}" style="min-height:120px"><div class="loading">loading</div></div>' for index in range(1,6)) + '''</div>
<div role="tabpanel" id="other" hidden><div id="chart-id-6" style="min-height:120px"><div class="loading">loading</div></div></div>
<script>function activate(id, charts, control) {
document.querySelectorAll('[role=tab]').forEach(t=>t.setAttribute('aria-selected','false'));
document.querySelectorAll('[role=tabpanel]').forEach(t=>t.hidden=true);
control.setAttribute('aria-selected','true'); document.getElementById(id).hidden=false;
charts.forEach(chart => fetch('/api/v1/chart/data', {method:'POST', headers:{'Content-Type':'application/json'},
body:JSON.stringify({form_data:{slice_id:chart}, datasource:{id:2,type:'table'}, queries:[{filters:[]}]})})
.then(r=>r.json()).then(data=>{document.getElementById('chart-id-'+chart).innerHTML=chart===6?'<table><tbody></tbody></table>':'<div role="alert" class="alert-danger">'+data.errors[0].message+'</div>'}));
}</script>'''
# #region Test.ChartHealth.ConnectedWorker.Handler [C:2] [TYPE Class]
# @BRIEF Literal external protocol server; production observer/journal remain unmocked.
class Handler(BaseHTTPRequestHandler):
# #region Test.ChartHealth.ConnectedWorker.Handler.Get [C:2] [TYPE Function]
def do_GET(self):
body = json.dumps({'result':{'id':42,'position_json':json.dumps(LAYOUT)}}).encode() if self.path.startswith('/api/v1/dashboard/') else HTML.encode()
self.send_response(200)
self.end_headers()
self.wfile.write(body)
# #endregion Test.ChartHealth.ConnectedWorker.Handler.Get
# #region Test.ChartHealth.ConnectedWorker.Handler.Post [C:2] [TYPE Function]
def do_POST(self):
payload = json.loads(self.rfile.read(int(self.headers['Content-Length'])))
chart = payload['form_data']['slice_id']
body = {'result':[{'data':[]}]} if chart == 6 else {'errors':[{'error_type':'GENERIC_DB_ENGINE_ERROR','message':f'Code: 386. DB::Exception: incompatible types in chart {chart}'}]}
self.send_response(200 if chart == 6 else 500)
self.end_headers()
self.wfile.write(json.dumps(body).encode())
# #endregion Test.ChartHealth.ConnectedWorker.Handler.Post
# #region Test.ChartHealth.ConnectedWorker.Handler.Log [C:1] [TYPE Function]
# @BRIEF Suppress external fixture request logs, which are not evidence.
def log_message(self, *_args):
pass
# #endregion Test.ChartHealth.ConnectedWorker.Handler.Log
# #endregion Test.ChartHealth.ConnectedWorker.Handler
# #region Test.ChartHealth.ConnectedWorker.Fixture [C:3] [TYPE Function]
@pytest.fixture
def health_context(seeded_execution,registry_engine,monkeypatch,tmp_path):
db = seeded_execution
run = db.get(ScenarioRun,'run-exec-0001')
run.status,run.phase = 'running','executing'
inputs = {'per_chart_timeout_seconds':2,'whole_timeout_seconds':30}
meta = {'logical_step_id':'health-check','tool':'browser','action':'assert_chart_health','action_inputs':inputs}
plan = {'scenario_revision_id':run.scenario_revision_id,'scenario_content_hash':run.scenario_content_hash,'steps':[meta]}
plan['plan_hash'] = sha256(json.dumps(plan,sort_keys=True,separators=(',',':')).encode()).hexdigest()
run.runner_plan = plan
step = {'scenario_run_id':run.id,'logical_step_id':'health-check','action':'assert_chart_health','step_meta':deepcopy(meta),
**{key:deepcopy(getattr(run,key)) for key in ('target_snapshot','execution_principal_fingerprint','live_execution_binding_ref','live_execution_binding_snapshot')}}
db.add(ScenarioStepRun(run_id=run.id,logical_step_id='health-check',step_position=1,attempt=1,status='running'))
claim_step(db,run.id,'health-check',worker_id='health-test',side_effect_key=None,idempotent=True,retry_safe=True,lease_seconds=120)
cap = claim_capacity(db,environment_id='preprod',environment_class='PREPROD',workload_class='browser',provider_id='browser',run_id=run.id,logical_step_id='health-check',ttl_seconds=120)
db.commit()
factory = sessionmaker(bind=registry_engine)
for name in ('traversal_store','chart_health_store','chart_health_summary','chart_health_tab_store'):
monkeypatch.setattr(f'src.services.dashboard_testing.execution.{name}.SessionLocal',factory)
storage,limits = DraftStorage(tmp_path),ChartHealthInput(**inputs)
journal = ChartHealthJournal(step,storage,limits,capacity_lease_id=cap['lease_id'])
return db,journal,step,storage,limits,cap['lease_id']
# #endregion Test.ChartHealth.ConnectedWorker.Fixture
# #region Test.ChartHealth.ConnectedWorker.Browser [C:3] [TYPE Function]
# @TEST_INVARIANT Real HTTP/DOM query errors do not wait for ready content and do not skip the healthy next tab.
@pytest.mark.asyncio
async def test_real_browser_five_errors_continue_other_tab_and_owned_complete_manifest(health_context):
from playwright.async_api import async_playwright
db,journal,step,storage,limits,cap = health_context
server = ThreadingHTTPServer(('127.0.0.1',0),Handler)
thread = Thread(target=server.serve_forever,daemon=True)
thread.start()
try:
async with async_playwright() as runtime:
browser = await runtime.chromium.launch(headless=True)
page = await browser.new_page()
await page.goto(f'http://127.0.0.1:{server.server_port}/dashboard')
result = await sweep_health(page,42,journal,limits)
await browser.close()
finally:
server.shutdown()
server.server_close()
assert result['status'] == 'failed' and result['complete'] is True, result
assert result['coverage'] == {'expected':6,'observed':6,'checked':6,'errored':5,'unresolved':0,'timeout':0,'unvisited':0,'complete':True}
manifest = json.loads(storage.retrieve(result['manifest_ref']))
assert [item['chart_name'] for item in manifest['charts']] == NAMES
assert [item['database_code'] for item in manifest['charts'][:5]] == ['386']*5
assert manifest['charts'][5]['status'] == 'healthy'
assert all(db.get(ScenarioArtifact,item['artifact_id']).owner_id == step['scenario_run_id'] for item in manifest['charts'])
resumed = ChartHealthJournal(step,storage,limits,capacity_lease_id=cap)
assert resumed.finish('passed','CHART_HEALTH_COMPLETE')['status'] == 'failed'
# #endregion Test.ChartHealth.ConnectedWorker.Browser
# #region Test.ChartHealth.ConnectedWorker.Partial [C:2] [TYPE Function]
# @TEST_INVARIANT A retained confirmed error dominates interrupted coverage without claiming the five unvisited charts were checked.
def test_one_error_then_interrupt_remains_failed_with_exact_partial_coverage(health_context):
from src.services.dashboard_testing.execution.providers.browser_health_manifest import parse_health_manifest
_,journal,_,storage,_,_ = health_context
journal.freeze_source(parse_health_manifest(LAYOUT))
journal.append_chart({'chart_id':1,'placement_id':'CHART-1','chart_name':'Debt','tab_path':['TAB-daily'],
'status':'failed','reason_code':'CHART_QUERY_FAILED','origin':'dom','observed_at':'2026-10-02T00:00:00Z',
'duration_seconds':0.0,'message':'Code: 386. DB::Exception: incompatible types','database_code':'386','terminal_state':'error'})
result = journal.finish('inconclusive','BROWSER_TRAVERSAL_CANCELLED')
assert result['status'] == 'failed' and result['complete'] is False
assert result['coverage']['errored'] == 1 and result['coverage']['checked'] == 1
assert result['coverage']['unvisited'] == 5
body = json.loads(storage.retrieve(result['manifest_ref']))
assert body['interruption_reason'] == 'BROWSER_TRAVERSAL_CANCELLED'
assert body['charts'][0]['chart_name'] == 'Debt'
# #endregion Test.ChartHealth.ConnectedWorker.Partial
# #region Test.ChartHealth.ConnectedWorker.Empty [C:2] [TYPE Function]
# @TEST_INVARIANT An unobserved source cannot PASS even when finish is called with requested_status passed.
def test_empty_health_observations_cannot_be_promoted_to_pass(health_context):
from src.services.dashboard_testing.execution.providers.browser_health_manifest import parse_health_manifest
_,journal,*_ = health_context
journal.freeze_source(parse_health_manifest(LAYOUT))
result = journal.finish('passed','CHART_HEALTH_COMPLETE')
assert result['status'] == 'inconclusive' and result['complete'] is False
assert result['coverage']['unvisited'] == 6 and result['coverage']['checked'] == 0
# #endregion Test.ChartHealth.ConnectedWorker.Empty
# #endregion Test.ChartHealth.ConnectedWorker

View File

@@ -0,0 +1,167 @@
# #region Test.ChartHealth.Independent [C:2] [TYPE Module] [SEMANTICS health,query-errors,async,attribution]
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Responses]
# @TEST_INVARIANT Unknown or pending query state cannot establish deterministic success or failure.
import asyncio
import json
from types import SimpleNamespace
from unittest.mock import Mock
import pytest
from src.services.dashboard_testing.query_failure import response_error
from src.services.dashboard_testing.execution.providers.browser_chart_responses import ChartResponseCollector
# #region Test.ChartHealth.Independent.Request [C:1] [TYPE Class]
class ExternalRequest:
# #region Test.ChartHealth.Independent.Request.Init [C:1] [TYPE Function]
def __init__(self, chart=11, metric='sales'):
self.url='http://fixture.invalid/api/v1/chart/data'
self.method='POST'
self.post_data_json={'form_data':{'slice_id':chart},'datasource':{'id':4,'type':'table'},
'queries':[{'filters':[{'col':'metric_selector','op':'IN','val':[metric]}]}]}
# #endregion Test.ChartHealth.Independent.Request.Init
# #endregion Test.ChartHealth.Independent.Request
# #region Test.ChartHealth.Independent.Response [C:1] [TYPE Class]
class ExternalResponse:
# #region Test.ChartHealth.Independent.Response.Init [C:1] [TYPE Function]
def __init__(self, request, payload, status=200):
self.request,self.url,self.status=request,request.url,status
self.raw=json.dumps(payload).encode()
# #endregion Test.ChartHealth.Independent.Response.Init
# #region Test.ChartHealth.Independent.Response.Body [C:1] [TYPE Function]
async def body(self):
return self.raw
# #endregion Test.ChartHealth.Independent.Response.Body
# #endregion Test.ChartHealth.Independent.Response
# #region Test.ChartHealth.Independent.Errors [C:2] [TYPE Function]
@pytest.mark.parametrize('status,payload',[(500,{'errors':[{'error_type':'GENERIC_DB_ENGINE_ERROR','message':'Code: 386. NO_COMMON_TYPE'}]}),
(200,{'result':[{'status':'failed','error':'Code: 386. NO_COMMON_TYPE'}]})])
@pytest.mark.asyncio
async def test_http_and_application_query_failures_are_attributed_to_exact_chart(status,payload):
collector=ChartResponseCollector(None,[11])
request=ExternalRequest()
collector.on_request(request)
await collector.read_response(ExternalResponse(request,payload,status))
error=collector.current_error(11)
assert error['message']=='Code: 386. NO_COMMON_TYPE'
assert error['attribution']['http_status']==status and error['attribution']['load_generation']==1
assert collector.current_error(12) is None
# #endregion Test.ChartHealth.Independent.Errors
# #region Test.ChartHealth.Independent.Healthy [C:2] [TYPE Function]
@pytest.mark.parametrize('payload',[{'result':[{'data':[]}]},{'result':[{'data':[{'value':0}]}]}])
def test_successful_empty_and_zero_are_not_query_errors(payload):
assert response_error(payload,200) is None
# #endregion Test.ChartHealth.Independent.Healthy
# #region Test.ChartHealth.Independent.AsyncPending [C:2] [TYPE Function]
@pytest.mark.asyncio
async def test_accepted_async_job_is_pending_until_terminal_owned_result():
collector=ChartResponseCollector(None,[11])
request=ExternalRequest()
collector.on_request(request)
await collector.read_response(ExternalResponse(request,{'result':[{'job_id':'job-current','status':'pending'}]},202))
assert collector.latest[11] in collector.pending
assert collector.current_error(11) is None
# #endregion Test.ChartHealth.Independent.AsyncPending
# #region Test.ChartHealth.Independent.AsyncResult [C:2] [TYPE Function]
# @TEST_INVARIANT A done notification is not completed data; only its exact owned result response clears waiting.
@pytest.mark.asyncio
async def test_async_done_waits_for_owned_result_payload_not_foreign_result():
collector = ChartResponseCollector(None, [11])
request = ExternalRequest()
collector.on_request(request)
await collector.read_response(ExternalResponse(request, {'result': [{'job_id': 'job-current', 'status': 'pending'}]}, 202))
event = ExternalResponse(ExternalRequest(99), {'result': [{'job_id': 'job-current', 'status': 'done', 'result_url': '/api/v1/chart/data/owned-result'}]})
event.url = 'http://fixture.invalid/api/v1/async_event/'
await collector.read_response(event)
assert collector.latest[11] in collector.pending
result_request = ExternalRequest(99)
result_request.url = 'http://fixture.invalid/api/v1/chart/data/owned-result'
result_request.method = 'GET'
collector.on_request(result_request)
await collector.read_response(ExternalResponse(result_request, {'result': [{'data': [{'value': 0}]}]}))
assert collector.latest[11] not in collector.pending
assert collector.current_error(11) is None
# #endregion Test.ChartHealth.Independent.AsyncResult
# #region Test.ChartHealth.Independent.Malformed [C:2] [TYPE Function]
@pytest.mark.asyncio
async def test_malformed_http200_is_unresolved_not_confirmed_query_failure():
collector=ChartResponseCollector(None,[11])
request=ExternalRequest()
collector.on_request(request)
await collector.read_response(ExternalResponse(request,{'result':None},200))
assert collector.current_error(11) is None
assert collector.latest[11] in collector.pending
# #endregion Test.ChartHealth.Independent.Malformed
# #region Test.ChartHealth.Independent.Retry [C:2] [TYPE Function]
@pytest.mark.asyncio
async def test_new_same_filter_retry_supersedes_late_old_error():
collector=ChartResponseCollector(None,[11])
first,retry=ExternalRequest(),ExternalRequest()
collector.on_request(first)
collector.on_request(retry)
await collector.read_response(ExternalResponse(first,{'errors':[{'message':'stale Code: 386'}]},500))
assert collector.current_error(11) is None and collector.latest[11] in collector.pending
await collector.read_response(ExternalResponse(retry,{'result':[{'data':[]}]},200))
assert collector.current_error(11) is None and collector.latest[11] not in collector.pending
# #endregion Test.ChartHealth.Independent.Retry
# #region Test.ChartHealth.Independent.Close [C:2] [TYPE Function]
@pytest.mark.asyncio
async def test_close_removes_listeners_and_drains_blocked_external_response_body():
page=SimpleNamespace(on=Mock(),remove_listener=Mock())
collector=ChartResponseCollector(page,[11])
request=ExternalRequest()
collector.on_request(request)
response=ExternalResponse(request,{'result':[{'data':[]}]})
event=asyncio.Event()
# #region Test.ChartHealth.Independent.Close.ExternalBody [C:1] [TYPE Function]
async def external_body():
await event.wait()
return response.raw
# #endregion Test.ChartHealth.Independent.Close.ExternalBody
response.body=external_body
collector.start()
collector.on_response(response)
await asyncio.sleep(0)
tasks=list(collector.tasks)
await collector.close()
assert tasks and all(task.done() for task in tasks)
assert page.remove_listener.call_count==3 and not collector.requests and not collector.pending
assert {call.args[0] for call in page.remove_listener.call_args_list} == {'request', 'response', 'requestfailed'}
# #endregion Test.ChartHealth.Independent.Close
# #region Test.ChartHealth.Independent.TransportFailure [C:2] [TYPE Function]
# @TEST_INVARIANT Current transport failure is unresolved evidence; stale and foreign failures cannot contaminate its chart.
def test_transport_failure_stays_unconfirmed_and_obeys_latest_request_owner():
collector = ChartResponseCollector(None, [11])
old, current, foreign = ExternalRequest(), ExternalRequest(), ExternalRequest(99)
old.failure = current.failure = foreign.failure = 'net::ERR_CONNECTION_RESET'
collector.on_request(old)
collector.on_request(current)
collector.on_request(foreign)
collector.on_request_failed(old)
collector.on_request_failed(foreign)
assert collector.current_error(11) is None
collector.on_request_failed(current)
error = collector.current_error(11)
assert error['confirmed'] is False and error['message'] == 'net::ERR_CONNECTION_RESET'
assert error['attribution']['load_generation'] == 2
assert collector.current_error(99) is None
# #endregion Test.ChartHealth.Independent.TransportFailure
# #endregion Test.ChartHealth.Independent

View File

@@ -0,0 +1,70 @@
# #region Test.ChartHealth.LegacyWorker [C:3] [TYPE Module] [SEMANTICS legacy,registry,normalized,pin,journal]
# @TEST_INVARIANT Pinned0386/0387 channels and journal input digests retain their original pre-health canonical fields.
# @RELATION BINDS_TO -> [ScenarioExecution.Traversal.Inputs.LegacyTabs]
# @RATIONALE Immutable54bdbcc4 input-module SHA256=1b6a3f5875f886cebd8e0a9d9b57eb69e5d44e51c6689352709cc5f8f398a87e; literal defaults below are its legacy DTO.
import json
from copy import deepcopy
from hashlib import sha256
import pytest
from sqlalchemy.orm import sessionmaker
from src.models.scenario_traversal import ScenarioTraversal
from src.models.scenario_run import ScenarioRun
from src.services.dashboard_testing.execution.providers.browser_traversal_inputs import parse_traversal_input,AllTabsTraversalInput
from src.services.dashboard_testing.execution.providers.browser_pinned_inputs import resolve_pinned_browser_inputs
from src.services.dashboard_testing.execution.traversal_tabs_store import TabsJournal
from test_chart_health_connected_worker import health_context # noqa: F401
LEGACY = {'per_tab_timeout_seconds':70,'whole_timeout_seconds':7200,'max_bytes':268435456}
# #region Test.ChartHealth.LegacyWorker.Channel [C:2] [TYPE Function]
@pytest.mark.parametrize('version',['038.6.0','038.7.0'])
def test_legacy_normalized_channel_has_exact_pre_health_fields(version):
assert parse_traversal_input('navigate_tabs',{},registry_version=version) == LEGACY
with pytest.raises(ValueError):
parse_traversal_input('navigate_tabs',{'chart_health':True},registry_version=version)
# #endregion Test.ChartHealth.LegacyWorker.Channel
# #region Test.ChartHealth.LegacyWorker.Pinned [C:3] [TYPE Function]
@pytest.mark.parametrize('version',['038.6.0','038.7.0'])
def test_real_pinned_projection_and_journal_resume_preserve_old_digest(health_context,registry_engine,monkeypatch,version):
db,_,step,storage,_,capacity = health_context
run = db.get(ScenarioRun,step['scenario_run_id'])
meta = {'logical_step_id':'health-check','tool':'browser','action':'navigate_tabs','action_inputs':{}}
body = {'scenario_revision_id':run.scenario_revision_id,'scenario_content_hash':run.scenario_content_hash,
'action_registry_version':version,'steps':[meta]}
body['plan_hash'] = sha256(json.dumps(body,sort_keys=True,separators=(',',':')).encode()).hexdigest()
run.runner_plan = deepcopy(body)
db.commit()
before = json.dumps(body,sort_keys=True)
step = {**step,'action':'navigate_tabs','step_meta':deepcopy(meta)}
monkeypatch.setattr('src.core.database.SessionLocal',sessionmaker(bind=registry_engine))
inputs = resolve_pinned_browser_inputs(step)
assert inputs == LEGACY
# Existing unbound fixture journal belongs to another action; this is a new owned logical step attempt.
db.query(ScenarioTraversal).filter_by(run_id=run.id).delete()
db.commit()
journal = TabsJournal(step,storage,AllTabsTraversalInput(**inputs),capacity_lease_id=capacity)
row = db.get(ScenarioTraversal,journal.id)
assert row.input_digest == sha256(b'{"max_bytes":268435456,"per_tab_timeout_seconds":70,"whole_timeout_seconds":7200}').hexdigest()
resumed = TabsJournal(step,storage,AllTabsTraversalInput(**inputs),capacity_lease_id=capacity)
assert resumed.id == journal.id and row.action == 'navigate_tabs'
db.refresh(run)
assert json.dumps(run.runner_plan,sort_keys=True) == before
# #endregion Test.ChartHealth.LegacyWorker.Pinned
# #region Test.ChartHealth.LegacyWorker.LayoutVersion [C:2] [TYPE Function]
# @TEST_INVARIANT Superset's actual v2 layout header is metadata, not a node; exact tab/chart identity remains authoritative.
def test_native_superset_layout_version_header_is_not_treated_as_node():
from src.services.dashboard_testing.execution.providers.browser_health_manifest import parse_health_manifest
source = parse_health_manifest({'DASHBOARD_VERSION_KEY':'v2',
'ROOT_ID':{'type':'ROOT','children':['TABS-main']},
'TABS-main':{'type':'TABS','children':['TAB-health']},
'TAB-health':{'type':'TAB','children':['CHART-19'],'meta':{'text':'Actual tab'}},
'CHART-19':{'type':'CHART','children':[],'meta':{'chartId':19,'sliceName':'Actual chart'}}})
assert source['source_total'] == 1
assert source['placements'] == [{'id':'CHART-19','chart_id':19,'name':'Actual chart','tab_path':['TAB-health']}]
# #endregion Test.ChartHealth.LegacyWorker.LayoutVersion
# #endregion Test.ChartHealth.LegacyWorker

View File

@@ -0,0 +1,188 @@
# #region Test.ChartHealth.MatrixPlacements [C:2] [TYPE Module] [SEMANTICS health,placement,tabs,timeout,partial]
# @TEST_INVARIANT Repeated chart IDs retain separate placement/tab identities; inaccessible or timed-out placements cannot erase an earlier confirmed failure.
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Sweep]
import json
import asyncio
import gc
from copy import deepcopy
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from threading import Thread
from unittest.mock import AsyncMock, Mock
import pytest
from test_chart_health_connected_worker import health_context # noqa: F401
from src.services.dashboard_testing.execution.providers.browser_health_sweep import sweep_health
REPEATED_LAYOUT = {'ROOT_ID': {'type': 'ROOT', 'children': ['TABS-main']},
'TABS-main': {'type': 'TABS', 'children': ['TAB-first', 'TAB-second']},
'TAB-first': {'type': 'TAB', 'meta': {'text': 'Same title'}, 'children': ['CHART-first']},
'TAB-second': {'type': 'TAB', 'meta': {'text': 'Same title'}, 'children': ['CHART-second']},
'CHART-first': {'type': 'CHART', 'meta': {'chartId': 11, 'sliceName': 'Repeated chart'}},
'CHART-second': {'type': 'CHART', 'meta': {'chartId': 11, 'sliceName': 'Repeated chart'}}}
# #region Test.ChartHealth.MatrixPlacements.Document [C:1] [TYPE Function]
def external_document(mode):
second_chart = 11 if mode == 'repeat' else 12
disabled = 'disabled' if mode == 'unopenable' else ''
return f'''<button role="tab" id="TABS-main-tab-TAB-first" aria-controls="first" aria-selected="false"
onclick="activate('first',11,this)">Same title</button>
<button {disabled} role="tab" id="TABS-main-tab-TAB-second" aria-controls="second" aria-selected="false"
onclick="activate('second',{second_chart},this)">Same title</button>
<div id="first" role="tabpanel" hidden style="min-height:120px"><div id="chart-id-11" style="min-height:120px"><div class="loading">loading</div></div></div>
<div id="second" role="tabpanel" hidden style="min-height:120px"><div id="chart-id-{second_chart}" style="min-height:120px"><div class="loading">loading</div></div></div>
<script>function activate(id, chart, control) {{
document.querySelectorAll('[role=tab]').forEach(t=>t.setAttribute('aria-selected','false'));
document.querySelectorAll('[role=tabpanel]').forEach(t=>t.hidden=true);
control.setAttribute('aria-selected','true');document.getElementById(id).hidden=false;
if (id==='second' && '{mode}'==='renderer_timeout') return;
fetch('/api/v1/chart/data',{{method:'POST',headers:{{'Content-Type':'application/json'}},
body:JSON.stringify({{form_data:{{slice_id:chart}},datasource:{{id:4,type:'table'}},queries:[{{filters:[{{col:'tab_scope',op:'IN',val:[id]}}]}}]}})}})
.then(r=>r.json()).then(data=>{{const root=document.getElementById(id).querySelector('#chart-id-'+chart);
root.innerHTML=id==='first'?'<div class="alert-danger">'+data.errors[0].message+'</div>':'<table><tbody></tbody></table>';}});
}}</script>'''
# #endregion Test.ChartHealth.MatrixPlacements.Document
# #region Test.ChartHealth.MatrixPlacements.Handler [C:1] [TYPE Class]
class ExternalHandler(BaseHTTPRequestHandler):
# #region Test.ChartHealth.MatrixPlacements.Handler.Get [C:1] [TYPE Function]
def do_GET(self):
mode = self.server.fixture_mode
layout = deepcopy(REPEATED_LAYOUT)
if mode != 'repeat':
layout['CHART-second']['meta']['chartId'] = 12
body = json.dumps({'result': {'id': 42, 'position_json': layout}}).encode() if self.path.startswith('/api/v1/dashboard/') else external_document(mode).encode()
self.send_response(200)
self.end_headers()
self.wfile.write(body)
# #endregion Test.ChartHealth.MatrixPlacements.Handler.Get
# #region Test.ChartHealth.MatrixPlacements.Handler.Post [C:1] [TYPE Function]
def do_POST(self):
request = json.loads(self.rfile.read(int(self.headers['Content-Length'])))
tab = request['queries'][0]['filters'][0]['val'][0]
payload = {'errors': [{'error_type': 'GENERIC_DB_ENGINE_ERROR', 'message': 'Code: 386. DB::Exception: NO_COMMON_TYPE'}]} if tab == 'first' else {'result': [{'data': []}]}
self.send_response(500 if tab == 'first' else 200)
self.end_headers()
self.wfile.write(json.dumps(payload).encode())
# #endregion Test.ChartHealth.MatrixPlacements.Handler.Post
# #region Test.ChartHealth.MatrixPlacements.Handler.Log [C:1] [TYPE Function]
def log_message(self, *_args):
pass
# #endregion Test.ChartHealth.MatrixPlacements.Handler.Log
# #endregion Test.ChartHealth.MatrixPlacements.Handler
# #region Test.ChartHealth.MatrixPlacements.Execute [C:1] [TYPE Function]
async def execute_external_fixture(context, mode):
from playwright.async_api import async_playwright
_, journal, _, storage, limits, _ = context
server = ThreadingHTTPServer(('127.0.0.1', 0), ExternalHandler)
server.fixture_mode = mode
Thread(target=server.serve_forever, daemon=True).start()
try:
async with async_playwright() as runtime:
browser = await runtime.chromium.launch(headless=True)
try:
page = await browser.new_page()
await page.goto(f'http://127.0.0.1:{server.server_port}/{mode}')
result = await sweep_health(page, 42, journal, limits)
finally:
await browser.close()
finally:
server.shutdown()
server.server_close()
return result, json.loads(storage.retrieve(result['manifest_ref']))
# #endregion Test.ChartHealth.MatrixPlacements.Execute
# #region Test.ChartHealth.MatrixPlacements.Repeated [C:2] [TYPE Function]
# @TEST_INVARIANT Same chart11 appears twice and newest tab-specific successful query cannot rewrite the already committed first-tab failure.
@pytest.mark.asyncio
async def test_same_chart_id_two_tabs_has_distinct_owned_placements(health_context):
result, manifest = await execute_external_fixture(health_context, 'repeat')
assert result['status'] == 'failed' and result['complete'] is True
assert result['coverage'] == {'expected': 2, 'observed': 2, 'checked': 2, 'errored': 1,
'unresolved': 0, 'timeout': 0, 'unvisited': 0, 'complete': True}
first, second = manifest['charts']
assert [(item['chart_id'], item['placement_id'], item['tab_path'], item['status']) for item in [first, second]] == [
(11, 'CHART-first', ['TAB-first'], 'failed'), (11, 'CHART-second', ['TAB-second'], 'healthy')]
assert first['database_code'] == '386'
assert first['artifact_id'] != second['artifact_id'] and first['sha256'] != second['sha256']
assert first['run_id'] == second['run_id'] == health_context[2]['scenario_run_id']
assert first['attempt'] == second['attempt'] == 1
# #endregion Test.ChartHealth.MatrixPlacements.Repeated
# #region Test.ChartHealth.MatrixPlacements.Unopenable [C:2] [TYPE Function]
# @TEST_INVARIANT A disabled second tab after a retained Code386 produces unresolved coverage and cannot downgrade confirmed FAILED.
@pytest.mark.asyncio
async def test_unopenable_tab_after_confirmed_error_remains_failed_partial(health_context):
result, manifest = await execute_external_fixture(health_context, 'unopenable')
assert result['status'] == 'failed' and result['complete'] is False
assert result['coverage'] == {'expected': 2, 'observed': 2, 'checked': 1, 'errored': 1,
'unresolved': 1, 'timeout': 1, 'unvisited': 0, 'complete': False}
assert [(item['placement_id'], item['tab_path'], item['status']) for item in manifest['charts']] == [
('CHART-first', ['TAB-first'], 'failed'), ('CHART-second', ['TAB-second'], 'inconclusive')]
assert manifest['charts'][0]['database_code'] == '386'
assert manifest['charts'][1]['reason_code'] == 'CHART_LOAD_TIMEOUT'
# #endregion Test.ChartHealth.MatrixPlacements.Unopenable
# #region Test.ChartHealth.MatrixPlacements.RenderTimeout [C:2] [TYPE Function]
# @TEST_INVARIANT An activated second panel that stays loading exhausts only its own budget and records the exact unchecked placement.
@pytest.mark.asyncio
async def test_renderer_timeout_after_confirmed_error_remains_failed_partial(health_context):
loop = asyncio.get_running_loop()
previous_handler = loop.get_exception_handler()
unhandled = []
# #region Test.ChartHealth.MatrixPlacements.RenderTimeout.Exception [C:1] [TYPE Function]
def observe_unhandled(active_loop, context):
unhandled.append(context)
active_loop.default_exception_handler(context) # Record normally; never hide the actual warning.
# #endregion Test.ChartHealth.MatrixPlacements.RenderTimeout.Exception
loop.set_exception_handler(observe_unhandled)
try:
result, manifest = await execute_external_fixture(health_context, 'renderer_timeout')
await asyncio.sleep(0.05)
gc.collect()
await asyncio.sleep(0.05)
gc.collect()
finally:
loop.set_exception_handler(previous_handler)
assert result['status'] == 'failed' and result['complete'] is False
assert result['coverage'] == {'expected': 2, 'observed': 2, 'checked': 1, 'errored': 1,
'unresolved': 1, 'timeout': 1, 'unvisited': 0, 'complete': False}
assert manifest['charts'][0]['status'] == 'failed' and manifest['charts'][0]['database_code'] == '386'
assert manifest['charts'][1]['chart_id'] == 12 and manifest['charts'][1]['tab_path'] == ['TAB-second']
assert manifest['charts'][1]['status'] == 'inconclusive' and manifest['charts'][1]['reason_code'] == 'CHART_LOAD_TIMEOUT'
assert unhandled == [], [str(item.get('exception') or item.get('message')) for item in unhandled]
# #endregion Test.ChartHealth.MatrixPlacements.RenderTimeout
# #region Test.ChartHealth.MatrixPlacements.ExpiredRpc [C:2] [TYPE Function]
# @TEST_INVARIANT If the exact monotonic budget expires during control checking, no new Playwright RPC may start beyond that deadline.
@pytest.mark.asyncio
async def test_expired_remaining_budget_does_not_start_one_ms_browser_rpc(health_context, monkeypatch):
from src.services.dashboard_testing.execution.providers.browser_chart_health import observe_chart
from src.services.dashboard_testing.execution.providers.browser_chart_responses import ChartResponseCollector
from src.services.dashboard_testing.execution.providers.browser_health_manifest import parse_health_manifest
_, journal, *_ = health_context
layout = deepcopy(REPEATED_LAYOUT)
layout['CHART-second']['meta']['chartId'] = 12
source = parse_health_manifest(layout)
journal.freeze_source(source)
journal.append_chart({'chart_id': 11, 'placement_id': 'CHART-first', 'chart_name': 'Repeated chart',
'tab_path': ['TAB-first'], 'status': 'failed', 'reason_code': 'CHART_QUERY_FAILED', 'origin': 'dom',
'observed_at': '2026-10-02T00:00:00Z', 'duration_seconds': 0, 'message': 'Code: 386. NO_COMMON_TYPE',
'database_code': '386', 'terminal_state': 'error'})
# External monotonic clock: 2s deadline; check-control crosses1.9→2.1s.
clock = Mock(side_effect=[0.0, 0.0, 1.9, 2.1, 2.1, 2.1])
monkeypatch.setattr('src.services.dashboard_testing.execution.providers.browser_chart_health.monotonic', clock)
chart = Mock()
chart.wait_for = AsyncMock()
chart.count = AsyncMock(return_value=1)
chart.scroll_into_view_if_needed = AsyncMock()
chart.evaluate = AsyncMock(return_value={'error': '', 'ready': False, 'loading': True})
panel = Mock()
panel.locator.return_value = chart
value = await observe_chart(None, panel, source['placements'][1], 2, ChartResponseCollector(None, [11, 12]), journal)
assert value['status'] == 'inconclusive' and value['reason_code'] == 'CHART_LOAD_TIMEOUT'
assert chart.evaluate.await_count == 0
journal.append_chart(value)
assert journal.finish('passed', 'CHART_HEALTH_COMPLETE')['status'] == 'failed'
# #endregion Test.ChartHealth.MatrixPlacements.ExpiredRpc
# #endregion Test.ChartHealth.MatrixPlacements

View File

@@ -0,0 +1,82 @@
# #region Test.ChartHealth.NativeCanary [C:4] [TYPE Module] [SEMANTICS actual,superset,clickhouse,error,recovery]
# @TEST_INVARIANT Actual Superset/ClickHouse query errors yield five386 diagnostics plus checked healthy other-tab; own metric recovery yields health PASS.
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Sweep]
import json
import os
from pathlib import Path
from copy import deepcopy
from hashlib import sha256
import pytest
from src.models.scenario_run import ScenarioRun
from src.models.scenario_traversal import ScenarioTraversal
from src.services.dashboard_testing.execution.chart_health_store import ChartHealthJournal
from src.services.dashboard_testing.execution.providers.browser_health_sweep import sweep_health
from src.services.dashboard_testing.execution.providers.browser_traversal_inputs import ChartHealthInput
from test_chart_health_connected_worker import health_context # noqa: F401
# #region Test.ChartHealth.NativeCanary.Private [C:2] [TYPE Function]
# @INVARIANT Local lab password is consumed only for login and never printed or included in the report.
def lab_password():
for line in Path(os.environ['CHART_HEALTH_NATIVE_ENV_FILE']).read_text().splitlines():
if line.startswith('FIXTURE_PASSWORD='):
return line.split('=',1)[1].strip().strip('\"\'')
raise RuntimeError('CHART_HEALTH_CANARY_PASSWORD_UNAVAILABLE')
# #endregion Test.ChartHealth.NativeCanary.Private
# #region Test.ChartHealth.NativeCanary.Journal [C:3] [TYPE Function]
# @POST Real owned fixture projection names the actual isolated PREPROD dashboard; it makes no public authoring/release claim.
def native_journal(context, dashboard_id):
db,_,step,storage,_,capacity = context
run = db.get(ScenarioRun,step['scenario_run_id'])
limits = ChartHealthInput(per_chart_timeout_seconds=70,whole_timeout_seconds=600)
meta = {**step['step_meta'],'action_inputs':{'per_chart_timeout_seconds':70,'whole_timeout_seconds':600}}
body = {key:deepcopy(value) for key,value in run.runner_plan.items() if key != 'plan_hash'}
body['steps'],body['action_registry_version'] = [meta],'038.8.0'
body['plan_hash'] = sha256(json.dumps(body,sort_keys=True,separators=(',',':')).encode()).hexdigest()
target = {**run.target_snapshot,'dashboard_id':dashboard_id}
run.runner_plan,run.target_snapshot = body,target
db.query(ScenarioTraversal).filter_by(run_id=run.id).delete()
db.commit()
step = {**step,'step_meta':meta,'target_snapshot':deepcopy(target)}
return ChartHealthJournal(step,storage,limits,capacity_lease_id=capacity),limits
# #endregion Test.ChartHealth.NativeCanary.Journal
# #region Test.ChartHealth.NativeCanary.Run [C:4] [TYPE Function]
# @POST Source-backed typed health is retained for both phases without mutating any source data or finance assets.
@pytest.mark.integration
@pytest.mark.asyncio
async def test_actual_clickhouse386_and_own_metric_recovery(health_context):
if not os.getenv('CHART_HEALTH_NATIVE_DASHBOARD_ID'):
pytest.skip('explicit isolated native canary configuration required')
from playwright.async_api import async_playwright
dashboard_id = int(os.environ['CHART_HEALTH_NATIVE_DASHBOARD_ID'])
journal,limits = native_journal(health_context,dashboard_id)
async with async_playwright() as runtime:
browser = await runtime.chromium.launch(headless=True)
try:
page = await browser.new_page(viewport={'width':1600,'height':1000})
await page.goto(os.environ['CHART_HEALTH_NATIVE_URL']+'/login/',wait_until='domcontentloaded')
await page.locator('#username').fill('admin')
await page.locator('#password').fill(lab_password())
await page.locator('input[type="submit"],button[type="submit"]').click()
await page.wait_for_url(lambda url:'/login' not in url,timeout=30000)
await page.goto(os.environ['CHART_HEALTH_NATIVE_URL']+f'/superset/dashboard/{dashboard_id}/?force=true',wait_until='domcontentloaded')
result = await sweep_health(page,dashboard_id,journal,limits)
finally:
await browser.close()
output = Path(os.environ['CHART_HEALTH_NATIVE_OUTPUT'])
output.mkdir(parents=True,exist_ok=False)
(output/'report.json').write_text(json.dumps({'scope':'Actual Superset/ClickHouse observer+owned journal integration, not public release lifecycle',
'phase':os.environ['CHART_HEALTH_NATIVE_PHASE'],'dashboard_id':dashboard_id,'result':result},indent=2))
assert result['complete'] is True and result['coverage']['checked'] == 6, result
if os.environ['CHART_HEALTH_NATIVE_PHASE'] == 'error':
assert result['status'] == 'failed' and result['coverage']['errored'] == 5, result
assert [item['database_code'] for item in result['chart_health']['errors']] == ['386']*5
assert result['chart_health']['charts'][5]['status'] == 'healthy'
else:
assert result['status'] == 'passed' and result['coverage']['errored'] == 0, result
# #endregion Test.ChartHealth.NativeCanary.Run
# #endregion Test.ChartHealth.NativeCanary

View File

@@ -0,0 +1,47 @@
# #region Test.ChartHealth.NativeDOMProbe [C:3] [TYPE Module] [SEMANTICS native,diagnostic,DOM,identity]
# @BRIEF Inventory real isolated Superset error/healthy card DOM; this is a diagnostic, not health acceptance.
import json
import os
from pathlib import Path
import pytest
from test_chart_health_native_canary import lab_password
SCRIPT = '''() => {
const labels=['Canary Debt','Canary Cash','Canary Margin','Canary Forecast','Canary Reserve','Canary Healthy'];
const attrs=e=>Object.fromEntries([...e.attributes].map(a=>[a.name,a.value]));
return labels.map(label=>({label,matches:[...document.querySelectorAll('*')].filter(e=>e.children.length===0 && e.textContent.trim()===label)
.map(e=>{const ancestors=[];let node=e;for(let i=0;i<7 && node;i++,node=node.parentElement){ancestors.push({tag:node.tagName,attributes:attrs(node),text:node.innerText.slice(0,10000),html:node.outerHTML.slice(0,18000)})}return ancestors})}));
}'''
# #region Test.ChartHealth.NativeDOMProbe.Run [C:3] [TYPE Function]
# @POST Both phases retain bounded actual card ancestors without observer fallbacks or synthetic query outcomes.
@pytest.mark.integration
@pytest.mark.asyncio
async def test_native_saved_card_structure_diagnostic_only():
if not os.getenv('CHART_HEALTH_NATIVE_DASHBOARD_ID'):
pytest.skip('explicit native diagnostic configuration required')
from playwright.async_api import async_playwright
dashboard_id = int(os.environ['CHART_HEALTH_NATIVE_DASHBOARD_ID'])
async with async_playwright() as runtime:
browser = await runtime.chromium.launch(headless=True)
try:
page = await browser.new_page(viewport={'width':1600,'height':1000})
await page.goto(os.environ['CHART_HEALTH_NATIVE_URL']+'/login/',wait_until='domcontentloaded')
await page.locator('#username').fill('admin')
await page.locator('#password').fill(lab_password())
await page.locator('input[type="submit"],button[type="submit"]').click()
await page.wait_for_url(lambda url:'/login' not in url,timeout=30000)
await page.goto(os.environ['CHART_HEALTH_NATIVE_URL']+f'/superset/dashboard/{dashboard_id}/?force=true',wait_until='domcontentloaded')
await page.get_by_text('Canary Debt',exact=True).wait_for(state='visible',timeout=30000)
await page.wait_for_timeout(3000)
report = await page.evaluate(SCRIPT)
finally:
await browser.close()
output = Path(os.environ['CHART_HEALTH_NATIVE_OUTPUT'])
output.mkdir(parents=True,exist_ok=False)
(output/'dom-diagnostic.json').write_text(json.dumps({'scope':'native card DOM inventory only, not health PASS',
'phase':os.environ['CHART_HEALTH_NATIVE_PHASE'],'dashboard_id':dashboard_id,'cards':report},indent=2))
assert len(report) == 6 and all(item['matches'] for item in report)
# #endregion Test.ChartHealth.NativeDOMProbe.Run
# #endregion Test.ChartHealth.NativeDOMProbe

View File

@@ -0,0 +1,76 @@
# #region Test.ChartHealth.NativeScope [C:2] [TYPE Module] [SEMANTICS health,native,scope,error,identity]
# @TEST_INVARIANT Native saved-card identity and chart body, never header SVG or foreign cards, establish health; attributed HTTP errors need no mounted renderer.
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Scope]
import pytest
from test_chart_health_connected_worker import health_context # noqa: F401
from test_chart_health_owned_evaluation_authority import LAYOUT
from test_chart_health_independent_authority import ExternalRequest, ExternalResponse
from src.services.dashboard_testing.execution.providers.browser_health_manifest import parse_health_manifest
from src.services.dashboard_testing.execution.providers.browser_chart_health import observe_chart
from src.services.dashboard_testing.execution.providers.browser_chart_responses import ChartResponseCollector
ERROR_CARD = '''<div data-test="chart-grid-component" data-test-chart-id="11" data-test-chart-name="Same title">
<header><svg width="24" height="24"></svg></header><div class="dashboard-chart" style="height:120px">
<div class="alert-danger">Code: 386. NO_COMMON_TYPE</div></div></div>'''
HEADER_ONLY = '''<div data-test="chart-grid-component" data-test-chart-id="11" data-test-chart-name="Same title">
<header><svg width="24" height="24"></svg></header><div class="dashboard-chart" style="height:120px"></div></div>'''
FOREIGN_CARD = '''<div data-test="chart-grid-component" data-test-chart-id="99" data-test-chart-name="Same title">
<div class="dashboard-chart" style="height:120px"><table><tbody><tr><td>0</td></tr></tbody></table></div></div>'''
# #region Test.ChartHealth.NativeScope.Observe [C:2] [TYPE Function]
# @TEST_INVARIANT Literal native error cards without ChartRenderer ID fail; header-only, duplicate, foreign and wrong-name observations cannot PASS.
@pytest.mark.parametrize('html,status,reason', [
(ERROR_CARD, 'failed', 'CHART_QUERY_FAILED'),
(HEADER_ONLY, 'inconclusive', 'CHART_LOAD_TIMEOUT'),
(ERROR_CARD + ERROR_CARD, 'inconclusive', 'CHART_PLACEMENT_AMBIGUOUS'),
(FOREIGN_CARD, 'inconclusive', 'CHART_LOAD_TIMEOUT'),
(ERROR_CARD.replace('Same title', 'Wrong title'), 'inconclusive', 'CHART_PLACEMENT_IDENTITY_MISMATCH'),
])
@pytest.mark.asyncio
async def test_native_body_scope_rejects_unowned_or_unready_visuals(health_context, html, status, reason):
from playwright.async_api import async_playwright
_, journal, *_ = health_context
source = parse_health_manifest(LAYOUT)
journal.freeze_source(source)
async with async_playwright() as runtime:
browser = await runtime.chromium.launch(headless=True)
try:
page = await browser.new_page()
await page.set_content(html)
value = await observe_chart(page, page, source['placements'][0], 2, ChartResponseCollector(page, [11]), journal)
finally:
await browser.close()
assert value['chart_id'] == 11 and value['placement_id'] == 'CHART-a'
assert value['status'] == status and value['reason_code'] == reason
if status == 'failed':
assert value['database_code'] == '386'
journal.append_chart(value)
assert journal.finish('inconclusive', 'CHART_HEALTH_INTERRUPTED')['complete'] is False
# #endregion Test.ChartHealth.NativeScope.Observe
# #region Test.ChartHealth.NativeScope.EarlyHttp [C:2] [TYPE Function]
# @TEST_INVARIANT A latest server-attributed Code386 HTTP failure fails before a missing ChartRenderer can consume its load deadline.
@pytest.mark.asyncio
async def test_owned_http_error_does_not_wait_for_missing_chart_renderer(health_context):
from playwright.async_api import async_playwright
_, journal, *_ = health_context
source = parse_health_manifest(LAYOUT)
journal.freeze_source(source)
collector = ChartResponseCollector(None, [11])
request = ExternalRequest()
collector.on_request(request)
await collector.read_response(ExternalResponse(request, {'errors': [{'message': 'Code: 386. NO_COMMON_TYPE'}]}, 500))
async with async_playwright() as runtime:
browser = await runtime.chromium.launch(headless=True)
try:
page = await browser.new_page()
await page.set_content('<div>Renderer absent</div>')
value = await observe_chart(page, page, source['placements'][0], 2, collector, journal)
finally:
await browser.close()
assert value['status'] == 'failed' and value['reason_code'] == 'CHART_QUERY_FAILED'
assert value['origin'] == 'http_response' and value['http_status'] == 500
assert value['request_identity'] == '1' and value['load_generation'] == 1
assert value['database_code'] == '386' and value['duration_seconds'] < 1
# #endregion Test.ChartHealth.NativeScope.EarlyHttp
# #endregion Test.ChartHealth.NativeScope

View File

@@ -0,0 +1,73 @@
# #region Test.ChartHealth.NestedDom [C:2] [TYPE Module] [SEMANTICS health,nested,tabs,offscreen,placement]
# @TEST_INVARIANT Nested paths and below-viewport charts use exact identities and remain independently retained.
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Sweep]
import json
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from threading import Thread
import pytest
from test_chart_health_connected_worker import health_context # noqa: F401
from src.services.dashboard_testing.execution.providers.browser_health_sweep import sweep_health
LAYOUT = {'ROOT_ID': {'type': 'ROOT', 'children': ['TABS-outer']},
'TABS-outer': {'type': 'TABS', 'children': ['TAB-parent']},
'TAB-parent': {'type': 'TAB', 'meta': {'text': 'Repeated title'}, 'children': ['CHART-top', 'TABS-inner']},
'CHART-top': {'type': 'CHART', 'meta': {'chartId': 11, 'sliceName': 'Same title'}},
'TABS-inner': {'type': 'TABS', 'children': ['TAB-child']},
'TAB-child': {'type': 'TAB', 'meta': {'text': 'Repeated title'}, 'children': ['CHART-bottom']},
'CHART-bottom': {'type': 'CHART', 'meta': {'chartId': 12, 'sliceName': 'Same title'}}}
HTML = '''<button role="tab" id="TABS-outer-tab-TAB-parent" aria-controls="parent" aria-selected="false"
onclick="this.setAttribute('aria-selected','true');document.getElementById('parent').hidden=false">Repeated title</button>
<div id="parent" role="tabpanel" hidden style="min-height:120px">
<div id="chart-id-11"><div class="alert-danger">Code: 386. NO_COMMON_TYPE</div></div>
<button role="tab" id="TABS-inner-tab-TAB-child" aria-controls="child" aria-selected="false"
onclick="this.setAttribute('aria-selected','true');document.getElementById('child').hidden=false">Repeated title</button>
<div id="child" role="tabpanel" hidden style="min-height:120px">
<div style="height:1800px"></div><div id="chart-id-12" style="height:120px"><table><tbody></tbody></table></div>
</div></div>'''
# #region Test.ChartHealth.NestedDom.Handler [C:1] [TYPE Class]
class ExternalHandler(BaseHTTPRequestHandler):
# #region Test.ChartHealth.NestedDom.Handler.Get [C:1] [TYPE Function]
def do_GET(self):
body = json.dumps({'result': {'id': 42, 'position_json': LAYOUT}}).encode() if self.path.startswith('/api/v1/dashboard/') else HTML.encode()
self.send_response(200)
self.end_headers()
self.wfile.write(body)
# #endregion Test.ChartHealth.NestedDom.Handler.Get
# #region Test.ChartHealth.NestedDom.Handler.Log [C:1] [TYPE Function]
def log_message(self, *_args):
pass
# #endregion Test.ChartHealth.NestedDom.Handler.Log
# #endregion Test.ChartHealth.NestedDom.Handler
# #region Test.ChartHealth.NestedDom.Sweep [C:2] [TYPE Function]
# @TEST_INVARIANT Real Chromium opens parent/child stable IDs and scrolls the healthy empty chart into view after recording the error.
@pytest.mark.asyncio
async def test_nested_duplicate_names_offscreen_empty_chart_complete_owned_coverage(health_context):
from playwright.async_api import async_playwright
_, journal, _, storage, limits, _ = health_context
server = ThreadingHTTPServer(('127.0.0.1', 0), ExternalHandler)
thread = Thread(target=server.serve_forever, daemon=True)
thread.start()
try:
async with async_playwright() as runtime:
browser = await runtime.chromium.launch(headless=True)
page = await browser.new_page(viewport={'width': 1000, 'height': 600})
await page.goto(f'http://127.0.0.1:{server.server_port}/dashboard')
result = await sweep_health(page, 42, journal, limits)
position = await page.locator('#chart-id-12').bounding_box()
assert position is not None and 0 <= position['y'] < 600
assert await page.locator('#TABS-inner-tab-TAB-child').get_attribute('aria-selected') == 'true'
await browser.close()
finally:
server.shutdown()
server.server_close()
assert result['status'] == 'failed' and result['complete'] is True
assert result['coverage'] == {'expected': 2, 'observed': 2, 'checked': 2, 'errored': 1,
'unresolved': 0, 'timeout': 0, 'unvisited': 0, 'complete': True}
manifest = json.loads(storage.retrieve(result['manifest_ref']))
assert [(item['chart_id'], item['tab_path'], item['status']) for item in manifest['charts']] == [
(11, ['TAB-parent'], 'failed'), (12, ['TAB-parent', 'TAB-child'], 'healthy')]
assert [item['chart_name'] for item in manifest['charts']] == ['Same title', 'Same title']
# #endregion Test.ChartHealth.NestedDom.Sweep
# #endregion Test.ChartHealth.NestedDom

View File

@@ -0,0 +1,122 @@
# #region Test.ChartHealth.NotificationWorker [C:4] [TYPE Module] [SEMANTICS configured,notification,receipt,CAS]
# @TEST_INVARIANT Explicit routing sends bounded named causes once; disabled configuration and rollback send nothing.
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Delivery]
from copy import deepcopy
import pytest
from src.models.config import AppConfigRecord
from src.services.dashboard_testing.automation.notify import persist_notification
from src.services.dashboard_testing.execution.chart_health_delivery import deliver_receipt
from src.services.dashboard_testing.execution.chart_health_delivery_hook import arm_health_delivery
SUMMARY = {'chart_query_errors':{'count':5,'omitted':0,'items':[
{'chart_name':name,'tab_path':['TAB-daily'],'database_code':'386','message':'Code: 386. NO_COMMON_TYPE'}
for name in ('Debt','Cash','Margin','Forecast','Reserve')]}}
# #region Test.ChartHealth.NotificationWorker.Receipt [C:2] [TYPE Function]
# @BRIEF Real durable receipt and committed global settings, independent of notification transport.
def notification_fixture(db, *, enabled=True, channels=None):
config = db.get(AppConfigRecord,'global')
payload = {'notifications':{'scenario_alerts':{'enabled':enabled,'channels':channels if channels is not None else [{'type':'SLACK','target':'test-only-destination'}]}}}
if config:
config.payload = payload
else:
db.add(AppConfigRecord(id='global',payload=payload))
row = persist_notification(db,event_type='failed',scenario_id='scn-exec-0001',run_id='run-exec-0001',
payload=deepcopy(SUMMARY),idempotency_key='health-notification-test')
return row
# #endregion Test.ChartHealth.NotificationWorker.Receipt
# #region Test.ChartHealth.NotificationWorker.Provider [C:2] [TYPE Class]
# @BRIEF Configured fake boundary records messages without touching any external destination.
class FakeProvider:
# #region Test.ChartHealth.NotificationWorker.Provider.Init [C:1] [TYPE Function]
def __init__(self, calls, sent=True):
self.calls,self.sent = calls,sent
# #endregion Test.ChartHealth.NotificationWorker.Provider.Init
# #region Test.ChartHealth.NotificationWorker.Provider.Send [C:1] [TYPE Function]
async def send(self, target, subject, body):
self.calls.append((target,subject,body))
return self.sent
# #endregion Test.ChartHealth.NotificationWorker.Provider.Send
# #endregion Test.ChartHealth.NotificationWorker.Provider
# #region Test.ChartHealth.NotificationWorker.Once [C:3] [TYPE Function]
@pytest.mark.asyncio
async def test_committed_named_errors_send_once_with_atomic_receipt_claim(seeded_execution):
db,calls = seeded_execution,[]
row = notification_fixture(db)
row.payload = {**row.payload,'health_delivery':{'status':'pending'}}
db.commit()
provider = FakeProvider(calls)
assert await deliver_receipt(db,row.id,provider_factory=lambda *_:provider) == 'delivered'
assert await deliver_receipt(db,row.id,provider_factory=lambda *_:provider) == 'already_claimed'
assert len(calls) == 1 and calls[0][0] == 'test-only-destination'
assert all(name in calls[0][2] for name in ('Debt','Cash','Margin','Forecast','Reserve'))
assert 'Code 386' in calls[0][2] and 'run-exec-0001' in calls[0][2]
db.refresh(row)
assert row.payload['health_delivery']['status'] == 'delivered'
assert 'test-only-destination' not in str(row.payload)
# #endregion Test.ChartHealth.NotificationWorker.Once
# #region Test.ChartHealth.NotificationWorker.Failure [C:2] [TYPE Function]
@pytest.mark.asyncio
async def test_false_transport_retains_delivery_failure_without_retry(seeded_execution):
db,calls = seeded_execution,[]
row = notification_fixture(db)
row.payload = {**row.payload,'health_delivery':{'status':'pending'}}
db.commit()
provider = FakeProvider(calls,False)
assert await deliver_receipt(db,row.id,provider_factory=lambda *_:provider) == 'failed'
assert await deliver_receipt(db,row.id,provider_factory=lambda *_:provider) == 'already_claimed'
assert len(calls) == 1
assert row.payload['health_delivery']['routes'][0]['reason_code'] == 'DELIVERY_FAILED'
# #endregion Test.ChartHealth.NotificationWorker.Failure
# #region Test.ChartHealth.NotificationWorker.Disabled [C:2] [TYPE Function]
@pytest.mark.parametrize('enabled,channels,reason',[(False,[],'DELIVERY_DISABLED'),(True,[],'DESTINATION_UNCONFIGURED')])
def test_disabled_or_missing_destination_has_explicit_skip_and_no_dispatch(seeded_execution,enabled,channels,reason):
db = seeded_execution
row = notification_fixture(db,enabled=enabled,channels=channels)
arm_health_delivery(db,row)
db.commit()
assert row.payload['health_delivery'] == {'status':'skipped','reason_code':reason}
assert not db.info.get('chart_health_deliveries')
# #endregion Test.ChartHealth.NotificationWorker.Disabled
# #region Test.ChartHealth.NotificationWorker.Commit [C:3] [TYPE Function]
def test_after_commit_dispatch_is_armed_once_and_rollback_cannot_start(seeded_execution,monkeypatch):
from src.services.dashboard_testing.execution import chart_health_delivery_hook as seam
db,started = seeded_execution,[]
# #region Test.ChartHealth.NotificationWorker.Commit.Thread [C:1] [TYPE Class]
class ExternalThread:
# #region Test.ChartHealth.NotificationWorker.Commit.Thread.Init [C:1] [TYPE Function]
def __init__(self, **kwargs):
self.identifiers = kwargs['args'][0]
# #endregion Test.ChartHealth.NotificationWorker.Commit.Thread.Init
# #region Test.ChartHealth.NotificationWorker.Commit.Thread.Start [C:1] [TYPE Function]
def start(self):
started.append(self.identifiers)
seam._slot.release()
# #endregion Test.ChartHealth.NotificationWorker.Commit.Thread.Start
# #endregion Test.ChartHealth.NotificationWorker.Commit.Thread
monkeypatch.setattr(seam,'Thread',ExternalThread)
row = notification_fixture(db)
arm_health_delivery(db,row)
assert not started
db.rollback()
assert not started and not db.info.get('chart_health_deliveries')
row = notification_fixture(db)
arm_health_delivery(db,row)
arm_health_delivery(db,row)
assert not started
db.commit()
assert started == [[row.id]]
# #endregion Test.ChartHealth.NotificationWorker.Commit
# #endregion Test.ChartHealth.NotificationWorker

View File

@@ -0,0 +1,125 @@
# #region Test.ChartHealth.OwnedIndependent [C:3] [TYPE Module]
# @TEST_INVARIANT Committed chart failures dominate advisory verdicts; model inputs require exact owned bytes and attempts.
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Store]
import json
import pytest
from src.models.scenario_artifact import ScenarioArtifact
from src.services.dashboard_testing.execution.chart_health_artifact import verify_health_artifact
from src.services.dashboard_testing.execution.chart_health_tab_store import append_tab_evaluation, store_health_artifact
from src.services.dashboard_testing.execution.evaluation_health_payloads import load_health_payloads
from src.services.dashboard_testing.execution.providers.browser_health_manifest import parse_health_manifest
# Infrastructure only: committed run, step lease, capacity, real DraftStorage and fresh DB sessions.
from test_chart_health_connected_worker import health_context # noqa: F401
LAYOUT = {'ROOT_ID': {'type': 'ROOT', 'children': ['CHART-a', 'CHART-b']},
'CHART-a': {'type': 'CHART', 'meta': {'chartId': 11, 'sliceName': 'Same title'}},
'CHART-b': {'type': 'CHART', 'meta': {'chartId': 12, 'sliceName': 'Same title'}}}
# #region Test.ChartHealth.OwnedIndependent.Append [C:1] [TYPE Function]
def append(journal, chart, status='failed'):
journal.append_chart({'chart_id': chart, 'placement_id': 'CHART-a' if chart == 11 else 'CHART-b',
'chart_name': 'Same title', 'tab_path': [], 'status': status,
'reason_code': 'CHART_QUERY_FAILED' if status == 'failed' else 'CHART_READY',
'origin': 'dom', 'observed_at': '2026-10-02T00:00:00Z', 'duration_seconds': 0,
'message': 'Code: 386. DB::Exception: incompatible types' if status == 'failed' else '',
'database_code': '386' if status == 'failed' else None,
'terminal_state': 'error' if status == 'failed' else 'ready'})
# #endregion Test.ChartHealth.OwnedIndependent.Append
# #region Test.ChartHealth.OwnedIndependent.Partial [C:2] [TYPE Function]
def test_partial_error_has_named_owned_bytes_and_exact_unvisited(health_context):
db, journal, step, storage, *_ = health_context
journal.freeze_source(parse_health_manifest(LAYOUT))
append(journal, 11)
result = journal.finish('inconclusive', 'BROWSER_TRAVERSAL_CANCELLED')
assert result['status'] == 'failed' and result['complete'] is False
assert result['coverage'] == {'expected': 2, 'observed': 1, 'checked': 1, 'errored': 1,
'unresolved': 0, 'timeout': 0, 'unvisited': 1, 'complete': False}
chart = result['chart_health']['charts'][0]
artifact = db.get(ScenarioArtifact, chart['artifact_id'])
payload = load_health_payloads([{'artifact_id': chart['content_ref'], 'sha256': chart['sha256'],
'byte_length': artifact.byte_length, 'content_type': 'application/json'}], storage, db=db, run_id=step['scenario_run_id'])
assert payload['items'][0]['content']['chart_id'] == 11
assert payload['items'][0]['content']['database_code'] == '386'
assert 'incompatible types' in payload['items'][0]['content']['message']
assert payload['truncated'] is False
# #endregion Test.ChartHealth.OwnedIndependent.Partial
# #region Test.ChartHealth.OwnedIndependent.Advisory [C:2] [TYPE Function]
def test_advisory_pass_never_erases_complete_deterministic_failure(health_context):
_, journal, _, storage, *_ = health_context
journal.freeze_source(parse_health_manifest(LAYOUT))
append(journal, 11)
append(journal, 12, 'healthy')
receipt = store_health_artifact(journal, b'{"verdict":"pass"}', 'judge.json')
append_tab_evaluation(journal, {'tab_path': [], 'status': 'succeeded', 'coverage_complete': True,
'verdict': 'pass', 'artifacts': [receipt]})
result = journal.finish('passed', 'CHART_HEALTH_COMPLETE')
assert result['status'] == 'failed' and result['complete'] is True
assert result['coverage']['errored'] == 1
assert [item['chart_id'] for item in result['chart_health']['charts']] == [11, 12]
assert result['chart_health']['tab_evaluations'][0]['verdict'] == 'pass'
assert json.loads(storage.retrieve(result['manifest_ref']))['status'] == 'failed'
# #endregion Test.ChartHealth.OwnedIndependent.Advisory
# #region Test.ChartHealth.OwnedIndependent.Ownership [C:2] [TYPE Function]
@pytest.mark.parametrize('field,value', [('owner_id', 'foreign-run'), ('attempt', 2), ('is_active', False)])
def test_foreign_attempt_or_inactive_artifact_refuses_before_evaluation(health_context, field, value):
db, journal, *_ = health_context
receipt = store_health_artifact(journal, b'{"verdict":"pass"}', 'judge.json')
artifact = db.get(ScenarioArtifact, receipt['artifact_id'])
setattr(artifact, field, value)
db.commit()
with pytest.raises(ValueError, match='CHART_HEALTH_ARTIFACT_INVALID'):
verify_health_artifact(journal, db, receipt)
# #endregion Test.ChartHealth.OwnedIndependent.Ownership
# #region Test.ChartHealth.OwnedIndependent.Auxiliary [C:2] [TYPE Function]
def test_tab_capture_between_chart_observations_preserves_exact_frontier(health_context):
_, journal, *_ = health_context
journal.freeze_source(parse_health_manifest(LAYOUT))
append(journal, 11)
store_health_artifact(journal, b'{"verdict":"pass"}', 'first-tab-judge.json')
append(journal, 12, 'healthy')
result = journal.finish('passed', 'CHART_HEALTH_COMPLETE')
assert result['status'] == 'failed' and result['coverage']['observed'] == 2
# #endregion Test.ChartHealth.OwnedIndependent.Auxiliary
# #region Test.ChartHealth.OwnedIndependent.Budget [C:2] [TYPE Function]
# @TEST_INVARIANT Omitted diagnostic text is explicit and cannot be presented as complete model coverage.
def test_diagnostic_byte_budget_reports_omission_without_trusting_partial_payload(health_context):
db, journal, step, storage, *_ = health_context
journal.freeze_source(parse_health_manifest(LAYOUT))
append(journal, 11)
result = journal.finish('inconclusive', 'BROWSER_TRAVERSAL_CANCELLED')
chart = result['chart_health']['charts'][0]
artifact = db.get(ScenarioArtifact, chart['artifact_id'])
payload = load_health_payloads([{'artifact_id': chart['content_ref'], 'sha256': chart['sha256'],
'byte_length': artifact.byte_length, 'content_type': 'application/json'}], storage,
db=db, run_id=step['scenario_run_id'], max_bytes=1)
assert payload['items'] == [] and payload['byte_length'] == 0 and payload['truncated'] is True
assert payload['omitted'] == [{'artifact_id': chart['content_ref'], 'reason': 'TEXT_BUDGET'}]
# #endregion Test.ChartHealth.OwnedIndependent.Budget
# #region Test.ChartHealth.OwnedIndependent.Tampered [C:2] [TYPE Function]
# @TEST_INVARIANT A forged receipt hash cannot authorize actual diagnostic bytes for the model.
def test_changed_diagnostic_digest_refuses_model_input(health_context):
db, journal, step, storage, *_ = health_context
journal.freeze_source(parse_health_manifest(LAYOUT))
append(journal, 11)
result = journal.finish('inconclusive', 'BROWSER_TRAVERSAL_CANCELLED')
chart = result['chart_health']['charts'][0]
artifact = db.get(ScenarioArtifact, chart['artifact_id'])
with pytest.raises(RuntimeError, match='CHART_HEALTH_EVIDENCE_INVALID'):
load_health_payloads([{'artifact_id': chart['content_ref'], 'sha256': '0' * 64,
'byte_length': artifact.byte_length, 'content_type': 'application/json'}], storage,
db=db, run_id=step['scenario_run_id'])
# #endregion Test.ChartHealth.OwnedIndependent.Tampered
# #endregion Test.ChartHealth.OwnedIndependent

View File

@@ -0,0 +1,71 @@
# #region Test.ChartHealth.ResponseWorker [C:3] [TYPE Module] [SEMANTICS http,async,filters,stale]
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Responses]
# @TEST_INVARIANT Late errors cannot cross the newest saved-chart filter generation, and foreign async jobs are ignored.
import json
from hashlib import sha256
import pytest
from src.services.dashboard_testing.execution.providers.browser_chart_responses import ChartResponseCollector
# #region Test.ChartHealth.ResponseWorker.Request [C:1] [TYPE Class]
class Request:
# #region Test.ChartHealth.ResponseWorker.Request.Init [C:1] [TYPE Function]
# @BRIEF Literal external request snapshot, independent of production attribution helpers.
def __init__(self, chart, value):
self.url = 'http://superset.invalid/api/v1/chart/data'
self.method = 'POST'
self.post_data_json = {'form_data':{'slice_id':chart},'datasource':{'id':2,'type':'table'},
'queries':[{'filters':[{'col':'metric','op':'IN','val':[value]}]}]}
# #endregion Test.ChartHealth.ResponseWorker.Request.Init
# #endregion Test.ChartHealth.ResponseWorker.Request
# #region Test.ChartHealth.ResponseWorker.Response [C:1] [TYPE Class]
class Response:
# #region Test.ChartHealth.ResponseWorker.Response.Init [C:1] [TYPE Function]
def __init__(self, request, payload, status=200, url=None):
self.request,self.raw,self.status = request,json.dumps(payload).encode(),status
self.url = url or request.url
# #endregion Test.ChartHealth.ResponseWorker.Response.Init
# #region Test.ChartHealth.ResponseWorker.Response.Body [C:1] [TYPE Function]
async def body(self):
return self.raw
# #endregion Test.ChartHealth.ResponseWorker.Response.Body
# #endregion Test.ChartHealth.ResponseWorker.Response
# #region Test.ChartHealth.ResponseWorker.Stale [C:2] [TYPE Function]
@pytest.mark.asyncio
async def test_late_old_filter_error_cannot_contaminate_new_success_or_other_chart():
collector = ChartResponseCollector(None,[1])
old,new,foreign = Request(1,'old'),Request(1,'new'),Request(99,'new')
collector.on_request(old)
collector.on_request(new)
collector.on_request(foreign)
await collector.read_response(Response(old,{'errors':[{'message':'Code: 386. old-filter failure'}]},500))
await collector.read_response(Response(new,{'result':[{'data':[{'value':0}]}]}))
await collector.read_response(Response(foreign,{'errors':[{'message':'foreign failure'}]},500))
assert collector.current_error(1) is None and collector.current_error(99) is None
assert len(collector.requests) == 2
# #endregion Test.ChartHealth.ResponseWorker.Stale
# #region Test.ChartHealth.ResponseWorker.Async [C:2] [TYPE Function]
@pytest.mark.asyncio
async def test_polling_errors_require_exact_current_job_and_keep_original_response_digest():
collector = ChartResponseCollector(None,[1])
request = Request(1,'sales')
collector.on_request(request)
await collector.read_response(Response(request,{'result':[{'job_id':'owned-job','status':'pending'}]},202))
polling = Response(Request(99,'ignored'),{'result':[{'job_id':'foreign-job','status':'error','errors':[{'message':'foreign'}]}]},url='http://superset.invalid/api/v1/async_event/')
await collector.read_response(polling)
assert collector.current_error(1) is None
current = Response(Request(99,'ignored'),{'result':[{'job_id':'owned-job','status':'error','errors':[{'error_type':'GENERIC_DB_ENGINE_ERROR','message':'Code: 386. DB::Exception: incompatible types'}]}]},url='http://superset.invalid/api/v1/async_event/')
await collector.read_response(current)
error = collector.current_error(1)
assert error['message'] == 'Code: 386. DB::Exception: incompatible types'
assert error['attribution']['load_generation'] == 1
assert error['attribution']['original_response_sha256'] == sha256(current.raw).hexdigest()
assert error['attribution']['original_byte_length'] == len(current.raw)
# #endregion Test.ChartHealth.ResponseWorker.Async
# #endregion Test.ChartHealth.ResponseWorker

View File

@@ -0,0 +1,77 @@
# #region Test.ChartHealth.SdkWorker [C:3] [TYPE Module] [SEMANTICS physical-request,SDK,budget,credentials]
# @TEST_INVARIANT The real encrypted credential boundary feeds one external SDK request with retries disabled and declared budgets.
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.BoundedClient]
import json
from types import SimpleNamespace
import pytest
from src.core.encryption import get_encryption_manager
from src.models.llm import LLMProvider
from src.services.dashboard_testing.execution.providers.browser_health_llm_client import BoundedHealthClient
# #region Test.ChartHealth.SdkWorker.Context [C:2] [TYPE Function]
@pytest.fixture
def sdk_context(seeded_execution,monkeypatch):
from src.services.dashboard_testing.execution.providers import browser_health_llm_client as seam
db,calls = seeded_execution,[]
provider = LLMProvider(id='bounded-health-sdk',name='Fake SDK only',provider_type='openai',
base_url='http://fixture.invalid/v1',api_key=get_encryption_manager().encrypt('fake-key-never-sent'),
default_model='vision-fixture',is_active=True,is_multimodal=True,supports_json_object=True)
db.add(provider)
db.commit()
response = SimpleNamespace(choices=[SimpleNamespace(finish_reason='stop',message=SimpleNamespace(content=json.dumps({'verdict':'pass','usage':{'input_tokens':9999}})))],
usage=SimpleNamespace(prompt_tokens=17,completion_tokens=9))
# #region Test.ChartHealth.SdkWorker.Context.Boundary [C:2] [TYPE Class]
class ExternalSdk:
# #region Test.ChartHealth.SdkWorker.Context.Boundary.Init [C:1] [TYPE Function]
def __init__(self, provider_type,key,url,model):
calls.append(('credential',key))
self.client,self.chat,self.completions = self,self,self
# #endregion Test.ChartHealth.SdkWorker.Context.Boundary.Init
# #region Test.ChartHealth.SdkWorker.Context.Boundary.Options [C:1] [TYPE Function]
def with_options(self, **options):
calls.append(('options',options))
return self
# #endregion Test.ChartHealth.SdkWorker.Context.Boundary.Options
# #region Test.ChartHealth.SdkWorker.Context.Boundary.Create [C:1] [TYPE Function]
async def create(self, **options):
calls.append(('request',options))
return response
# #endregion Test.ChartHealth.SdkWorker.Context.Boundary.Create
# #region Test.ChartHealth.SdkWorker.Context.Boundary.Close [C:1] [TYPE Function]
async def close(self):
calls.append(('closed',True))
# #endregion Test.ChartHealth.SdkWorker.Context.Boundary.Close
# #endregion Test.ChartHealth.SdkWorker.Context.Boundary
monkeypatch.setattr(seam,'LLMClient',ExternalSdk)
spec = SimpleNamespace(model_id='vision-fixture',limits=SimpleNamespace(timeout_ms=1500,max_output_tokens=128))
return BoundedHealthClient(db,provider,spec),calls,response
# #endregion Test.ChartHealth.SdkWorker.Context
# #region Test.ChartHealth.SdkWorker.Budgets [C:2] [TYPE Function]
@pytest.mark.asyncio
async def test_single_physical_request_uses_declared_limits_and_transport_usage(sdk_context):
client,calls,_ = sdk_context
result = await client.get_json_completion([{'role':'user','content':'Code: 386'}])
assert calls[0] == ('credential','fake-key-never-sent')
assert ('options',{'max_retries':0,'timeout':1.5}) in calls
requests = [item for kind,item in calls if kind == 'request']
assert len(requests) == 1
assert requests[0]['max_tokens'] == 128 and requests[0]['response_format'] == {'type':'json_object'}
assert result['usage'] == {'input_tokens':17,'output_tokens':9}
assert calls[-1] == ('closed',True)
# #endregion Test.ChartHealth.SdkWorker.Budgets
# #region Test.ChartHealth.SdkWorker.Truncation [C:2] [TYPE Function]
@pytest.mark.asyncio
async def test_truncated_response_refuses_without_second_request_and_closes(sdk_context):
client,calls,response = sdk_context
response.choices[0].finish_reason = 'length'
with pytest.raises(RuntimeError,match='EVALUATION_RESPONSE_TRUNCATED'):
await client.get_json_completion([])
assert sum(kind == 'request' for kind,_ in calls) == 1
assert calls[-1] == ('closed',True)
# #endregion Test.ChartHealth.SdkWorker.Truncation
# #endregion Test.ChartHealth.SdkWorker

View File

@@ -0,0 +1,71 @@
# #region Test.ChartHealth.SlowClock [C:2] [TYPE Module] [SEMANTICS health,slow,timeout,coverage]
# @TEST_INVARIANT Slow healthy charts have independent deadlines; an immediate error remains a durable failure.
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Observer]
from unittest.mock import AsyncMock, Mock
import pytest
from test_chart_health_connected_worker import health_context # noqa: F401
from test_chart_health_owned_evaluation_authority import LAYOUT
from src.services.dashboard_testing.execution.providers.browser_chart_health import observe_chart
from src.services.dashboard_testing.execution.providers.browser_chart_responses import ChartResponseCollector
from src.services.dashboard_testing.execution.providers.browser_health_manifest import parse_health_manifest
# #region Test.ChartHealth.SlowClock.Clock [C:1] [TYPE Class]
class ExternalClock:
# #region Test.ChartHealth.SlowClock.Clock.Init [C:1] [TYPE Function]
def __init__(self):
self.now = 0.0
# #endregion Test.ChartHealth.SlowClock.Clock.Init
# #region Test.ChartHealth.SlowClock.Clock.Read [C:1] [TYPE Function]
def __call__(self):
return self.now
# #endregion Test.ChartHealth.SlowClock.Clock.Read
# #endregion Test.ChartHealth.SlowClock.Clock
# #region Test.ChartHealth.SlowClock.Chart [C:1] [TYPE Function]
def external_chart(state):
chart = Mock()
chart.wait_for = AsyncMock()
chart.count = AsyncMock(return_value=1)
chart.scroll_into_view_if_needed = AsyncMock()
chart.evaluate = AsyncMock(side_effect=state)
return chart
# #endregion Test.ChartHealth.SlowClock.Chart
# #region Test.ChartHealth.SlowClock.Success [C:2] [TYPE Function]
# @TEST_INVARIANT Simulated 45/60-second loading does not delay immediate Code386 retention or consume another chart's deadline.
@pytest.mark.parametrize('seconds', [45.0, 60.0])
@pytest.mark.asyncio
async def test_slow_chart_and_immediate_error_keep_separate_observation_budgets(health_context, monkeypatch, seconds):
_, journal, *_ = health_context
source = parse_health_manifest(LAYOUT)
journal.freeze_source(source)
clock = ExternalClock()
monkeypatch.setattr('src.services.dashboard_testing.execution.providers.browser_chart_health.monotonic', clock)
error = external_chart(lambda *_args, **_kwargs: {'error': 'Code: 386. NO_COMMON_TYPE', 'loading': False, 'ready': False})
# #region Test.ChartHealth.SlowClock.Success.Render [C:1] [TYPE Function]
def render(*_args, **_kwargs):
clock.now += seconds / 3
return {'error': '', 'loading': clock.now < seconds, 'ready': clock.now >= seconds}
# #endregion Test.ChartHealth.SlowClock.Success.Render
slow = external_chart(render)
absent_native = Mock()
absent_native.count = AsyncMock(return_value=0)
panel = Mock()
panel.locator.side_effect = {
'[data-test="chart-grid-component"][data-test-chart-id="11"]': absent_native,
'[data-test="chart-grid-component"][data-test-chart-id="12"]': absent_native,
'#chart-id-11': error, '#chart-id-12': slow,
}.__getitem__
collector = ChartResponseCollector(None, [11, 12])
first = await observe_chart(None, panel, source['placements'][0], 70, collector, journal)
assert first['status'] == 'failed' and first['database_code'] == '386' and first['duration_seconds'] == 0
journal.append_chart(first)
second = await observe_chart(None, panel, source['placements'][1], 70, collector, journal)
assert second['status'] == 'healthy' and second['duration_seconds'] == seconds
assert error.evaluate.await_count == 1 and slow.evaluate.await_count == 3
journal.append_chart(second)
result = journal.finish('passed', 'CHART_HEALTH_COMPLETE')
assert result['status'] == 'failed' and result['complete'] is True
assert result['coverage']['checked'] == 2 and result['coverage']['errored'] == 1
# #endregion Test.ChartHealth.SlowClock.Success
# #endregion Test.ChartHealth.SlowClock

View File

@@ -0,0 +1,168 @@
# #region Test.ChartHealth.VlmIndependent [C:2] [TYPE Module] [SEMANTICS health,provider,pin,owned,images]
# @TEST_INVARIANT Fresh persisted visual provider configuration and owned diagnostic/image bytes precede model work.
# @RELATION BINDS_TO -> [ScenarioExecution.ChartHealth.Provider]
from types import SimpleNamespace
import base64
import json
from copy import deepcopy
from unittest.mock import AsyncMock, Mock
from sqlalchemy.orm import sessionmaker
from src.models.llm import LLMProvider
from src.services.dashboard_testing.execution.providers.browser_health_provider import admit_health_provider
import pytest
from test_chart_health_connected_worker import health_context # noqa: F401
from test_chart_health_owned_evaluation_authority import append, LAYOUT
from src.services.dashboard_testing.execution.chart_health_tab_store import store_health_artifact, tab_diagnostics
from src.services.dashboard_testing.execution.providers.browser_health_manifest import parse_health_manifest
from src.services.dashboard_testing.execution.providers.browser_health_evaluation import evaluation_inputs
from test_chart_health_sdk_worker import sdk_context # noqa: F401
# #region Test.ChartHealth.VlmIndependent.UuidJournal [C:1] [TYPE Function]
def uuid_journal(context):
from src.models.scenario_run import ScenarioRun, ScenarioStepRun
from src.services.dashboard_testing.execution.worker import claim_step
from src.services.dashboard_testing.execution.capacity import claim_capacity
from src.services.dashboard_testing.execution.chart_health_store import ChartHealthJournal
db, old, old_step, storage, limits, _ = context
original = db.get(ScenarioRun, old.run_id)
fields = {column.name: deepcopy(getattr(original, column.name)) for column in ScenarioRun.__table__.columns if column.name != 'id'}
fields['idempotency_key'] = 'independent-whole-health-evaluation'
run = ScenarioRun(id='126b6988-28fc-4697-8f6a-954d000443c1', **fields)
db.add(run)
db.flush()
step = deepcopy(old_step)
step['scenario_run_id'] = run.id
run.runner_plan = deepcopy(original.runner_plan)
db.add(ScenarioStepRun(run_id=run.id, logical_step_id='health-check', step_position=1, attempt=1, status='running'))
claim_step(db, run.id, 'health-check', worker_id='whole-test', side_effect_key=None, idempotent=True, retry_safe=True, lease_seconds=120)
capacity = claim_capacity(db, environment_id='preprod', environment_class='PREPROD', workload_class='browser', provider_id='browser', run_id=run.id, logical_step_id='health-check', ttl_seconds=120)
db.commit()
return ChartHealthJournal(step, storage, limits, capacity_lease_id=capacity['lease_id'])
# #endregion Test.ChartHealth.VlmIndependent.UuidJournal
PIN = 'config_sha256:01a89824f8992e6a9b79cc577e88d48ea1e8057ad0dec367d7650a1726b76a29'
# #region Test.ChartHealth.VlmIndependent.Provider [C:1] [TYPE Function]
@pytest.fixture
def provider_context(health_context):
db = health_context[0]
provider = LLMProvider(id='health-provider', name='Fixture', provider_type='openai',
base_url='http://fixture.invalid/v1', api_key='fixture-unused-no-decryption',
default_model='vision-fixture', is_active=True, is_multimodal=True, max_images=2,
context_window=8192, max_output_tokens=512, supports_json_object=True)
db.add(provider)
db.commit()
policy = SimpleNamespace(spec=SimpleNamespace(provider_id='health-provider', provider_version=PIN,
model_id='vision-fixture', model_version='configured-model:vision-fixture',
limits=SimpleNamespace(max_images=2)))
return db, provider, policy
# #endregion Test.ChartHealth.VlmIndependent.Provider
# #region Test.ChartHealth.VlmIndependent.Positive [C:2] [TYPE Function]
# @TEST_INVARIANT Literal authored public configuration pin admits the matching persisted provider.
def test_exact_visual_provider_pin_admits_without_network_or_key_read(provider_context):
db, provider, policy = provider_context
assert admit_health_provider(db, policy) is provider
# #endregion Test.ChartHealth.VlmIndependent.Positive
# #region Test.ChartHealth.VlmIndependent.Changed [C:2] [TYPE Function]
# @TEST_INVARIANT Changed URL/model or unavailable vision authority refuses prior to charged work.
@pytest.mark.parametrize('field,value,code', [
('base_url', 'http://different.invalid/v1', 'CHART_HEALTH_VLM_PROVIDER_CHANGED'),
('default_model', 'different-model', 'CHART_HEALTH_VLM_PROVIDER_CHANGED'),
('is_active', False, 'CHART_HEALTH_VLM_PROVIDER_UNAVAILABLE'),
('is_multimodal', False, 'CHART_HEALTH_VLM_PROVIDER_UNAVAILABLE')])
def test_changed_provider_configuration_rejects_old_authored_pin(provider_context, field, value, code):
db, provider, policy = provider_context
setattr(provider, field, value)
db.commit()
with pytest.raises(ValueError, match=code):
admit_health_provider(db, policy)
# #endregion Test.ChartHealth.VlmIndependent.Changed
# #region Test.ChartHealth.VlmIndependent.ImageBudget [C:2] [TYPE Function]
# @TEST_INVARIANT Caller cannot enlarge image budget past pinned provider capability.
def test_image_budget_exceeding_matching_provider_is_rejected(provider_context):
db, _, policy = provider_context
policy.spec.limits.max_images = 3
with pytest.raises(ValueError, match='CHART_HEALTH_VLM_IMAGE_BUDGET'):
admit_health_provider(db, policy)
# #endregion Test.ChartHealth.VlmIndependent.ImageBudget
# #region Test.ChartHealth.VlmIndependent.OwnedImages [C:2] [TYPE Function]
# @TEST_INVARIANT Actual same-attempt PNG bytes accompany actual Code386 diagnostic contents, not references alone.
def test_owned_image_and_error_contents_cross_real_evaluation_input_boundary(health_context, registry_engine, monkeypatch):
_, journal, _, _, *_ = health_context
monkeypatch.setattr('src.services.dashboard_testing.execution.providers.browser_health_evaluation.SessionLocal', sessionmaker(bind=registry_engine))
journal.freeze_source(parse_health_manifest(LAYOUT))
append(journal, 11)
image = base64.b64decode('iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8/x8AAwMCAO+aRZsAAAAASUVORK5CYII=')
receipt = store_health_artifact(journal, image, 'owned-tab.png', 'image/png')
manifest, payload, images, artifacts = evaluation_inputs(journal, tab_diagnostics(journal, []), [{'receipt': receipt}])
assert images == [(image, 'image/png')]
assert len(manifest) == 2 and len(artifacts) == 2
assert payload['items'][0]['content']['chart_id'] == 11
assert payload['items'][0]['content']['database_code'] == '386'
assert 'incompatible types' in payload['items'][0]['content']['message']
# #endregion Test.ChartHealth.VlmIndependent.OwnedImages
# #region Test.ChartHealth.VlmIndependent.WholeTab [C:2] [TYPE Function]
# @TEST_INVARIANT Real tab evaluation retains image/content/raw/record and permits later chart while deterministic FAIL dominates model PASS.
@pytest.mark.asyncio
async def test_whole_tab_evaluation_then_next_chart_preserves_failure(health_context, sdk_context, registry_engine, monkeypatch):
from src.services.dashboard_testing.scenario.models import AgentEvaluationSpec
from src.services.dashboard_testing.scenario.metric_evaluation_provider import provider_public_config_digest
from src.services.dashboard_testing.execution.providers.browser_health_evaluation import evaluate_tab
from src.services.dashboard_testing.execution.providers.browser_health_evaluation_inputs import HealthEvaluationPolicy
journal = uuid_journal(health_context)
sdk, calls, response = sdk_context
monkeypatch.setattr('src.services.dashboard_testing.execution.providers.browser_health_evaluation.SessionLocal', sessionmaker(bind=registry_engine))
spec = AgentEvaluationSpec.model_validate({'schema_version': 1,
'spec_id': 'e5e5e5e5-e5e5-4e5e-8e5e-e5e5e5e5e5e5', 'provider_id': sdk.provider.id,
'provider_version': 'config_sha256:' + provider_public_config_digest(sdk.provider),
'model_id': 'vision-fixture', 'model_version': 'configured-model:vision-fixture',
'prompt_template_id': 'health-prompt', 'prompt_template_version': '1.0.0', 'prompt_template_hash': 'a' * 64,
'evidence_refs': ['health-check'], 'comparison_refs': [], 'output_schema': 'agent-evaluation.schema.json',
'decision_policy': {'policy_id': 'baseline-semantic', 'version': '1.0.0'},
'limits': {'timeout_ms': 1500, 'max_images': 2, 'max_input_tokens': 32000, 'max_output_tokens': 128, 'max_cost': '1.00', 'currency': 'USD'},
'trust_policy_hash': 'b' * 64,
'criteria': [{'criterion_id': 'visual', 'criterion_kind': 'semantic', 'description': 'Explain visible chart errors', 'comparison_id': None}]})
policy = HealthEvaluationPolicy(mode='all_visited', spec=spec)
journal.freeze_source(parse_health_manifest(LAYOUT))
append(journal, 11)
diagnostic_ref = tab_diagnostics(journal, [])[0]['receipt']['content_ref']
response.choices[0].message.content = json.dumps({'verdict': 'pass', 'confidence': 0.9,
'findings': [{'criterion_id': 'visual', 'severity': 'info', 'message': 'Error card text is readable',
'evidence_artifact_ids': [diagnostic_ref]}]})
# External browser screenshot protocol only; local capture/evaluation/authority are unmocked.
chart = Mock()
chart.count = AsyncMock(return_value=1)
chart.scroll_into_view_if_needed = AsyncMock()
chart.screenshot = AsyncMock(return_value=base64.b64decode('iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8/x8AAwMCAO+aRZsAAAAASUVORK5CYII='))
absent_native = Mock()
absent_native.count = AsyncMock(return_value=0)
panel = Mock()
panel.locator.side_effect = {
'[data-test="chart-grid-component"][data-test-chart-id="11"]': absent_native,
'#chart-id-11': chart,
}.__getitem__
await evaluate_tab(None, panel, [], [{'id': 'CHART-a', 'chart_id': 11}], journal, policy, 5)
append(journal, 12, 'healthy')
result = journal.finish('passed', 'CHART_HEALTH_COMPLETE')
assert result['status'] == 'failed' and result['complete'] is True
evaluation = result['chart_health']['tab_evaluations'][0]
assert evaluation['status'] == 'succeeded' and evaluation['advisory_verdict'] == 'pass', evaluation
assert len(evaluation['artifacts']) == 4 # diagnostic, actual image, SDK response and typed record
requests = [options for kind, options in calls if kind == 'request']
assert len(requests) == 1
assert 'Code: 386' in json.dumps(requests[0]['messages'])
assert 'data:image/png;base64,' in json.dumps(requests[0]['messages'])
from src.models.provider_capacity import CapacityLease
leases = health_context[0].query(CapacityLease).filter_by(run_id=journal.run_id, workload_class='agent_evaluation').all()
assert len(leases) == 1 and leases[0].status == 'released'
assert leases[0].provider_version == spec.provider_version.removeprefix('config_sha256:')
assert len(leases[0].provider_version) == 64
record = json.loads(journal.storage.retrieve(evaluation['artifacts'][-1]['content_ref']))
assert record['provider_version'] == spec.provider_version and record['comparison_ids'] == []
# #endregion Test.ChartHealth.VlmIndependent.WholeTab
# #endregion Test.ChartHealth.VlmIndependent

View File

@@ -30,6 +30,8 @@ from src.services.dashboard_testing.query_executor import execute_dashboard_quer
# ── Helpers ────────────────────────────────────────────────────────
# #region Test.DashboardTesting.QueryExecutor.Own._chart_data_response [C:3] [TYPE Function]
# @BRIEF Build fixture response bytes and their digest from the supplied parsed result.
def _chart_data_response(result_dict: dict) -> ChartDataResponse:
"""Build a ChartDataResponse from a result dict (simulates httpx raw_response)."""
raw = json.dumps(result_dict, sort_keys=True, default=str).encode("utf-8")
@@ -39,6 +41,7 @@ def _chart_data_response(result_dict: dict) -> ChartDataResponse:
raw_bytes=raw,
source_response_hash=hashlib.sha256(raw).hexdigest(),
)
# #endregion Test.DashboardTesting.QueryExecutor.Own._chart_data_response
# #region Test.DashboardTesting.QueryExecutor.NoSQLRejection [C:3] [TYPE Function] [SEMANTICS testing,baseline,no-sql,security]
@@ -123,6 +126,7 @@ async def test_superset_error_preserved():
assert "SUPERSET" in result.warnings[0].code
# #region Test.DashboardTesting.QueryExecutor.Own.test_chart_data_error_body_cannot_become_empty_baseline_value [C:3] [TYPE Function]
@pytest.mark.asyncio
async def test_chart_data_error_body_cannot_become_empty_baseline_value():
client = AsyncMock()
@@ -133,8 +137,15 @@ async def test_chart_data_error_body_cannot_become_empty_baseline_value():
environment_id="dev", dashboard_id=42, chart_id=128, result_key="revenue",
normalized_filters=NormalizedFilterContext(filters=[], filters_hash="sha256:empty"),
)
with pytest.raises(ValueError, match="chart-data rejected"):
await execute_dashboard_query_envelope(client, request)
result = await execute_dashboard_query_envelope(client, request)
assert result.normalized_value.kind == ValueKind.UNKNOWN
assert result.diagnostic["status"] == "failed"
assert result.diagnostic["message"] == "invalid query"
assert result.payload_kind == "sanitized_diagnostic"
assert result.diagnostic["original_response_sha256"] != result.source_response_hash
assert b"invalid query" in result.raw_response_content
# #endregion Test.DashboardTesting.QueryExecutor.Own.test_chart_data_error_body_cannot_become_empty_baseline_value
# #endregion Test.DashboardTesting.QueryExecutor.SupersetErrorTaxonomy
# #region Test.DashboardTesting.QueryExecutor.TemporalFilterMapping [C:3] [TYPE Function] [SEMANTICS testing,baseline,temporal,filter]
@@ -179,6 +190,8 @@ async def test_temporal_filter_mapped_correctly():
# ── Authoritative model test helpers ─────────────────────────────────
# #region Test.DashboardTesting.QueryExecutor.Own._make_basic_query_model [C:3] [TYPE Function]
# @BRIEF Build a minimal approved query model with the requested charts and filter targets.
def _make_basic_query_model(
chart_ids: list[int] | None = None,
filter_targets: dict[str, list[int]] | None = None,
@@ -236,6 +249,7 @@ def _make_basic_query_model(
native_filters=native_filters,
query_model_fingerprint=fingerprint,
)
# #endregion Test.DashboardTesting.QueryExecutor.Own._make_basic_query_model
# ── Authoritative model tests ────────────────────────────────────────
@@ -431,26 +445,35 @@ from src.services.dashboard_testing.query_executor import (
)
# #region Test.DashboardTesting.QueryExecutor.Own.test_check_forbidden_fields_rejects_sql [C:1] [TYPE Function]
def test_check_forbidden_fields_rejects_sql():
with pytest.raises(ValueError, match="Forbidden fields"):
_check_forbidden_fields({"sql": "SELECT 1", "chart_id": 1})
# #endregion Test.DashboardTesting.QueryExecutor.Own.test_check_forbidden_fields_rejects_sql
# #region Test.DashboardTesting.QueryExecutor.Own.test_check_forbidden_fields_clean_passes [C:1] [TYPE Function]
def test_check_forbidden_fields_clean_passes():
_check_forbidden_fields({"chart_id": 1, "result_key": "k"}) # no raise
# #endregion Test.DashboardTesting.QueryExecutor.Own.test_check_forbidden_fields_clean_passes
# #region Test.DashboardTesting.QueryExecutor.Own.test_extract_query_result_non_dict_first_record [C:1] [TYPE Function]
def test_extract_query_result_non_dict_first_record():
raw_value, query_id = _extract_query_result({"result": [[1, 2]], "query_id": "q-1"}, "k")
assert raw_value == [1, 2]
assert query_id == "q-1"
# #endregion Test.DashboardTesting.QueryExecutor.Own.test_extract_query_result_non_dict_first_record
# #region Test.DashboardTesting.QueryExecutor.Own.test_extract_query_result_empty [C:1] [TYPE Function]
def test_extract_query_result_empty():
raw_value, query_id = _extract_query_result({"result": []}, "k")
assert raw_value is None
# #endregion Test.DashboardTesting.QueryExecutor.Own.test_extract_query_result_empty
# #region Test.DashboardTesting.QueryExecutor.Own.test_build_filters_in_operator_with_values [C:3] [TYPE Function]
def test_build_filters_in_operator_with_values():
from src.schemas.dashboard_testing import FilterValue, NormalizedFilter, NormalizedFilterContext
nf = NormalizedFilterContext(
@@ -465,8 +488,10 @@ def test_build_filters_in_operator_with_values():
out = _build_chart_data_filters(nf)
assert out[0]["operator"] == "IN"
assert out[0]["comparator"] == ["a", "b"]
# #endregion Test.DashboardTesting.QueryExecutor.Own.test_build_filters_in_operator_with_values
# #region Test.DashboardTesting.QueryExecutor.Own.test_build_filters_generic_fallback [C:3] [TYPE Function]
def test_build_filters_generic_fallback():
from src.schemas.dashboard_testing import FilterValue, NormalizedFilter, NormalizedFilterContext
nf = NormalizedFilterContext(
@@ -481,8 +506,10 @@ def test_build_filters_generic_fallback():
out = _build_chart_data_filters(nf)
assert out[0]["operator"] == "=="
assert out[0]["comparator"] == "x"
# #endregion Test.DashboardTesting.QueryExecutor.Own.test_build_filters_generic_fallback
# #region Test.DashboardTesting.QueryExecutor.Own.test_envelope_requires_chart_or_dataset [C:3] [TYPE Function]
@pytest.mark.asyncio
async def test_envelope_requires_chart_or_dataset():
request = ExecuteQueryRequest.model_construct(
@@ -491,4 +518,5 @@ async def test_envelope_requires_chart_or_dataset():
)
with pytest.raises(ValueError, match="Either chart_id or dataset_id"):
await execute_dashboard_query_envelope(AsyncMock(), request)
# #endregion Test.DashboardTesting.QueryExecutor.Own.test_envelope_requires_chart_or_dataset
# #endregion Test.DashboardTesting.QueryExecutor.Branches

View File

@@ -0,0 +1,116 @@
#!/usr/bin/env python3
# #region FullFlow.ChartHealth.CanarySeed [C:4] [TYPE Module] [SEMANTICS superset,clickhouse,fixture,type-conflict,recovery]
# @PRE Existing isolatedDEV Superset and finance ClickHouse binding/source are available.
# @POST Six isolated native charts expose five actual ClickHouse386 errors then recover by changing only their own metric.
# @INVARIANT Existing finance dashboards/datasets/data/release evidence are never modified.
import argparse
import json
from uuid import UUID,uuid5
from superset.app import create_app
NAMESPACE = UUID('496dfb4d-c02b-4cdb-8bcc-bd7c17fe851b')
BROKEN = "sum(if(amount_cents > 0, toInt64(1), 'chart-health-type-conflict'))"
HEALTHY = 'sum(amount_cents)'
NAMES = ['Canary Debt','Canary Cash','Canary Margin','Canary Forecast','Canary Reserve','Canary Healthy']
# #region FullFlow.ChartHealth.CanarySeed.Identity [C:1] [TYPE Function]
# @POST Namespace isolates every fixture asset from existing finance identities.
def identity(key):
return uuid5(NAMESPACE,key)
# #endregion FullFlow.ChartHealth.CanarySeed.Identity
# #region FullFlow.ChartHealth.CanarySeed.Source [C:3] [TYPE Function]
# @POST The own virtual dataset reads one real existing source row; no ClickHouse DDL/DML is issued.
def own_source(session, owner):
from superset.models.core import Database
from superset.connectors.sqla.models import SqlaTable,SqlMetric
database = session.query(Database).filter_by(uuid=uuid5(UUID('a75d5902-c347-490c-a282-f00acc1e1230'),'database')).one()
source = session.query(SqlaTable).filter_by(uuid=identity('dataset')).first()
if source is None:
source = SqlaTable(table_name='ss_tools_chart_health_canary',database=database,owners=[owner],uuid=identity('dataset'),
sql='SELECT amount_cents FROM finance_lab.finance_positions LIMIT 1')
session.add(source)
source.fetch_metadata()
source.metrics = [SqlMetric(metric_name='canary_value',expression=BROKEN),SqlMetric(metric_name='healthy_value',expression=HEALTHY)]
session.flush()
if source.sql != 'SELECT amount_cents FROM finance_lab.finance_positions LIMIT 1':
raise ValueError('CHART_HEALTH_CANARY_SOURCE_CONFLICT')
return source
# #endregion FullFlow.ChartHealth.CanarySeed.Source
# #region FullFlow.ChartHealth.CanarySeed.Charts [C:3] [TYPE Function]
# @POST Only namespaced chart metadata is created; saved query errors come from actual engine metric typing.
def own_charts(session, owner, source):
from superset.models.slice import Slice
charts = []
for index,name in enumerate(NAMES):
chart = session.query(Slice).filter_by(uuid=identity(f'chart:{index}')).first()
if chart is None:
params = {'viz_type':'big_number_total','datasource':f'{source.id}__table','metric':'healthy_value' if index == 5 else 'canary_value',
'time_range':'No filter','adhoc_filters':[],'row_limit':1,'number_format':',.0f'}
chart = Slice(slice_name=name,viz_type='big_number_total',datasource_id=source.id,datasource_type='table',
params=json.dumps(params),owners=[owner],uuid=identity(f'chart:{index}'))
session.add(chart)
session.flush()
if chart.datasource_id != source.id:
raise ValueError('CHART_HEALTH_CANARY_CHART_CONFLICT')
charts.append(chart)
return charts
# #endregion FullFlow.ChartHealth.CanarySeed.Charts
# #region FullFlow.ChartHealth.CanarySeed.Layout [C:3] [TYPE Function]
# @POST Exact saved placement IDs and tab ancestry match real Superset layout semantics.
def own_layout(charts):
tabs = ['TAB-health-errors','TAB-health-healthy']
layout = {'DASHBOARD_VERSION_KEY':'v2','ROOT_ID':{'id':'ROOT_ID','type':'ROOT','children':['GRID_ID']},
'GRID_ID':{'id':'GRID_ID','type':'GRID','children':['TABS-health'],'parents':['ROOT_ID']},
'TABS-health':{'id':'TABS-health','type':'TABS','children':tabs,'parents':['ROOT_ID','GRID_ID'],'meta':{}}}
for tab_id,group in zip(tabs,(charts[:5],charts[5:])):
parents = ['ROOT_ID','GRID_ID','TABS-health']
rows = [f'{tab_id}-row-{index}' for index in range((len(group)+1)//2)]
layout[tab_id] = {'id':tab_id,'type':'TAB','children':rows,'parents':parents,'meta':{'text':'Daily summary' if len(group) == 5 else 'Healthy other tab'}}
for index,row_id in enumerate(rows):
pair = group[index*2:index*2+2]
layout[row_id] = {'id':row_id,'type':'ROW','children':[f'CHART-{item.id}' for item in pair],'parents':parents+[tab_id],'meta':{'background':'BACKGROUND_TRANSPARENT'}}
for chart in pair:
layout[f'CHART-{chart.id}'] = {'id':f'CHART-{chart.id}','type':'CHART','children':[],'parents':parents+[tab_id,row_id],
'meta':{'chartId':chart.id,'uuid':str(chart.uuid),'sliceName':chart.slice_name,'width':6 if len(pair) == 2 else 12,'height':50}}
return layout
# #endregion FullFlow.ChartHealth.CanarySeed.Layout
# #region FullFlow.ChartHealth.CanarySeed.Main [C:4] [TYPE Function]
# @SIDE_EFFECT Changes only namespaced native fixture metric/dashboard metadata, never existing data or execution receipts.
def main():
phase = argparse.ArgumentParser()
phase.add_argument('--phase',choices=['error','recovered'],required=True)
args = phase.parse_args()
with create_app().app_context():
from superset import db,security_manager
from superset.models.dashboard import Dashboard
owner = security_manager.find_user(username='admin')
if owner is None:
raise ValueError('CHART_HEALTH_CANARY_ADMIN_UNAVAILABLE')
source = own_source(db.session,owner)
metric = next(item for item in source.metrics if item.metric_name == 'canary_value')
if metric.expression not in {BROKEN,HEALTHY}:
raise ValueError('CHART_HEALTH_CANARY_METRIC_CONFLICT')
metric.expression = BROKEN if args.phase == 'error' else HEALTHY
charts = own_charts(db.session,owner,source)
dashboard = db.session.query(Dashboard).filter_by(uuid=identity('dashboard')).first()
if dashboard is None:
dashboard = Dashboard(dashboard_title='Isolated chart health canary',slug='chart-health-canary',published=True,owners=[owner],slices=charts,
uuid=identity('dashboard'),position_json=json.dumps(own_layout(charts)),json_metadata=json.dumps({'native_filter_configuration':[],
'timed_refresh_immune_slices':[],'expanded_slices':{},'refresh_frequency':0,'color_scheme':'supersetColors','label_colors':{},'shared_label_colors':{}}))
db.session.add(dashboard)
db.session.commit()
print(json.dumps({'phase':args.phase,'dashboard_id':dashboard.id,'slug':dashboard.slug,'chart_ids':[item.id for item in charts],'dataset_id':source.id}))
# #endregion FullFlow.ChartHealth.CanarySeed.Main
if __name__ == '__main__':
main()
# #endregion FullFlow.ChartHealth.CanarySeed

View File

@@ -0,0 +1,70 @@
<!-- #region ScenarioRunMonitor.Component.ChartHealth [C:4] [TYPE Component] [SEMANTICS chart,health,errors,coverage,diagnostic] -->
<!-- @BRIEF Show named failed charts, tab context and partial coverage before technical evidence disclosures. -->
<!-- @UX_STATE absent -> hidden by parent; observed -> counts and stable tab groups; partial -> explicit unchecked/unresolved counts alongside FAILED; expanded -> per-placement cause, filters/time/digest and owned evidence. -->
<!-- @UX_STATE TabFilter(all|stableTabId): local display filter never mutates verdict, coverage or stored observations. -->
<!-- @UX_REACTIVITY groups/runId derive only from the parent-selected run; keyed run selection remounts local filters/disclosures. -->
<!-- @INVARIANT Model explanation never overwrites confirmed query failure and query failures never render numeric delta. -->
<script lang="ts">
import { chartErrorCauses } from "./chart-health";
import type { ChartHealthGroup } from "./chart-health";
import EvidenceViewer from "./EvidenceViewer.svelte";
import { Button } from "$lib/ui";
let { groups, runId, onselectstep }: { groups: ChartHealthGroup[]; runId: string; onselectstep?: (_id: string) => void } = $props();
let tab = $state("");
let chart = $state("");
let showAll = $state(false);
</script>
<section class="mt-4 rounded-lg border border-border bg-surface p-3" aria-label="Chart query health">
<h3 class="font-semibold text-text">Chart query health</h3>
{#each groups as group (group.stepId)}
<div class="mt-3 rounded border border-border p-3">
<p class="font-medium text-destructive">{group.errors.length} charts failed <span class="text-text-muted">· {group.status}</span></p>
{#each chartErrorCauses(group.errors) as cause (cause.cause)}
<p class="mt-1 text-sm">{cause.cause} · {cause.charts.length} charts: {cause.charts.map(item => item.chart_name).join(", ")}</p>
{/each}
{#if Object.keys(group.coverage).length}
<p class="mt-1 text-sm text-text-muted">Checked {String(group.coverage.checked ?? 0)} / {String(group.coverage.expected ?? 0)} · unresolved {String(group.coverage.unresolved ?? 0)} · unvisited {String(group.coverage.unvisited ?? 0)} · {group.coverage.complete === true ? "Complete coverage" : "Partial coverage"}</p>
{/if}
{#if group.tabs.length}
<label class="mt-3 block text-sm">Tab
<select class="ml-2 rounded border border-border bg-surface p-1" bind:value={tab}>
<option value="">All tabs</option>
{#each group.tabs as item (String(item.tab_id))}<option value={String(item.tab_id)}>{String(item.name)} · {String(item.errored)} errors</option>{/each}
</select>
</label>
{/if}
<label class="mt-2 block text-sm">Chart
<select class="ml-2 rounded border border-border bg-surface p-1" bind:value={chart}>
<option value="">All charts</option>{#each group.charts as item (item.placement_id)}<option value={item.placement_id}>{item.chart_name} · {item.status ?? "unavailable"}</option>{/each}
</select>
</label>
<label class="mt-2 block text-sm"><input type="checkbox" bind:checked={showAll} /> Show healthy and unresolved observations</label>
<ul class="mt-2 space-y-2">
{#each group.charts.filter(error => (showAll || error.status === "failed" || chart === error.placement_id) && (!tab || error.tab_path.includes(tab)) && (!chart || error.placement_id === chart)) as error (error.placement_id)}
<li class="rounded border border-border p-2">
<strong>{error.chart_name}</strong><span class="ml-2 text-xs">{error.status ?? "failed"}</span><span class="ml-2 text-xs text-text-muted">{group.tabs.find(item => item.tab_id === error.tab_path.at(-1))?.name ?? error.tab_path.join(" / ")}</span>
<p class="mt-1 whitespace-pre-wrap break-words text-sm">{error.message}</p>
{#if error.database_code}<p class="text-xs text-text-muted">Database code {error.database_code}</p>{/if}
<details class="mt-2 text-xs text-text-muted"><summary>Diagnostic evidence</summary>
<p class="mt-1 break-all">{error.placement_id} · chart {error.chart_id ?? "unavailable"} · {error.origin ?? "unavailable"}</p>
<p class="break-all">{error.observed_at ?? "unavailable"} · filters {error.filter_fingerprint ?? "unobserved"}</p>
{#if error.truncated}<p>Message truncated; retained coverage remains explicit.</p>{/if}
<p class="break-all">{error.sha256 ?? "See step evidence"}</p>
{#if error.artifact_id && runId}<EvidenceViewer {runId} artifactId={error.artifact_id} contentType="application/json" sha256={error.sha256} label={error.chart_name} />{/if}
</details>
</li>
{/each}
</ul>
{#if group.evaluations.length}
<details class="mt-3 text-sm"><summary>Model review by tab (separate from query health)</summary>
{#each group.evaluations as evaluation, index (index)}
<p class="mt-2">{Array.isArray(evaluation.tab_path) ? evaluation.tab_path.join(" / ") : "Tab"} · {String(evaluation.status)} · {String(evaluation.advisory_verdict ?? "inconclusive")} · {evaluation.coverage_complete === true ? "Captured requested frames" : "Incomplete visual evidence"}</p>
{#if Array.isArray(evaluation.findings)}{#each evaluation.findings as finding, index (index)}<p class="text-text-muted">{String(finding.message ?? "")}</p>{/each}{/if}
{/each}
</details>
{/if}
<Button variant="ghost" class="mt-2" onclick={() => onselectstep?.(group.stepId)}>Open step evidence</Button>
</div>
{/each}
</section>
<!-- #endregion ScenarioRunMonitor.Component.ChartHealth -->

View File

@@ -12,9 +12,12 @@
<!-- @UX_STATE Selection(none|selected): failure/deviation controls emit onselectstep; the parent-owned selectedStepId drives the run-bound inspector and suppresses the matching duplicate deviation card. -->
<!-- @UX_STATE Disclosures(collapsed|expanded): provenance and failure technical fields expand locally; failed checks and unavailable evidence retain authoritative statuses. -->
<!-- @UX_REACTIVITY result, steps, plan, runId and selectedStepId come from the same parent-selected run; selection callbacks do not fetch or alter runtime results. -->
<!-- @UX_STATE ChartHealth(absent|observed|partial): named persisted chart query failures render separately from numeric comparisons; partial coverage remains visible beside FAILED. -->
<script lang="ts">
import { t } from "$lib/i18n/index.svelte.js";
import type { ScenarioExecutionResult, ScenarioStepRun } from "$lib/types/scenario-run";
import ChartHealthCard from "./ChartHealthCard.svelte";
import { chartHealthGroups } from "./chart-health";
import RunStepInspector from "./RunStepInspector.svelte";
import MetricDeviationCard from "./MetricDeviationCard.svelte";
import { Button } from "$lib/ui";
@@ -40,6 +43,7 @@
} = $props();
const dt = $derived($t.dashboard_testing ?? {});
const healthGroups = $derived(chartHealthGroups(steps));
const differences = $derived(metricDifferences(steps));
const hasPinnedBaselines = $derived(hasPinnedMetricBaselines(plan));
const failedComparisons = $derived(steps.filter((step) => persistedComparison(step)?.status === "fail").length);
@@ -75,6 +79,7 @@
<p class="mt-1 text-sm text-text-muted" role="status">{dt.result_metric_unproven}</p>
{:else}<p class="mt-1 text-sm text-text-muted">{dt.result_metric_no_mismatch}</p>{/if}
</section>
{#if healthGroups.length}{#key runId}<ChartHealthCard groups={healthGroups} {runId} {onselectstep} />{/key}{/if}
{#if result.failures && result.failures.length > 0}
<section class="mt-4" aria-label={dt.result_failed_aria}>
<h3 class="text-sm font-medium text-text">{dt.result_failed_summary} ({result.failures.length})</h3>

View File

@@ -0,0 +1,46 @@
// #region Test.ScenarioRunMonitor.ChartHealth [C:3] [TYPE Module] [SEMANTICS named-errors,coverage,model,context]
// @TEST_INVARIANT Named query failures and partial coverage survive model PASS and never become numeric deltas.
// @RELATION BINDS_TO -> [ScenarioRunMonitor.Component.ChartHealth]
import { fireEvent, render, screen, within } from "@testing-library/svelte";
import { expect, it } from "vitest";
import ChartHealthCard from "../ChartHealthCard.svelte";
import type { ChartHealthGroup } from "../chart-health";
const names = ["Debt", "Cash", "Margin", "Forecast", "Reserve"];
const failed = names.map((chart_name, index) => ({ status: "failed", placement_id: `CHART-${index+1}`,
chart_id: index+1, chart_name, tab_path: ["TAB-daily"], message: "Code: 386. NO_COMMON_TYPE", database_code: "386" }));
const healthy = { status: "healthy", placement_id: "CHART-6", chart_id: 6, chart_name: "Healthy zero", tab_path: ["TAB-other"], message: "" };
const group: ChartHealthGroup = { stepId: "chart-health", status: "failed", errors: failed, charts: [...failed, healthy],
coverage: { expected: 7, checked: 6, unresolved: 0, unvisited: 1, complete: false },
tabs: [{ tab_id: "TAB-daily", name: "Daily summary", errored: 5 }, { tab_id: "TAB-other", name: "Other", errored: 0 }],
evaluations: [{ tab_path: ["TAB-daily"], status: "succeeded", advisory_verdict: "pass", coverage_complete: false,
findings: [{ message: "All visible cards look normal" }] }] };
// #region Test.ScenarioRunMonitor.ChartHealth.Causes [C:2] [TYPE Function]
it("shows five named Code386 placements and explicit unchecked coverage despite advisory PASS", () => {
render(ChartHealthCard, { props: { groups: [group], runId: "run-a" } });
const region = screen.getByRole("region", { name: "Chart query health" });
const items = within(region).getAllByRole("listitem");
expect(items).toHaveLength(5);
names.forEach((name, index) => expect(within(items[index]).getByText(name, { exact: true })).toBeTruthy());
expect(within(region).getByText(/Checked 6 \/ 7.*unvisited 1.*Partial coverage/)).toBeTruthy();
expect(within(region).getByText(/5 charts failed/)).toBeTruthy();
expect(within(region).getByText(/pass.*Incomplete visual evidence/)).toBeTruthy();
expect(within(region).queryByText(/delta|expected value|actual value/i)).toBeNull();
});
// #endregion Test.ScenarioRunMonitor.ChartHealth.Causes
// #region Test.ScenarioRunMonitor.ChartHealth.Selection [C:2] [TYPE Function]
it("selects the actual healthy placement and its tab without changing failed coverage", async () => {
render(ChartHealthCard, { props: { groups: [group], runId: "run-a" } });
await fireEvent.change(screen.getByLabelText("Tab"), { target: { value: "TAB-other" } });
await fireEvent.change(screen.getByLabelText("Chart"), { target: { value: "CHART-6" } });
const items = screen.getAllByRole("listitem");
expect(items).toHaveLength(1);
expect(within(items[0]).getByText("Healthy zero", { exact: true })).toBeTruthy();
expect(within(items[0]).getByText("healthy", { exact: true })).toBeTruthy();
expect(screen.getByText(/5 charts failed/)).toBeTruthy();
expect(screen.getByText(/Partial coverage/)).toBeTruthy();
});
// #endregion Test.ScenarioRunMonitor.ChartHealth.Selection
// #endregion Test.ScenarioRunMonitor.ChartHealth

View File

@@ -0,0 +1,92 @@
// #region ScenarioRunMonitor.ChartHealth [C:3] [TYPE Module] [SEMANTICS health,errors,coverage,projection]
// @BRIEF Project persisted chart errors and coverage without inferring numeric deviations or changing verdicts.
import type { ScenarioStepRun } from "$lib/types/scenario-run";
export interface ChartError {
status?: string; chart_id?: number; placement_id: string; chart_name: string; tab_path: string[];
message: string; database_code?: string; application_code?: string; observed_at?: string;
filter_fingerprint?: string; origin?: string; truncated?: boolean; artifact_id?: string; sha256?: string;
}
export interface ChartHealthGroup {
stepId: string; status: string; errors: ChartError[]; charts: ChartError[]; coverage: Record<string, unknown>;
tabs: Record<string, unknown>[]; evaluations: Record<string, unknown>[];
}
// #region ScenarioRunMonitor.ChartHealth.Record [C:1] [TYPE Function]
// @BRIEF Accept object envelopes only.
function record(value: unknown): Record<string, unknown> | null {
return value !== null && typeof value === "object" && !Array.isArray(value) ? value as Record<string, unknown> : null;
}
// #endregion ScenarioRunMonitor.ChartHealth.Record
// #region ScenarioRunMonitor.ChartHealth.Groups [C:3] [TYPE Function]
// @POST Only persisted explicit failed diagnostics establish query errors; healthy/uncertain observations remain separate.
export function chartHealthGroups(steps: ScenarioStepRun[]): ChartHealthGroup[] {
const groups: ChartHealthGroup[] = [];
for (const step of steps) {
const group = healthGroup(step);
if (group) groups.push(group);
}
return groups;
}
// #endregion ScenarioRunMonitor.ChartHealth.Groups
// #region ScenarioRunMonitor.ChartHealth.Array [C:1] [TYPE Function]
// @BRIEF Persisted non-array fields project as absent collections.
function array(value: unknown): unknown[] {
return Array.isArray(value) ? value : [];
}
// #endregion ScenarioRunMonitor.ChartHealth.Array
// #region ScenarioRunMonitor.ChartHealth.Failed [C:2] [TYPE Function]
// @POST A query error requires explicit persisted failed status and placement/name identity.
function failedDiagnostic(value: unknown): boolean {
const item = record(value);
return item?.status === "failed" && typeof item.placement_id === "string" && typeof item.chart_name === "string";
}
// #endregion ScenarioRunMonitor.ChartHealth.Failed
// #region ScenarioRunMonitor.ChartHealth.Group [C:3] [TYPE Function]
// @POST The selected step keeps its own verdict, chart observations and independent visual/checked coverage.
function healthGroup(step: ScenarioStepRun): ChartHealthGroup | null {
const source = healthSource(step);
if (!source) return null;
const errors = array(source.diagnostics).filter(failedDiagnostic) as ChartError[];
return { stepId: step.logical_step_id, status: step.status, errors, ...healthFields(source.health, errors) };
}
// #endregion ScenarioRunMonitor.ChartHealth.Group
// #region ScenarioRunMonitor.ChartHealth.Source [C:2] [TYPE Function]
// @POST Nested persisted outcomes take precedence; absent health evidence remains hidden.
function healthSource(step: ScenarioStepRun): { health: Record<string, unknown> | null; diagnostics: unknown } | null {
const outer = record(step.step_outcome);
const outcome = record(outer?.step_outcome) ?? outer;
const health = record(outcome?.chart_health);
const diagnostics = health?.errors ?? outcome?.chart_diagnostics;
if (!health && !Array.isArray(diagnostics)) return null;
return { health, diagnostics };
}
// #endregion ScenarioRunMonitor.ChartHealth.Source
// #region ScenarioRunMonitor.ChartHealth.Fields [C:2] [TYPE Function]
// @POST Additive observed collections never reinterpret the selected step verdict.
function healthFields(health: Record<string, unknown> | null, errors: ChartError[]): Omit<ChartHealthGroup, "stepId" | "status" | "errors"> {
return {
charts: (Array.isArray(health?.charts) ? health.charts : errors) as ChartError[],
coverage: record(health?.coverage) ?? {}, tabs: array(health?.tabs) as Record<string, unknown>[],
evaluations: array(health?.tab_evaluations) as Record<string, unknown>[] };
}
// #endregion ScenarioRunMonitor.ChartHealth.Fields
// #region ScenarioRunMonitor.ChartHealth.Causes [C:2] [TYPE Function]
// @POST Shared error codes group cause summaries without losing any affected placement or its individual message.
export function chartErrorCauses(errors: ChartError[]): { cause: string; charts: ChartError[] }[] {
const groups = new Map<string, ChartError[]>();
for (const error of errors) {
const cause = error.database_code ? `Database code ${error.database_code}` : error.application_code ?? error.message;
groups.set(cause, [...(groups.get(cause) ?? []), error]);
}
return [...groups.entries()].map(([cause, charts]) => ({ cause, charts }));
}
// #endregion ScenarioRunMonitor.ChartHealth.Causes
// #endregion ScenarioRunMonitor.ChartHealth

View File

@@ -0,0 +1,70 @@
#!/usr/bin/env python3
# #region Tooling.ChartHealth.NativeCanary [C:4] [TYPE Module] [SEMANTICS local,isolated,recovery,privileged]
# @PRE Parent authorizes access to the existing isolated laboratory; no build/deploy/data mutation is performed.
# @POST Error and recovered phases run real production observer checks; recovery is attempted in finally even if error check fails.
import argparse
import json
import os
from pathlib import Path
import subprocess
import sys
ROOT = Path(__file__).resolve().parents[1]
# #region Tooling.ChartHealth.NativeCanary.Seed [C:3] [TYPE Function]
# @POST Only native namespaced metric metadata changes; captured Superset startup logs/credentials are never emitted.
def seed(env_file, phase):
command = ['docker','compose','-p','ss-tools-full-flow','--env-file',str(env_file),
'-f','docker-compose.full-flow.yml','-f','docker-compose.full-flow.finance.yml',
'exec','-T','superset-preprod','/app/.venv/bin/python','/fixture/chart_health_canary_seed.py','--phase',phase]
result = subprocess.run(command,cwd=ROOT,capture_output=True,text=True,timeout=120)
if result.returncode:
raise RuntimeError(f'CHART_HEALTH_NATIVE_SEED_FAILED_{result.returncode}')
records = [json.loads(line) for line in result.stdout.splitlines() if line.startswith('{"phase":')]
if len(records) != 1:
raise RuntimeError('CHART_HEALTH_NATIVE_SEED_IDENTITY_UNAVAILABLE')
return records[0]
# #endregion Tooling.ChartHealth.NativeCanary.Seed
# #region Tooling.ChartHealth.NativeCanary.Verify [C:3] [TYPE Function]
# @POST Test output and phase result are retained separately; a failure is not relabeled PASS.
def verify(args, record):
environment = {**os.environ,'CHART_HEALTH_NATIVE_ENV_FILE':str(args.env_file),
'CHART_HEALTH_NATIVE_URL':'http://127.0.0.1:18112','CHART_HEALTH_NATIVE_DASHBOARD_ID':str(record['dashboard_id']),
'CHART_HEALTH_NATIVE_PHASE':record['phase'],'CHART_HEALTH_NATIVE_OUTPUT':str(args.output/record['phase'])}
leaf = 'test_chart_health_native_dom_worker.py' if args.diagnose else 'test_chart_health_native_canary.py'
command = [str(ROOT/'backend/.venv/bin/python'),'-m','pytest','-q','--run-integration',
f'tests/services/dashboard_testing/registry/{leaf}']
result = subprocess.run(command,cwd=ROOT/'backend',env=environment,capture_output=True,text=True,timeout=700)
(args.output/f'{record["phase"]}-pytest.log').write_text(result.stdout+result.stderr)
return result.returncode
# #endregion Tooling.ChartHealth.NativeCanary.Verify
# #region Tooling.ChartHealth.NativeCanary.Main [C:4] [TYPE Function]
# @POST A fresh evidence directory preserves both phases; only both actual checks yield exit0.
def main():
parser = argparse.ArgumentParser()
parser.add_argument('--env-file',type=Path,required=True)
parser.add_argument('--output',type=Path,required=True)
parser.add_argument('--diagnose',action='store_true')
args = parser.parse_args()
args.output = args.output.resolve()
args.output.mkdir(parents=True,exist_ok=False)
results = {'error':None,'recovered':None}
try:
record = seed(args.env_file,'error')
results['error'] = verify(args,record)
finally:
record = seed(args.env_file,'recovered')
results['recovered'] = verify(args,record)
(args.output/'phase-results.json').write_text(json.dumps(results,indent=2))
print(json.dumps({'scope':'native DOM diagnostic only' if args.diagnose else 'isolated native canary','phase_exit_codes':results}))
return 0 if results == {'error':0,'recovered':0} else 1
# #endregion Tooling.ChartHealth.NativeCanary.Main
if __name__ == '__main__':
sys.exit(main())
# #endregion Tooling.ChartHealth.NativeCanary

View File

@@ -41,7 +41,7 @@
# @INVARIANT Acquire load semaphore before request reaches shared client semaphore; never manually re-acquire shared semaphore.
# @INVARIANT Breaker/stop prevents new queue intake; in-flight requests drain within deadline.
# @DATA_CONTRACT ValidatedLoadProfile -> LoadRun + LoadExecution[] + LoadRunAggregate
# @RELATION CALLS -> [SupersetClient.ChartData.Execute]
# @REJECTED Historical design-stub CALLS -> [SupersetClient.ChartData.Execute] is retired: the canonical RunnerPool invokes an injected executor, so a direct chart-data call is not established by this module.
# @RELATION DEPENDS_ON -> [Core.Manager.CreateTask]
# @RELATION DEPENDS_ON -> [Core.ClientRegistry.GetSemaphore]
# @RATIONALE Async workers directly model executions; HTTP pool alone hides queue pressure.

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,650 @@
{
"baseline": "54bdbcc4",
"historical_source_triples": 123,
"valid_required_triples": 122,
"valid_indexed_triples": 122,
"missing_valid_indexed_triples": [],
"required_triples": [
[
"Api.DashboardTesting",
"DEPENDS_ON",
"BaselineEngine.QueryExecutor.ExecuteQuery"
],
[
"Api.DashboardTesting.ExecuteQuery",
"CALLS",
"BaselineEngine.QueryExecutor.ExecuteQuery"
],
[
"Api.DashboardTesting.Scenario",
"DEPENDS_ON",
"ScenarioGraph.Compiler.Compile"
],
[
"Api.DashboardTesting.Scenario",
"DEPENDS_ON",
"ScenarioGraph.Validator.Validate"
],
[
"Api.ScenarioLiveBindings",
"DEPENDS_ON",
"ScenarioExecution.LiveBinding.Identity"
],
[
"BaselineEngine.Api",
"DEPENDS_ON",
"BaselineEngine.QueryExecutor.Execute"
],
[
"BaselineEngine.Candidates.Capture",
"DEPENDS_ON",
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
],
[
"BaselineEngine.Inheritance",
"DEPENDS_ON",
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
],
[
"BaselineEngine.Inheritance.Execute",
"DEPENDS_ON",
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
],
[
"BaselineEngine.QueryExecutor.Execute",
"DEPENDS_ON",
"SupersetClient.ChartData.Execute"
],
[
"BaselineEngine.Verification.ExecutorMetric.Async",
"DEPENDS_ON",
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
],
[
"Core.ConfigModels.ScenarioLiveExecutionBinding",
"DEPENDS_ON",
"ScenarioExecution.LiveBinding.Identity"
],
[
"LoadTesting.Executor.ExecuteSupersetChart",
"CALLS",
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
],
[
"McpInterface.ScenarioPipeline",
"DEPENDS_ON",
"ScenarioGraph.Compiler.Compile"
],
[
"McpServer.ScenarioTools",
"CALLS",
"ScenarioGraph.Compiler.Compile"
],
[
"McpServer.ScenarioTools",
"CALLS",
"ScenarioGraph.Validator.Validate"
],
[
"McpServer.ToolsScenario",
"CALLS",
"ScenarioGraph.Compiler.Compile"
],
[
"McpServer.TraversalGuidance.Build",
"DEPENDS_ON",
"ScenarioGraph.Templates"
],
[
"ScenarioEditor.Apply",
"DEPENDS_ON",
"ScenarioGraph.Validator.Validate"
],
[
"ScenarioExecution.BrowserProvider",
"DEPENDS_ON",
"ScenarioExecution.BrowserProvider.Admission"
],
[
"ScenarioExecution.BrowserProvider",
"DEPENDS_ON",
"ScenarioExecution.BrowserProvider.Transport"
],
[
"ScenarioExecution.BrowserProvider.Admission",
"DEPENDS_ON",
"ScenarioExecution.LiveBinding.Identity"
],
[
"ScenarioExecution.BrowserProvider.NativeFilter",
"IMPLEMENTS",
"ScenarioExecution.BrowserProvider.Transport"
],
[
"ScenarioExecution.BrowserProvider.ReadOnlyActions",
"IMPLEMENTS",
"ScenarioExecution.BrowserProvider.Transport"
],
[
"ScenarioExecution.BrowserProvider.ReadOnlyActions.ObserveFlows",
"IMPLEMENTS",
"ScenarioExecution.BrowserProvider.Transport"
],
[
"ScenarioExecution.BrowserProvider.Reconciler",
"DEPENDS_ON",
"ScenarioExecution.BrowserProvider.Transport"
],
[
"ScenarioExecution.BrowserProvider.Session",
"DEPENDS_ON",
"ScenarioExecution.BrowserProvider.Transport"
],
[
"ScenarioExecution.BrowserProvider.Session.Registry",
"DEPENDS_ON",
"ScenarioExecution.BrowserProvider.Transport"
],
[
"ScenarioExecution.BrowserProvider.TableEvidence",
"DEPENDS_ON",
"ScenarioExecution.BrowserProvider.Admission.Evidence"
],
[
"ScenarioExecution.BrowserProvider.TableEvidence.Observation",
"CALLS",
"ScenarioExecution.BrowserProvider.Admission.Evidence"
],
[
"ScenarioExecution.BrowserProvider.Transport.Factory",
"DEPENDS_ON",
"ScenarioExecution.BrowserProvider.Transport"
],
[
"ScenarioExecution.BrowserProvider.Transport.MutationFlow",
"DEPENDS_ON",
"ScenarioExecution.BrowserProvider.Transport"
],
[
"ScenarioExecution.DecisionPolicy",
"CALLED_BY",
"ScenarioExecution.Runner.Walker"
],
[
"ScenarioExecution.EvaluationAdapter",
"CALLS",
"ScenarioExecution.AgentEvaluation"
],
[
"ScenarioExecution.EvaluationBinding",
"CALLED_BY",
"ScenarioExecution.Runner.Walker"
],
[
"ScenarioExecution.EvaluationImages",
"CALLED_BY",
"ScenarioExecution.EvaluationAdapter"
],
[
"ScenarioExecution.EvaluationPrompt",
"CALLED_BY",
"ScenarioExecution.EvaluationAdapter"
],
[
"ScenarioExecution.Executors.SqlEvidence",
"DEPENDS_ON",
"ScenarioExecution.LiveBinding.SqlEvidenceAdapter"
],
[
"ScenarioExecution.LiveBinding",
"CALLS",
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
],
[
"ScenarioExecution.LiveBinding.SqlEvidenceAdapter",
"CALLS",
"ScenarioExecution.LiveBinding.Execute"
],
[
"ScenarioExecution.LiveCanaryV5",
"VERIFIES",
"ScenarioExecution.Runner.Walker"
],
[
"ScenarioExecution.LiveCompositionRoot",
"CALLS",
"ScenarioExecution.EvaluationAdapter.Factory"
],
[
"ScenarioExecution.LiveCompositionRoot",
"CALLS",
"ScenarioExecution.LiveBinding.SupersetAdapter"
],
[
"ScenarioExecution.LiveCompositionRoot",
"DEPENDS_ON",
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
],
[
"ScenarioExecution.LiveCompositionRoot.EvaluationAdapter",
"CALLS",
"ScenarioExecution.EvaluationAdapter.Factory"
],
[
"ScenarioExecution.ProviderPreflight",
"DEPENDS_ON",
"ScenarioGraph.Templates.ActionRegistry"
],
[
"ScenarioExecution.Runner",
"CALLS",
"ScenarioExecution.LiveBinding.SupersetAdapter"
],
[
"ScenarioExecution.Runner.ConfiguredBinding",
"DEPENDS_ON",
"ScenarioExecution.LiveBinding.Identity"
],
[
"ScenarioExecution.Runner.CrashRecovery",
"CALLS",
"ScenarioExecution.Runner.TerminalSignal"
],
[
"ScenarioExecution.Runner.CrashRecovery",
"CALLS",
"ScenarioExecution.Runner.Walker"
],
[
"ScenarioExecution.Runner.CrashRecovery",
"DEPENDS_ON",
"ScenarioExecution.LiveBinding.Identity"
],
[
"ScenarioExecution.Runner.DefaultRegistry",
"CALLS",
"ScenarioExecution.LiveBinding.SupersetAdapter"
],
[
"ScenarioExecution.Runner.QueuedDispatch",
"CALLS",
"ScenarioExecution.Runner.TerminalSignal"
],
[
"ScenarioExecution.Runner.QueuedDispatch",
"CALLS",
"ScenarioExecution.Runner.Walker"
],
[
"ScenarioExecution.Traversal.PinnedInputs.Resolve",
"CALLS",
"ScenarioExecution.Traversal.Inputs.Parse"
],
[
"ScenarioExecution.Traversal.PinnedInputs.Resolve",
"CALLS",
"ScenarioExecution.Traversal.PinnedInputs.Projection"
],
[
"ScenarioGraph.Api",
"DEPENDS_ON",
"ScenarioGraph.Compiler.Compile"
],
[
"ScenarioGraph.Api",
"DEPENDS_ON",
"ScenarioGraph.Validator.Validate"
],
[
"ScenarioGraph.Compiler.ChainEmission",
"CALLED_BY",
"ScenarioGraph.Compiler.CompileGraph"
],
[
"ScenarioGraph.Compiler.ChainEmission",
"DEPENDS_ON",
"ScenarioGraph.Compiler.BuildStep"
],
[
"ScenarioGraph.Compiler.ChainEmission",
"DEPENDS_ON",
"ScenarioGraph.Templates"
],
[
"ScenarioGraph.MetricEvaluationProvider",
"DEPENDS_ON",
"ScenarioGraph.Models.MetricTextEvaluationRecipe"
],
[
"ScenarioGraph.MetricEvaluationRecipe",
"DEPENDS_ON",
"ScenarioGraph.Models.MetricTextEvaluationRecipe"
],
[
"ScenarioGraph.MetricRecipeStepInputs.Validate",
"CALLS",
"ScenarioGraph.StepInputs.EmbeddedLiterals"
],
[
"ScenarioGraph.MetricResultSchema",
"DEPENDS_ON",
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
],
[
"ScenarioGraph.ServerOwnedPipeline",
"DEPENDS_ON",
"ScenarioGraph.Compiler.Compile"
],
[
"ScenarioGraph.ServerOwnedPipeline",
"DEPENDS_ON",
"ScenarioGraph.Validator.Validate"
],
[
"ScenarioGraph.StepInputs",
"CALLED_BY",
"ScenarioGraph.Validator.CheckStepInputs"
],
[
"Stage6.TableTextRecipeSlice",
"DEPENDS_ON",
"ScenarioGraph.Models.MetricTextEvaluationRecipe"
],
[
"SupersetBaselineEngine.QA.Audit",
"VERIFIES",
"BaselineEngine.QueryExecutor.ExecuteQuery"
],
[
"Test.DashboardTesting.CandidateCapture",
"BINDS_TO",
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
],
[
"Test.DashboardTesting.ChartDataRaw",
"BINDS_TO",
"SupersetClient.ChartData.Execute"
],
[
"Test.DashboardTesting.MetricExecutorCatalog",
"VERIFIES",
"BaselineEngine.QueryExecutor.Envelope"
],
[
"Test.DashboardTesting.MetricExecutorCatalog",
"VERIFIES",
"BaselineEngine.QueryExecutor.ExecuteQueryEnvelope"
],
[
"Test.DashboardTesting.QueryExecutor",
"VERIFIES",
"BaselineEngine.QueryExecutor.ExecuteQuery"
],
[
"Test.EvaluationText.BrowserCommitFrontier",
"BINDS_TO",
"ScenarioExecution.Runner.Walker"
],
[
"Test.EvaluationText.SubmitRuntime",
"BINDS_TO",
"ScenarioExecution.AgentEvaluation"
],
[
"Test.MetricBrowser.InputAuthority",
"BINDS_TO",
"ScenarioExecution.BrowserProvider.Admission.Gate"
],
[
"Test.MetricTextRecipe.TokenContext",
"BINDS_TO",
"ScenarioGraph.Validator.Validate"
],
[
"Test.MetricTextRecipe.ValidatorAuthority",
"BINDS_TO",
"ScenarioGraph.Validator.Validate"
],
[
"Test.Scenario.Compiler",
"BINDS_TO",
"ScenarioGraph.Compiler.Compile"
],
[
"Test.Scenario.MetricGraphV2Authority",
"BINDS_TO",
"ScenarioGraph.Compiler.Compile"
],
[
"Test.Scenario.Validator",
"BINDS_TO",
"ScenarioGraph.Validator.Validate"
],
[
"Test.Scenario.ValidatorBelief",
"BINDS_TO",
"ScenarioGraph.Validator.Validate"
],
[
"Test.Scenario.ValidatorProperties",
"BINDS_TO",
"ScenarioGraph.Validator.Validate"
],
[
"Test.ScenarioAutomation.NotificationsWiring",
"VERIFIES",
"ScenarioExecution.Runner.TerminalSignal"
],
[
"Test.ScenarioEditor.StepInputs",
"BINDS_TO",
"ScenarioGraph.StepInputs"
],
[
"Test.ScenarioExecution.AgentEvaluation",
"BINDS_TO",
"ScenarioExecution.AgentEvaluation"
],
[
"Test.ScenarioExecution.AgentEvaluationStore",
"BINDS_TO",
"ScenarioExecution.AgentEvaluation"
],
[
"Test.ScenarioExecution.BrowserLimitsCleanup",
"BINDS_TO",
"ScenarioExecution.BrowserProvider.Transport"
],
[
"Test.ScenarioExecution.BrowserReadOnlyActions",
"BINDS_TO",
"ScenarioExecution.BrowserProvider.Admission"
],
[
"Test.ScenarioExecution.BrowserReadOnlyActions",
"BINDS_TO",
"ScenarioExecution.BrowserProvider.Transport"
],
[
"Test.ScenarioExecution.BrowserTableRetainedAuthority",
"BINDS_TO",
"ScenarioExecution.BrowserProvider.Transport"
],
[
"Test.ScenarioExecution.CancelTimeout",
"BINDS_TO",
"ScenarioExecution.Runner.Walker"
],
[
"Test.ScenarioExecution.CrashRecovery",
"BINDS_TO",
"ScenarioExecution.Runner.TerminalSignal"
],
[
"Test.ScenarioExecution.DueAdmissionAtomicity",
"VERIFIES",
"ScenarioExecution.Runner.TerminalSignal"
],
[
"Test.ScenarioExecution.EvaluationBinding",
"BINDS_TO",
"ScenarioExecution.Runner.Walker"
],
[
"Test.ScenarioExecution.LiveBinding",
"BINDS_TO",
"ScenarioExecution.LiveBinding"
],
[
"Test.ScenarioExecution.LiveBinding",
"VERIFIES",
"ScenarioExecution.LiveBinding.Evidence"
],
[
"Test.ScenarioExecution.LiveBinding",
"VERIFIES",
"ScenarioExecution.LiveBinding.Execute"
],
[
"Test.ScenarioExecution.LlmInjectionOffline",
"BINDS_TO",
"ScenarioExecution.EvaluationAdapter"
],
[
"Test.ScenarioExecution.MetricDispatchContinuation",
"BINDS_TO",
"ScenarioExecution.Runner.Walker"
],
[
"Test.ScenarioExecution.QueuedDispatch",
"BINDS_TO",
"ScenarioExecution.Runner.Walker"
],
[
"Test.ScenarioExecution.RetryClosure",
"BINDS_TO",
"ScenarioExecution.Runner.Walker"
],
[
"Test.ScenarioExecution.RetryDispatch",
"BINDS_TO",
"ScenarioExecution.Runner.TerminalSignal"
],
[
"Test.ScenarioExecution.SupersetProviderContract",
"BINDS_TO",
"ScenarioExecution.LiveBinding"
],
[
"Test.ScenarioExecution.SupersetProviderContract",
"VERIFIES",
"ScenarioExecution.LiveBinding.Evidence"
],
[
"Test.ScenarioExecution.SupersetProviderContract",
"VERIFIES",
"ScenarioExecution.LiveBinding.Execute"
],
[
"Test.ScenarioExecution.SupersetProviderContract",
"VERIFIES",
"ScenarioExecution.LiveBinding.Payload"
],
[
"Test.ScenarioExecution.TerminalSignals",
"BINDS_TO",
"ScenarioExecution.Runner.Walker"
],
[
"Test.ScenarioExecution.TerminalSignals",
"VERIFIES",
"ScenarioExecution.Runner.TerminalSignal"
],
[
"Test.ScenarioExecution.TraversalInputs",
"BINDS_TO",
"ScenarioExecution.Traversal.Inputs"
],
[
"Test.ScenarioExecution.TraversalJournal",
"BINDS_TO",
"ScenarioExecution.Traversal.Store"
],
[
"Test.ScenarioExecution.Walker",
"BINDS_TO",
"ScenarioExecution.Runner.Walker"
],
[
"Test.ScenarioExecution.Walker",
"VERIFIES",
"ScenarioExecution.Runner.Walker"
],
[
"Test.ScenarioGraph.AgentEvaluationModels",
"BINDS_TO",
"ScenarioGraph.Models.AgentEvaluationSpec"
],
[
"Test.ScenarioRunMonitor.InspectionResult",
"BINDS_TO",
"ScenarioRunMonitor.Component.Result"
],
[
"Test.SecurityOpsOffline",
"VERIFIES",
"ScenarioExecution.EvaluationAdapter.Raw"
],
[
"Test.Traversal.SampleJournal",
"BINDS_TO",
"ScenarioExecution.Traversal.Store"
],
[
"Tooling.FinanceSemanticCuration.Report",
"VERIFIES",
"ScenarioExecution.Traversal.Runtime.Provider.Factory"
],
[
"Tooling.FinanceSemanticCuration.Report",
"VERIFIES",
"ScenarioExecution.Traversal.Runtime.Provider.Submit"
],
[
"VerificationProgram.Reconciliation",
"DEPENDS_ON",
"ScenarioGraph.Templates.ActionRegistry"
]
],
"retired_documentation_assertion": {
"triple": [
"LoadTesting.RunnerPool",
"CALLS",
"SupersetClient.ChartData.Execute"
],
"authorization": "Root explicitly accepted122trueindexed +1 retired invaliddocumentation assertion",
"reason": "Duplicate fenced design contract says CALLS; canonical production pool invokes an injected executor. Direct chart-data call is not established. No production CALLS relation manufactured.",
"canonical_ID_preserved": "LoadTesting.RunnerPool",
"canonical_source_untouched": "backend/src/services/load_testing/runner_pool.py",
"incoming_references_unchanged": true
},
"metadata_scope_extension": [
{
"path": "specs/050-mcp-interface/plans/T029a-table-text-recipe-slice-2026-10-01.md",
"before_sha256": "8d53d368d499f0899338d191b5696537d5091b84d0e6da8c3ec13b779e86c5e1",
"after_sha256": "b7c27b3d8014227c7a470b50d9f27297ab58c901d1f4cffd02242706857e6ead",
"diff": "--- specs/050-mcp-interface/plans/T029a-table-text-recipe-slice-2026-10-01.md\n+++ specs/050-mcp-interface/plans/T029a-table-text-recipe-slice-2026-10-01.md\n@@ -6,7 +6,6 @@\n and heavy 40\u201360-second dashboard acceptance remain open.\n \n ## @{ Stage6.TableTextRecipeSlice [C:5] [TYPE ADR] [SEMANTICS recipe,text,context,provider,token-budget]\n-\n @BRIEF Closed owned metric/table recipe with exact filter provenance, honest provider aliases and nonmonetary limits.\n @RELATION DEPENDS_ON -> [Stage6.OwnedTextEvidenceSlice]\n @RELATION DEPENDS_ON -> [ScenarioGraph.MetricEvaluationRecipe.Compile]\n@@ -137,7 +136,6 @@\n ## @} Stage6.TableTextRecipeSlice\n \n ## @{ Stage6.TableTextCapacityNativeDecision [C:5] [TYPE ADR] [SEMANTICS recipe,capacity,native-filter,transactions]\n-\n @BRIEF Pending native scope and provider-wide capacity authority for the owned text recipe.\n @RELATION DEPENDS_ON -> [Stage6.TableTextRecipeSlice]\n @PRE Recipe scope and public provider identity are verified before a committed capacity claim or HTTP request.\n@@ -265,7 +263,6 @@\n ## @} Stage6.TableTextCapacityNativeDecision\n \n ## @{ Stage6.TableTextRuntimeInputDecision [C:5] [TYPE ADR] [SEMANTICS recipe,runtime,inputs,authority,snapshot]\n-\n @BRIEF Preserve registry snapshot identity while conveying admitted recipe inputs to the real browser provider.\n @RELATION DEPENDS_ON -> [Stage6.TableTextRecipeSlice]\n @PRE Runtime graph/step identity is proved against the persisted immutable plan before recipe input projection.\n"
},
{
"path": "specs/040-dashboard-load-testing/contracts/modules.md",
"before_sha256": "eb376a4eeddefdf4982daef439cc0661bdab70f890f0219bc0055067741dfa1f",
"after_sha256": "d33c10b84e7a00af4cdb14fb9dd03bd66838a77f13c2393edfe22b0884131bf9",
"diff": "--- specs/040-dashboard-load-testing/contracts/modules.md\n+++ specs/040-dashboard-load-testing/contracts/modules.md\n@@ -41,7 +41,7 @@\n # @INVARIANT Acquire load semaphore before request reaches shared client semaphore; never manually re-acquire shared semaphore.\n # @INVARIANT Breaker/stop prevents new queue intake; in-flight requests drain within deadline.\n # @DATA_CONTRACT ValidatedLoadProfile -> LoadRun + LoadExecution[] + LoadRunAggregate\n-# @RELATION CALLS -> [SupersetClient.ChartData.Execute]\n+# @REJECTED Historical design-stub CALLS -> [SupersetClient.ChartData.Execute] is retired: the canonical RunnerPool invokes an injected executor, so a direct chart-data call is not established by this module.\n # @RELATION DEPENDS_ON -> [Core.Manager.CreateTask]\n # @RELATION DEPENDS_ON -> [Core.ClientRegistry.GetSemaphore]\n # @RATIONALE Async workers directly model executions; HTTP pool alone hides queue pressure.\n"
}
],
"config_changes": null,
"limitations": [
"Snapshot after metadata-only retirement/header normalization; final combined owner source freeze, legacy parity and navigation receipts follow separately.",
"This is not123indexedPASS and not globalGO."
]
}

View File

@@ -6,7 +6,6 @@ Intermediate BLOCKED/pending entries remain chronological evidence. Global GO
and heavy 40–60-second dashboard acceptance remain open.
## @{ Stage6.TableTextRecipeSlice [C:5] [TYPE ADR] [SEMANTICS recipe,text,context,provider,token-budget]
@BRIEF Closed owned metric/table recipe with exact filter provenance, honest provider aliases and nonmonetary limits.
@RELATION DEPENDS_ON -> [Stage6.OwnedTextEvidenceSlice]
@RELATION DEPENDS_ON -> [ScenarioGraph.MetricEvaluationRecipe.Compile]
@@ -137,7 +136,6 @@ isolated Docker contour.
## @} Stage6.TableTextRecipeSlice
## @{ Stage6.TableTextCapacityNativeDecision [C:5] [TYPE ADR] [SEMANTICS recipe,capacity,native-filter,transactions]
@BRIEF Pending native scope and provider-wide capacity authority for the owned text recipe.
@RELATION DEPENDS_ON -> [Stage6.TableTextRecipeSlice]
@PRE Recipe scope and public provider identity are verified before a committed capacity claim or HTTP request.
@@ -265,7 +263,6 @@ inconclusive; its diagnosed runtime-input correction is described below.
## @} Stage6.TableTextCapacityNativeDecision
## @{ Stage6.TableTextRuntimeInputDecision [C:5] [TYPE ADR] [SEMANTICS recipe,runtime,inputs,authority,snapshot]
@BRIEF Preserve registry snapshot identity while conveying admitted recipe inputs to the real browser provider.
@RELATION DEPENDS_ON -> [Stage6.TableTextRecipeSlice]
@PRE Runtime graph/step identity is proved against the persisted immutable plan before recipe input projection.

View File

@@ -0,0 +1,285 @@
# Chart health, query errors and per-tab VLM — implementation handoff
## @{ ScenarioGraph.ChartHealth.Plan [C:4] [TYPE ADR] [SEMANTICS chart-health,query-errors,tabs,vlm,evidence]
@BRIEF Preserve Superset query failures as owned evidence, check every visited tab and expose actionable chart diagnostics to analysts and VLM.
@PRE Existing sampling, ownership, cancellation, registry pinning and baseline bindings remain authoritative.
@POST Confirmed chart failures are durable and attributable; partial coverage is explicit; VLM cannot erase a deterministic failure.
@INVARIANT Evidence identifies run, attempt, tab path, chart placement, filter state and load generation; stale or unrelated responses cannot establish a verdict.
@RATIONALE SQL-error cards can currently become generic timeouts, and the API adapter returns before retaining error evidence. Shared observation logic supports both a graph action and composite tab traversal without introducing a general nested-graph engine.
@REJECTED Screenshot-only detection, HTTP-status-only success, global unscoped error text searches and arbitrary nested graph execution were rejected for this slice.
## Authority and scope
### Implementation checkpoint — 2026-10-02
S1–S5 are connected in source and the focused runtime gates pass; acceptance is
still pending the independent semantic audit. The new authoring registry is `038.8.0`; historical
`038.6` full traversal and `038.7` sampling inputs/journals keep their meanings.
New `navigate_tabs` authoring explicitly enables chart health. Per-tab visual
evaluation is opt-in and requires a pinned active visual provider configuration.
Retained evidence so far:
- Independent literal HTTP/application/async/stale/transport and real owned
artifact/provider/whole-tab evaluation checks: **26 passed**. Three causal
failures were fixed: async202 must remain pending; malformed200 cannot confirm
query failure; auxiliary frame bytes must not be added twice to diagnostic
frontier bytes. Whole evaluation now captures actual PNG+Code386 contents,
retains raw/record artifacts and continues to a later chart. Advisory PASS
cannot erase a confirmed chart FAIL.
- Existing evaluation/models compatibility: **28 passed**. Semantic review has
no invented comparison IDs. The full authored provider pin stays in evidence;
only its exact validated64hex digest enters the existing64char capacity lease.
The actual retained lease uses the64char digest while the owned record keeps
the full78char pin. These checks use SQLite; actual PostgreSQL admission is
subsequently checked against an isolated real PostgreSQL Testcontainers instance:
**1 passed in4.86s** with authored78→committed64→released lease identity.
- Fake external SDK boundary: **2 passed**, with actual encrypted DB credential,
one physical request, retries disabled, declared timeout/output bounds,
transport usage and closure. No paid provider call occurred.
- Configured fake notifications: **5 passed**; analyst DOM: **2 passed**.
Durable summary is always retained. Optional delivery uses the existing global
`notifications.scenario_alerts` namespace and existing SMTP/TELEGRAM/SLACK
providers, after terminal commit. It is disabled by default; destinations
are explicit. Atomic pending-to-dispatching CAS gives at-most-once attempts,
with explicit skipped/failed results and no automatic ambiguous-send retry.
- Final combined focused backend boundary suite: **81 passed in3.07s**;
root privileged connected Chromium suite: **3 passed in3.66s**, including
all five named Code386 cards plus the healthy empty chart on another tab.
Checked coverage is6/6, errors5, complete=true, overall FAILED. There were
no late TargetClosed/unretrieved Future warnings in the returned output.
- All current touched Python surfaces pass explicit F/C901≤10 and production
module length<400. Final anchor,
relation and source-hash closure remains required before readiness claims.
Bounded post-review fixes and proofs:
- Frontend projection was split into cohesive source/field helpers. The explicit
TypeScript parser ESLint gate enforces complexity≤10 with zero errors/warnings;
analyst DOM checks remain **2 passed**. The repository's default ESLint config
does not match ordinary `.ts` files, so an ignored-file run is not accepted as
this proof.
- `038.6.0`/`038.7.0` pinned tab inputs now use their frozen original three-field
DTO instead of receiving additive health fields. **4 passed** cover exact
normalized old defaults, rejection of new flags, real committed projection,
unchanged persisted plan bytes, old input digest and journal resume. Literal
baseline derives from immutable `54bdbcc4`, sourceSHA
`1b6a3f5875f886cebd8e0a9d9b57eb69e5d44e51c6689352709cc5f8f398a87e`.
- Independent nested/offscreen Chromium: **1 passed in2.80s**. Simulated45/60s
budget checks: **2 passed**. These supplement rather than replace the complete
five-error connected Chromium proof.
- Native Superset/ClickHouse canary is prepared in
`docker/full-flow/chart_health_canary_seed.py` and
`scripts/chart-health-native-canary.py`. It creates only namespaced PREPROD
metadata, reads one existing real finance source row, provokes an actual
Int64/String metric type conflict, and recovers only its own metric in finally.
Both error/recovered phase checks retain actual observer+owned journal results;
they make no public release/authoring claim. Actual execution remains pending.
The first connected Chromium fixture retained five confirmed errors but reported
incomplete coverage: its empty table collapsed the next panel to zero height.
The fixture now gives cards their real Superset-style minimum height. The complete
coverage assertion is unchanged; the fresh privileged browser gate now passes.
Earlier failed receipts/results are retained and must not be counted as PASS.
Declared operational limits: chart health does not fix the separate expensive
mid-page navigation limitation of default pagination quantiles. Server logs and
unobserved async WebSocket bodies are not available evidence. Unresolved response
bodies/loading/transport stay INCONCLUSIVE; confirmed errors stay FAILED even
when chart or visual coverage is partial. No real notification or deployment has
been performed during this implementation packet.
User request: create an agent plan and delegate implementation of the discussed
query-error/chart-health/VLM flow. Work from current HEAD and preserve unrelated
maintenance, Git UI, login and translation changes. Follow INV1–7 and canonical
skills: self-implementation, semantics-core/python/testing; semantics-svelte for
UI changes. Required gates: real contract boundaries, preserved original IDs and
incoming relations, strict C901 <=10, production Python modules <400 lines.
## Current gaps to reproduce first
- `query_executor.py`: API errors become UNKNOWN values with warnings and an
error-context envelope; this is not necessarily the original HTTP body.
- `execution/live_binding.py`: UNKNOWN returns `SUPERSET_QUERY_FAILED` before
`_evidence_result`; other exceptions become generic execution errors.
- `execution/providers/browser_all_tabs.py`: waits for chart content before
checking an alert; catches the failure as INCONCLUSIVE and stops the sweep.
- `execution/evaluation_adapter.py`, `evaluation_manifest.py`,
`evaluation_prompt.py`: VLM gets images/manifest; ordinary JSON references
do not automatically provide the actual diagnostic contents to the model.
- No implemented `assert_chart_health` action or general nested graph runtime
has been established. Verify exact action/schema/registry seams before edits.
## Ordered implementation slices
### S1 — Typed diagnostic evidence and failure retention
Define a versioned diagnostic schema before implementation. Include run/step/
attempt, environment/dashboard, stable chart ID and placement/tab path, effective
filter fingerprint, load generation/request identity, timestamps/duration,
HTTP status, application/DB error code, bounded message, observation origin,
terminal loading state, evidence references/digests and truncation indicators.
Distinguish original response, sanitized retained payload and synthetic exception
context; never label a reconstructed JSON object as original wire evidence.
Retain evidence before returning FAILED/INCONCLUSIVE. Propagate references and
typed diagnostics through provider receipts, step outcomes and persistence.
Preserve real failure details on API exceptions; missing/unreadable evidence
must be explicit. Detect errors inside HTTP-200 payloads as well as non-2xx.
### S2 — Shared chart-health observation and graph action
Implement one common observer, used by a new admitted `assert_chart_health`
action and tab traversal. Add descriptor, input/output schemas, registry version,
executor routing and MCP explanations through existing authoritative mechanisms.
Never reinterpret an already pinned registry snapshot or existing saved plan.
Observe scoped DOM error cards and matched chart-data requests. Install network
listeners before the corresponding load/filter action; bound buffers and remove
listeners on completion/cancellation. A request must match the expected chart,
filter state and current load generation. Handle asynchronous query polling.
Check errors while waiting for readiness, not after requiring rendered content.
Exclude unrelated HTTP failures and stale responses from chart verdicts.
Successful empty results and zeros are health successes; data-presence/business
assertions belong to their own steps. Expected-error scenarios require explicit
typed expectations. Confirmed query failure is FAILED; unresolved loading or
unattributed/infrastructure-only observations are INCONCLUSIVE with evidence.
### S3 — Composite `navigate_tabs` integration and coverage
Keep `navigate_tabs` as a composite driver for this slice. For each stable tab:
activate ancestors/panel -> observe its chart placements -> capture evidence ->
optional VLM evaluation while the tab is active -> advance. Reuse S2; do not
duplicate detectors or add arbitrary graph callbacks/recursive graph execution.
Visit charts below the viewport and nested panels using bounded owned scrolling.
Resolve placements by stable IDs and paths, not titles. Continue after individual
confirmed chart failures to collect other findings. Persist per-chart/per-tab
results incrementally; cancellation, leases and ownership remain enforced.
Separate verdict from coverage: observed FAILED dominates even when coverage is
partial; otherwise incomplete coverage is INCONCLUSIVE; PASS requires complete
requested coverage. Show expected/checked/error/timeout/unvisited counts. Keep
pagination sampling independent from chart/tab health coverage. Document that
many tabs and 40–60 second queries can make this step take a long time.
### S4 — Per-tab VLM diagnostic context
When enabled, support explicit all-visited/selected/problem-area scope. Capture
and evaluate the chosen tab before leaving it; long tabs need multiple frames
with explicit coverage. Attach bounded *contents* of owned diagnostic artifacts
alongside screenshots and criteria, using the existing evidence-as-data boundary.
Artifact references alone are insufficient. Validate ownership/hashes and budget;
report omitted/truncated diagnostics or images and insufficient coverage.
The evaluation path must accept diagnostic evidence from failed observations
without allowing comparison of fabricated metric values. VLM may explain the
failure and assess appearance; it cannot change a confirmed deterministic FAIL
to PASS. Preserve deterministic and VLM verdicts separately. Do not silently
convert every existing visual evaluation into a paid per-tab VLM call.
Server-side Superset logs are optional future enrichment requiring a separate
correlated connector. HTTP responses, scoped DOM and browser request failures
are this slice's sources; do not claim access to server logs. Exclude credentials,
cookies and unnecessary sensitive response/query content from model payloads.
### S5 — Analyst results, graph and configured notifications
Show a summary such as “5 charts failed — Daily summary”, with tab/chart names,
cause, filters, time and proof links. Group equal causes while retaining every
affected placement. Expand/filter the composite graph node by tab/chart; keep
partial coverage visible alongside FAILED. Error charts have no numeric baseline
delta; preserve baseline comparisons for successful producers.
Use the existing automation notification machinery for run failures, with a
bounded chart-error summary and deduplication. Delivery depends on configured
channels; tests use fake delivery. Do not send real external messages during
implementation. Add UX_STATE before component code and keep technical details
collapsible beneath actionable explanations.
## Acceptance matrix
### Current acceptance ledger — 2026-10-02
All ten functional rows now have focused executable assertions passing. Final
acceptance remains open: timeout cleanup emits an unhandled Playwright future,
and native error detection fails. This ledger does not establish release GO.
| Row | Status | Actual proof / remaining subcase |
|---|---|---|
| 1 | PASS, protocol fixture | Real Chromium: five Code386 cards retained, healthy next tab checked, 6/6 coverage and deterministic FAILED. |
| 2 | PASS | Independent HTTP500, HTTP200 application error and owned asynchronous terminal-result fixtures. |
| 3 | PASS | Two independent 45/60-second simulated-clock cases; immediate error recorded at duration zero while slow chart uses its own budget. |
| 4 | PASS | Stale/retried/foreign request ownership and request-failure attribution; listener/task cleanup. |
| 5 | PASS | Successful zero and empty results remain healthy, including real browser empty table. |
| 6 | PASS | Real Chromium proves duplicate titles, nested paths, offscreen scrolling and same chart ID in distinct placements with failed/healthy owned observations. |
| 7 | PASS assertions; cleanup OPEN | Cancellation, unopenable-tab and renderer-timeout after confirmed FAIL retain exact partial coverage. The three placement cases pass, but renderer timeout emits an unhandled Playwright future; a strict cleanup oracle is pending. |
| 8 | PASS, fake external SDK | Owned diagnostic contents plus actual PNG cross the real evaluation boundary; digest/ownership/budget guards and advisory-PASS isolation proven. |
| 9 | PASS, focused compatibility | Version-pinned legacy DTO fixtures prove old plan bytes, normalized fields, input digest and journal resume; existing focused compatibility/ownership gates pass. |
| 10 | PASS, fake delivery | Two analyst DOM cases and five configured fake notification cases; no real external messages. |
Additional required native gate: existing Superset/ClickHouse stand is available.
The first actual error/recovery canary failed both phases before observations:
Superset layout contains a scalar `DASHBOARD_VERSION_KEY: v2`, which the health
manifest parser incorrectly treated as a node. The owner added a dictionary-node
guard and a literal regression case (legacy/layout combined gate: five PASS).
The second native canary completed: recovery PASS, error phase FAIL. Six charts
were observed, only one checked; five error-card locators timed out and produced
INCONCLUSIVE instead of confirmed failure. The owner is repairing native error
DOM handling and timeout cleanup. Output paths are retained separately per attempt.
The bounded repair checks current attributed API errors before waiting for
successful chart-renderer DOM. Native Superset5 saved-card identity is resolved
through its exact chart-grid-component/chart-id attributes; readiness is scoped
to the chart body, excluding header icons, while screenshots retain the whole
card. The existing per-chart deadline reserves 250ms for SDK drainage and never
starts a new RPC after the remaining budget expires. The independent expired-RPC
regression passes after its retained causal failure; response/owned-storage and
legacy/layout gates pass 23 cases. Actual native DOM inventory, strict native
error/recovery rerun and real Chromium timeout-cleanup verification remain open.
Additional completed gates: real isolated PostgreSQL capacity admission/release
one PASS, nested/offscreen Chromium one PASS, strict TypeScript complexity <=10
and Python C901 <=10. Final source hash/index closure follows the necessary native
manifest fix. Historical documentation edge retirement is explicit in the
semantic edge-policy receipt; it must not be counted as a preserved runtime CALLS.
Use meaningful fixtures with production paths and @TEST_INVARIANT traceability:
1. Five error cards like the user's screenshot: all five attributable findings
retained; other charts checked; overall FAILED, never generic PASS.
2. HTTP 500; HTTP 200 with application error; asynchronous job failure.
3. Slow success (40–60s simulated clock) beside an immediate failure; independent
budgets, prompt failure observation and cancellation responsiveness.
4. Late old-filter response, duplicate/retried request, unrelated failed endpoint.
5. Successful zero and empty data; no automatic query-error classification.
6. Duplicate titles, repeated chart placement, nested tabs, offscreen chart.
7. Unopenable tab/renderer timeout/cancellation after one confirmed failure:
durable FAILED plus partial coverage and exact unchecked counts.
8. VLM payload contains actual diagnostic text and scoped screenshots; missing
artifacts/budget truncation explicit; model PASS cannot erase query FAIL.
9. Existing pinned registry plans, metric/baseline flows and ownership/cleanup
gates retain behavior; no weakening old fixtures to make unrelated failures pass.
10. UI makes each affected chart/cause discoverable; fake notification receives
one grouped summary, with no live external delivery.
Run focused backend/frontend checks. Add a local Superset/ClickHouse fixture with
an intentional type-conflict query and recovery if the existing test stand is
available; preserve datasets and existing release evidence. Do not run a new
625-page sweep or real LLM billing just to validate these diagnostics. Record
actual commands/results and distinguish inherited failures from regressions.
## Agent result and review gates
Return <RESULT> with exact changed paths, completed S1–S5 status, evidence/test
commands, API/registry compatibility, semantic index/ID/edge receipts and remaining
limitations. Update this plan and main handoff with actual statuses. Delegate
independent verification and semantic curation at integration; retain review
evidence. No automatic global GO, commit, deployment or external notification.
Root handles privilege-only checks when sandbox blocks TestClient/Docker.
## @} ScenarioGraph.ChartHealth.Plan
## Commit checkpoint — 2026-10-02
User explicitly authorized commit and push. Post-fix real Chromium matrix: four PASS in 10.54s, including cleanup and expired-deadline guards, with no unhandled future warning. Affected independent checks: 28 PASS. Native Superset error/recovery acceptance and final semantic freeze remain open; this commit is not release GO.