T0: absorb shared/ into backend — cot_logger→src/core, CotJsonFormatter→src/core/cot_formatter.py, _llm_http/_llm_health/ssl→src/core/utils; imports rewritten (26 prod + tests, patch targets); run.sh/backend.Dockerfile/requirements/.axiom source_dirs/semantic_health/AGENTS/INSTALL cleaned; ADR-0022 supersedes ADR-0015; fixed latent CI defects (ss_tools ImportError, record.message in logger tests, same-name test-module collision). ADR-0021 wire enrichment (additive): contract_id/claim/error_code/loc fields; _contract_id ContextVar + resolve_contract_id (explicit > belief_scope > declared-src mirror, derived src never mirrors); EXPLORE auto-loc via single frame walk; facade error auto-fill; 2KB payload cap with payload_truncated/payload_bytes markers; migrated 85 error="CODE" sites to error_code= (12 files); pilot editor/load.py; superset preview payload-bomb inlined bodies removed. Analytics SSOT src/core/log_stats.py (bond transition matrix, orphan-EXPLORE ratio, REFLECT pairing, intent families, coverage, insufficient-sample flag); pretty_cot.py --stats/--digest/--trajectory/--story over one engine; log_gap_service three-tier ground-truth triangulation (FAILED w/o EXPLORE etc.) + GET /api/reports/log-stats|task-log-gaps (polling-suppressed); scripts/cot_audit.py CLI; enriched fields persisted into task_logs.payload for tier queries. Frontend: ReportsAnalyticsModel + AnalyticsStatsPanel (Logs tab) + TaskGapPanel and per-row T1/T2/T3 gap badges (Tasks tab); cot-logger.ts ADR-0021 opts; i18n en/ru. Scheduler console spam fixed: apscheduler logger demoted to WARNING via LoggingConfig.scheduler_log_level. .axiom belief patterns -> $OBJ.* (alias undercount). molecular-cot-logging skill updated (fields, decision rules, tie-break, CLI) and synced. Reviewed orthogonally: F1 cot_span contract pollution, F2 cap boundary accounting, F3 digest over-dedup, F4 trace-state bound, F5 tier metadata — fixed with regression tests. Validation: backend 11287 passed + ruff + compileall; frontend 3446 passed + lint + build; CLI smoke on live app.log.
109 lines
4.7 KiB
Python
109 lines
4.7 KiB
Python
#!/usr/bin/env python3
|
|
# #region Scripts.CotAudit [C:2] [TYPE Script] [SEMANTICS audit,triangulation,gap,task-logs,cli]
|
|
# @BRIEF Agent CLI for the ground-truth invisible-failure triangulation (ADR-0021/D2).
|
|
# Thin offline wrapper around the SSOT service — NO parser, NO duplicated SQL:
|
|
# the same three-tier query powers GET /api/reports/task-log-gaps (frontend) and
|
|
# this script (agent without a running server). T1: FAILED task with no EXPLORE
|
|
# in task_logs (provably invisible failure); T2: FAILED with EXPLORE lacking
|
|
# contract_id/error_code binding; T3: SUCCESS with EXPLORE (silent fallbacks).
|
|
# @RELATION DEPENDS_ON -> [Services.Reports.LogGapService.ComputeGaps]
|
|
# @NOTE
|
|
# cd backend && .venv/bin/python ../scripts/cot_audit.py --task-log-gaps [--since-days 7] [--json]
|
|
# Requires DATABASE_URL (loads backend/.env like neighbouring scripts). Advisory: exit 0 always.
|
|
"""
|
|
cot_audit.py — ground-truth task-log gap report for agents.
|
|
|
|
Why SQL, not AST: whether a branch "should" have logged is semantically undecidable
|
|
statically (Rice); whether a FAILED task produced zero EXPLORE markers is a fact.
|
|
Detection of syntactic silence (swallowed exceptions) lives in axiom-mcp
|
|
(audit_belief_runtime, finding codes per ADR-0021) — single implementation there.
|
|
"""
|
|
# #endregion Scripts.CotAudit
|
|
|
|
import argparse
|
|
import json
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
_REPO_ROOT = Path(__file__).resolve().parent.parent
|
|
_BACKEND = _REPO_ROOT / "backend"
|
|
|
|
|
|
def _bootstrap() -> None:
|
|
"""Put backend/src on sys.path and load backend/.env (mirrors run.sh behavior)."""
|
|
sys.path.insert(0, str(_BACKEND))
|
|
env_file = _BACKEND / ".env"
|
|
if env_file.exists():
|
|
import os
|
|
|
|
for line in env_file.read_text(encoding="utf-8", errors="replace").splitlines():
|
|
line = line.strip()
|
|
if not line or line.startswith("#") or "=" not in line:
|
|
continue
|
|
key, _, value = line.partition("=")
|
|
os.environ.setdefault(key.strip(), value.strip().strip('"').strip("'"))
|
|
|
|
|
|
def task_log_gaps(since_days: int, limit: int, as_json: bool) -> None:
|
|
_bootstrap()
|
|
try:
|
|
from src.core.database import SessionLocal
|
|
from src.services.reports.log_gap_service import compute_task_log_gaps
|
|
|
|
db = SessionLocal()
|
|
except Exception as exc: # friendly operator/agent-facing failure, not a traceback
|
|
print(f"cot_audit: database unavailable ({type(exc).__name__}: {exc})", file=sys.stderr)
|
|
print(
|
|
"hint: run with backend/.env present (DATABASE_URL) — "
|
|
"cd backend && .venv/bin/python ../scripts/cot_audit.py --task-log-gaps",
|
|
file=sys.stderr,
|
|
)
|
|
if as_json:
|
|
print(json.dumps({"error": "database_unavailable"}, ensure_ascii=False))
|
|
return
|
|
|
|
try:
|
|
report = compute_task_log_gaps(db, since_days=since_days, limit=limit)
|
|
finally:
|
|
db.close()
|
|
|
|
if as_json:
|
|
print(json.dumps(report, ensure_ascii=False, indent=2))
|
|
return
|
|
|
|
icons = {
|
|
"tier1_invisible_failures": ("T1 INVISIBLE", "FAILED task, zero EXPLORE markers — fix the silent branch"),
|
|
"tier2_unbound_explore": ("T2 UNBOUND", "EXPLORE present but no contract_id/error_code — adopt ADR-0021 fields"),
|
|
"tier3_silent_fallbacks": ("T3 FALLBACK", "SUCCESS task with EXPLORE — systematic fallbacks under green status"),
|
|
}
|
|
print(f"== task-log gap report (since {report['since_days']}d, generated {report['generated_at']}) ==")
|
|
for key, (label, hint) in icons.items():
|
|
rows = report.get(key) or []
|
|
print(f"\n[{label}] {len(rows)} — {hint}")
|
|
for row in rows[:20]:
|
|
err = f" error={row['error']}" if row.get("error") else ""
|
|
print(f" {row['created_at']} {row['type']:<24} {row['task_id']}{err}")
|
|
if len(rows) > 20:
|
|
print(f" ... {len(rows) - 20} more (use --json for full report)")
|
|
|
|
|
|
def main() -> None:
|
|
parser = argparse.ArgumentParser(description="CoT ground-truth audits for agents")
|
|
parser.add_argument("--task-log-gaps", action="store_true",
|
|
help="Three-tier invisible-failure report from task_records/task_logs")
|
|
parser.add_argument("--since-days", type=int, default=7, help="Window in days (default 7)")
|
|
parser.add_argument("--limit", type=int, default=200, help="Max tasks per tier (default 200)")
|
|
parser.add_argument("--json", action="store_true", help="Machine-readable output")
|
|
args = parser.parse_args()
|
|
|
|
if args.task_log_gaps:
|
|
task_log_gaps(args.since_days, args.limit, args.json)
|
|
else:
|
|
parser.print_help()
|
|
# Advisory tool: never gates anything (decision D4).
|
|
sys.exit(0)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|