feat(logging): self-diagnosing EXPLORE + shared/ absorption + belief analytics (ADR-0021/0022)
T0: absorb shared/ into backend — cot_logger→src/core, CotJsonFormatter→src/core/cot_formatter.py, _llm_http/_llm_health/ssl→src/core/utils; imports rewritten (26 prod + tests, patch targets); run.sh/backend.Dockerfile/requirements/.axiom source_dirs/semantic_health/AGENTS/INSTALL cleaned; ADR-0022 supersedes ADR-0015; fixed latent CI defects (ss_tools ImportError, record.message in logger tests, same-name test-module collision). ADR-0021 wire enrichment (additive): contract_id/claim/error_code/loc fields; _contract_id ContextVar + resolve_contract_id (explicit > belief_scope > declared-src mirror, derived src never mirrors); EXPLORE auto-loc via single frame walk; facade error auto-fill; 2KB payload cap with payload_truncated/payload_bytes markers; migrated 85 error="CODE" sites to error_code= (12 files); pilot editor/load.py; superset preview payload-bomb inlined bodies removed. Analytics SSOT src/core/log_stats.py (bond transition matrix, orphan-EXPLORE ratio, REFLECT pairing, intent families, coverage, insufficient-sample flag); pretty_cot.py --stats/--digest/--trajectory/--story over one engine; log_gap_service three-tier ground-truth triangulation (FAILED w/o EXPLORE etc.) + GET /api/reports/log-stats|task-log-gaps (polling-suppressed); scripts/cot_audit.py CLI; enriched fields persisted into task_logs.payload for tier queries. Frontend: ReportsAnalyticsModel + AnalyticsStatsPanel (Logs tab) + TaskGapPanel and per-row T1/T2/T3 gap badges (Tasks tab); cot-logger.ts ADR-0021 opts; i18n en/ru. Scheduler console spam fixed: apscheduler logger demoted to WARNING via LoggingConfig.scheduler_log_level. .axiom belief patterns -> $OBJ.* (alias undercount). molecular-cot-logging skill updated (fields, decision rules, tie-break, CLI) and synced. Reviewed orthogonally: F1 cot_span contract pollution, F2 cap boundary accounting, F3 digest over-dedup, F4 trace-state bound, F5 tier metadata — fixed with regression tests. Validation: backend 11287 passed + ruff + compileall; frontend 3446 passed + lint + build; CLI smoke on live app.log.
This commit is contained in:
108
scripts/cot_audit.py
Normal file
108
scripts/cot_audit.py
Normal file
@@ -0,0 +1,108 @@
|
||||
#!/usr/bin/env python3
|
||||
# #region Scripts.CotAudit [C:2] [TYPE Script] [SEMANTICS audit,triangulation,gap,task-logs,cli]
|
||||
# @BRIEF Agent CLI for the ground-truth invisible-failure triangulation (ADR-0021/D2).
|
||||
# Thin offline wrapper around the SSOT service — NO parser, NO duplicated SQL:
|
||||
# the same three-tier query powers GET /api/reports/task-log-gaps (frontend) and
|
||||
# this script (agent without a running server). T1: FAILED task with no EXPLORE
|
||||
# in task_logs (provably invisible failure); T2: FAILED with EXPLORE lacking
|
||||
# contract_id/error_code binding; T3: SUCCESS with EXPLORE (silent fallbacks).
|
||||
# @RELATION DEPENDS_ON -> [Services.Reports.LogGapService.ComputeGaps]
|
||||
# @NOTE
|
||||
# cd backend && .venv/bin/python ../scripts/cot_audit.py --task-log-gaps [--since-days 7] [--json]
|
||||
# Requires DATABASE_URL (loads backend/.env like neighbouring scripts). Advisory: exit 0 always.
|
||||
"""
|
||||
cot_audit.py — ground-truth task-log gap report for agents.
|
||||
|
||||
Why SQL, not AST: whether a branch "should" have logged is semantically undecidable
|
||||
statically (Rice); whether a FAILED task produced zero EXPLORE markers is a fact.
|
||||
Detection of syntactic silence (swallowed exceptions) lives in axiom-mcp
|
||||
(audit_belief_runtime, finding codes per ADR-0021) — single implementation there.
|
||||
"""
|
||||
# #endregion Scripts.CotAudit
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
_REPO_ROOT = Path(__file__).resolve().parent.parent
|
||||
_BACKEND = _REPO_ROOT / "backend"
|
||||
|
||||
|
||||
def _bootstrap() -> None:
|
||||
"""Put backend/src on sys.path and load backend/.env (mirrors run.sh behavior)."""
|
||||
sys.path.insert(0, str(_BACKEND))
|
||||
env_file = _BACKEND / ".env"
|
||||
if env_file.exists():
|
||||
import os
|
||||
|
||||
for line in env_file.read_text(encoding="utf-8", errors="replace").splitlines():
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#") or "=" not in line:
|
||||
continue
|
||||
key, _, value = line.partition("=")
|
||||
os.environ.setdefault(key.strip(), value.strip().strip('"').strip("'"))
|
||||
|
||||
|
||||
def task_log_gaps(since_days: int, limit: int, as_json: bool) -> None:
|
||||
_bootstrap()
|
||||
try:
|
||||
from src.core.database import SessionLocal
|
||||
from src.services.reports.log_gap_service import compute_task_log_gaps
|
||||
|
||||
db = SessionLocal()
|
||||
except Exception as exc: # friendly operator/agent-facing failure, not a traceback
|
||||
print(f"cot_audit: database unavailable ({type(exc).__name__}: {exc})", file=sys.stderr)
|
||||
print(
|
||||
"hint: run with backend/.env present (DATABASE_URL) — "
|
||||
"cd backend && .venv/bin/python ../scripts/cot_audit.py --task-log-gaps",
|
||||
file=sys.stderr,
|
||||
)
|
||||
if as_json:
|
||||
print(json.dumps({"error": "database_unavailable"}, ensure_ascii=False))
|
||||
return
|
||||
|
||||
try:
|
||||
report = compute_task_log_gaps(db, since_days=since_days, limit=limit)
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
if as_json:
|
||||
print(json.dumps(report, ensure_ascii=False, indent=2))
|
||||
return
|
||||
|
||||
icons = {
|
||||
"tier1_invisible_failures": ("T1 INVISIBLE", "FAILED task, zero EXPLORE markers — fix the silent branch"),
|
||||
"tier2_unbound_explore": ("T2 UNBOUND", "EXPLORE present but no contract_id/error_code — adopt ADR-0021 fields"),
|
||||
"tier3_silent_fallbacks": ("T3 FALLBACK", "SUCCESS task with EXPLORE — systematic fallbacks under green status"),
|
||||
}
|
||||
print(f"== task-log gap report (since {report['since_days']}d, generated {report['generated_at']}) ==")
|
||||
for key, (label, hint) in icons.items():
|
||||
rows = report.get(key) or []
|
||||
print(f"\n[{label}] {len(rows)} — {hint}")
|
||||
for row in rows[:20]:
|
||||
err = f" error={row['error']}" if row.get("error") else ""
|
||||
print(f" {row['created_at']} {row['type']:<24} {row['task_id']}{err}")
|
||||
if len(rows) > 20:
|
||||
print(f" ... {len(rows) - 20} more (use --json for full report)")
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="CoT ground-truth audits for agents")
|
||||
parser.add_argument("--task-log-gaps", action="store_true",
|
||||
help="Three-tier invisible-failure report from task_records/task_logs")
|
||||
parser.add_argument("--since-days", type=int, default=7, help="Window in days (default 7)")
|
||||
parser.add_argument("--limit", type=int, default=200, help="Max tasks per tier (default 200)")
|
||||
parser.add_argument("--json", action="store_true", help="Machine-readable output")
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.task_log_gaps:
|
||||
task_log_gaps(args.since_days, args.limit, args.json)
|
||||
else:
|
||||
parser.print_help()
|
||||
# Advisory tool: never gates anything (decision D4).
|
||||
sys.exit(0)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -6,7 +6,12 @@
|
||||
# python scripts/pretty_cot.py backend/logs/app.log --last 80
|
||||
# tail -f logs/app.log | python scripts/pretty_cot.py
|
||||
# python scripts/pretty_cot.py backend/logs/app.log --follow
|
||||
# python scripts/pretty_cot.py backend/logs/app.log* --stats [--json]
|
||||
# python scripts/pretty_cot.py backend/logs/app.log* --digest [--top 20] [--json]
|
||||
# python scripts/pretty_cot.py backend/logs/app.log* --trajectory ScenarioExecution.BrowserProvider
|
||||
# python scripts/pretty_cot.py backend/logs/app.log --trace <id> --story
|
||||
# @INVARIANT Output is always a readable narrative grouped by trace_id.
|
||||
# @RELATION DEPENDS_ON -> [Core.LogStats]
|
||||
"""
|
||||
pretty_cot.py — Agent-centric pretty printer for Molecular CoT logs.
|
||||
|
||||
@@ -33,8 +38,8 @@ from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Any, Iterable, Optional
|
||||
|
||||
# ── Central suppression (mirrors shared/ss_tools/shared/cot_logger.py) ──
|
||||
# This is the canonical list. If you update this, update shared too.
|
||||
# ── Central suppression (mirrors backend/src/core/cot_logger.py) ──
|
||||
# This is the canonical list. If you update this, update the backend module too.
|
||||
_ROUTINE_PHRASES = (
|
||||
"Reusing cached Superset auth tokens",
|
||||
"Resolve authenticated user principal",
|
||||
@@ -57,6 +62,26 @@ def _is_routine(intent: str) -> bool:
|
||||
return any(p in intent for p in _ROUTINE_PHRASES)
|
||||
|
||||
|
||||
# ── Analytics engine (SSOT: backend/src/core/log_stats.py — never duplicate) ──
|
||||
_REPO_ROOT = Path(__file__).resolve().parent.parent
|
||||
|
||||
|
||||
def _load_engine():
|
||||
"""Load log_stats.py by file path — stdlib-only module, zero package side effects
|
||||
(works from any python3 without the backend venv; zombie-mode friendly)."""
|
||||
import importlib.util
|
||||
|
||||
engine_path = _REPO_ROOT / "backend" / "src" / "core" / "log_stats.py"
|
||||
try:
|
||||
spec = importlib.util.spec_from_file_location("cot_log_stats", engine_path)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
except Exception as exc: # pragma: no cover - environment dependent
|
||||
print(f"Error: cannot load {engine_path} ({exc})", file=sys.stderr)
|
||||
return None
|
||||
|
||||
|
||||
ICONS = {
|
||||
"REASON": "→",
|
||||
"REFLECT": "✓",
|
||||
@@ -126,6 +151,16 @@ def format_record(rec: dict[str, Any], compact: bool = False) -> str:
|
||||
if error:
|
||||
base += f" | error={error}"
|
||||
|
||||
error_code = rec.get("error_code")
|
||||
claim = rec.get("claim")
|
||||
loc = rec.get("loc")
|
||||
contract = rec.get("contract_id")
|
||||
|
||||
if error_code:
|
||||
base += f" | #{error_code}"
|
||||
if claim:
|
||||
base += f" | claim={claim}"
|
||||
|
||||
if payload and not compact:
|
||||
try:
|
||||
p = json.dumps(payload, ensure_ascii=False, default=str)[:180]
|
||||
@@ -133,6 +168,11 @@ def format_record(rec: dict[str, Any], compact: bool = False) -> str:
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
if loc and not compact:
|
||||
base += f" | loc={loc}"
|
||||
if contract and contract != src and not compact:
|
||||
base += f" | cid={contract}"
|
||||
|
||||
if trace_short and not compact:
|
||||
base = f"[{trace_short}] {base}"
|
||||
|
||||
@@ -204,11 +244,23 @@ def follow_file(file_path: str) -> None:
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Pretty-print Molecular CoT logs for agents")
|
||||
parser.add_argument("files", nargs="*", help="Log files (JSON lines). Use - for stdin.")
|
||||
parser.add_argument("--last", type=int, default=200, help="Only process last N lines")
|
||||
parser.add_argument("--last", type=int, default=None,
|
||||
help="Only process last N lines (narrative default: 200; analytics: all)")
|
||||
parser.add_argument("--trace", help="Filter to a specific trace_id")
|
||||
parser.add_argument("--compact", action="store_true", help="Less verbose output")
|
||||
parser.add_argument("--no-group", action="store_true", help="Do not group by trace_id")
|
||||
parser.add_argument("--follow", action="store_true", help="Follow (like tail -f)")
|
||||
parser.add_argument("--stats", action="store_true",
|
||||
help="Log-economy + bond-structure stats (engine: backend/src/core/log_stats.py)")
|
||||
parser.add_argument("--digest", action="store_true",
|
||||
help="Prioritized EXPLORE fix-map digest")
|
||||
parser.add_argument("--top", type=int, default=20, help="Digest group limit (default 20)")
|
||||
parser.add_argument("--trajectory", metavar="CONTRACT_ID",
|
||||
help="Belief trajectory (ts, marker, claim) for one contract")
|
||||
parser.add_argument("--story", action="store_true",
|
||||
help="With --trace: chronological story ending in a VERDICT echo of EXPLOREs")
|
||||
parser.add_argument("--json", action="store_true",
|
||||
help="Machine-readable JSON output (with --stats/--digest/--trajectory)")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
@@ -218,6 +270,9 @@ def main() -> None:
|
||||
follow_file(f)
|
||||
return
|
||||
|
||||
analytics_mode = bool(args.stats or args.digest or args.trajectory or args.story)
|
||||
effective_last = args.last if args.last else (None if analytics_mode else 200)
|
||||
|
||||
# ── Batch mode ─────────────────────────────────────────────────────────
|
||||
sources: list[Any] = []
|
||||
if not args.files:
|
||||
@@ -233,9 +288,8 @@ def main() -> None:
|
||||
for src in sources:
|
||||
try:
|
||||
lines = src if hasattr(src, "readline") else src.read_text(encoding="utf-8", errors="replace").splitlines()
|
||||
if args.last and hasattr(lines, "__iter__") and not args.follow:
|
||||
if isinstance(lines, list):
|
||||
lines = lines[-args.last:]
|
||||
if effective_last and isinstance(lines, list):
|
||||
lines = lines[-effective_last:]
|
||||
for line in lines:
|
||||
rec = parse_line(line if isinstance(line, str) else line)
|
||||
if rec:
|
||||
@@ -245,8 +299,6 @@ def main() -> None:
|
||||
except Exception as e:
|
||||
print(f"Error reading: {e}", file=sys.stderr)
|
||||
|
||||
pretty_print(all_records, group_by_trace=not args.no_group, compact=args.compact)
|
||||
|
||||
# Close files
|
||||
for s in sources:
|
||||
if hasattr(s, "close") and s is not sys.stdin:
|
||||
@@ -255,6 +307,46 @@ def main() -> None:
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# ── Analytics modes (single engine implementation — log_stats.py) ─────
|
||||
if args.stats or args.digest or args.trajectory:
|
||||
engine = _load_engine()
|
||||
if engine is None:
|
||||
sys.exit(2)
|
||||
if args.stats:
|
||||
stats = engine.compute_stats(all_records)
|
||||
print(json.dumps(stats, ensure_ascii=False, indent=2) if args.json
|
||||
else engine.render_stats(stats))
|
||||
elif args.digest:
|
||||
groups = engine.compute_digest(all_records, top=args.top)
|
||||
print(json.dumps(groups, ensure_ascii=False, indent=2) if args.json
|
||||
else engine.render_digest(groups))
|
||||
else:
|
||||
rows = engine.compute_trajectory(all_records, args.trajectory)
|
||||
print(json.dumps(rows, ensure_ascii=False, indent=2) if args.json
|
||||
else engine.render_trajectory(args.trajectory, rows))
|
||||
return
|
||||
|
||||
if args.story:
|
||||
# Chronological causal story; ends with a VERDICT echo (recency for
|
||||
# constrained attention — the falsification must be the last thing read).
|
||||
all_records.sort(key=lambda r: (r.get("ts") or ""))
|
||||
for r in all_records:
|
||||
line = format_record(r, compact=args.compact)
|
||||
if line:
|
||||
print(line)
|
||||
verdicts = [r for r in all_records if r.get("marker") == "EXPLORE"]
|
||||
if verdicts:
|
||||
print("\n== VERDICT (falsified beliefs) ==")
|
||||
for r in verdicts:
|
||||
line = format_record(r, compact=False)
|
||||
if line:
|
||||
print(" " + line)
|
||||
else:
|
||||
print("\n== VERDICT: no EXPLORE in window — no falsified beliefs observed ==")
|
||||
return
|
||||
|
||||
pretty_print(all_records, group_by_trace=not args.no_group, compact=args.compact)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
@@ -35,13 +35,10 @@ SKIP_PARTS = {
|
||||
PROD_ZONES = (
|
||||
ROOT / "backend" / "src",
|
||||
ROOT / "frontend" / "src",
|
||||
ROOT / "agent" / "src",
|
||||
ROOT / "shared" / "src",
|
||||
)
|
||||
ALL_ZONES = PROD_ZONES + (
|
||||
ROOT / "backend" / "tests",
|
||||
ROOT / "frontend" / "tests",
|
||||
ROOT / "agent" / "tests",
|
||||
)
|
||||
|
||||
REGION_OPEN = re.compile(
|
||||
|
||||
Reference in New Issue
Block a user