feat(logging): self-diagnosing EXPLORE + shared/ absorption + belief analytics (ADR-0021/0022)

T0: absorb shared/ into backend — cot_logger→src/core, CotJsonFormatter→src/core/cot_formatter.py, _llm_http/_llm_health/ssl→src/core/utils; imports rewritten (26 prod + tests, patch targets); run.sh/backend.Dockerfile/requirements/.axiom source_dirs/semantic_health/AGENTS/INSTALL cleaned; ADR-0022 supersedes ADR-0015; fixed latent CI defects (ss_tools ImportError, record.message in logger tests, same-name test-module collision).

ADR-0021 wire enrichment (additive): contract_id/claim/error_code/loc fields; _contract_id ContextVar + resolve_contract_id (explicit > belief_scope > declared-src mirror, derived src never mirrors); EXPLORE auto-loc via single frame walk; facade error auto-fill; 2KB payload cap with payload_truncated/payload_bytes markers; migrated 85 error="CODE" sites to error_code= (12 files); pilot editor/load.py; superset preview payload-bomb inlined bodies removed.

Analytics SSOT src/core/log_stats.py (bond transition matrix, orphan-EXPLORE ratio, REFLECT pairing, intent families, coverage, insufficient-sample flag); pretty_cot.py --stats/--digest/--trajectory/--story over one engine; log_gap_service three-tier ground-truth triangulation (FAILED w/o EXPLORE etc.) + GET /api/reports/log-stats|task-log-gaps (polling-suppressed); scripts/cot_audit.py CLI; enriched fields persisted into task_logs.payload for tier queries.

Frontend: ReportsAnalyticsModel + AnalyticsStatsPanel (Logs tab) + TaskGapPanel and per-row T1/T2/T3 gap badges (Tasks tab); cot-logger.ts ADR-0021 opts; i18n en/ru. Scheduler console spam fixed: apscheduler logger demoted to WARNING via LoggingConfig.scheduler_log_level. .axiom belief patterns -> $OBJ.* (alias undercount). molecular-cot-logging skill updated (fields, decision rules, tie-break, CLI) and synced.

Reviewed orthogonally: F1 cot_span contract pollution, F2 cap boundary accounting, F3 digest over-dedup, F4 trace-state bound, F5 tier metadata — fixed with regression tests. Validation: backend 11287 passed + ruff + compileall; frontend 3446 passed + lint + build; CLI smoke on live app.log.
This commit is contained in:
2026-09-04 20:56:41 +03:00
parent 34a6507fa3
commit 65121cac6b
97 changed files with 3390 additions and 686 deletions

108
scripts/cot_audit.py Normal file
View File

@@ -0,0 +1,108 @@
#!/usr/bin/env python3
# #region Scripts.CotAudit [C:2] [TYPE Script] [SEMANTICS audit,triangulation,gap,task-logs,cli]
# @BRIEF Agent CLI for the ground-truth invisible-failure triangulation (ADR-0021/D2).
# Thin offline wrapper around the SSOT service — NO parser, NO duplicated SQL:
# the same three-tier query powers GET /api/reports/task-log-gaps (frontend) and
# this script (agent without a running server). T1: FAILED task with no EXPLORE
# in task_logs (provably invisible failure); T2: FAILED with EXPLORE lacking
# contract_id/error_code binding; T3: SUCCESS with EXPLORE (silent fallbacks).
# @RELATION DEPENDS_ON -> [Services.Reports.LogGapService.ComputeGaps]
# @NOTE
# cd backend && .venv/bin/python ../scripts/cot_audit.py --task-log-gaps [--since-days 7] [--json]
# Requires DATABASE_URL (loads backend/.env like neighbouring scripts). Advisory: exit 0 always.
"""
cot_audit.py — ground-truth task-log gap report for agents.
Why SQL, not AST: whether a branch "should" have logged is semantically undecidable
statically (Rice); whether a FAILED task produced zero EXPLORE markers is a fact.
Detection of syntactic silence (swallowed exceptions) lives in axiom-mcp
(audit_belief_runtime, finding codes per ADR-0021) — single implementation there.
"""
# #endregion Scripts.CotAudit
import argparse
import json
import sys
from pathlib import Path
_REPO_ROOT = Path(__file__).resolve().parent.parent
_BACKEND = _REPO_ROOT / "backend"
def _bootstrap() -> None:
"""Put backend/src on sys.path and load backend/.env (mirrors run.sh behavior)."""
sys.path.insert(0, str(_BACKEND))
env_file = _BACKEND / ".env"
if env_file.exists():
import os
for line in env_file.read_text(encoding="utf-8", errors="replace").splitlines():
line = line.strip()
if not line or line.startswith("#") or "=" not in line:
continue
key, _, value = line.partition("=")
os.environ.setdefault(key.strip(), value.strip().strip('"').strip("'"))
def task_log_gaps(since_days: int, limit: int, as_json: bool) -> None:
_bootstrap()
try:
from src.core.database import SessionLocal
from src.services.reports.log_gap_service import compute_task_log_gaps
db = SessionLocal()
except Exception as exc: # friendly operator/agent-facing failure, not a traceback
print(f"cot_audit: database unavailable ({type(exc).__name__}: {exc})", file=sys.stderr)
print(
"hint: run with backend/.env present (DATABASE_URL) — "
"cd backend && .venv/bin/python ../scripts/cot_audit.py --task-log-gaps",
file=sys.stderr,
)
if as_json:
print(json.dumps({"error": "database_unavailable"}, ensure_ascii=False))
return
try:
report = compute_task_log_gaps(db, since_days=since_days, limit=limit)
finally:
db.close()
if as_json:
print(json.dumps(report, ensure_ascii=False, indent=2))
return
icons = {
"tier1_invisible_failures": ("T1 INVISIBLE", "FAILED task, zero EXPLORE markers — fix the silent branch"),
"tier2_unbound_explore": ("T2 UNBOUND", "EXPLORE present but no contract_id/error_code — adopt ADR-0021 fields"),
"tier3_silent_fallbacks": ("T3 FALLBACK", "SUCCESS task with EXPLORE — systematic fallbacks under green status"),
}
print(f"== task-log gap report (since {report['since_days']}d, generated {report['generated_at']}) ==")
for key, (label, hint) in icons.items():
rows = report.get(key) or []
print(f"\n[{label}] {len(rows)} — {hint}")
for row in rows[:20]:
err = f" error={row['error']}" if row.get("error") else ""
print(f" {row['created_at']} {row['type']:<24} {row['task_id']}{err}")
if len(rows) > 20:
print(f" ... {len(rows) - 20} more (use --json for full report)")
def main() -> None:
parser = argparse.ArgumentParser(description="CoT ground-truth audits for agents")
parser.add_argument("--task-log-gaps", action="store_true",
help="Three-tier invisible-failure report from task_records/task_logs")
parser.add_argument("--since-days", type=int, default=7, help="Window in days (default 7)")
parser.add_argument("--limit", type=int, default=200, help="Max tasks per tier (default 200)")
parser.add_argument("--json", action="store_true", help="Machine-readable output")
args = parser.parse_args()
if args.task_log_gaps:
task_log_gaps(args.since_days, args.limit, args.json)
else:
parser.print_help()
# Advisory tool: never gates anything (decision D4).
sys.exit(0)
if __name__ == "__main__":
main()

View File

@@ -6,7 +6,12 @@
# python scripts/pretty_cot.py backend/logs/app.log --last 80
# tail -f logs/app.log | python scripts/pretty_cot.py
# python scripts/pretty_cot.py backend/logs/app.log --follow
# python scripts/pretty_cot.py backend/logs/app.log* --stats [--json]
# python scripts/pretty_cot.py backend/logs/app.log* --digest [--top 20] [--json]
# python scripts/pretty_cot.py backend/logs/app.log* --trajectory ScenarioExecution.BrowserProvider
# python scripts/pretty_cot.py backend/logs/app.log --trace <id> --story
# @INVARIANT Output is always a readable narrative grouped by trace_id.
# @RELATION DEPENDS_ON -> [Core.LogStats]
"""
pretty_cot.py — Agent-centric pretty printer for Molecular CoT logs.
@@ -33,8 +38,8 @@ from datetime import datetime
from pathlib import Path
from typing import Any, Iterable, Optional
# ── Central suppression (mirrors shared/ss_tools/shared/cot_logger.py) ──
# This is the canonical list. If you update this, update shared too.
# ── Central suppression (mirrors backend/src/core/cot_logger.py) ──
# This is the canonical list. If you update this, update the backend module too.
_ROUTINE_PHRASES = (
"Reusing cached Superset auth tokens",
"Resolve authenticated user principal",
@@ -57,6 +62,26 @@ def _is_routine(intent: str) -> bool:
return any(p in intent for p in _ROUTINE_PHRASES)
# ── Analytics engine (SSOT: backend/src/core/log_stats.py — never duplicate) ──
_REPO_ROOT = Path(__file__).resolve().parent.parent
def _load_engine():
"""Load log_stats.py by file path — stdlib-only module, zero package side effects
(works from any python3 without the backend venv; zombie-mode friendly)."""
import importlib.util
engine_path = _REPO_ROOT / "backend" / "src" / "core" / "log_stats.py"
try:
spec = importlib.util.spec_from_file_location("cot_log_stats", engine_path)
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module
except Exception as exc: # pragma: no cover - environment dependent
print(f"Error: cannot load {engine_path} ({exc})", file=sys.stderr)
return None
ICONS = {
"REASON": "→",
"REFLECT": "✓",
@@ -126,6 +151,16 @@ def format_record(rec: dict[str, Any], compact: bool = False) -> str:
if error:
base += f" | error={error}"
error_code = rec.get("error_code")
claim = rec.get("claim")
loc = rec.get("loc")
contract = rec.get("contract_id")
if error_code:
base += f" | #{error_code}"
if claim:
base += f" | claim={claim}"
if payload and not compact:
try:
p = json.dumps(payload, ensure_ascii=False, default=str)[:180]
@@ -133,6 +168,11 @@ def format_record(rec: dict[str, Any], compact: bool = False) -> str:
except Exception:
pass
if loc and not compact:
base += f" | loc={loc}"
if contract and contract != src and not compact:
base += f" | cid={contract}"
if trace_short and not compact:
base = f"[{trace_short}] {base}"
@@ -204,11 +244,23 @@ def follow_file(file_path: str) -> None:
def main() -> None:
parser = argparse.ArgumentParser(description="Pretty-print Molecular CoT logs for agents")
parser.add_argument("files", nargs="*", help="Log files (JSON lines). Use - for stdin.")
parser.add_argument("--last", type=int, default=200, help="Only process last N lines")
parser.add_argument("--last", type=int, default=None,
help="Only process last N lines (narrative default: 200; analytics: all)")
parser.add_argument("--trace", help="Filter to a specific trace_id")
parser.add_argument("--compact", action="store_true", help="Less verbose output")
parser.add_argument("--no-group", action="store_true", help="Do not group by trace_id")
parser.add_argument("--follow", action="store_true", help="Follow (like tail -f)")
parser.add_argument("--stats", action="store_true",
help="Log-economy + bond-structure stats (engine: backend/src/core/log_stats.py)")
parser.add_argument("--digest", action="store_true",
help="Prioritized EXPLORE fix-map digest")
parser.add_argument("--top", type=int, default=20, help="Digest group limit (default 20)")
parser.add_argument("--trajectory", metavar="CONTRACT_ID",
help="Belief trajectory (ts, marker, claim) for one contract")
parser.add_argument("--story", action="store_true",
help="With --trace: chronological story ending in a VERDICT echo of EXPLOREs")
parser.add_argument("--json", action="store_true",
help="Machine-readable JSON output (with --stats/--digest/--trajectory)")
args = parser.parse_args()
@@ -218,6 +270,9 @@ def main() -> None:
follow_file(f)
return
analytics_mode = bool(args.stats or args.digest or args.trajectory or args.story)
effective_last = args.last if args.last else (None if analytics_mode else 200)
# ── Batch mode ─────────────────────────────────────────────────────────
sources: list[Any] = []
if not args.files:
@@ -233,9 +288,8 @@ def main() -> None:
for src in sources:
try:
lines = src if hasattr(src, "readline") else src.read_text(encoding="utf-8", errors="replace").splitlines()
if args.last and hasattr(lines, "__iter__") and not args.follow:
if isinstance(lines, list):
lines = lines[-args.last:]
if effective_last and isinstance(lines, list):
lines = lines[-effective_last:]
for line in lines:
rec = parse_line(line if isinstance(line, str) else line)
if rec:
@@ -245,8 +299,6 @@ def main() -> None:
except Exception as e:
print(f"Error reading: {e}", file=sys.stderr)
pretty_print(all_records, group_by_trace=not args.no_group, compact=args.compact)
# Close files
for s in sources:
if hasattr(s, "close") and s is not sys.stdin:
@@ -255,6 +307,46 @@ def main() -> None:
except Exception:
pass
# ── Analytics modes (single engine implementation — log_stats.py) ─────
if args.stats or args.digest or args.trajectory:
engine = _load_engine()
if engine is None:
sys.exit(2)
if args.stats:
stats = engine.compute_stats(all_records)
print(json.dumps(stats, ensure_ascii=False, indent=2) if args.json
else engine.render_stats(stats))
elif args.digest:
groups = engine.compute_digest(all_records, top=args.top)
print(json.dumps(groups, ensure_ascii=False, indent=2) if args.json
else engine.render_digest(groups))
else:
rows = engine.compute_trajectory(all_records, args.trajectory)
print(json.dumps(rows, ensure_ascii=False, indent=2) if args.json
else engine.render_trajectory(args.trajectory, rows))
return
if args.story:
# Chronological causal story; ends with a VERDICT echo (recency for
# constrained attention — the falsification must be the last thing read).
all_records.sort(key=lambda r: (r.get("ts") or ""))
for r in all_records:
line = format_record(r, compact=args.compact)
if line:
print(line)
verdicts = [r for r in all_records if r.get("marker") == "EXPLORE"]
if verdicts:
print("\n== VERDICT (falsified beliefs) ==")
for r in verdicts:
line = format_record(r, compact=False)
if line:
print(" " + line)
else:
print("\n== VERDICT: no EXPLORE in window — no falsified beliefs observed ==")
return
pretty_print(all_records, group_by_trace=not args.no_group, compact=args.compact)
if __name__ == "__main__":
main()

View File

@@ -35,13 +35,10 @@ SKIP_PARTS = {
PROD_ZONES = (
ROOT / "backend" / "src",
ROOT / "frontend" / "src",
ROOT / "agent" / "src",
ROOT / "shared" / "src",
)
ALL_ZONES = PROD_ZONES + (
ROOT / "backend" / "tests",
ROOT / "frontend" / "tests",
ROOT / "agent" / "tests",
)
REGION_OPEN = re.compile(