T0: absorb shared/ into backend — cot_logger→src/core, CotJsonFormatter→src/core/cot_formatter.py, _llm_http/_llm_health/ssl→src/core/utils; imports rewritten (26 prod + tests, patch targets); run.sh/backend.Dockerfile/requirements/.axiom source_dirs/semantic_health/AGENTS/INSTALL cleaned; ADR-0022 supersedes ADR-0015; fixed latent CI defects (ss_tools ImportError, record.message in logger tests, same-name test-module collision). ADR-0021 wire enrichment (additive): contract_id/claim/error_code/loc fields; _contract_id ContextVar + resolve_contract_id (explicit > belief_scope > declared-src mirror, derived src never mirrors); EXPLORE auto-loc via single frame walk; facade error auto-fill; 2KB payload cap with payload_truncated/payload_bytes markers; migrated 85 error="CODE" sites to error_code= (12 files); pilot editor/load.py; superset preview payload-bomb inlined bodies removed. Analytics SSOT src/core/log_stats.py (bond transition matrix, orphan-EXPLORE ratio, REFLECT pairing, intent families, coverage, insufficient-sample flag); pretty_cot.py --stats/--digest/--trajectory/--story over one engine; log_gap_service three-tier ground-truth triangulation (FAILED w/o EXPLORE etc.) + GET /api/reports/log-stats|task-log-gaps (polling-suppressed); scripts/cot_audit.py CLI; enriched fields persisted into task_logs.payload for tier queries. Frontend: ReportsAnalyticsModel + AnalyticsStatsPanel (Logs tab) + TaskGapPanel and per-row T1/T2/T3 gap badges (Tasks tab); cot-logger.ts ADR-0021 opts; i18n en/ru. Scheduler console spam fixed: apscheduler logger demoted to WARNING via LoggingConfig.scheduler_log_level. .axiom belief patterns -> $OBJ.* (alias undercount). molecular-cot-logging skill updated (fields, decision rules, tie-break, CLI) and synced. Reviewed orthogonally: F1 cot_span contract pollution, F2 cap boundary accounting, F3 digest over-dedup, F4 trace-state bound, F5 tier metadata — fixed with regression tests. Validation: backend 11287 passed + ruff + compileall; frontend 3446 passed + lint + build; CLI smoke on live app.log.
353 lines
13 KiB
Python
Executable File
353 lines
13 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
# #region Scripts.PrettyCoT [C:2] [TYPE Script] [SEMANTICS logging,pretty,agent,cli]
|
|
# @BRIEF Agent-first pretty printer and trace visualizer for Molecular CoT JSON logs.
|
|
# @RELATION IMPLEMENTS -> [molecular-cot-logging:CLI Reader]
|
|
# @NOTE
|
|
# python scripts/pretty_cot.py backend/logs/app.log --last 80
|
|
# tail -f logs/app.log | python scripts/pretty_cot.py
|
|
# python scripts/pretty_cot.py backend/logs/app.log --follow
|
|
# python scripts/pretty_cot.py backend/logs/app.log* --stats [--json]
|
|
# python scripts/pretty_cot.py backend/logs/app.log* --digest [--top 20] [--json]
|
|
# python scripts/pretty_cot.py backend/logs/app.log* --trajectory ScenarioExecution.BrowserProvider
|
|
# python scripts/pretty_cot.py backend/logs/app.log --trace <id> --story
|
|
# @INVARIANT Output is always a readable narrative grouped by trace_id.
|
|
# @RELATION DEPENDS_ON -> [Core.LogStats]
|
|
"""
|
|
pretty_cot.py — Agent-centric pretty printer for Molecular CoT logs.
|
|
|
|
Usage:
|
|
python scripts/pretty_cot.py backend/logs/app.log --last 100
|
|
python scripts/pretty_cot.py --trace <trace_id> < log.txt
|
|
python scripts/pretty_cot.py backend/logs/app.log --follow
|
|
tail -f backend/logs/app.log | python scripts/pretty_cot.py
|
|
|
|
Makes raw JSON CoT lines human- and agent-readable with icons, grouping,
|
|
and optional filtering. Designed so an agent can quickly understand a trace.
|
|
|
|
This is the reference tool for "agent view" of logs (see molecular-cot-logging skill).
|
|
"""
|
|
# #endregion Scripts.PrettyCoT
|
|
|
|
import argparse
|
|
import json
|
|
import os
|
|
import sys
|
|
import time
|
|
from collections import defaultdict
|
|
from datetime import datetime
|
|
from pathlib import Path
|
|
from typing import Any, Iterable, Optional
|
|
|
|
# ── Central suppression (mirrors backend/src/core/cot_logger.py) ──
|
|
# This is the canonical list. If you update this, update the backend module too.
|
|
_ROUTINE_PHRASES = (
|
|
"Reusing cached Superset auth tokens",
|
|
"Resolve authenticated user principal",
|
|
"User principal resolved",
|
|
"Resolving current user preference",
|
|
"Loading current user's dashboard preference",
|
|
"Validated ENCRYPTION_KEY",
|
|
"Superset client ready",
|
|
"Superset client initialized",
|
|
"SupersetClientRegistry.get_client",
|
|
"Initialized ResourceService",
|
|
"ResourceService initialized",
|
|
"Created shared HTTP client",
|
|
"Ensured directory",
|
|
)
|
|
|
|
|
|
def _is_routine(intent: str) -> bool:
|
|
"""Check if intent matches known infrastructure noise (defense-in-depth)."""
|
|
return any(p in intent for p in _ROUTINE_PHRASES)
|
|
|
|
|
|
# ── Analytics engine (SSOT: backend/src/core/log_stats.py — never duplicate) ──
|
|
_REPO_ROOT = Path(__file__).resolve().parent.parent
|
|
|
|
|
|
def _load_engine():
|
|
"""Load log_stats.py by file path — stdlib-only module, zero package side effects
|
|
(works from any python3 without the backend venv; zombie-mode friendly)."""
|
|
import importlib.util
|
|
|
|
engine_path = _REPO_ROOT / "backend" / "src" / "core" / "log_stats.py"
|
|
try:
|
|
spec = importlib.util.spec_from_file_location("cot_log_stats", engine_path)
|
|
module = importlib.util.module_from_spec(spec)
|
|
spec.loader.exec_module(module)
|
|
return module
|
|
except Exception as exc: # pragma: no cover - environment dependent
|
|
print(f"Error: cannot load {engine_path} ({exc})", file=sys.stderr)
|
|
return None
|
|
|
|
|
|
ICONS = {
|
|
"REASON": "→",
|
|
"REFLECT": "✓",
|
|
"EXPLORE": "⚠",
|
|
}
|
|
|
|
LEVEL_COLOR = {
|
|
"INFO": "",
|
|
"WARNING": "⚡ ",
|
|
"ERROR": "🔥 ",
|
|
"DEBUG": "… ",
|
|
}
|
|
|
|
|
|
def parse_line(line: str) -> Optional[dict[str, Any]]:
|
|
line = line.strip()
|
|
if not line:
|
|
return None
|
|
try:
|
|
# Handle lines that may be prefixed by docker timestamps etc.
|
|
if line.startswith("{"):
|
|
return json.loads(line)
|
|
# Try to find embedded JSON
|
|
start = line.find("{")
|
|
if start != -1:
|
|
candidate = line[start:]
|
|
end = candidate.rfind("}") + 1
|
|
if end > 1:
|
|
return json.loads(candidate[:end])
|
|
except Exception:
|
|
pass
|
|
return None
|
|
|
|
|
|
def format_record(rec: dict[str, Any], compact: bool = False) -> str:
|
|
ts = rec.get("ts", "")
|
|
if isinstance(ts, str) and "T" in ts:
|
|
try:
|
|
dt = datetime.fromisoformat(ts.replace("Z", "+00:00").split(".")[0])
|
|
ts = dt.strftime("%H:%M:%S")
|
|
except Exception:
|
|
ts = ts[-12:]
|
|
|
|
level = rec.get("level", "INFO")
|
|
marker = rec.get("marker", "REASON")
|
|
src = rec.get("src", "?")
|
|
intent = rec.get("intent", "")
|
|
payload = rec.get("payload")
|
|
error = rec.get("error")
|
|
elapsed = rec.get("elapsed_ms")
|
|
|
|
# Suppress routine infra noise (defense-in-depth; call-site cleanup is primary)
|
|
if _is_routine(intent):
|
|
return ""
|
|
|
|
icon = ICONS.get(marker, "·")
|
|
lvl = LEVEL_COLOR.get(level, "") + level[:4]
|
|
|
|
trace = rec.get("trace_id", "")
|
|
trace_short = trace[:8] if trace and trace != "no-trace" else ""
|
|
|
|
base = f"{ts} {icon} {lvl:5} {src:30.30} {intent}"
|
|
|
|
if elapsed is not None:
|
|
base += f" ⏱{elapsed}ms"
|
|
|
|
if error:
|
|
base += f" | error={error}"
|
|
|
|
error_code = rec.get("error_code")
|
|
claim = rec.get("claim")
|
|
loc = rec.get("loc")
|
|
contract = rec.get("contract_id")
|
|
|
|
if error_code:
|
|
base += f" | #{error_code}"
|
|
if claim:
|
|
base += f" | claim={claim}"
|
|
|
|
if payload and not compact:
|
|
try:
|
|
p = json.dumps(payload, ensure_ascii=False, default=str)[:180]
|
|
base += f" | {p}"
|
|
except Exception:
|
|
pass
|
|
|
|
if loc and not compact:
|
|
base += f" | loc={loc}"
|
|
if contract and contract != src and not compact:
|
|
base += f" | cid={contract}"
|
|
|
|
if trace_short and not compact:
|
|
base = f"[{trace_short}] {base}"
|
|
|
|
return base
|
|
|
|
|
|
def pretty_print(
|
|
records: Iterable[dict[str, Any]],
|
|
group_by_trace: bool = True,
|
|
compact: bool = False,
|
|
max_per_trace: int = 60,
|
|
) -> None:
|
|
if not group_by_trace:
|
|
for r in records:
|
|
line = format_record(r, compact=compact)
|
|
if line:
|
|
print(line)
|
|
return
|
|
|
|
# Group and limit per trace to keep signal high for agent
|
|
by_trace: dict[str, list[dict]] = defaultdict(list)
|
|
for r in records:
|
|
tid = r.get("trace_id") or "no-trace"
|
|
by_trace[tid].append(r)
|
|
|
|
for tid, items in by_trace.items():
|
|
filtered = [r for r in items if format_record(r, compact=compact)]
|
|
if not filtered:
|
|
continue
|
|
if tid != "no-trace":
|
|
print(f"\n=== TRACE {tid} ({len(filtered)} events) ===")
|
|
shown = 0
|
|
for r in filtered:
|
|
if shown >= max_per_trace:
|
|
print(f" ... ({len(filtered) - shown} more events truncated for agent focus)")
|
|
break
|
|
line = format_record(r, compact=compact)
|
|
if line:
|
|
print(" " + line)
|
|
shown += 1
|
|
|
|
|
|
def follow_file(file_path: str) -> None:
|
|
"""Tail -f equivalent for a single log file, processing new lines in real time."""
|
|
with open(file_path, encoding="utf-8", errors="replace") as f:
|
|
# Seek to end
|
|
f.seek(0, os.SEEK_END)
|
|
print(f"Following {file_path}... (Ctrl-C to stop)", file=sys.stderr)
|
|
buf = ""
|
|
try:
|
|
while True:
|
|
chunk = f.read()
|
|
if chunk:
|
|
buf += chunk
|
|
lines = buf.split("\n")
|
|
buf = lines[-1] # keep incomplete last line
|
|
for line in lines[:-1]:
|
|
rec = parse_line(line)
|
|
if rec:
|
|
line_out = format_record(rec)
|
|
if line_out:
|
|
print(line_out)
|
|
else:
|
|
time.sleep(0.25)
|
|
except KeyboardInterrupt:
|
|
pass
|
|
|
|
|
|
def main() -> None:
|
|
parser = argparse.ArgumentParser(description="Pretty-print Molecular CoT logs for agents")
|
|
parser.add_argument("files", nargs="*", help="Log files (JSON lines). Use - for stdin.")
|
|
parser.add_argument("--last", type=int, default=None,
|
|
help="Only process last N lines (narrative default: 200; analytics: all)")
|
|
parser.add_argument("--trace", help="Filter to a specific trace_id")
|
|
parser.add_argument("--compact", action="store_true", help="Less verbose output")
|
|
parser.add_argument("--no-group", action="store_true", help="Do not group by trace_id")
|
|
parser.add_argument("--follow", action="store_true", help="Follow (like tail -f)")
|
|
parser.add_argument("--stats", action="store_true",
|
|
help="Log-economy + bond-structure stats (engine: backend/src/core/log_stats.py)")
|
|
parser.add_argument("--digest", action="store_true",
|
|
help="Prioritized EXPLORE fix-map digest")
|
|
parser.add_argument("--top", type=int, default=20, help="Digest group limit (default 20)")
|
|
parser.add_argument("--trajectory", metavar="CONTRACT_ID",
|
|
help="Belief trajectory (ts, marker, claim) for one contract")
|
|
parser.add_argument("--story", action="store_true",
|
|
help="With --trace: chronological story ending in a VERDICT echo of EXPLOREs")
|
|
parser.add_argument("--json", action="store_true",
|
|
help="Machine-readable JSON output (with --stats/--digest/--trajectory)")
|
|
|
|
args = parser.parse_args()
|
|
|
|
# ── Follow mode ────────────────────────────────────────────────────────
|
|
if args.follow and args.files:
|
|
for f in args.files:
|
|
follow_file(f)
|
|
return
|
|
|
|
analytics_mode = bool(args.stats or args.digest or args.trajectory or args.story)
|
|
effective_last = args.last if args.last else (None if analytics_mode else 200)
|
|
|
|
# ── Batch mode ─────────────────────────────────────────────────────────
|
|
sources: list[Any] = []
|
|
if not args.files:
|
|
sources.append(sys.stdin)
|
|
else:
|
|
for f in args.files:
|
|
if f == "-":
|
|
sources.append(sys.stdin)
|
|
else:
|
|
sources.append(Path(f).open(encoding="utf-8", errors="replace"))
|
|
|
|
all_records: list[dict] = []
|
|
for src in sources:
|
|
try:
|
|
lines = src if hasattr(src, "readline") else src.read_text(encoding="utf-8", errors="replace").splitlines()
|
|
if effective_last and isinstance(lines, list):
|
|
lines = lines[-effective_last:]
|
|
for line in lines:
|
|
rec = parse_line(line if isinstance(line, str) else line)
|
|
if rec:
|
|
if args.trace and args.trace not in (rec.get("trace_id") or ""):
|
|
continue
|
|
all_records.append(rec)
|
|
except Exception as e:
|
|
print(f"Error reading: {e}", file=sys.stderr)
|
|
|
|
# Close files
|
|
for s in sources:
|
|
if hasattr(s, "close") and s is not sys.stdin:
|
|
try:
|
|
s.close()
|
|
except Exception:
|
|
pass
|
|
|
|
# ── Analytics modes (single engine implementation — log_stats.py) ─────
|
|
if args.stats or args.digest or args.trajectory:
|
|
engine = _load_engine()
|
|
if engine is None:
|
|
sys.exit(2)
|
|
if args.stats:
|
|
stats = engine.compute_stats(all_records)
|
|
print(json.dumps(stats, ensure_ascii=False, indent=2) if args.json
|
|
else engine.render_stats(stats))
|
|
elif args.digest:
|
|
groups = engine.compute_digest(all_records, top=args.top)
|
|
print(json.dumps(groups, ensure_ascii=False, indent=2) if args.json
|
|
else engine.render_digest(groups))
|
|
else:
|
|
rows = engine.compute_trajectory(all_records, args.trajectory)
|
|
print(json.dumps(rows, ensure_ascii=False, indent=2) if args.json
|
|
else engine.render_trajectory(args.trajectory, rows))
|
|
return
|
|
|
|
if args.story:
|
|
# Chronological causal story; ends with a VERDICT echo (recency for
|
|
# constrained attention — the falsification must be the last thing read).
|
|
all_records.sort(key=lambda r: (r.get("ts") or ""))
|
|
for r in all_records:
|
|
line = format_record(r, compact=args.compact)
|
|
if line:
|
|
print(line)
|
|
verdicts = [r for r in all_records if r.get("marker") == "EXPLORE"]
|
|
if verdicts:
|
|
print("\n== VERDICT (falsified beliefs) ==")
|
|
for r in verdicts:
|
|
line = format_record(r, compact=False)
|
|
if line:
|
|
print(" " + line)
|
|
else:
|
|
print("\n== VERDICT: no EXPLORE in window — no falsified beliefs observed ==")
|
|
return
|
|
|
|
pretty_print(all_records, group_by_trace=not args.no_group, compact=args.compact)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|