Files
ss-tools/scripts/semantic_health.py
busya 65121cac6b feat(logging): self-diagnosing EXPLORE + shared/ absorption + belief analytics (ADR-0021/0022)
T0: absorb shared/ into backend — cot_logger→src/core, CotJsonFormatter→src/core/cot_formatter.py, _llm_http/_llm_health/ssl→src/core/utils; imports rewritten (26 prod + tests, patch targets); run.sh/backend.Dockerfile/requirements/.axiom source_dirs/semantic_health/AGENTS/INSTALL cleaned; ADR-0022 supersedes ADR-0015; fixed latent CI defects (ss_tools ImportError, record.message in logger tests, same-name test-module collision).

ADR-0021 wire enrichment (additive): contract_id/claim/error_code/loc fields; _contract_id ContextVar + resolve_contract_id (explicit > belief_scope > declared-src mirror, derived src never mirrors); EXPLORE auto-loc via single frame walk; facade error auto-fill; 2KB payload cap with payload_truncated/payload_bytes markers; migrated 85 error="CODE" sites to error_code= (12 files); pilot editor/load.py; superset preview payload-bomb inlined bodies removed.

Analytics SSOT src/core/log_stats.py (bond transition matrix, orphan-EXPLORE ratio, REFLECT pairing, intent families, coverage, insufficient-sample flag); pretty_cot.py --stats/--digest/--trajectory/--story over one engine; log_gap_service three-tier ground-truth triangulation (FAILED w/o EXPLORE etc.) + GET /api/reports/log-stats|task-log-gaps (polling-suppressed); scripts/cot_audit.py CLI; enriched fields persisted into task_logs.payload for tier queries.

Frontend: ReportsAnalyticsModel + AnalyticsStatsPanel (Logs tab) + TaskGapPanel and per-row T1/T2/T3 gap badges (Tasks tab); cot-logger.ts ADR-0021 opts; i18n en/ru. Scheduler console spam fixed: apscheduler logger demoted to WARNING via LoggingConfig.scheduler_log_level. .axiom belief patterns -> $OBJ.* (alias undercount). molecular-cot-logging skill updated (fields, decision rules, tie-break, CLI) and synced.

Reviewed orthogonally: F1 cot_span contract pollution, F2 cap boundary accounting, F3 digest over-dedup, F4 trace-state bound, F5 tier metadata — fixed with regression tests. Validation: backend 11287 passed + ruff + compileall; frontend 3446 passed + lint + build; CLI smoke on live app.log.
2026-09-04 20:56:41 +03:00

184 lines
6.0 KiB
Python

#!/usr/bin/env python3
# #region Tooling.SemanticHealth [C:3] [TYPE Module] [SEMANTICS grace,health,zombie-mode]
# @BRIEF Count #region pairs, dual [C:N], duplicate IDs, and copy-paste @-tags without treating missing tags as errors (INV_9).
# @RELATION BINDS_TO -> [Std.Semantics.Core]
"""Structural GRACE health for zombie-mode (no Axiom required).
INV_9: missing @-tags are never errors. This script reports broken pairs,
dual [C:N], duplicate IDs, oversized files, and optional copy-paste tag text.
Exit codes:
0 ok (or warnings only)
1 --strict-pairs and at least one mismatched #region/#endregion in production src
"""
from __future__ import annotations
import argparse
import collections
import re
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
SKIP_PARTS = {
".git",
".venv",
"node_modules",
"__pycache__",
".axiom",
"coverage",
"coverage_html_frontend",
"dist",
"build",
}
PROD_ZONES = (
ROOT / "backend" / "src",
ROOT / "frontend" / "src",
)
ALL_ZONES = PROD_ZONES + (
ROOT / "backend" / "tests",
ROOT / "frontend" / "tests",
)
REGION_OPEN = re.compile(
r"^\s*#\s*#region\s+(\S+)(?:\s+\[C:(\d+)\])?",
re.I | re.M,
)
REGION_CLOSE = re.compile(r"^\s*#\s*#endregion\s+(\S+)", re.I | re.M)
HTML_OPEN = re.compile(
r"^\s*<!--\s*#region\s+(\S+)(?:\s+\[C:(\d+)\])?",
re.I | re.M,
)
HTML_CLOSE = re.compile(r"^\s*<!--\s*#endregion\s+(\S+)", re.I | re.M)
JS_OPEN = re.compile(
r"^\s*//\s*#region\s+(\S+)(?:\s+\[C:(\d+)\])?",
re.I | re.M,
)
JS_CLOSE = re.compile(r"^\s*//\s*#endregion\s+(\S+)", re.I | re.M)
DUAL_C = re.compile(r"#region[^\n]*\[C:\d+\][^\n]*\[C:\d+\]")
TAG_LINE = re.compile(
r"^\s*(?:#|//|<!--)?\s*@(BRIEF|RATIONALE|REJECTED|PRE|POST|TEST_EDGE)\s+(.*?)(?:-->)?\s*$",
re.I,
)
def _skip(path: Path) -> bool:
return any(part in SKIP_PARTS for part in path.parts) or "__tests__" in path.parts
def _iter_files(zones: tuple[Path, ...]) -> list[Path]:
out: list[Path] = []
for zone in zones:
if not zone.exists():
continue
for path in zone.rglob("*"):
if not path.is_file() or _skip(path):
continue
if path.suffix not in {".py", ".svelte", ".ts", ".js"}:
continue
out.append(path)
return out
def _opens_closes(path: Path, text: str) -> tuple[list[str], list[str]]:
if path.suffix == ".py":
opens = [m.group(1) for m in REGION_OPEN.finditer(text)]
closes = [m.group(1) for m in REGION_CLOSE.finditer(text)]
elif path.suffix == ".svelte":
opens = [m.group(1) for m in HTML_OPEN.finditer(text)] + [
m.group(1) for m in JS_OPEN.finditer(text)
]
closes = [m.group(1) for m in HTML_CLOSE.finditer(text)] + [
m.group(1) for m in JS_CLOSE.finditer(text)
]
else:
opens = [m.group(1) for m in JS_OPEN.finditer(text)]
closes = [m.group(1) for m in JS_CLOSE.finditer(text)]
return opens, closes
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument(
"--strict-pairs",
action="store_true",
help="exit 1 if production src has mismatched region pairs",
)
parser.add_argument(
"--all-zones",
action="store_true",
help="include tests (default: production src only)",
)
args = parser.parse_args()
zones = ALL_ZONES if args.all_zones else PROD_ZONES
mismatched: list[str] = []
dual_c: list[str] = []
oversized: list[str] = []
ids: list[str] = []
tag_text: dict[str, collections.Counter] = collections.defaultdict(collections.Counter)
for path in _iter_files(zones):
text = path.read_text(encoding="utf-8", errors="replace")
rel = str(path.relative_to(ROOT))
lines = text.count("\n") + 1
if lines > 400 and "__tests__" not in path.parts:
oversized.append(f"{lines:5d} {rel}")
opens, closes = _opens_closes(path, text)
ids.extend(opens)
if len(opens) != len(closes):
mismatched.append(f"{rel} open={len(opens)} close={len(closes)}")
for i, line in enumerate(text.splitlines(), 1):
if DUAL_C.search(line):
dual_c.append(f"{rel}:{i}")
m = TAG_LINE.match(line)
if m:
body = re.sub(r"\s+", " ", m.group(2)).strip().lower()
if len(body) >= 24:
tag_text[m.group(1).upper()][body] += 1
dup_ids = [(n, i) for i, n in collections.Counter(ids).items() if n > 1]
dup_ids.sort(reverse=True)
print("== GRACE structural health ==")
print(f"root: {ROOT}")
print(f"zones: {', '.join(str(z.relative_to(ROOT)) for z in zones if z.exists())}")
print(f"regions: {len(ids)}")
print(f"mismatched pairs: {len(mismatched)}")
for row in mismatched[:40]:
print(f" {row}")
if len(mismatched) > 40:
print(f" … {len(mismatched) - 40} more")
print(f"dual [C:N]: {len(dual_c)}")
for row in dual_c[:20]:
print(f" {row}")
print(f"files >400 LOC (non-__tests__): {len(oversized)}")
for row in sorted(oversized, reverse=True)[:15]:
print(f" {row}")
print(f"duplicate contract IDs: {len(dup_ids)}")
for n, cid in dup_ids[:15]:
print(f" {n} {cid}")
print("== copy-paste tag heuristic (advisory, not a failure) ==")
for tag, counter in tag_text.items():
clones = [(n, t) for t, n in counter.items() if n >= 3]
clones.sort(reverse=True)
if not clones:
continue
print(f"{tag}: {len(clones)} texts reused ≥3 times")
for n, t in clones[:5]:
print(f" {n} {t[:100]}")
print("INV_9: missing @-tags are not reported as errors.")
if args.strict_pairs and mismatched and not args.all_zones:
return 1
if args.strict_pairs and args.all_zones and mismatched:
return 1
return 0
if __name__ == "__main__":
sys.exit(main())
# #endregion Tooling.SemanticHealth