Copy the standalone code-review and terraform-review skills into
plugins/reviews as audit-code and audit-terraform. The rename separates the
automated, linter-driven audits from the guided review-pr walkthrough that
already lived here.
Resolve bundled script paths through ${SKILL_DIR}, exported in a new step 0.
CLAUDE_PLUGIN_ROOT is not set in the Bash tool environment, so the obvious
substitution would have expanded to nothing and broken every collection
script invocation.
Replace the PLAN and DESIGN docs with READMEs written from the current
SKILL.md and scripts. The old docs had drifted badly: they named semgrep
where the code calls opengrep, scoped five review agents where there are
now eight, and predated Lua, PowerShell, and GitHub Actions support.
Add CONSISTENCY_NORMS to the audit-terraform agent inputs. The collection
script writes consistency_norms.json and the agent prompt declares it, but
SKILL.md never listed it, leaving the variable unsubstituted.
Drop the --ingest-verdicts instruction from both skills. review_stats.py
parses no arguments, so the ref-mode verdict template it told users to feed
back could never be read.
Point audit-terraform's smoke test at README.md and resolve its fixture
paths relative to the test file rather than an absolute home directory.
Tests: 197 passing (audit-code), 106 passing (audit-terraform).
71 lines
2.1 KiB
Python
71 lines
2.1 KiB
Python
"""Aggregate runs.jsonl into precision + token-cost stats."""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
|
|
def compute_stats(log_path: Path) -> dict:
|
|
if not log_path.exists():
|
|
return {"by_agent": {}, "by_rule": {}, "runs": 0}
|
|
|
|
by_agent: dict[str, dict] = {}
|
|
by_rule: dict[str, dict] = {}
|
|
runs: set[str] = set()
|
|
|
|
for line in log_path.read_text(encoding="utf-8").splitlines():
|
|
if not line.strip():
|
|
continue
|
|
rec = json.loads(line)
|
|
agent = rec.get("agent", "?")
|
|
|
|
if rec["kind"] == "subagent_run":
|
|
runs.add(rec["run_id"])
|
|
a = by_agent.setdefault(agent, _empty_agent())
|
|
a["tokens"] += rec.get("input_tokens", 0) + rec.get("output_tokens", 0)
|
|
a["duration_ms"] += rec.get("duration_ms", 0)
|
|
a["runs"] += 1
|
|
|
|
elif rec["kind"] == "verdict":
|
|
verdict = rec["verdict"]
|
|
a = by_agent.setdefault(agent, _empty_agent())
|
|
a["total"] += 1
|
|
a[verdict] = a.get(verdict, 0) + 1
|
|
|
|
rule_key = f"{agent}/{rec['rule_id']}"
|
|
r = by_rule.setdefault(rule_key, _empty_rule())
|
|
r["total"] += 1
|
|
r[verdict] = r.get(verdict, 0) + 1
|
|
|
|
for a in by_agent.values():
|
|
a["precision"] = a["kept"] / a["total"] if a["total"] else 0.0
|
|
a["tokens_per_kept"] = a["tokens"] / a["kept"] if a["kept"] else float("inf")
|
|
|
|
for r in by_rule.values():
|
|
r["precision"] = r["kept"] / r["total"] if r["total"] else 0.0
|
|
|
|
return {"by_agent": by_agent, "by_rule": by_rule, "runs": len(runs)}
|
|
|
|
|
|
def _empty_agent() -> dict:
|
|
return {
|
|
"tokens": 0, "duration_ms": 0, "runs": 0,
|
|
"total": 0, "kept": 0, "dismissed": 0, "false_positive": 0,
|
|
}
|
|
|
|
|
|
def _empty_rule() -> dict:
|
|
return {"total": 0, "kept": 0, "dismissed": 0, "false_positive": 0}
|
|
|
|
|
|
def main(argv: list[str]) -> int:
|
|
log = Path.home() / ".claude/cache/audit-terraform/runs.jsonl"
|
|
stats = compute_stats(log)
|
|
print(json.dumps(stats, indent=2))
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main(sys.argv[1:]))
|