Move the code and terraform audits into the reviews plugin
Copy the standalone code-review and terraform-review skills into
plugins/reviews as audit-code and audit-terraform. The rename separates the
automated, linter-driven audits from the guided review-pr walkthrough that
already lived here.
Resolve bundled script paths through ${SKILL_DIR}, exported in a new step 0.
CLAUDE_PLUGIN_ROOT is not set in the Bash tool environment, so the obvious
substitution would have expanded to nothing and broken every collection
script invocation.
Replace the PLAN and DESIGN docs with READMEs written from the current
SKILL.md and scripts. The old docs had drifted badly: they named semgrep
where the code calls opengrep, scoped five review agents where there are
now eight, and predated Lua, PowerShell, and GitHub Actions support.
Add CONSISTENCY_NORMS to the audit-terraform agent inputs. The collection
script writes consistency_norms.json and the agent prompt declares it, but
SKILL.md never listed it, leaving the variable unsubstituted.
Drop the --ingest-verdicts instruction from both skills. review_stats.py
parses no arguments, so the ref-mode verdict template it told users to feed
back could never be read.
Point audit-terraform's smoke test at README.md and resolve its fixture
paths relative to the test file rather than an absolute home directory.
Tests: 197 passing (audit-code), 106 passing (audit-terraform).
This commit is contained in:
@@ -0,0 +1,28 @@
|
||||
"""Adapter: actionlint -format '{{json .}}' normalized to Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
def parse_actionlint_output(stdout: str) -> list[Finding]:
|
||||
try:
|
||||
payload = json.loads(stdout)
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
if not isinstance(payload, list):
|
||||
return []
|
||||
out: list[Finding] = []
|
||||
for item in payload:
|
||||
line = item.get("line", 0)
|
||||
out.append(Finding(
|
||||
tool="actionlint",
|
||||
rule_id=item.get("kind", "unknown"),
|
||||
severity="high",
|
||||
file=item.get("filepath", ""),
|
||||
line=line,
|
||||
end_line=line,
|
||||
message=item.get("message", ""),
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,29 @@
|
||||
"""Adapter: bandit -f json normalized to Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
_SEVERITY = {"HIGH": "high", "MEDIUM": "medium", "LOW": "low"}
|
||||
|
||||
|
||||
def parse_bandit_output(stdout: str) -> list[Finding]:
|
||||
try:
|
||||
payload = json.loads(stdout)
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
out: list[Finding] = []
|
||||
for result in payload.get("results", []):
|
||||
line_range = result.get("line_range") or [result.get("line_number"), result.get("line_number")]
|
||||
out.append(Finding(
|
||||
tool="bandit",
|
||||
rule_id=result.get("test_id", ""),
|
||||
severity=_SEVERITY.get(result.get("issue_severity", ""), "medium"),
|
||||
file=result.get("filename", ""),
|
||||
line=line_range[0],
|
||||
end_line=line_range[-1],
|
||||
message=result.get("issue_text", ""),
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,51 @@
|
||||
"""Adapter: dotnet build text output normalized to Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
_LINE_RE = re.compile(
|
||||
r"^(?P<file>[^(]+)\((?P<line>\d+),\d+\):\s*"
|
||||
r"(?P<severity>error|warning)\s+"
|
||||
r"(?P<code>[A-Z]+\d+):\s*"
|
||||
r"(?P<message>.+?)\s*"
|
||||
r"(?:\[[^\]]+\])?\s*$"
|
||||
)
|
||||
|
||||
|
||||
def _severity(code: str, severity_word: str) -> str:
|
||||
if code.startswith("SCS"):
|
||||
return "high"
|
||||
if severity_word == "error":
|
||||
return "medium"
|
||||
return "low"
|
||||
|
||||
|
||||
def parse_dotnet_build_output(stdout: str, repo_root: str) -> list[Finding]:
|
||||
out: list[Finding] = []
|
||||
seen: set[tuple[str, int, str]] = set()
|
||||
for raw in stdout.splitlines():
|
||||
m = _LINE_RE.match(raw)
|
||||
if not m:
|
||||
continue
|
||||
file_abs = m.group("file")
|
||||
rel = os.path.relpath(file_abs, repo_root) if file_abs.startswith(repo_root) else file_abs
|
||||
code = m.group("code")
|
||||
line = int(m.group("line"))
|
||||
key = (rel, line, code)
|
||||
if key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
out.append(Finding(
|
||||
tool="dotnet",
|
||||
rule_id=code,
|
||||
severity=_severity(code, m.group("severity")),
|
||||
file=rel,
|
||||
line=line,
|
||||
end_line=line,
|
||||
message=m.group("message"),
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,40 @@
|
||||
"""Adapter: eslint -f json normalized to Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
def _severity(rule_id: str, sev_num: int) -> str:
|
||||
if rule_id and "security" in rule_id:
|
||||
return "high"
|
||||
if sev_num == 2:
|
||||
return "medium"
|
||||
return "low"
|
||||
|
||||
|
||||
def parse_eslint_output(stdout: str, repo_root: str) -> list[Finding]:
|
||||
try:
|
||||
payload = json.loads(stdout)
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
if not isinstance(payload, list):
|
||||
return []
|
||||
out: list[Finding] = []
|
||||
for entry in payload:
|
||||
file_path = entry.get("filePath", "")
|
||||
rel = os.path.relpath(file_path, repo_root) if file_path.startswith(repo_root) else file_path
|
||||
for msg in entry.get("messages", []):
|
||||
rule_id = msg.get("ruleId") or ""
|
||||
out.append(Finding(
|
||||
tool="eslint",
|
||||
rule_id=rule_id,
|
||||
severity=_severity(rule_id, msg.get("severity", 1)),
|
||||
file=rel,
|
||||
line=msg.get("line", 0),
|
||||
end_line=msg.get("endLine", msg.get("line", 0)),
|
||||
message=msg.get("message", ""),
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,27 @@
|
||||
"""Adapter: gitleaks --report-format json normalized to Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
def parse_gitleaks_output(stdout: str) -> list[Finding]:
|
||||
try:
|
||||
payload = json.loads(stdout)
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
if not isinstance(payload, list):
|
||||
return []
|
||||
out: list[Finding] = []
|
||||
for r in payload:
|
||||
out.append(Finding(
|
||||
tool="gitleaks",
|
||||
rule_id=r.get("RuleID", ""),
|
||||
severity="critical",
|
||||
file=r.get("File", ""),
|
||||
line=r.get("StartLine", 0),
|
||||
end_line=r.get("EndLine", r.get("StartLine", 0)),
|
||||
message=r.get("Description", ""),
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,41 @@
|
||||
"""Adapter: interrogate --quiet --output-format=json output → Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
def parse_interrogate_output(stdout: str) -> list[Finding]:
|
||||
try:
|
||||
payload = json.loads(stdout)
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
if not isinstance(payload, dict):
|
||||
return []
|
||||
files = payload.get("files")
|
||||
if not isinstance(files, dict):
|
||||
return []
|
||||
out: list[Finding] = []
|
||||
for file_path, file_data in files.items():
|
||||
if not isinstance(file_data, dict):
|
||||
continue
|
||||
for entry in file_data.get("missing", []):
|
||||
if entry.get("private"):
|
||||
continue
|
||||
full_name = entry.get("name", "")
|
||||
symbol = full_name.split(":", 1)[-1] if ":" in full_name else full_name
|
||||
if symbol.startswith("_"):
|
||||
continue
|
||||
kind = entry.get("type", "symbol")
|
||||
line = entry.get("lineno", 0)
|
||||
out.append(Finding(
|
||||
tool="interrogate",
|
||||
rule_id="interrogate:missing-docstring",
|
||||
severity="low",
|
||||
file=file_path,
|
||||
line=line,
|
||||
end_line=line,
|
||||
message=f"missing docstring for public {kind} {symbol}",
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,34 @@
|
||||
"""Adapter: jscpd JSON output → Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
def parse_jscpd_output(stdout: str) -> list[Finding]:
|
||||
try:
|
||||
payload = json.loads(stdout)
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
if not isinstance(payload, dict):
|
||||
return []
|
||||
out: list[Finding] = []
|
||||
for dup in payload.get("duplicates", []):
|
||||
first = dup.get("firstFile", {})
|
||||
second = dup.get("secondFile", {})
|
||||
lines = dup.get("lines", 0)
|
||||
severity = "medium" if lines >= 30 else "low"
|
||||
out.append(Finding(
|
||||
tool="jscpd",
|
||||
rule_id=f"jscpd:clone-{lines}lines",
|
||||
severity=severity,
|
||||
file=first.get("name", ""),
|
||||
line=first.get("start", 0),
|
||||
end_line=first.get("end", 0),
|
||||
message=(
|
||||
f"duplicate of {second.get('name', '')}:"
|
||||
f"{second.get('start', 0)}-{second.get('end', 0)} ({lines} lines)"
|
||||
),
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,31 @@
|
||||
"""Adapter: knip --reporter json output → Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
def parse_knip_output(stdout: str) -> list[Finding]:
|
||||
try:
|
||||
payload = json.loads(stdout)
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
if not isinstance(payload, dict):
|
||||
return []
|
||||
out: list[Finding] = []
|
||||
for issue in payload.get("issues", []):
|
||||
file_path = issue.get("file", "")
|
||||
for export in issue.get("exports", []):
|
||||
name = export.get("name", "")
|
||||
line = export.get("line", 0)
|
||||
out.append(Finding(
|
||||
tool="knip",
|
||||
rule_id="knip:dead-export",
|
||||
severity="medium",
|
||||
file=file_path,
|
||||
line=line,
|
||||
end_line=line,
|
||||
message=f"unused export '{name}'",
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,46 @@
|
||||
"""Adapter: lizard --csv output → Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
def _severity_for_ccn(ccn: int) -> str | None:
|
||||
if ccn >= 20:
|
||||
return "high"
|
||||
if ccn >= 10:
|
||||
return "medium"
|
||||
return None
|
||||
|
||||
|
||||
def parse_lizard_output(stdout: str) -> list[Finding]:
|
||||
out: list[Finding] = []
|
||||
for raw in stdout.splitlines():
|
||||
parts = raw.strip().split(",")
|
||||
if len(parts) < 6:
|
||||
continue
|
||||
try:
|
||||
ccn = int(parts[1])
|
||||
except ValueError:
|
||||
continue
|
||||
severity = _severity_for_ccn(ccn)
|
||||
if severity is None:
|
||||
continue
|
||||
location = parts[5]
|
||||
loc_parts = location.split("@")
|
||||
if len(loc_parts) < 3:
|
||||
continue
|
||||
name, line_str, file_path = loc_parts[0], loc_parts[1], loc_parts[2]
|
||||
try:
|
||||
line = int(line_str)
|
||||
except ValueError:
|
||||
continue
|
||||
out.append(Finding(
|
||||
tool="lizard",
|
||||
rule_id=f"lizard:ccn={ccn}",
|
||||
severity=severity,
|
||||
file=file_path,
|
||||
line=line,
|
||||
end_line=line,
|
||||
message=f"function {name} has CCN {ccn}",
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,32 @@
|
||||
"""Adapter: luac -p stderr normalized to Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
_PATTERN = re.compile(r"^luac:\s+(.+?):(\d+):\s+(.+)$")
|
||||
|
||||
|
||||
def parse_luac_output(stderr: str, repo_root: str = "") -> list[Finding]:
|
||||
out: list[Finding] = []
|
||||
prefix = repo_root.rstrip("/") + "/" if repo_root else ""
|
||||
for line in stderr.splitlines():
|
||||
m = _PATTERN.match(line.strip())
|
||||
if not m:
|
||||
continue
|
||||
filepath, lineno_str, message = m.group(1), m.group(2), m.group(3)
|
||||
if prefix and filepath.startswith(prefix):
|
||||
filepath = filepath[len(prefix):]
|
||||
lineno = int(lineno_str)
|
||||
out.append(Finding(
|
||||
tool="luac",
|
||||
rule_id="syntax-error",
|
||||
severity="critical",
|
||||
file=filepath,
|
||||
line=lineno,
|
||||
end_line=lineno,
|
||||
message=message,
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,37 @@
|
||||
"""Adapter: mypy text output normalized to Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
_LINE_RE = re.compile(
|
||||
r"^(?P<file>[^:]+):(?P<line>\d+):\s*"
|
||||
r"(?P<severity>error|warning|note):\s*"
|
||||
r"(?P<message>.+?)"
|
||||
r"(?:\s+\[(?P<code>[a-z\-]+)\])?\s*$"
|
||||
)
|
||||
|
||||
_SEVERITY = {"error": "medium", "warning": "low", "note": None}
|
||||
|
||||
|
||||
def parse_mypy_output(stdout: str) -> list[Finding]:
|
||||
out: list[Finding] = []
|
||||
for raw in stdout.splitlines():
|
||||
m = _LINE_RE.match(raw)
|
||||
if not m:
|
||||
continue
|
||||
sev = _SEVERITY.get(m.group("severity"))
|
||||
if sev is None:
|
||||
continue
|
||||
out.append(Finding(
|
||||
tool="mypy",
|
||||
rule_id=m.group("code") or "",
|
||||
severity=sev,
|
||||
file=m.group("file"),
|
||||
line=int(m.group("line")),
|
||||
end_line=int(m.group("line")),
|
||||
message=m.group("message"),
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,51 @@
|
||||
"""Adapter: opengrep --json normalized to Finding[]. Same JSON schema as semgrep (opengrep is a semgrep fork)."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
_SEVERITY = {"ERROR": "high", "WARNING": "medium", "INFO": "low"}
|
||||
_CWE_RE = re.compile(r"(CWE-\d+)")
|
||||
_LOCAL_RULES_MARKER = "scripts.rules."
|
||||
|
||||
|
||||
def _clean_rule_id(check_id: str) -> str:
|
||||
if _LOCAL_RULES_MARKER in check_id:
|
||||
return check_id.rsplit(_LOCAL_RULES_MARKER, 1)[-1]
|
||||
return check_id
|
||||
|
||||
|
||||
def _cwe(meta: dict) -> str | None:
|
||||
raw = meta.get("cwe")
|
||||
if isinstance(raw, list) and raw:
|
||||
m = _CWE_RE.search(str(raw[0]))
|
||||
return m.group(1) if m else None
|
||||
if isinstance(raw, str):
|
||||
m = _CWE_RE.search(raw)
|
||||
return m.group(1) if m else None
|
||||
return None
|
||||
|
||||
|
||||
def parse_opengrep_output(stdout: str) -> list[Finding]:
|
||||
try:
|
||||
payload = json.loads(stdout)
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
out: list[Finding] = []
|
||||
for r in payload.get("results", []):
|
||||
extra = r.get("extra", {})
|
||||
metadata = extra.get("metadata", {})
|
||||
out.append(Finding(
|
||||
tool="opengrep",
|
||||
rule_id=_clean_rule_id(r.get("check_id", "")),
|
||||
severity=_SEVERITY.get(extra.get("severity", ""), "medium"),
|
||||
file=r.get("path", ""),
|
||||
line=r.get("start", {}).get("line", 0),
|
||||
end_line=r.get("end", {}).get("line", r.get("start", {}).get("line", 0)),
|
||||
message=extra.get("message", ""),
|
||||
cwe=_cwe(metadata),
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,31 @@
|
||||
"""Adapter: osv-scanner --format json normalized to Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
def parse_osv_scanner_output(stdout: str) -> list[Finding]:
|
||||
try:
|
||||
payload = json.loads(stdout)
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
out: list[Finding] = []
|
||||
for r in payload.get("results", []):
|
||||
manifest_path = r.get("source", {}).get("path", "")
|
||||
for pkg in r.get("packages", []):
|
||||
p = pkg.get("package", {})
|
||||
name = p.get("name", "")
|
||||
version = p.get("version", "")
|
||||
for v in pkg.get("vulnerabilities", []):
|
||||
out.append(Finding(
|
||||
tool="osv-scanner",
|
||||
rule_id=v.get("id", ""),
|
||||
severity="high",
|
||||
file=manifest_path,
|
||||
line=1,
|
||||
end_line=1,
|
||||
message=f"{name} {version}: {v.get('summary', '')}",
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,35 @@
|
||||
"""Adapter: pip-audit -f json normalized to Finding[].
|
||||
|
||||
Findings are attached to the dependency manifest file rather than a source
|
||||
file, since the vulnerability is in a pinned dep, not in code.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
def parse_pip_audit_output(stdout: str, manifest_path: str = "requirements.txt") -> list[Finding]:
|
||||
try:
|
||||
payload = json.loads(stdout)
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
out: list[Finding] = []
|
||||
for dep in payload.get("dependencies", []):
|
||||
name = dep.get("name", "")
|
||||
version = dep.get("version", "")
|
||||
for v in dep.get("vulns", []):
|
||||
fix = v.get("fix_versions") or []
|
||||
fix_str = ", ".join(fix) if fix else None
|
||||
out.append(Finding(
|
||||
tool="pip-audit",
|
||||
rule_id=v.get("id", ""),
|
||||
severity="high",
|
||||
file=manifest_path,
|
||||
line=1,
|
||||
end_line=1,
|
||||
message=f"{name} {version}: {v.get('description', '')}",
|
||||
fix_suggestion=f"upgrade to {fix_str}" if fix_str else None,
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,56 @@
|
||||
"""Adapter: Invoke-ScriptAnalyzer JSON normalized to Finding[].
|
||||
|
||||
Shared by both PowerShell tools — PSScriptAnalyzer's built-in rules and the
|
||||
InjectionHunter custom rule pack — because both are Invoke-ScriptAnalyzer runs
|
||||
and emit the same DiagnosticRecord shape.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
_SEVERITY = {
|
||||
"ParseError": "critical",
|
||||
"Error": "high",
|
||||
"Warning": "medium",
|
||||
"Information": "low",
|
||||
}
|
||||
|
||||
_INJECTION_FLOOR = "high"
|
||||
|
||||
|
||||
def parse_psscriptanalyzer_output(
|
||||
stdout: str, repo_root: str, tool: str = "psscriptanalyzer",
|
||||
) -> list[Finding]:
|
||||
try:
|
||||
payload = json.loads(stdout)
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
if isinstance(payload, dict):
|
||||
payload = [payload]
|
||||
if not isinstance(payload, list):
|
||||
return []
|
||||
|
||||
out: list[Finding] = []
|
||||
for item in payload:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
path = item.get("ScriptPath") or item.get("ScriptName") or ""
|
||||
rel = os.path.relpath(path, repo_root) if path.startswith(repo_root) else path
|
||||
line = item.get("Line") or 0
|
||||
severity = _SEVERITY.get(item.get("Severity", ""), "medium")
|
||||
if tool == "injectionhunter":
|
||||
severity = _INJECTION_FLOOR
|
||||
out.append(Finding(
|
||||
tool=tool,
|
||||
rule_id=item.get("RuleName", "unknown"),
|
||||
severity=severity,
|
||||
file=rel,
|
||||
line=line,
|
||||
end_line=item.get("EndLine") or line,
|
||||
message=item.get("Message", ""),
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,46 @@
|
||||
"""Adapter: radon cc -j JSON output → Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
def _severity_for_complexity(cc: int) -> str | None:
|
||||
if cc >= 20:
|
||||
return "high"
|
||||
if cc >= 10:
|
||||
return "medium"
|
||||
return None
|
||||
|
||||
|
||||
def parse_radon_output(stdout: str) -> list[Finding]:
|
||||
try:
|
||||
payload = json.loads(stdout)
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
if not isinstance(payload, dict):
|
||||
return []
|
||||
out: list[Finding] = []
|
||||
for file_path, items in payload.items():
|
||||
if not isinstance(items, list):
|
||||
continue
|
||||
for item in items:
|
||||
cc = item.get("complexity", 0)
|
||||
severity = _severity_for_complexity(cc)
|
||||
if severity is None:
|
||||
continue
|
||||
line = item.get("lineno", 0)
|
||||
end_line = item.get("endline", line)
|
||||
name = item.get("name", "?")
|
||||
kind = item.get("type", "function")
|
||||
out.append(Finding(
|
||||
tool="radon",
|
||||
rule_id=f"radon:cc={cc}",
|
||||
severity=severity,
|
||||
file=file_path,
|
||||
line=line,
|
||||
end_line=end_line,
|
||||
message=f"high cyclomatic complexity (CCN={cc}) in {kind} {name}",
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,40 @@
|
||||
"""Adapter: ruff check --output-format=json normalized to Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
def _severity_for_rule(code: str) -> str:
|
||||
if not code:
|
||||
return "low"
|
||||
if code.startswith("S"):
|
||||
return "high"
|
||||
if code.startswith("B"):
|
||||
return "medium"
|
||||
return "low"
|
||||
|
||||
|
||||
def parse_ruff_output(stdout: str) -> list[Finding]:
|
||||
try:
|
||||
payload = json.loads(stdout)
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
if not isinstance(payload, list):
|
||||
return []
|
||||
out: list[Finding] = []
|
||||
for item in payload:
|
||||
loc = item.get("location") or {}
|
||||
end = item.get("end_location") or loc
|
||||
code = item.get("code", "")
|
||||
out.append(Finding(
|
||||
tool="ruff",
|
||||
rule_id=code,
|
||||
severity=_severity_for_rule(code),
|
||||
file=item.get("filename", ""),
|
||||
line=loc.get("row", 0),
|
||||
end_line=end.get("row", loc.get("row", 0)),
|
||||
message=item.get("message", ""),
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,38 @@
|
||||
"""Adapter: ruff check with idiom/simplification selectors → Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
def _severity_for_idiom_rule(code: str) -> str:
|
||||
if not code:
|
||||
return "low"
|
||||
if code.startswith(("PLR", "C90", "B")):
|
||||
return "medium"
|
||||
return "low"
|
||||
|
||||
|
||||
def parse_ruff_idiom_output(stdout: str) -> list[Finding]:
|
||||
try:
|
||||
payload = json.loads(stdout)
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
if not isinstance(payload, list):
|
||||
return []
|
||||
out: list[Finding] = []
|
||||
for item in payload:
|
||||
loc = item.get("location") or {}
|
||||
end = item.get("end_location") or loc
|
||||
code = item.get("code", "")
|
||||
out.append(Finding(
|
||||
tool="ruff-idiom",
|
||||
rule_id=code,
|
||||
severity=_severity_for_idiom_rule(code),
|
||||
file=item.get("filename", ""),
|
||||
line=loc.get("row", 0),
|
||||
end_line=end.get("row", loc.get("row", 0)),
|
||||
message=item.get("message", ""),
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,34 @@
|
||||
"""Adapter: selene --display-style=json normalized to Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
_SEVERITY = {"Error": "high", "Warning": "medium", "Note": "low"}
|
||||
|
||||
|
||||
def parse_selene_output(stdout: str) -> list[Finding]:
|
||||
try:
|
||||
payload = json.loads(stdout)
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
if not isinstance(payload, list):
|
||||
return []
|
||||
out: list[Finding] = []
|
||||
for item in payload:
|
||||
code = item.get("code", {})
|
||||
span = item.get("span", {})
|
||||
start = span.get("start", {})
|
||||
end_span = span.get("end", {})
|
||||
out.append(Finding(
|
||||
tool="selene",
|
||||
rule_id=code.get("name", "unknown"),
|
||||
severity=_SEVERITY.get(code.get("severity", ""), "medium"),
|
||||
file=item.get("filename", ""),
|
||||
line=start.get("line", 0),
|
||||
end_line=end_span.get("line", start.get("line", 0)),
|
||||
message=item.get("primary_label", ""),
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,33 @@
|
||||
"""Adapter: tsc --noEmit text output normalized to Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
_LINE_RE = re.compile(
|
||||
r"^(?P<file>[^(]+)\((?P<line>\d+),\d+\):\s*"
|
||||
r"(?P<severity>error|warning)\s+"
|
||||
r"(?P<code>TS\d+):\s*"
|
||||
r"(?P<message>.+?)\s*$"
|
||||
)
|
||||
|
||||
|
||||
def parse_tsc_output(stdout: str) -> list[Finding]:
|
||||
out: list[Finding] = []
|
||||
for raw in stdout.splitlines():
|
||||
m = _LINE_RE.match(raw)
|
||||
if not m:
|
||||
continue
|
||||
sev = "medium" if m.group("severity") == "error" else "low"
|
||||
out.append(Finding(
|
||||
tool="tsc",
|
||||
rule_id=m.group("code"),
|
||||
severity=sev,
|
||||
file=m.group("file"),
|
||||
line=int(m.group("line")),
|
||||
end_line=int(m.group("line")),
|
||||
message=m.group("message"),
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,31 @@
|
||||
"""Adapter: vulture text output → Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
_LINE_RE = re.compile(
|
||||
r"^(?P<file>[^:]+):(?P<line>\d+):\s*(?P<msg>.+?)\s*\((?P<conf>\d+)%\s*confidence\)$"
|
||||
)
|
||||
|
||||
|
||||
def parse_vulture_output(stdout: str) -> list[Finding]:
|
||||
out: list[Finding] = []
|
||||
for raw in stdout.splitlines():
|
||||
m = _LINE_RE.match(raw.strip())
|
||||
if not m:
|
||||
continue
|
||||
conf = int(m.group("conf"))
|
||||
severity = "medium" if conf >= 80 else "low"
|
||||
out.append(Finding(
|
||||
tool="vulture",
|
||||
rule_id=f"vulture:{conf}pct",
|
||||
severity=severity,
|
||||
file=m.group("file"),
|
||||
line=int(m.group("line")),
|
||||
end_line=int(m.group("line")),
|
||||
message=m.group("msg"),
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,48 @@
|
||||
"""Adapter: zizmor --format sarif normalized to Finding[]."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
|
||||
from scripts.manifest import Finding
|
||||
|
||||
|
||||
_LEVEL = {"error": "high", "warning": "medium", "note": "low"}
|
||||
_CWE_RE = re.compile(r"(CWE-\d+)")
|
||||
|
||||
|
||||
def parse_zizmor_output(stdout: str) -> list[Finding]:
|
||||
try:
|
||||
payload = json.loads(stdout)
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
out: list[Finding] = []
|
||||
for run in payload.get("runs", []):
|
||||
cwe_by_rule: dict[str, str] = {}
|
||||
for rule in run.get("tool", {}).get("driver", {}).get("rules", []):
|
||||
for tag in rule.get("properties", {}).get("tags", []):
|
||||
m = _CWE_RE.match(tag)
|
||||
if m:
|
||||
cwe_by_rule[rule["id"]] = m.group(1)
|
||||
break
|
||||
for result in run.get("results", []):
|
||||
rule_id = result.get("ruleId", "unknown")
|
||||
locations = result.get("locations", [])
|
||||
if not locations:
|
||||
continue
|
||||
phys = locations[0].get("physicalLocation", {})
|
||||
uri = phys.get("artifactLocation", {}).get("uri", "")
|
||||
region = phys.get("region", {})
|
||||
start_line = region.get("startLine", 0)
|
||||
end_line = region.get("endLine", start_line)
|
||||
out.append(Finding(
|
||||
tool="zizmor",
|
||||
rule_id=rule_id,
|
||||
severity=_LEVEL.get(result.get("level", "warning"), "medium"),
|
||||
file=uri,
|
||||
line=start_line,
|
||||
end_line=end_line,
|
||||
message=result.get("message", {}).get("text", ""),
|
||||
cwe=cwe_by_rule.get(rule_id),
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,464 @@
|
||||
"""collect-findings.py — produce a audit-code manifest.json."""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import glob
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
from collections import Counter
|
||||
from pathlib import Path
|
||||
|
||||
_HERE = Path(__file__).resolve().parent
|
||||
if str(_HERE.parent) not in sys.path:
|
||||
sys.path.insert(0, str(_HERE.parent))
|
||||
|
||||
from scripts.adapters import (
|
||||
bandit as ad_bandit,
|
||||
ruff as ad_ruff,
|
||||
ruff_idiom as ad_ruff_idiom,
|
||||
mypy as ad_mypy,
|
||||
eslint as ad_eslint,
|
||||
tsc as ad_tsc,
|
||||
opengrep as ad_opengrep,
|
||||
gitleaks as ad_gitleaks,
|
||||
pip_audit as ad_pip_audit,
|
||||
osv_scanner as ad_osv,
|
||||
dotnet as ad_dotnet,
|
||||
vulture as ad_vulture,
|
||||
radon as ad_radon,
|
||||
interrogate as ad_interrogate,
|
||||
lizard as ad_lizard,
|
||||
knip as ad_knip,
|
||||
jscpd as ad_jscpd,
|
||||
selene as ad_selene,
|
||||
luac as ad_luac,
|
||||
psscriptanalyzer as ad_pssa,
|
||||
actionlint as ad_actionlint,
|
||||
zizmor as ad_zizmor,
|
||||
)
|
||||
from scripts.diff_filter import filter_findings_by_diff
|
||||
from scripts.git_diff import (
|
||||
changed_files_with_ranges, resolve_base_ref, resolve_default_branch,
|
||||
)
|
||||
from scripts.language_detect import (
|
||||
detect_language, is_dep_manifest, is_gha_file, is_supported_source,
|
||||
known_unsupported_language,
|
||||
)
|
||||
from scripts.manifest import (
|
||||
ChangedFile, Finding, LanguageBreakdown, Manifest,
|
||||
PackageDiff, ToolStat,
|
||||
)
|
||||
from scripts.package_diff import diff_package_json, diff_requirements_txt
|
||||
from scripts.runner import run_tool, tool_available
|
||||
from scripts.slicing import slice_for_agent
|
||||
|
||||
|
||||
def _git_show(repo: str, ref: str, path: str) -> str:
|
||||
"""Return file content at `ref`, or empty string if not present (e.g. new file)."""
|
||||
r = subprocess.run(
|
||||
["git", "-C", repo, "show", f"{ref}:{path}"],
|
||||
capture_output=True, text=True, check=False,
|
||||
)
|
||||
return r.stdout if r.returncode == 0 else ""
|
||||
|
||||
|
||||
def _build_package_diffs(
|
||||
repo: str, base: str, dep_manifest_paths: list[str],
|
||||
) -> dict[str, PackageDiff]:
|
||||
diffs: dict[str, PackageDiff] = {
|
||||
"python": PackageDiff(),
|
||||
"javascript": PackageDiff(),
|
||||
"csharp": PackageDiff(),
|
||||
}
|
||||
for path in dep_manifest_paths:
|
||||
name = Path(path).name
|
||||
before = _git_show(repo, base, path)
|
||||
after_path = Path(repo) / path
|
||||
after = after_path.read_text() if after_path.is_file() else ""
|
||||
if name.endswith(".txt") and "requirements" in name:
|
||||
pd = diff_requirements_txt(before, after)
|
||||
_merge_package_diff(diffs["python"], pd)
|
||||
elif name == "package.json":
|
||||
pd = diff_package_json(before, after)
|
||||
_merge_package_diff(diffs["javascript"], pd)
|
||||
return diffs
|
||||
|
||||
|
||||
def _merge_package_diff(into: PackageDiff, src: PackageDiff) -> None:
|
||||
into.added.extend(src.added)
|
||||
into.removed.extend(src.removed)
|
||||
into.upgraded.extend(src.upgraded)
|
||||
|
||||
|
||||
def _build_breakdown(changed_files: list[ChangedFile], skipped: list[str]) -> LanguageBreakdown:
|
||||
n = Counter(cf.language for cf in changed_files)
|
||||
return LanguageBreakdown(
|
||||
python=n["python"], javascript=n["javascript"], typescript=n["typescript"],
|
||||
csharp=n["csharp"], lua=n["lua"], powershell=n["powershell"],
|
||||
github_actions=n["github-actions"], skipped_files=skipped,
|
||||
)
|
||||
|
||||
|
||||
def _run_python_tools(repo: str) -> tuple[list[Finding], dict[str, ToolStat]]:
|
||||
findings: list[Finding] = []
|
||||
stats: dict[str, ToolStat] = {}
|
||||
for binary, args, adapter in [
|
||||
("bandit", ["bandit", "-r", ".", "-f", "json", "-q"], ad_bandit.parse_bandit_output),
|
||||
("ruff", ["ruff", "check", "--output-format=json", "."], ad_ruff.parse_ruff_output),
|
||||
("ruff-idiom", [
|
||||
"ruff", "check",
|
||||
"--select", "SIM,PERF,UP,RET,PLR,C90,B",
|
||||
"--output-format=json",
|
||||
"--isolated",
|
||||
".",
|
||||
], ad_ruff_idiom.parse_ruff_idiom_output),
|
||||
("mypy", ["mypy", "."], ad_mypy.parse_mypy_output),
|
||||
]:
|
||||
if not tool_available(binary):
|
||||
stats[binary] = ToolStat(ran=False, reason="not on PATH")
|
||||
continue
|
||||
r = run_tool(args, cwd=repo)
|
||||
parsed = adapter(r.stdout)
|
||||
stats[binary] = ToolStat(ran=True, pre_filter=len(parsed), post_filter=0)
|
||||
findings.extend(parsed)
|
||||
return findings, stats
|
||||
|
||||
|
||||
def _run_js_tools(repo: str) -> tuple[list[Finding], dict[str, ToolStat]]:
|
||||
findings: list[Finding] = []
|
||||
stats: dict[str, ToolStat] = {}
|
||||
if tool_available("eslint"):
|
||||
r = run_tool(["eslint", ".", "-f", "json"], cwd=repo)
|
||||
parsed = ad_eslint.parse_eslint_output(r.stdout, repo_root=repo)
|
||||
stats["eslint"] = ToolStat(ran=True, pre_filter=len(parsed))
|
||||
findings.extend(parsed)
|
||||
else:
|
||||
stats["eslint"] = ToolStat(ran=False, reason="not on PATH")
|
||||
if tool_available("tsc"):
|
||||
r = run_tool(["tsc", "--noEmit"], cwd=repo)
|
||||
parsed = ad_tsc.parse_tsc_output(r.stdout)
|
||||
stats["tsc"] = ToolStat(ran=True, pre_filter=len(parsed))
|
||||
findings.extend(parsed)
|
||||
else:
|
||||
stats["tsc"] = ToolStat(ran=False, reason="not on PATH")
|
||||
return findings, stats
|
||||
|
||||
|
||||
def _run_dotnet_tools(repo: str) -> tuple[list[Finding], dict[str, ToolStat]]:
|
||||
findings: list[Finding] = []
|
||||
stats: dict[str, ToolStat] = {}
|
||||
if tool_available("dotnet"):
|
||||
r = run_tool(["dotnet", "build", "--no-incremental"], cwd=repo)
|
||||
parsed = ad_dotnet.parse_dotnet_build_output(r.stdout, repo_root=repo)
|
||||
stats["dotnet"] = ToolStat(ran=True, pre_filter=len(parsed))
|
||||
findings.extend(parsed)
|
||||
else:
|
||||
stats["dotnet"] = ToolStat(ran=False, reason="not on PATH")
|
||||
return findings, stats
|
||||
|
||||
|
||||
def _run_lua_tools(repo: str) -> tuple[list[Finding], dict[str, ToolStat]]:
|
||||
findings: list[Finding] = []
|
||||
stats: dict[str, ToolStat] = {}
|
||||
if tool_available("selene"):
|
||||
r = run_tool(["selene", "--display-style=json", "."], cwd=repo)
|
||||
parsed = ad_selene.parse_selene_output(r.stdout)
|
||||
stats["selene"] = ToolStat(ran=True, pre_filter=len(parsed), post_filter=0)
|
||||
findings.extend(parsed)
|
||||
else:
|
||||
stats["selene"] = ToolStat(ran=False, reason="not on PATH")
|
||||
if tool_available("luac"):
|
||||
lua_files = sorted(glob.glob(f"{repo}/**/*.lua", recursive=True))
|
||||
combined_stderr = ""
|
||||
for lua_file in lua_files:
|
||||
r = run_tool(["luac", "-p", lua_file], cwd=repo)
|
||||
if r.stderr:
|
||||
combined_stderr += r.stderr
|
||||
parsed = ad_luac.parse_luac_output(combined_stderr, repo_root=repo)
|
||||
stats["luac"] = ToolStat(ran=True, pre_filter=len(parsed), post_filter=0)
|
||||
findings.extend(parsed)
|
||||
else:
|
||||
stats["luac"] = ToolStat(ran=False, reason="not on PATH")
|
||||
return findings, stats
|
||||
|
||||
|
||||
_PSSA_PROJECT = (
|
||||
" | Select-Object RuleName,"
|
||||
"@{n='Severity';e={$_.Severity.ToString()}},"
|
||||
"ScriptPath,"
|
||||
"@{n='Line';e={$_.Extent.StartLineNumber}},"
|
||||
"@{n='EndLine';e={$_.Extent.EndLineNumber}},"
|
||||
"Message | ConvertTo-Json -Depth 3 -AsArray"
|
||||
)
|
||||
|
||||
_PSSA_COMMAND = "Invoke-ScriptAnalyzer -Path . -Recurse -ErrorAction SilentlyContinue"
|
||||
|
||||
_INJECTION_HUNTER_COMMAND = (
|
||||
"$m = (Get-Module -ListAvailable -Name InjectionHunter | Select-Object -First 1).Path; "
|
||||
"Invoke-ScriptAnalyzer -Path . -Recurse -CustomRulePath $m -ErrorAction SilentlyContinue"
|
||||
)
|
||||
|
||||
|
||||
def _pwsh_module_available(repo: str, module: str) -> bool:
|
||||
r = run_tool(
|
||||
["pwsh", "-NoProfile", "-NonInteractive", "-Command",
|
||||
f"if (Get-Module -ListAvailable -Name {module}) {{ exit 0 }} else {{ exit 1 }}"],
|
||||
cwd=repo,
|
||||
)
|
||||
return r.ran and r.exit_code == 0
|
||||
|
||||
|
||||
def _run_powershell_tools(repo: str) -> tuple[list[Finding], dict[str, ToolStat]]:
|
||||
findings: list[Finding] = []
|
||||
stats: dict[str, ToolStat] = {}
|
||||
if not tool_available("pwsh"):
|
||||
for name in ("psscriptanalyzer", "injectionhunter"):
|
||||
stats[name] = ToolStat(ran=False, reason="pwsh not on PATH")
|
||||
return findings, stats
|
||||
for tool, module, command in [
|
||||
("psscriptanalyzer", "PSScriptAnalyzer", _PSSA_COMMAND),
|
||||
("injectionhunter", "InjectionHunter", _INJECTION_HUNTER_COMMAND),
|
||||
]:
|
||||
if not _pwsh_module_available(repo, module):
|
||||
stats[tool] = ToolStat(ran=False, reason=f"{module} module not installed")
|
||||
continue
|
||||
r = run_tool(
|
||||
["pwsh", "-NoProfile", "-NonInteractive", "-Command", command + _PSSA_PROJECT],
|
||||
cwd=repo,
|
||||
)
|
||||
parsed = ad_pssa.parse_psscriptanalyzer_output(r.stdout, repo_root=repo, tool=tool)
|
||||
stats[tool] = ToolStat(ran=True, pre_filter=len(parsed), post_filter=0)
|
||||
findings.extend(parsed)
|
||||
return findings, stats
|
||||
|
||||
|
||||
def _run_gha_tools(repo: str) -> tuple[list[Finding], dict[str, ToolStat]]:
|
||||
findings: list[Finding] = []
|
||||
stats: dict[str, ToolStat] = {}
|
||||
if tool_available("actionlint"):
|
||||
r = run_tool(["actionlint", "-format", "{{json .}}", ".github/"], cwd=repo)
|
||||
parsed = ad_actionlint.parse_actionlint_output(r.stdout)
|
||||
stats["actionlint"] = ToolStat(ran=True, pre_filter=len(parsed), post_filter=0)
|
||||
findings.extend(parsed)
|
||||
else:
|
||||
stats["actionlint"] = ToolStat(ran=False, reason="not on PATH")
|
||||
if tool_available("zizmor"):
|
||||
r = run_tool(["zizmor", "--format", "sarif", ".github/"], cwd=repo)
|
||||
parsed = ad_zizmor.parse_zizmor_output(r.stdout)
|
||||
stats["zizmor"] = ToolStat(ran=True, pre_filter=len(parsed), post_filter=0)
|
||||
findings.extend(parsed)
|
||||
else:
|
||||
stats["zizmor"] = ToolStat(ran=False, reason="not on PATH")
|
||||
return findings, stats
|
||||
|
||||
|
||||
def _run_polyglot_tools(repo: str) -> tuple[list[Finding], dict[str, ToolStat]]:
|
||||
findings: list[Finding] = []
|
||||
stats: dict[str, ToolStat] = {}
|
||||
if tool_available("opengrep"):
|
||||
lua_rules = _HERE / "rules" / "lua-security.yaml"
|
||||
r = run_tool(
|
||||
["opengrep", "--config=auto", f"--config={lua_rules}", "--json", "--quiet"],
|
||||
cwd=repo,
|
||||
)
|
||||
parsed = ad_opengrep.parse_opengrep_output(r.stdout)
|
||||
stats["opengrep"] = ToolStat(ran=True, pre_filter=len(parsed))
|
||||
findings.extend(parsed)
|
||||
else:
|
||||
stats["opengrep"] = ToolStat(ran=False, reason="not on PATH")
|
||||
if tool_available("gitleaks"):
|
||||
r = run_tool(["gitleaks", "detect", "--report-format=json", "--report-path=/dev/stdout", "--no-banner"], cwd=repo)
|
||||
parsed = ad_gitleaks.parse_gitleaks_output(r.stdout)
|
||||
stats["gitleaks"] = ToolStat(ran=True, pre_filter=len(parsed))
|
||||
findings.extend(parsed)
|
||||
else:
|
||||
stats["gitleaks"] = ToolStat(ran=False, reason="not on PATH")
|
||||
return findings, stats
|
||||
|
||||
|
||||
def _run_maintainability_tools(
|
||||
repo: str, languages_present: set[str]
|
||||
) -> tuple[list[Finding], dict[str, ToolStat]]:
|
||||
findings: list[Finding] = []
|
||||
stats: dict[str, ToolStat] = {}
|
||||
is_python = "python" in languages_present
|
||||
is_js = "javascript" in languages_present or "typescript" in languages_present
|
||||
for binary, args, adapter, should_run in [
|
||||
("vulture", ["vulture", "."], ad_vulture.parse_vulture_output, is_python),
|
||||
("radon", ["radon", "cc", "-j", "."], ad_radon.parse_radon_output, is_python),
|
||||
("interrogate", ["interrogate", "--quiet", "--output-format=json", "."], ad_interrogate.parse_interrogate_output, is_python),
|
||||
("lizard", ["lizard", "--csv", "."], ad_lizard.parse_lizard_output, True),
|
||||
("npx", ["npx", "knip", "--reporter", "json"], ad_knip.parse_knip_output, is_js),
|
||||
("jscpd", ["jscpd", "--reporters", "json", "--silent", "."], ad_jscpd.parse_jscpd_output, is_js),
|
||||
]:
|
||||
tool_name = "knip" if binary == "npx" else binary
|
||||
if not should_run:
|
||||
continue
|
||||
if not tool_available(binary):
|
||||
stats[tool_name] = ToolStat(ran=False, reason="not on PATH")
|
||||
continue
|
||||
r = run_tool(args, cwd=repo)
|
||||
parsed = adapter(r.stdout)
|
||||
stats[tool_name] = ToolStat(ran=True, pre_filter=len(parsed), post_filter=0)
|
||||
findings.extend(parsed)
|
||||
return findings, stats
|
||||
|
||||
|
||||
def _run_dep_tools(repo: str) -> tuple[list[Finding], dict[str, ToolStat]]:
|
||||
findings: list[Finding] = []
|
||||
stats: dict[str, ToolStat] = {}
|
||||
if tool_available("pip-audit"):
|
||||
r = run_tool(["pip-audit", "-f", "json"], cwd=repo)
|
||||
parsed = ad_pip_audit.parse_pip_audit_output(r.stdout)
|
||||
stats["pip-audit"] = ToolStat(ran=True, pre_filter=len(parsed))
|
||||
findings.extend(parsed)
|
||||
else:
|
||||
stats["pip-audit"] = ToolStat(ran=False, reason="not on PATH")
|
||||
if tool_available("osv-scanner"):
|
||||
r = run_tool(["osv-scanner", "--recursive", "--format=json", "."], cwd=repo)
|
||||
parsed = ad_osv.parse_osv_scanner_output(r.stdout)
|
||||
stats["osv-scanner"] = ToolStat(ran=True, pre_filter=len(parsed))
|
||||
findings.extend(parsed)
|
||||
else:
|
||||
stats["osv-scanner"] = ToolStat(ran=False, reason="not on PATH")
|
||||
return findings, stats
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--repo", required=True)
|
||||
parser.add_argument("--base", default=None)
|
||||
parser.add_argument("--head", default="HEAD")
|
||||
parser.add_argument("--output-dir", required=True)
|
||||
parser.add_argument("--mode", choices=("local", "ref"), required=True)
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
repo = str(Path(args.repo).resolve())
|
||||
out_dir = Path(args.output_dir).resolve()
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
default_branch = resolve_default_branch(repo)
|
||||
base = args.base or resolve_base_ref(repo, default_branch)
|
||||
errors: list[str] = []
|
||||
|
||||
try:
|
||||
changed = changed_files_with_ranges(repo, base, args.head)
|
||||
except RuntimeError as e:
|
||||
manifest = Manifest(
|
||||
mode=args.mode, base_ref=base, head_ref=args.head,
|
||||
default_branch=default_branch,
|
||||
language_breakdown=LanguageBreakdown(),
|
||||
changed_files=[], findings=[],
|
||||
package_diffs={}, tool_stats={}, tools_unavailable=[],
|
||||
errors=[str(e)],
|
||||
)
|
||||
(out_dir / "manifest.json").write_text(manifest.to_json())
|
||||
return 1
|
||||
|
||||
changed_files: list[ChangedFile] = []
|
||||
skipped: list[str] = []
|
||||
dep_manifest_paths: list[str] = []
|
||||
unsupported_by_language: dict[str, list[str]] = {}
|
||||
for path, ranges in changed:
|
||||
if is_supported_source(path):
|
||||
changed_files.append(ChangedFile(
|
||||
path=path,
|
||||
language=detect_language(path),
|
||||
added_lines=ranges,
|
||||
))
|
||||
continue
|
||||
if is_gha_file(path):
|
||||
changed_files.append(ChangedFile(
|
||||
path=path,
|
||||
language="github-actions",
|
||||
added_lines=ranges,
|
||||
))
|
||||
continue
|
||||
if is_dep_manifest(path):
|
||||
dep_manifest_paths.append(path)
|
||||
continue
|
||||
unsupported = known_unsupported_language(path)
|
||||
if unsupported is not None:
|
||||
unsupported_by_language.setdefault(unsupported, []).append(path)
|
||||
skipped.append(path)
|
||||
|
||||
for language, paths in sorted(unsupported_by_language.items()):
|
||||
errors.append(
|
||||
f"unsupported language not reviewed: {language} ({', '.join(paths)})"
|
||||
)
|
||||
|
||||
breakdown = _build_breakdown(changed_files, skipped)
|
||||
|
||||
if not changed_files and not dep_manifest_paths:
|
||||
errors.append("no supported source files in change set; skipped: " + ", ".join(skipped))
|
||||
manifest = Manifest(
|
||||
mode=args.mode, base_ref=base, head_ref=args.head,
|
||||
default_branch=default_branch, language_breakdown=breakdown,
|
||||
changed_files=[], findings=[],
|
||||
package_diffs={}, tool_stats={}, tools_unavailable=[],
|
||||
errors=errors,
|
||||
)
|
||||
(out_dir / "manifest.json").write_text(manifest.to_json())
|
||||
return 1
|
||||
|
||||
findings: list[Finding] = []
|
||||
tool_stats: dict[str, ToolStat] = {}
|
||||
|
||||
def collect(runner, *args) -> None:
|
||||
f, s = runner(*args)
|
||||
findings.extend(f)
|
||||
tool_stats.update(s)
|
||||
|
||||
languages_present = {cf.language for cf in changed_files}
|
||||
if "python" in languages_present:
|
||||
collect(_run_python_tools, repo)
|
||||
if "javascript" in languages_present or "typescript" in languages_present:
|
||||
collect(_run_js_tools, repo)
|
||||
if "csharp" in languages_present:
|
||||
collect(_run_dotnet_tools, repo)
|
||||
|
||||
collect(_run_polyglot_tools, repo)
|
||||
collect(_run_dep_tools, repo)
|
||||
collect(_run_maintainability_tools, repo, languages_present)
|
||||
if "lua" in languages_present:
|
||||
collect(_run_lua_tools, repo)
|
||||
if "powershell" in languages_present:
|
||||
collect(_run_powershell_tools, repo)
|
||||
if "github-actions" in languages_present:
|
||||
collect(_run_gha_tools, repo)
|
||||
|
||||
filtered = filter_findings_by_diff(findings, changed_files, repo_root=repo)
|
||||
for tool_name, stat in tool_stats.items():
|
||||
if stat.ran:
|
||||
stat.post_filter = sum(1 for f in filtered if f.tool == tool_name)
|
||||
|
||||
tools_unavailable = [name for name, s in tool_stats.items() if not s.ran]
|
||||
|
||||
package_diffs = _build_package_diffs(repo, base, dep_manifest_paths)
|
||||
|
||||
manifest = Manifest(
|
||||
mode=args.mode,
|
||||
base_ref=base,
|
||||
head_ref=args.head,
|
||||
default_branch=default_branch,
|
||||
language_breakdown=breakdown,
|
||||
changed_files=changed_files,
|
||||
findings=filtered,
|
||||
package_diffs=package_diffs,
|
||||
tool_stats=tool_stats,
|
||||
tools_unavailable=tools_unavailable,
|
||||
errors=errors,
|
||||
)
|
||||
|
||||
(out_dir / "manifest.json").write_text(manifest.to_json())
|
||||
manifest_dict = manifest.to_dict()
|
||||
for agent in ("security-triage", "type-safety", "dependency", "consistency", "secrets", "maintainability", "walkthrough", "gha-reviewer"):
|
||||
sliced = slice_for_agent(manifest_dict, agent)
|
||||
(out_dir / f"manifest-{agent}.json").write_text(json.dumps(sliced, indent=2))
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,41 @@
|
||||
"""Filter Finding[] to those overlapping any changed-line range."""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
from scripts.manifest import ChangedFile, Finding
|
||||
|
||||
|
||||
def _overlaps(a: tuple[int, int], b: tuple[int, int]) -> bool:
|
||||
return not (a[1] < b[0] or b[1] < a[0])
|
||||
|
||||
|
||||
def _normalize(path: str, repo_root: str | None = None) -> str:
|
||||
"""Strip `./` prefix and (optionally) the repo root so adapter paths
|
||||
line up with git-diff paths regardless of which form the tool emitted.
|
||||
"""
|
||||
if repo_root and path.startswith(repo_root):
|
||||
path = os.path.relpath(path, repo_root)
|
||||
while path.startswith("./"):
|
||||
path = path[2:]
|
||||
return path
|
||||
|
||||
|
||||
def filter_findings_by_diff(
|
||||
findings: list[Finding], changed_files: list[ChangedFile],
|
||||
repo_root: str | None = None,
|
||||
) -> list[Finding]:
|
||||
by_path: dict[str, list[tuple[int, int]]] = {
|
||||
_normalize(cf.path, repo_root): list(cf.added_lines)
|
||||
for cf in changed_files
|
||||
}
|
||||
out: list[Finding] = []
|
||||
for f in findings:
|
||||
norm = _normalize(f.file, repo_root)
|
||||
ranges = by_path.get(norm)
|
||||
if not ranges:
|
||||
continue
|
||||
if any(_overlaps((f.line, f.end_line), r) for r in ranges):
|
||||
f.file = norm
|
||||
out.append(f)
|
||||
return out
|
||||
@@ -0,0 +1,122 @@
|
||||
"""Git diff scanning with origin-base resolution and defensive flags."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import subprocess
|
||||
|
||||
|
||||
_HUNK_RE = re.compile(r"^@@ -\d+(?:,\d+)? \+(\d+)(?:,(\d+))? @@")
|
||||
|
||||
|
||||
def resolve_default_branch(repo: str) -> str:
|
||||
try:
|
||||
r = subprocess.run(
|
||||
["git", "-C", repo, "symbolic-ref", "refs/remotes/origin/HEAD"],
|
||||
capture_output=True, text=True, check=False,
|
||||
)
|
||||
if r.returncode == 0:
|
||||
return r.stdout.strip().rsplit("/", 1)[-1]
|
||||
except FileNotFoundError:
|
||||
pass
|
||||
for candidate in ("main", "master"):
|
||||
r = subprocess.run(
|
||||
["git", "-C", repo, "rev-parse", f"origin/{candidate}"],
|
||||
capture_output=True, text=True, check=False,
|
||||
)
|
||||
if r.returncode == 0:
|
||||
return candidate
|
||||
return "main"
|
||||
|
||||
|
||||
def resolve_base_ref(repo: str, branch: str) -> str:
|
||||
"""Return the diff base for `branch`, fetching origin first."""
|
||||
try:
|
||||
subprocess.run(
|
||||
["git", "-C", repo, "fetch", "--quiet", "--no-tags",
|
||||
"origin", branch],
|
||||
capture_output=True, text=True, check=False, timeout=60,
|
||||
)
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired):
|
||||
pass
|
||||
check = subprocess.run(
|
||||
["git", "-C", repo, "rev-parse", "--verify", "--quiet",
|
||||
f"origin/{branch}"],
|
||||
capture_output=True, text=True, check=False,
|
||||
)
|
||||
return f"origin/{branch}" if check.returncode == 0 else branch
|
||||
|
||||
|
||||
def _run_git_diff(repo: str, base: str, head: str) -> str:
|
||||
result = subprocess.run(
|
||||
[
|
||||
"git", "-C", repo,
|
||||
"-c", "diff.noprefix=false",
|
||||
"-c", "color.ui=never",
|
||||
"diff", "--no-ext-diff", "--unified=0", f"{base}...{head}",
|
||||
],
|
||||
capture_output=True, text=True, check=False,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
raise RuntimeError(f"git diff failed: {result.stderr.strip()}")
|
||||
return result.stdout
|
||||
|
||||
|
||||
def _iter_file_blocks(diff_text: str):
|
||||
current: str | None = None
|
||||
buf: list[str] = []
|
||||
for line in diff_text.splitlines():
|
||||
if line.startswith("diff --git "):
|
||||
if current is not None:
|
||||
yield current, "\n".join(buf)
|
||||
current = None
|
||||
buf = []
|
||||
elif line.startswith("+++ b/"):
|
||||
current = line[len("+++ b/"):]
|
||||
if current is not None:
|
||||
buf.append(line)
|
||||
if current is not None:
|
||||
yield current, "\n".join(buf)
|
||||
|
||||
|
||||
def _added_ranges(block: str) -> list[tuple[int, int]]:
|
||||
ranges: list[tuple[int, int]] = []
|
||||
cur: int | None = None
|
||||
start: int | None = None
|
||||
end: int | None = None
|
||||
for line in block.splitlines():
|
||||
m = _HUNK_RE.match(line)
|
||||
if m:
|
||||
if start is not None:
|
||||
ranges.append((start, end)) # type: ignore[arg-type]
|
||||
start = end = None
|
||||
cur = int(m.group(1))
|
||||
continue
|
||||
if cur is None:
|
||||
continue
|
||||
if line.startswith("+") and not line.startswith("+++"):
|
||||
if start is None:
|
||||
start = cur
|
||||
end = cur
|
||||
cur += 1
|
||||
elif line.startswith("-") and not line.startswith("---"):
|
||||
continue
|
||||
else:
|
||||
if start is not None:
|
||||
ranges.append((start, end)) # type: ignore[arg-type]
|
||||
start = end = None
|
||||
cur += 1
|
||||
if start is not None:
|
||||
ranges.append((start, end)) # type: ignore[arg-type]
|
||||
return ranges
|
||||
|
||||
|
||||
def changed_files_with_ranges(
|
||||
repo: str, base: str, head: str,
|
||||
) -> list[tuple[str, list[tuple[int, int]]]]:
|
||||
"""Return [(path, added_line_ranges), ...] for every changed file."""
|
||||
out: list[tuple[str, list[tuple[int, int]]]] = []
|
||||
for path, block in _iter_file_blocks(_run_git_diff(repo, base, head)):
|
||||
ranges = _added_ranges(block)
|
||||
if ranges:
|
||||
out.append((path, ranges))
|
||||
return out
|
||||
+160
@@ -0,0 +1,160 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
SKILL_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
cd "$SKILL_DIR"
|
||||
|
||||
say() { printf '\n\033[1m▶ %s\033[0m\n' "$*"; }
|
||||
warn() { printf '\033[33m! %s\033[0m\n' "$*"; }
|
||||
|
||||
|
||||
if ! command -v uv >/dev/null 2>&1; then
|
||||
warn "uv not installed. Install: https://docs.astral.sh/uv/getting-started/installation/"
|
||||
warn "Falling back to plain pip — tools will install into the active environment."
|
||||
if ! command -v pip >/dev/null 2>&1; then
|
||||
echo "Neither uv nor pip available. Aborting Python-tools install."
|
||||
exit 1
|
||||
fi
|
||||
pip install bandit ruff mypy pip-audit
|
||||
else
|
||||
say "Installing Python tools into $SKILL_DIR/.venv/ via uv"
|
||||
uv sync --group tools
|
||||
fi
|
||||
|
||||
|
||||
install_opengrep() {
|
||||
if [[ -x "$SKILL_DIR/.venv/bin/opengrep" ]]; then
|
||||
echo " opengrep: already installed (.venv/bin)"
|
||||
return
|
||||
fi
|
||||
say "Installing opengrep into $SKILL_DIR/.venv/bin/"
|
||||
local os arch asset
|
||||
os="$(uname -s)"
|
||||
arch="$(uname -m)"
|
||||
case "$os-$arch" in
|
||||
Linux-x86_64) asset="opengrep_manylinux_x86" ;;
|
||||
Linux-aarch64) asset="opengrep_manylinux_aarch64" ;;
|
||||
Darwin-x86_64) asset="opengrep_osx_x86" ;;
|
||||
Darwin-arm64) asset="opengrep_osx_arm64" ;;
|
||||
*)
|
||||
warn "opengrep: unsupported platform $os-$arch, install manually from https://github.com/opengrep/opengrep/releases"
|
||||
return
|
||||
;;
|
||||
esac
|
||||
local tag url
|
||||
tag="$(curl -fsSL https://api.github.com/repos/opengrep/opengrep/releases/latest | grep -m1 '"tag_name"' | sed -E 's/.*"([^"]+)".*/\1/')"
|
||||
if [[ -z "$tag" ]]; then
|
||||
warn "opengrep: could not resolve latest release tag, install manually"
|
||||
return
|
||||
fi
|
||||
url="https://github.com/opengrep/opengrep/releases/download/$tag/$asset"
|
||||
mkdir -p "$SKILL_DIR/.venv/bin"
|
||||
curl -fsSL "$url" -o "$SKILL_DIR/.venv/bin/opengrep"
|
||||
chmod +x "$SKILL_DIR/.venv/bin/opengrep"
|
||||
}
|
||||
|
||||
install_opengrep
|
||||
|
||||
|
||||
install_powershell_modules() {
|
||||
if ! command -v pwsh >/dev/null 2>&1; then
|
||||
warn "pwsh not found — PowerShell review (PSScriptAnalyzer, InjectionHunter) will be skipped."
|
||||
warn " Install: https://learn.microsoft.com/powershell/scripting/install/installing-powershell"
|
||||
return
|
||||
fi
|
||||
say "Installing PowerShell modules for the current user"
|
||||
pwsh -NoProfile -NonInteractive -Command '
|
||||
foreach ($m in "PSScriptAnalyzer", "InjectionHunter") {
|
||||
if (Get-Module -ListAvailable -Name $m) {
|
||||
Write-Host " $m: already installed"
|
||||
} else {
|
||||
Install-Module -Name $m -Scope CurrentUser -Force -AcceptLicense -Repository PSGallery
|
||||
Write-Host " $m: installed"
|
||||
}
|
||||
}'
|
||||
}
|
||||
|
||||
install_powershell_modules
|
||||
|
||||
|
||||
install_native_brew() {
|
||||
say "Installing native tools via Homebrew"
|
||||
for pkg in gitleaks osv-scanner gh; do
|
||||
if brew list --formula | grep -qx "$pkg"; then
|
||||
echo " $pkg: already installed"
|
||||
else
|
||||
brew install "$pkg"
|
||||
fi
|
||||
done
|
||||
}
|
||||
|
||||
install_native_apt() {
|
||||
say "Installing native tools via apt-get"
|
||||
if ! command -v gh >/dev/null 2>&1; then
|
||||
warn "gh: follow https://github.com/cli/cli/blob/trunk/docs/install_linux.md"
|
||||
fi
|
||||
if ! command -v gitleaks >/dev/null 2>&1; then
|
||||
warn "gitleaks: download from https://github.com/gitleaks/gitleaks/releases"
|
||||
fi
|
||||
if ! command -v osv-scanner >/dev/null 2>&1; then
|
||||
warn "osv-scanner: install via 'go install github.com/google/osv-scanner/cmd/osv-scanner@latest' or download from https://github.com/google/osv-scanner/releases"
|
||||
fi
|
||||
}
|
||||
|
||||
install_native_arch() {
|
||||
local helper
|
||||
if command -v paru >/dev/null 2>&1; then
|
||||
helper="paru"
|
||||
elif command -v yay >/dev/null 2>&1; then
|
||||
helper="yay"
|
||||
else
|
||||
helper="pacman"
|
||||
fi
|
||||
say "Installing native tools via $helper"
|
||||
|
||||
local pkgs=(gitleaks github-cli osv-scanner)
|
||||
if [[ "$helper" == "pacman" ]]; then
|
||||
sudo pacman -S --needed --noconfirm gitleaks github-cli || true
|
||||
if ! command -v osv-scanner >/dev/null 2>&1; then
|
||||
warn "osv-scanner is AUR-only; pacman can't install it. Use paru/yay or install manually:"
|
||||
warn " go install github.com/google/osv-scanner/cmd/osv-scanner@latest"
|
||||
fi
|
||||
else
|
||||
"$helper" -S --needed --noconfirm "${pkgs[@]}"
|
||||
fi
|
||||
}
|
||||
|
||||
if command -v brew >/dev/null 2>&1; then
|
||||
install_native_brew
|
||||
elif command -v pacman >/dev/null 2>&1; then
|
||||
install_native_arch
|
||||
elif command -v apt-get >/dev/null 2>&1; then
|
||||
install_native_apt
|
||||
else
|
||||
warn "No supported native package manager found (brew/pacman/apt). Install gitleaks, osv-scanner, gh manually."
|
||||
fi
|
||||
|
||||
|
||||
say "Verifying tool availability"
|
||||
for tool in bandit ruff mypy pip-audit opengrep vulture radon interrogate lizard gitleaks osv-scanner gh pwsh; do
|
||||
if [[ -x "$SKILL_DIR/.venv/bin/$tool" ]]; then
|
||||
printf ' %-15s %s\n' "$tool" "(.venv/bin)"
|
||||
elif command -v "$tool" >/dev/null 2>&1; then
|
||||
printf ' %-15s %s\n' "$tool" "$(command -v "$tool")"
|
||||
else
|
||||
printf ' %-15s \033[31mmissing\033[0m\n' "$tool"
|
||||
fi
|
||||
done
|
||||
|
||||
cat <<EOF
|
||||
|
||||
Per-project tools (not installed here — must live in the target repo):
|
||||
- eslint + eslint-plugin-security (npm i -D)
|
||||
- typescript (tsc) (npm i -D)
|
||||
- SecurityCodeScan + dotnet (dotnet add package SecurityCodeScan.VS2019)
|
||||
- knip (npm i -D)
|
||||
- jscpd (npm i -D)
|
||||
|
||||
Run /audit-code to use the skill.
|
||||
EOF
|
||||
@@ -0,0 +1,111 @@
|
||||
"""Detect the language of a changed file from its path."""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import PurePosixPath
|
||||
|
||||
|
||||
SOURCE_EXTENSIONS: dict[str, str] = {
|
||||
".py": "python",
|
||||
".js": "javascript",
|
||||
".jsx": "javascript",
|
||||
".mjs": "javascript",
|
||||
".cjs": "javascript",
|
||||
".ts": "typescript",
|
||||
".tsx": "typescript",
|
||||
".cs": "csharp",
|
||||
".lua": "lua",
|
||||
".ps1": "powershell",
|
||||
".psm1": "powershell",
|
||||
".psd1": "powershell",
|
||||
}
|
||||
|
||||
|
||||
DEP_MANIFESTS: set[str] = {
|
||||
"requirements.txt", "pyproject.toml", "poetry.lock", "Pipfile.lock",
|
||||
"Pipfile", "setup.py", "setup.cfg",
|
||||
"package.json", "package-lock.json", "yarn.lock", "pnpm-lock.yaml",
|
||||
"packages.config",
|
||||
}
|
||||
|
||||
DEP_MANIFEST_PREFIXES: tuple[str, ...] = ("requirements",)
|
||||
DEP_MANIFEST_EXTENSIONS: tuple[str, ...] = (".csproj", ".sln")
|
||||
|
||||
|
||||
KNOWN_UNSUPPORTED_LANGUAGES: dict[str, str] = {
|
||||
".go": "Go",
|
||||
".rb": "Ruby",
|
||||
".java": "Java",
|
||||
".kt": "Kotlin",
|
||||
".kts": "Kotlin",
|
||||
".rs": "Rust",
|
||||
".swift": "Swift",
|
||||
".php": "PHP",
|
||||
".scala": "Scala",
|
||||
".sc": "Scala",
|
||||
".clj": "Clojure",
|
||||
".cljs": "ClojureScript",
|
||||
".ex": "Elixir",
|
||||
".exs": "Elixir",
|
||||
".erl": "Erlang",
|
||||
".hrl": "Erlang",
|
||||
".dart": "Dart",
|
||||
".r": "R",
|
||||
".jl": "Julia",
|
||||
".zig": "Zig",
|
||||
".nim": "Nim",
|
||||
".hs": "Haskell",
|
||||
".ml": "OCaml",
|
||||
".mli": "OCaml",
|
||||
".fs": "F#",
|
||||
".fsi": "F#",
|
||||
".fsx": "F#",
|
||||
".vb": "Visual Basic",
|
||||
".pl": "Perl",
|
||||
".pm": "Perl",
|
||||
".c": "C",
|
||||
".h": "C/C++ header",
|
||||
".cpp": "C++",
|
||||
".cc": "C++",
|
||||
".cxx": "C++",
|
||||
".hpp": "C++",
|
||||
".m": "Objective-C",
|
||||
".mm": "Objective-C++",
|
||||
".groovy": "Groovy",
|
||||
}
|
||||
|
||||
_GHA_YAML_EXTS: frozenset[str] = frozenset((".yml", ".yaml"))
|
||||
_GHA_PATH_PREFIXES: tuple[str, ...] = (".github/workflows/", ".github/actions/")
|
||||
|
||||
|
||||
def detect_language(path: str) -> str | None:
|
||||
suffix = PurePosixPath(path).suffix
|
||||
return SOURCE_EXTENSIONS.get(suffix)
|
||||
|
||||
|
||||
def is_supported_source(path: str) -> bool:
|
||||
return detect_language(path) is not None
|
||||
|
||||
|
||||
def is_gha_file(path: str) -> bool:
|
||||
"""Return True if path is a GitHub Actions workflow or action definition."""
|
||||
normalized = path.replace("\\", "/")
|
||||
if PurePosixPath(normalized).suffix.lower() not in _GHA_YAML_EXTS:
|
||||
return False
|
||||
return normalized.startswith(_GHA_PATH_PREFIXES)
|
||||
|
||||
|
||||
def known_unsupported_language(path: str) -> str | None:
|
||||
"""Return the display name of a recognized-but-unsupported source language,
|
||||
or None if the path isn't one we recognize as source code.
|
||||
"""
|
||||
suffix = PurePosixPath(path).suffix.lower()
|
||||
return KNOWN_UNSUPPORTED_LANGUAGES.get(suffix)
|
||||
|
||||
|
||||
def is_dep_manifest(path: str) -> bool:
|
||||
name = PurePosixPath(path).name
|
||||
if name in DEP_MANIFESTS:
|
||||
return True
|
||||
if any(name.startswith(p) and name.endswith(".txt") for p in DEP_MANIFEST_PREFIXES):
|
||||
return True
|
||||
return any(name.endswith(ext) for ext in DEP_MANIFEST_EXTENSIONS)
|
||||
@@ -0,0 +1,84 @@
|
||||
"""Append one subagent_run row per agent after the fan-out completes.
|
||||
|
||||
The orchestrator calls this once after all subagents return. It scans
|
||||
<output-dir>/findings-<agent>.json for each agent listed in the
|
||||
--usage-json payload, counts findings, and writes a subagent_run row to
|
||||
runs.jsonl using token / duration metadata supplied by the orchestrator.
|
||||
|
||||
Usage:
|
||||
|
||||
python scripts/log-run.py \\
|
||||
--output-dir <OUTPUT> --run-id <hex> --repo <path> --mode <local|ref> \\
|
||||
--usage-json - <<JSON
|
||||
{
|
||||
"walkthrough-reviewer": {"model":"sonnet","input_tokens":1234,"output_tokens":567,"duration_ms":4500},
|
||||
"security-triage-reviewer":{"model":"sonnet","input_tokens":2345,"output_tokens":678,"duration_ms":5200}
|
||||
}
|
||||
JSON
|
||||
|
||||
`--log-path` defaults to ~/.claude/cache/audit-code/runs.jsonl.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from scripts.telemetry import append_subagent_run
|
||||
|
||||
_DEFAULT_LOG = Path.home() / ".claude/cache/audit-code/runs.jsonl"
|
||||
|
||||
|
||||
def _count_findings(path: Path) -> int:
|
||||
if not path.exists():
|
||||
return 0
|
||||
try:
|
||||
data = json.loads(path.read_text(encoding="utf-8"))
|
||||
except json.JSONDecodeError:
|
||||
return 0
|
||||
if isinstance(data, dict):
|
||||
findings = data.get("findings")
|
||||
if isinstance(findings, list):
|
||||
return len(findings)
|
||||
elif isinstance(data, list):
|
||||
return len(data)
|
||||
return 0
|
||||
|
||||
|
||||
def main(argv: list[str]) -> int:
|
||||
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||
p.add_argument("--output-dir", required=True, type=Path)
|
||||
p.add_argument("--run-id", required=True)
|
||||
p.add_argument("--repo", required=True)
|
||||
p.add_argument("--mode", required=True, choices=["local", "ref"])
|
||||
p.add_argument("--log-path", type=Path, default=_DEFAULT_LOG)
|
||||
p.add_argument("--usage-json", required=True,
|
||||
help="Path to JSON file, or '-' for stdin.")
|
||||
args = p.parse_args(argv)
|
||||
|
||||
raw = sys.stdin.read() if args.usage_json == "-" else Path(args.usage_json).read_text(encoding="utf-8")
|
||||
usage = json.loads(raw)
|
||||
if not isinstance(usage, dict):
|
||||
print("usage-json must be a JSON object keyed by agent name", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
for agent, meta in usage.items():
|
||||
if not isinstance(meta, dict):
|
||||
print(f"skipping {agent}: usage entry not an object", file=sys.stderr)
|
||||
continue
|
||||
findings_path = args.output_dir / f"findings-{agent}.json"
|
||||
append_subagent_run(
|
||||
args.log_path,
|
||||
run_id=args.run_id, repo=args.repo, mode=args.mode, agent=agent,
|
||||
model=str(meta.get("model", "?")),
|
||||
input_tokens=int(meta.get("input_tokens", 0)),
|
||||
output_tokens=int(meta.get("output_tokens", 0)),
|
||||
duration_ms=int(meta.get("duration_ms", 0)),
|
||||
finding_count=_count_findings(findings_path),
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main(sys.argv[1:]))
|
||||
@@ -0,0 +1,146 @@
|
||||
"""Manifest dataclasses for collect-findings.py output.
|
||||
|
||||
The manifest is the contract between the script and the review subagents.
|
||||
Schema mirrors DESIGN.md.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass, field, asdict
|
||||
from typing import Literal
|
||||
|
||||
|
||||
Mode = Literal["local", "ref"]
|
||||
Severity = Literal["critical", "high", "medium", "low", "info"]
|
||||
Language = Literal[
|
||||
"python", "javascript", "typescript", "csharp", "lua", "powershell",
|
||||
"github-actions",
|
||||
]
|
||||
|
||||
|
||||
@dataclass
|
||||
class Finding:
|
||||
tool: str
|
||||
rule_id: str
|
||||
severity: Severity
|
||||
file: str
|
||||
line: int
|
||||
end_line: int
|
||||
message: str
|
||||
cwe: str | None = None
|
||||
fix_suggestion: str | None = None
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
@dataclass
|
||||
class ChangedFile:
|
||||
path: str
|
||||
language: str
|
||||
added_lines: list[tuple[int, int]]
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {
|
||||
"path": self.path,
|
||||
"language": self.language,
|
||||
"added_lines": [list(r) for r in self.added_lines],
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class LanguageBreakdown:
|
||||
python: int = 0
|
||||
javascript: int = 0
|
||||
typescript: int = 0
|
||||
csharp: int = 0
|
||||
lua: int = 0
|
||||
powershell: int = 0
|
||||
github_actions: int = 0
|
||||
skipped_files: list[str] = field(default_factory=list)
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
@dataclass
|
||||
class PackageEntry:
|
||||
name: str
|
||||
version: str
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
@dataclass
|
||||
class PackageUpgrade:
|
||||
name: str
|
||||
from_version: str
|
||||
to_version: str
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {"name": self.name, "from": self.from_version, "to": self.to_version}
|
||||
|
||||
|
||||
@dataclass
|
||||
class PackageDiff:
|
||||
added: list[PackageEntry] = field(default_factory=list)
|
||||
removed: list[PackageEntry] = field(default_factory=list)
|
||||
upgraded: list[PackageUpgrade] = field(default_factory=list)
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {
|
||||
"added": [p.to_dict() for p in self.added],
|
||||
"removed": [p.to_dict() for p in self.removed],
|
||||
"upgraded": [u.to_dict() for u in self.upgraded],
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class ToolStat:
|
||||
ran: bool
|
||||
pre_filter: int = 0
|
||||
post_filter: int = 0
|
||||
reason: str | None = None
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
d: dict = {"ran": self.ran}
|
||||
if self.ran:
|
||||
d["pre_filter"] = self.pre_filter
|
||||
d["post_filter"] = self.post_filter
|
||||
else:
|
||||
d["reason"] = self.reason or ""
|
||||
return d
|
||||
|
||||
|
||||
@dataclass
|
||||
class Manifest:
|
||||
mode: Mode
|
||||
base_ref: str
|
||||
head_ref: str
|
||||
default_branch: str
|
||||
language_breakdown: LanguageBreakdown
|
||||
changed_files: list[ChangedFile]
|
||||
findings: list[Finding]
|
||||
package_diffs: dict[str, PackageDiff]
|
||||
tool_stats: dict[str, ToolStat]
|
||||
tools_unavailable: list[str]
|
||||
errors: list[str]
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {
|
||||
"mode": self.mode,
|
||||
"base_ref": self.base_ref,
|
||||
"head_ref": self.head_ref,
|
||||
"default_branch": self.default_branch,
|
||||
"language_breakdown": self.language_breakdown.to_dict(),
|
||||
"changed_files": [c.to_dict() for c in self.changed_files],
|
||||
"findings": [f.to_dict() for f in self.findings],
|
||||
"package_diffs": {k: v.to_dict() for k, v in self.package_diffs.items()},
|
||||
"tool_stats": {k: v.to_dict() for k, v in self.tool_stats.items()},
|
||||
"tools_unavailable": list(self.tools_unavailable),
|
||||
"errors": list(self.errors),
|
||||
}
|
||||
|
||||
def to_json(self, indent: int = 2) -> str:
|
||||
return json.dumps(self.to_dict(), indent=indent, sort_keys=False)
|
||||
@@ -0,0 +1,62 @@
|
||||
"""Diff dependency manifests to produce PackageDiff."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
|
||||
from scripts.manifest import PackageDiff, PackageEntry, PackageUpgrade
|
||||
|
||||
|
||||
_REQ_LINE_RE = re.compile(
|
||||
r"^\s*(?P<name>[A-Za-z0-9_\-.]+)\s*==\s*(?P<version>[^\s#]+)"
|
||||
)
|
||||
|
||||
|
||||
def _parse_requirements(text: str) -> dict[str, str]:
|
||||
out: dict[str, str] = {}
|
||||
for line in text.splitlines():
|
||||
line = line.split("#", 1)[0]
|
||||
m = _REQ_LINE_RE.match(line)
|
||||
if m:
|
||||
out[m.group("name").lower()] = m.group("version")
|
||||
return out
|
||||
|
||||
|
||||
def _diff_maps(before: dict[str, str], after: dict[str, str]) -> PackageDiff:
|
||||
added = [
|
||||
PackageEntry(name=n, version=v)
|
||||
for n, v in sorted(after.items()) if n not in before
|
||||
]
|
||||
removed = [
|
||||
PackageEntry(name=n, version=v)
|
||||
for n, v in sorted(before.items()) if n not in after
|
||||
]
|
||||
upgraded = [
|
||||
PackageUpgrade(name=n, from_version=before[n], to_version=after[n])
|
||||
for n in sorted(before.keys() & after.keys())
|
||||
if before[n] != after[n]
|
||||
]
|
||||
return PackageDiff(added=added, removed=removed, upgraded=upgraded)
|
||||
|
||||
|
||||
def diff_requirements_txt(before: str, after: str) -> PackageDiff:
|
||||
return _diff_maps(_parse_requirements(before), _parse_requirements(after))
|
||||
|
||||
|
||||
def _parse_package_json(text: str) -> dict[str, str]:
|
||||
try:
|
||||
doc = json.loads(text)
|
||||
except json.JSONDecodeError:
|
||||
return {}
|
||||
out: dict[str, str] = {}
|
||||
for key in ("dependencies", "devDependencies"):
|
||||
section = doc.get(key) or {}
|
||||
if isinstance(section, dict):
|
||||
for name, version in section.items():
|
||||
if isinstance(version, str):
|
||||
out[name.lower()] = version
|
||||
return out
|
||||
|
||||
|
||||
def diff_package_json(before: str, after: str) -> PackageDiff:
|
||||
return _diff_maps(_parse_package_json(before), _parse_package_json(after))
|
||||
@@ -0,0 +1,70 @@
|
||||
"""Aggregate runs.jsonl into precision + token-cost stats."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def compute_stats(log_path: Path) -> dict:
|
||||
if not log_path.exists():
|
||||
return {"by_agent": {}, "by_rule": {}, "runs": 0}
|
||||
|
||||
by_agent: dict[str, dict] = {}
|
||||
by_rule: dict[str, dict] = {}
|
||||
runs: set[str] = set()
|
||||
|
||||
for line in log_path.read_text(encoding="utf-8").splitlines():
|
||||
if not line.strip():
|
||||
continue
|
||||
rec = json.loads(line)
|
||||
agent = rec.get("agent", "?")
|
||||
|
||||
if rec["kind"] == "subagent_run":
|
||||
runs.add(rec["run_id"])
|
||||
a = by_agent.setdefault(agent, _empty_agent())
|
||||
a["tokens"] += rec.get("input_tokens", 0) + rec.get("output_tokens", 0)
|
||||
a["duration_ms"] += rec.get("duration_ms", 0)
|
||||
a["runs"] += 1
|
||||
|
||||
elif rec["kind"] == "verdict":
|
||||
verdict = rec["verdict"]
|
||||
a = by_agent.setdefault(agent, _empty_agent())
|
||||
a["total"] += 1
|
||||
a[verdict] = a.get(verdict, 0) + 1
|
||||
|
||||
rule_key = f"{agent}/{rec['rule_id']}"
|
||||
r = by_rule.setdefault(rule_key, _empty_rule())
|
||||
r["total"] += 1
|
||||
r[verdict] = r.get(verdict, 0) + 1
|
||||
|
||||
for a in by_agent.values():
|
||||
a["precision"] = a["kept"] / a["total"] if a["total"] else 0.0
|
||||
a["tokens_per_kept"] = a["tokens"] / a["kept"] if a["kept"] else float("inf")
|
||||
|
||||
for r in by_rule.values():
|
||||
r["precision"] = r["kept"] / r["total"] if r["total"] else 0.0
|
||||
|
||||
return {"by_agent": by_agent, "by_rule": by_rule, "runs": len(runs)}
|
||||
|
||||
|
||||
def _empty_agent() -> dict:
|
||||
return {
|
||||
"tokens": 0, "duration_ms": 0, "runs": 0,
|
||||
"total": 0, "kept": 0, "dismissed": 0, "false_positive": 0,
|
||||
}
|
||||
|
||||
|
||||
def _empty_rule() -> dict:
|
||||
return {"total": 0, "kept": 0, "dismissed": 0, "false_positive": 0}
|
||||
|
||||
|
||||
def main(argv: list[str]) -> int:
|
||||
log = Path.home() / ".claude/cache/audit-code/runs.jsonl"
|
||||
stats = compute_stats(log)
|
||||
print(json.dumps(stats, indent=2))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main(sys.argv[1:]))
|
||||
@@ -0,0 +1,29 @@
|
||||
rules:
|
||||
- id: lua-os-execute
|
||||
pattern-either:
|
||||
- pattern: os.execute($CMD)
|
||||
- pattern: io.popen($CMD)
|
||||
message: >-
|
||||
Shell command executed via os.execute/io.popen. If $CMD includes any
|
||||
externally-influenced data (arguments, env vars, network/file input),
|
||||
this is command injection. Verify the command is a fixed literal or
|
||||
properly escaped/allowlisted.
|
||||
languages: [lua]
|
||||
severity: WARNING
|
||||
metadata:
|
||||
cwe: ["CWE-78: Improper Neutralization of Special Elements used in an OS Command ('OS Command Injection')"]
|
||||
category: security
|
||||
|
||||
- id: lua-dynamic-code-load
|
||||
pattern-either:
|
||||
- pattern: load($CODE)
|
||||
- pattern: loadstring($CODE)
|
||||
message: >-
|
||||
Dynamic code loaded via load/loadstring. If $CODE is derived from
|
||||
external input, this allows arbitrary code execution. Verify the
|
||||
source is trusted and not attacker-influenced.
|
||||
languages: [lua]
|
||||
severity: WARNING
|
||||
metadata:
|
||||
cwe: ["CWE-94: Improper Control of Generation of Code ('Code Injection')"]
|
||||
category: security
|
||||
@@ -0,0 +1,70 @@
|
||||
"""Subprocess wrapper for external linters + parallel dispatch."""
|
||||
from __future__ import annotations
|
||||
|
||||
import concurrent.futures as cf
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
_DEFAULT_TIMEOUT = 180
|
||||
|
||||
_SKILL_VENV_BIN = Path(__file__).resolve().parent.parent / ".venv" / "bin"
|
||||
|
||||
|
||||
@dataclass
|
||||
class ToolResult:
|
||||
binary: str
|
||||
ran: bool
|
||||
exit_code: int = -1
|
||||
stdout: str = ""
|
||||
stderr: str = ""
|
||||
reason: str = ""
|
||||
|
||||
|
||||
def resolve_tool(binary: str) -> str | None:
|
||||
"""Resolve a tool name to an absolute path, preferring the skill's
|
||||
`.venv/bin/` over the user's PATH. Returns None if not found anywhere.
|
||||
"""
|
||||
local = _SKILL_VENV_BIN / binary
|
||||
if local.is_file() and os.access(local, os.X_OK):
|
||||
return str(local)
|
||||
return shutil.which(binary)
|
||||
|
||||
|
||||
def tool_available(binary: str) -> bool:
|
||||
return resolve_tool(binary) is not None
|
||||
|
||||
|
||||
def run_tool(args: list[str], cwd: str, timeout: int = _DEFAULT_TIMEOUT) -> ToolResult:
|
||||
binary = args[0]
|
||||
resolved = resolve_tool(binary)
|
||||
if resolved is None:
|
||||
return ToolResult(binary=binary, ran=False, reason="not on PATH")
|
||||
try:
|
||||
result = subprocess.run(
|
||||
[resolved, *args[1:]], cwd=cwd, capture_output=True, text=True,
|
||||
check=False, timeout=timeout,
|
||||
)
|
||||
except subprocess.TimeoutExpired:
|
||||
return ToolResult(binary=binary, ran=False, reason=f"timeout after {timeout}s")
|
||||
return ToolResult(
|
||||
binary=binary, ran=True,
|
||||
exit_code=result.returncode,
|
||||
stdout=result.stdout, stderr=result.stderr,
|
||||
)
|
||||
|
||||
|
||||
def run_tools_parallel(
|
||||
jobs: list[tuple[list[str], str]], max_workers: int = 8,
|
||||
) -> list[ToolResult]:
|
||||
"""Run a list of (args, cwd) jobs in parallel. Order preserved."""
|
||||
results: list[ToolResult] = [None] * len(jobs) # type: ignore[list-item]
|
||||
with cf.ThreadPoolExecutor(max_workers=max_workers) as ex:
|
||||
futures = {ex.submit(run_tool, a, c): i for i, (a, c) in enumerate(jobs)}
|
||||
for fut in cf.as_completed(futures):
|
||||
i = futures[fut]
|
||||
results[i] = fut.result()
|
||||
return results
|
||||
@@ -0,0 +1,111 @@
|
||||
"""Per-agent manifest slicing."""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
_TYPE_TOOLS = {"mypy", "tsc"}
|
||||
_DEP_TOOLS = {"pip-audit", "osv-scanner"}
|
||||
_SECRET_TOOLS = {"gitleaks"}
|
||||
_MAINTAINABILITY_TOOLS = {"vulture", "radon", "interrogate", "lizard", "knip", "jscpd", "selene"}
|
||||
_GHA_TOOLS = {"actionlint", "zizmor"}
|
||||
|
||||
_PSSA_SECURITY_RULES = {
|
||||
"PSAvoidUsingPlainTextForPassword",
|
||||
"PSAvoidUsingConvertToSecureStringWithPlainText",
|
||||
"PSAvoidUsingUsernameAndPasswordParams",
|
||||
"PSUsePSCredentialType",
|
||||
"PSAvoidUsingInvokeExpression",
|
||||
"PSAvoidUsingComputerNameHardcoded",
|
||||
"PSAvoidUsingBrokenHashAlgorithms",
|
||||
}
|
||||
|
||||
|
||||
def slice_for_agent(manifest: dict, agent: str) -> dict:
|
||||
"""Return a subset of the manifest scoped to a specific reviewer."""
|
||||
base = {
|
||||
"mode": manifest["mode"],
|
||||
"base_ref": manifest["base_ref"],
|
||||
"head_ref": manifest["head_ref"],
|
||||
"default_branch": manifest["default_branch"],
|
||||
"language_breakdown": manifest["language_breakdown"],
|
||||
"changed_files": list(manifest["changed_files"]),
|
||||
"tool_stats": dict(manifest["tool_stats"]),
|
||||
"tools_unavailable": list(manifest["tools_unavailable"]),
|
||||
"errors": list(manifest["errors"]),
|
||||
}
|
||||
findings = manifest["findings"]
|
||||
|
||||
if agent == "security-triage":
|
||||
kept: list[dict] = []
|
||||
for f in findings:
|
||||
tool = f["tool"]
|
||||
if tool in {"bandit", "ruff", "opengrep", "luac", "injectionhunter"}:
|
||||
kept.append(f)
|
||||
elif tool == "psscriptanalyzer" and f.get("rule_id", "") in _PSSA_SECURITY_RULES:
|
||||
kept.append(f)
|
||||
elif tool == "eslint" and "security" in f.get("rule_id", ""):
|
||||
kept.append(f)
|
||||
elif tool == "dotnet" and f.get("rule_id", "").startswith("SCS"):
|
||||
kept.append(f)
|
||||
return {**base, "findings": kept}
|
||||
|
||||
if agent == "type-safety":
|
||||
kept = []
|
||||
for f in findings:
|
||||
tool = f["tool"]
|
||||
if tool in _TYPE_TOOLS:
|
||||
kept.append(f)
|
||||
elif tool == "eslint" and "security" not in f.get("rule_id", ""):
|
||||
kept.append(f)
|
||||
elif tool == "dotnet" and not f.get("rule_id", "").startswith("SCS"):
|
||||
kept.append(f)
|
||||
return {**base, "findings": kept}
|
||||
|
||||
if agent == "dependency":
|
||||
return {
|
||||
**base,
|
||||
"findings": [f for f in findings if f["tool"] in _DEP_TOOLS],
|
||||
"package_diffs": dict(manifest["package_diffs"]),
|
||||
}
|
||||
|
||||
if agent == "secrets":
|
||||
return {**base, "findings": [f for f in findings if f["tool"] in _SECRET_TOOLS]}
|
||||
|
||||
if agent == "consistency":
|
||||
kept: list[dict] = []
|
||||
for f in findings:
|
||||
if f["tool"] != "ruff-idiom":
|
||||
continue
|
||||
rid = f.get("rule_id", "")
|
||||
if rid.startswith(("C901", "PLR0915")):
|
||||
continue
|
||||
kept.append(f)
|
||||
return {**base, "findings": kept}
|
||||
|
||||
if agent == "walkthrough":
|
||||
return {
|
||||
"mode": manifest["mode"],
|
||||
"base_ref": manifest["base_ref"],
|
||||
"head_ref": manifest["head_ref"],
|
||||
"default_branch": manifest["default_branch"],
|
||||
"language_breakdown": manifest["language_breakdown"],
|
||||
"changed_files": list(manifest["changed_files"]),
|
||||
"errors": list(manifest["errors"]),
|
||||
}
|
||||
|
||||
if agent == "maintainability":
|
||||
kept = []
|
||||
for f in findings:
|
||||
if f["tool"] in _MAINTAINABILITY_TOOLS:
|
||||
kept.append(f)
|
||||
elif f["tool"] == "luac":
|
||||
kept.append(f)
|
||||
elif f["tool"] == "psscriptanalyzer" and f.get("rule_id", "") not in _PSSA_SECURITY_RULES:
|
||||
kept.append(f)
|
||||
elif f["tool"] == "ruff-idiom" and f.get("rule_id", "").startswith(("C901", "PLR0915")):
|
||||
kept.append(f)
|
||||
return {**base, "findings": kept}
|
||||
|
||||
if agent == "gha-reviewer":
|
||||
return {**base, "findings": [f for f in findings if f["tool"] in _GHA_TOOLS]}
|
||||
|
||||
return manifest
|
||||
@@ -0,0 +1,53 @@
|
||||
"""Append-only JSONL telemetry for audit-code runs."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def _now() -> str:
|
||||
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
||||
|
||||
|
||||
def _append(log_path: Path, record: dict) -> None:
|
||||
log_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with log_path.open("a", encoding="utf-8") as f:
|
||||
f.write(json.dumps(record) + "\n")
|
||||
|
||||
|
||||
def append_subagent_run(
|
||||
log_path: Path, *, run_id: str, repo: str, mode: str, agent: str,
|
||||
model: str, input_tokens: int, output_tokens: int,
|
||||
duration_ms: int, finding_count: int,
|
||||
) -> None:
|
||||
_append(log_path, {
|
||||
"kind": "subagent_run", "ts": _now(),
|
||||
"run_id": run_id, "repo": repo, "mode": mode, "agent": agent,
|
||||
"model": model,
|
||||
"input_tokens": input_tokens, "output_tokens": output_tokens,
|
||||
"duration_ms": duration_ms, "finding_count": finding_count,
|
||||
})
|
||||
|
||||
|
||||
def append_verdict(
|
||||
log_path: Path, *, run_id: str, agent: str, rule_id: str,
|
||||
file: str, line: int, verdict: str, notes: str = "",
|
||||
) -> None:
|
||||
if verdict not in {"kept", "dismissed", "false_positive"}:
|
||||
raise ValueError(f"invalid verdict: {verdict!r}")
|
||||
_append(log_path, {
|
||||
"kind": "verdict", "ts": _now(),
|
||||
"run_id": run_id, "agent": agent, "rule_id": rule_id,
|
||||
"file": file, "line": line, "verdict": verdict, "notes": notes,
|
||||
})
|
||||
|
||||
|
||||
def read_runs(log_path: Path) -> list[dict]:
|
||||
if not log_path.exists():
|
||||
return []
|
||||
return [
|
||||
json.loads(line)
|
||||
for line in log_path.read_text(encoding="utf-8").splitlines()
|
||||
if line.strip()
|
||||
]
|
||||
Reference in New Issue
Block a user