Move the code and terraform audits into the reviews plugin
Copy the standalone code-review and terraform-review skills into
plugins/reviews as audit-code and audit-terraform. The rename separates the
automated, linter-driven audits from the guided review-pr walkthrough that
already lived here.
Resolve bundled script paths through ${SKILL_DIR}, exported in a new step 0.
CLAUDE_PLUGIN_ROOT is not set in the Bash tool environment, so the obvious
substitution would have expanded to nothing and broken every collection
script invocation.
Replace the PLAN and DESIGN docs with READMEs written from the current
SKILL.md and scripts. The old docs had drifted badly: they named semgrep
where the code calls opengrep, scoped five review agents where there are
now eight, and predated Lua, PowerShell, and GitHub Actions support.
Add CONSISTENCY_NORMS to the audit-terraform agent inputs. The collection
script writes consistency_norms.json and the agent prompt declares it, but
SKILL.md never listed it, leaving the variable unsubstituted.
Drop the --ingest-verdicts instruction from both skills. review_stats.py
parses no arguments, so the ref-mode verdict template it told users to feed
back could never be read.
Point audit-terraform's smoke test at README.md and resolve its fixture
paths relative to the test file rather than an absolute home directory.
Tests: 197 passing (audit-code), 106 passing (audit-terraform).
This commit is contained in:
@@ -0,0 +1,132 @@
|
||||
"""Reconcile plan-detected and diff-detected resources into a catalog.
|
||||
|
||||
Schema: one entry per (source_dir, local_address). Each entry carries a list
|
||||
of CatalogInstance (plan_dir, address_at_plan, action) so a module reviewed
|
||||
at N callsites yields one entry, not N.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
from scripts.manifest import CatalogEntry, CatalogInstance, ModuleGraphEntry
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PlanHit:
|
||||
address: str
|
||||
type: str
|
||||
action: str
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class DiffHit:
|
||||
source_dir: str
|
||||
local_address: str
|
||||
type: str
|
||||
|
||||
|
||||
def _strip_module_prefix(address: str) -> tuple[list[str], str]:
|
||||
parts = address.split(".")
|
||||
locals_: list[str] = []
|
||||
while len(parts) >= 2 and parts[0] == "module":
|
||||
locals_.append(parts[1])
|
||||
parts = parts[2:]
|
||||
return locals_, ".".join(parts)
|
||||
|
||||
|
||||
def _plan_source_dir(plan_dir: str, address: str,
|
||||
module_graph: dict[str, ModuleGraphEntry]) -> str:
|
||||
module_locals, _ = _strip_module_prefix(address)
|
||||
if not module_locals:
|
||||
return plan_dir
|
||||
current_dir = plan_dir
|
||||
for module_local in module_locals:
|
||||
next_dir = None
|
||||
has_explicit_mapping = False
|
||||
for entry in module_graph.values():
|
||||
local_names = entry.callsite_local_names.get(current_dir)
|
||||
if local_names is None:
|
||||
continue
|
||||
has_explicit_mapping = True
|
||||
if module_local in local_names:
|
||||
next_dir = local_names[module_local]
|
||||
break
|
||||
if next_dir is None:
|
||||
if has_explicit_mapping:
|
||||
return current_dir
|
||||
unique_candidates = sorted(
|
||||
mod_dir for mod_dir, entry in module_graph.items()
|
||||
if current_dir in entry.callsites
|
||||
)
|
||||
if len(unique_candidates) != 1:
|
||||
return current_dir
|
||||
next_dir = unique_candidates[0]
|
||||
current_dir = next_dir
|
||||
return current_dir
|
||||
|
||||
|
||||
def build_catalog(
|
||||
plan_hits: dict[str, list[PlanHit]],
|
||||
diff_hits: list[DiffHit],
|
||||
module_graph: dict[str, ModuleGraphEntry],
|
||||
) -> list[CatalogEntry]:
|
||||
by_key: dict[tuple[str, str], CatalogEntry] = {}
|
||||
|
||||
def _get(source_dir: str, local_addr: str, type_: str) -> CatalogEntry:
|
||||
k = (source_dir, local_addr)
|
||||
if k not in by_key:
|
||||
by_key[k] = CatalogEntry(
|
||||
source_dir=source_dir,
|
||||
local_address=local_addr,
|
||||
type=type_,
|
||||
source="plan",
|
||||
instances=[],
|
||||
)
|
||||
return by_key[k]
|
||||
|
||||
for plan_dir, hits in plan_hits.items():
|
||||
for h in hits:
|
||||
source_dir = _plan_source_dir(plan_dir, h.address, module_graph)
|
||||
_, local_addr = _strip_module_prefix(h.address)
|
||||
entry = _get(source_dir, local_addr, h.type)
|
||||
entry.instances.append(CatalogInstance(
|
||||
plan_dir=plan_dir, address_at_plan=h.address, action=h.action,
|
||||
))
|
||||
|
||||
for d in diff_hits:
|
||||
k = (d.source_dir, d.local_address)
|
||||
if k in by_key:
|
||||
existing = by_key[k]
|
||||
by_key[k] = CatalogEntry(
|
||||
source_dir=existing.source_dir,
|
||||
local_address=existing.local_address,
|
||||
type=existing.type,
|
||||
source="both",
|
||||
instances=list(existing.instances),
|
||||
)
|
||||
continue
|
||||
graph_entry = module_graph.get(d.source_dir)
|
||||
if graph_entry and graph_entry.callsites:
|
||||
instances = [
|
||||
CatalogInstance(
|
||||
plan_dir=cs,
|
||||
address_at_plan=f"module.<{d.source_dir}>.{d.local_address}",
|
||||
action="unknown",
|
||||
)
|
||||
for cs in graph_entry.callsites
|
||||
]
|
||||
else:
|
||||
instances = [CatalogInstance(
|
||||
plan_dir=d.source_dir,
|
||||
address_at_plan=d.local_address,
|
||||
action="unknown",
|
||||
)]
|
||||
by_key[k] = CatalogEntry(
|
||||
source_dir=d.source_dir,
|
||||
local_address=d.local_address,
|
||||
type=d.type,
|
||||
source="diff",
|
||||
instances=instances,
|
||||
)
|
||||
|
||||
return sorted(by_key.values(), key=lambda e: (e.source_dir, e.local_address))
|
||||
@@ -0,0 +1,309 @@
|
||||
"""collect-changes.py — produce a audit-terraform manifest.json."""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import concurrent.futures as cf
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
_HERE = Path(__file__).resolve().parent
|
||||
if str(_HERE.parent) not in sys.path:
|
||||
sys.path.insert(0, str(_HERE.parent))
|
||||
|
||||
from scripts.catalog import build_catalog, DiffHit, PlanHit
|
||||
from scripts.git_diff import (
|
||||
changed_files,
|
||||
changed_line_ranges,
|
||||
is_terragrunt_file,
|
||||
)
|
||||
from scripts.hcl_diff import touched_resources
|
||||
from scripts.manifest import Manifest, PlanResult, PlanUnit
|
||||
from scripts.module_graph import build_module_graph
|
||||
from scripts.scanners import _run_trivy_config, _run_tflint
|
||||
from scripts.slicing import _slice_for_agent
|
||||
from scripts.plan_output import parse_plan_resources
|
||||
from scripts.plan_runner import detect_tool, run_init, run_plan
|
||||
from scripts.reference_set import compute_consistency_norms, compute_reference_sets
|
||||
from scripts.resolve_plan_units import resolve_plan_units
|
||||
from scripts.source_lookup import find_block
|
||||
|
||||
|
||||
_PLAN_CONCURRENCY = 8
|
||||
|
||||
|
||||
def _resolve_default_branch(repo: Path) -> str:
|
||||
try:
|
||||
r = subprocess.run(
|
||||
["git", "-C", str(repo), "symbolic-ref", "refs/remotes/origin/HEAD"],
|
||||
capture_output=True, text=True, check=False,
|
||||
)
|
||||
if r.returncode == 0:
|
||||
return r.stdout.strip().rsplit("/", 1)[-1]
|
||||
except FileNotFoundError:
|
||||
pass
|
||||
for candidate in ("main", "master"):
|
||||
r = subprocess.run(
|
||||
["git", "-C", str(repo), "rev-parse", f"origin/{candidate}"],
|
||||
capture_output=True, text=True, check=False,
|
||||
)
|
||||
if r.returncode == 0:
|
||||
return candidate
|
||||
return "main"
|
||||
|
||||
|
||||
def _resolve_base_ref(repo: Path, branch: str) -> str:
|
||||
"""Return the diff base for `branch`, fetching origin first so the
|
||||
comparison is against the canonical remote tip rather than a stale local
|
||||
branch. Falls back to the local branch name if origin isn't reachable.
|
||||
"""
|
||||
try:
|
||||
subprocess.run(
|
||||
["git", "-C", str(repo), "fetch", "--quiet", "--no-tags",
|
||||
"origin", branch],
|
||||
capture_output=True, text=True, check=False, timeout=60,
|
||||
)
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired):
|
||||
pass
|
||||
check = subprocess.run(
|
||||
["git", "-C", str(repo), "rev-parse", "--verify", "--quiet",
|
||||
f"origin/{branch}"],
|
||||
capture_output=True, text=True, check=False,
|
||||
)
|
||||
return f"origin/{branch}" if check.returncode == 0 else branch
|
||||
|
||||
|
||||
def _safe_plan_filename(plan_dir: str) -> str:
|
||||
"""Map a plan_dir like '.' or 'live/prod/app' to a stable, filename-safe stem.
|
||||
|
||||
'.' becomes 'root' (otherwise we'd produce '..txt' on disk).
|
||||
"""
|
||||
cleaned = plan_dir.replace("/", "_").strip(".")
|
||||
return cleaned or "root"
|
||||
|
||||
|
||||
def _run_one(
|
||||
repo: Path,
|
||||
plan_dir: str,
|
||||
triggered_by: list[str],
|
||||
changed_files_for_unit: list[str],
|
||||
out_dir: Path,
|
||||
) -> tuple[PlanUnit, list[PlanHit], str | None]:
|
||||
full = repo / plan_dir
|
||||
tool = detect_tool(full)
|
||||
terragrunt_changed = any(
|
||||
is_terragrunt_file(changed_file)
|
||||
for changed_file in changed_files_for_unit
|
||||
)
|
||||
init_res = run_init(full, tool)
|
||||
if not init_res.ok:
|
||||
plan_res = PlanResult(ok=False, stdout_path="", exit_code=-1,
|
||||
summary="init-failed")
|
||||
err = f"init failed in {plan_dir}: {init_res.stderr_tail.strip()}"
|
||||
return (
|
||||
PlanUnit(plan_dir=plan_dir, tool=tool, init=init_res,
|
||||
plan=plan_res, triggered_by=triggered_by,
|
||||
terragrunt_changed=terragrunt_changed,
|
||||
changed_files=changed_files_for_unit),
|
||||
[], err,
|
||||
)
|
||||
|
||||
stdout_path = out_dir / "plans" / f"{_safe_plan_filename(plan_dir)}.txt"
|
||||
plan_res = run_plan(full, tool, stdout_path)
|
||||
if not plan_res.ok:
|
||||
captured = Path(plan_res.stdout_path).read_text()[-1000:].strip()
|
||||
err = f"plan failed in {plan_dir} (exit {plan_res.exit_code}): {captured}"
|
||||
return (
|
||||
PlanUnit(plan_dir=plan_dir, tool=tool, init=init_res,
|
||||
plan=plan_res, triggered_by=triggered_by,
|
||||
terragrunt_changed=terragrunt_changed,
|
||||
changed_files=changed_files_for_unit),
|
||||
[], err,
|
||||
)
|
||||
|
||||
parsed = parse_plan_resources(Path(plan_res.stdout_path).read_text())
|
||||
hits = [
|
||||
PlanHit(address=p.address, type=p.type, action=p.action) for p in parsed
|
||||
]
|
||||
return (
|
||||
PlanUnit(plan_dir=plan_dir, tool=tool, init=init_res,
|
||||
plan=plan_res, triggered_by=triggered_by,
|
||||
terragrunt_changed=terragrunt_changed,
|
||||
changed_files=changed_files_for_unit),
|
||||
hits, None,
|
||||
)
|
||||
|
||||
|
||||
def _changed_files_for_plan_unit(
|
||||
all_changed_files: list[str],
|
||||
triggered_by: list[str],
|
||||
) -> list[str]:
|
||||
triggered_dirs = set(triggered_by)
|
||||
return sorted(
|
||||
changed_file
|
||||
for changed_file in all_changed_files
|
||||
if str(Path(changed_file).parent.as_posix()) in triggered_dirs
|
||||
)
|
||||
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--repo", required=True)
|
||||
parser.add_argument("--base", default=None)
|
||||
parser.add_argument("--head", default="HEAD")
|
||||
parser.add_argument("--output-dir", required=True)
|
||||
parser.add_argument("--mode", choices=("local", "ref"), required=True)
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
repo = Path(args.repo).resolve()
|
||||
out_dir = Path(args.output_dir).resolve()
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
default_branch = _resolve_default_branch(repo)
|
||||
base = args.base or _resolve_base_ref(repo, default_branch)
|
||||
|
||||
errors: list[str] = []
|
||||
|
||||
try:
|
||||
files = changed_files(str(repo), base, args.head)
|
||||
dirs = {str(Path(path).parent.as_posix()) or "." for path in files}
|
||||
terragrunt_files = {path for path in files if is_terragrunt_file(path)}
|
||||
except RuntimeError as e:
|
||||
manifest = Manifest(
|
||||
base_ref=base, head_ref=args.head, mode=args.mode,
|
||||
default_branch=default_branch,
|
||||
changed_source_dirs=[], plan_units=[], catalog=[],
|
||||
trivy_findings=[],
|
||||
module_graph={}, errors=[str(e)],
|
||||
)
|
||||
(out_dir / "manifest.json").write_text(manifest.to_json())
|
||||
return 1
|
||||
|
||||
if not files:
|
||||
manifest = Manifest(
|
||||
base_ref=base, head_ref=args.head, mode=args.mode,
|
||||
default_branch=default_branch,
|
||||
changed_source_dirs=[], plan_units=[], catalog=[],
|
||||
trivy_findings=[],
|
||||
module_graph={}, errors=["no terraform/hcl files changed"],
|
||||
)
|
||||
(out_dir / "manifest.json").write_text(manifest.to_json())
|
||||
return 0
|
||||
|
||||
module_graph = build_module_graph(repo)
|
||||
plan_units_map, orphan_modules = resolve_plan_units(repo, dirs, module_graph)
|
||||
for orphan in orphan_modules:
|
||||
errors.append(f"module has no callsites; diff-only review: {orphan}")
|
||||
|
||||
plan_hits: dict[str, list[PlanHit]] = {}
|
||||
plan_unit_records: list[PlanUnit] = []
|
||||
|
||||
if plan_units_map:
|
||||
with cf.ThreadPoolExecutor(max_workers=_PLAN_CONCURRENCY) as ex:
|
||||
futures = {
|
||||
ex.submit(
|
||||
_run_one,
|
||||
repo,
|
||||
pd,
|
||||
tb,
|
||||
_changed_files_for_plan_unit(files, tb),
|
||||
out_dir,
|
||||
): pd
|
||||
for pd, tb in plan_units_map.items()
|
||||
}
|
||||
for fut in cf.as_completed(futures):
|
||||
record, hits, err = fut.result()
|
||||
plan_unit_records.append(record)
|
||||
if err:
|
||||
errors.append(err)
|
||||
else:
|
||||
plan_hits[record.plan_dir] = hits
|
||||
|
||||
diff_hits: list[DiffHit] = []
|
||||
for f in files:
|
||||
if not f.endswith(".tf"):
|
||||
continue
|
||||
try:
|
||||
ranges = changed_line_ranges(str(repo), base, args.head, f)
|
||||
except RuntimeError as e:
|
||||
errors.append(f"diff scan failed for {f}: {e}")
|
||||
continue
|
||||
if not ranges:
|
||||
continue
|
||||
blocks = touched_resources(repo / f, ranges)
|
||||
src_dir = str(Path(f).parent.as_posix()) or "."
|
||||
for b in blocks:
|
||||
diff_hits.append(DiffHit(
|
||||
source_dir=src_dir,
|
||||
local_address=f"{b.type}.{b.name}",
|
||||
type=b.type,
|
||||
))
|
||||
|
||||
if terragrunt_files:
|
||||
changed_plan_dirs = {pu.plan_dir for pu in plan_unit_records}
|
||||
for terragrunt_file in sorted(terragrunt_files):
|
||||
src_dir = str(Path(terragrunt_file).parent.as_posix()) or "."
|
||||
if src_dir not in changed_plan_dirs:
|
||||
errors.append(
|
||||
f"terragrunt change has no planned unit context: {terragrunt_file}"
|
||||
)
|
||||
|
||||
trivy_payload, trivy_findings, trivy_error = _run_trivy_config(repo, files)
|
||||
(out_dir / "trivy-findings.json").write_text(json.dumps(trivy_payload, indent=2))
|
||||
if trivy_error:
|
||||
errors.append(trivy_error)
|
||||
|
||||
tflint_payload, tflint_findings, tflint_error = _run_tflint(repo, files)
|
||||
(out_dir / "tflint-findings.json").write_text(json.dumps(tflint_payload, indent=2))
|
||||
if tflint_error:
|
||||
errors.append(tflint_error)
|
||||
|
||||
catalog = build_catalog(plan_hits, diff_hits, module_graph)
|
||||
for entry in catalog:
|
||||
loc = find_block(repo / entry.source_dir, entry.local_address)
|
||||
if loc is None:
|
||||
continue
|
||||
entry.block_header = loc.header
|
||||
entry.evidence_line = loc.evidence_line
|
||||
entry.key_attributes = loc.key_attributes
|
||||
entry.review_context = loc.review_context
|
||||
entry.block_file = loc.file.name
|
||||
entry.block_start = loc.start_line
|
||||
entry.block_end = loc.end_line
|
||||
ref_sets = compute_reference_sets(repo, dirs, module_graph)
|
||||
(out_dir / "reference_sets.json").write_text(
|
||||
json.dumps(ref_sets, indent=2)
|
||||
)
|
||||
consistency_norms = compute_consistency_norms(repo, ref_sets)
|
||||
(out_dir / "consistency_norms.json").write_text(
|
||||
json.dumps(consistency_norms, indent=2)
|
||||
)
|
||||
|
||||
manifest = Manifest(
|
||||
base_ref=base,
|
||||
head_ref=args.head,
|
||||
mode=args.mode,
|
||||
default_branch=default_branch,
|
||||
changed_source_dirs=sorted(dirs),
|
||||
plan_units=sorted(plan_unit_records, key=lambda p: p.plan_dir),
|
||||
catalog=catalog,
|
||||
trivy_findings=trivy_findings,
|
||||
tflint_findings=tflint_findings,
|
||||
module_graph=module_graph,
|
||||
errors=errors,
|
||||
)
|
||||
(out_dir / "manifest.json").write_text(manifest.to_json())
|
||||
manifest_dict = manifest.to_dict()
|
||||
for agent in ("fsbp", "cis", "aws-bp", "consistency", "tf-hygiene", "walkthrough"):
|
||||
sliced = _slice_for_agent(manifest_dict, agent)
|
||||
(out_dir / f"manifest-{agent}.json").write_text(
|
||||
json.dumps(sliced, indent=2)
|
||||
)
|
||||
|
||||
return 1 if any(not pu.plan.ok for pu in plan_unit_records) else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,147 @@
|
||||
"""Map Security Hub control-ID prefixes to terraform resource types.
|
||||
|
||||
This table is hand-curated and conservative — when a control could apply to
|
||||
multiple terraform resource types, list them all. Update as new control
|
||||
families ship.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
PREFIX_TO_TERRAFORM_TYPES: dict[str, list[str]] = {
|
||||
"Account": ["aws_account_alternate_contact"],
|
||||
"ACM": ["aws_acm_certificate"],
|
||||
"APIGateway": ["aws_api_gateway_rest_api", "aws_apigatewayv2_api",
|
||||
"aws_api_gateway_stage", "aws_api_gateway_method"],
|
||||
"AppSync": ["aws_appsync_graphql_api"],
|
||||
"Athena": ["aws_athena_workgroup"],
|
||||
"AutoScaling": ["aws_autoscaling_group", "aws_launch_template",
|
||||
"aws_launch_configuration"],
|
||||
"Autoscaling": ["aws_autoscaling_group", "aws_launch_template",
|
||||
"aws_launch_configuration"],
|
||||
"Backup": ["aws_backup_plan", "aws_backup_vault"],
|
||||
"BedrockAgentCore": [],
|
||||
"CloudFormation": ["aws_cloudformation_stack",
|
||||
"aws_cloudformation_stack_set"],
|
||||
"CloudFront": ["aws_cloudfront_distribution"],
|
||||
"CloudTrail": ["aws_cloudtrail"],
|
||||
"CodeBuild": ["aws_codebuild_project"],
|
||||
"Cognito": ["aws_cognito_user_pool",
|
||||
"aws_cognito_identity_pool"],
|
||||
"Config": ["aws_config_configuration_recorder",
|
||||
"aws_config_delivery_channel",
|
||||
"aws_config_config_rule"],
|
||||
"Connect": ["aws_connect_instance"],
|
||||
"DataFirehose": ["aws_kinesis_firehose_delivery_stream"],
|
||||
"DataSync": ["aws_datasync_task"],
|
||||
"DMS": ["aws_dms_endpoint",
|
||||
"aws_dms_replication_instance"],
|
||||
"DocumentDB": ["aws_docdb_cluster", "aws_docdb_cluster_instance"],
|
||||
"DynamoDB": ["aws_dynamodb_table"],
|
||||
"EC2": ["aws_instance", "aws_security_group",
|
||||
"aws_security_group_rule", "aws_vpc", "aws_subnet",
|
||||
"aws_network_acl", "aws_ebs_volume",
|
||||
"aws_ebs_default_kms_key", "aws_eip", "aws_nat_gateway",
|
||||
"aws_internet_gateway", "aws_route_table",
|
||||
"aws_vpc_endpoint", "aws_flow_log",
|
||||
"aws_default_security_group"],
|
||||
"ECR": ["aws_ecr_repository",
|
||||
"aws_ecr_repository_policy"],
|
||||
"ECS": ["aws_ecs_cluster", "aws_ecs_service",
|
||||
"aws_ecs_task_definition"],
|
||||
"EFS": ["aws_efs_file_system",
|
||||
"aws_efs_file_system_policy",
|
||||
"aws_efs_access_point"],
|
||||
"EKS": ["aws_eks_cluster", "aws_eks_node_group"],
|
||||
"ELB": ["aws_lb", "aws_alb", "aws_elb",
|
||||
"aws_lb_listener", "aws_alb_listener",
|
||||
"aws_lb_target_group"],
|
||||
"ElastiCache": ["aws_elasticache_cluster",
|
||||
"aws_elasticache_replication_group"],
|
||||
"ElasticBeanstalk": ["aws_elastic_beanstalk_environment"],
|
||||
"ElasticSearch": ["aws_elasticsearch_domain",
|
||||
"aws_opensearch_domain"],
|
||||
"EMR": ["aws_emr_cluster"],
|
||||
"ES": ["aws_elasticsearch_domain"],
|
||||
"EventBridge": ["aws_cloudwatch_event_rule",
|
||||
"aws_cloudwatch_event_bus"],
|
||||
"FSx": ["aws_fsx_lustre_file_system",
|
||||
"aws_fsx_windows_file_system",
|
||||
"aws_fsx_openzfs_file_system",
|
||||
"aws_fsx_ontap_file_system"],
|
||||
"Glue": ["aws_glue_catalog_database",
|
||||
"aws_glue_crawler", "aws_glue_job"],
|
||||
"GuardDuty": ["aws_guardduty_detector"],
|
||||
"IAM": ["aws_iam_user", "aws_iam_role", "aws_iam_policy",
|
||||
"aws_iam_user_policy", "aws_iam_role_policy",
|
||||
"aws_iam_access_key",
|
||||
"aws_iam_account_password_policy",
|
||||
"aws_iam_group", "aws_iam_user_policy_attachment",
|
||||
"aws_iam_role_policy_attachment"],
|
||||
"Inspector": ["aws_inspector2_enabler"],
|
||||
"Kinesis": ["aws_kinesis_stream"],
|
||||
"KMS": ["aws_kms_key", "aws_kms_alias"],
|
||||
"Lambda": ["aws_lambda_function",
|
||||
"aws_lambda_permission",
|
||||
"aws_lambda_function_url"],
|
||||
"Macie": ["aws_macie2_account"],
|
||||
"MQ": ["aws_mq_broker"],
|
||||
"MSK": ["aws_msk_cluster"],
|
||||
"Neptune": ["aws_neptune_cluster",
|
||||
"aws_neptune_cluster_instance"],
|
||||
"NetworkFirewall": ["aws_networkfirewall_firewall",
|
||||
"aws_networkfirewall_firewall_policy",
|
||||
"aws_networkfirewall_rule_group"],
|
||||
"Opensearch": ["aws_opensearch_domain"],
|
||||
"PCA": ["aws_acmpca_certificate_authority"],
|
||||
"RDS": ["aws_db_instance", "aws_rds_cluster",
|
||||
"aws_db_subnet_group", "aws_db_parameter_group",
|
||||
"aws_rds_cluster_instance",
|
||||
"aws_db_snapshot",
|
||||
"aws_db_event_subscription"],
|
||||
"Redshift": ["aws_redshift_cluster",
|
||||
"aws_redshift_parameter_group"],
|
||||
"RedshiftServerless": ["aws_redshiftserverless_namespace",
|
||||
"aws_redshiftserverless_workgroup"],
|
||||
"Route53": ["aws_route53_zone"],
|
||||
"S3": ["aws_s3_bucket", "aws_s3_bucket_policy",
|
||||
"aws_s3_bucket_public_access_block",
|
||||
"aws_s3_bucket_versioning",
|
||||
"aws_s3_bucket_server_side_encryption_configuration",
|
||||
"aws_s3_bucket_logging",
|
||||
"aws_s3_bucket_lifecycle_configuration",
|
||||
"aws_s3_bucket_acl"],
|
||||
"SageMaker": ["aws_sagemaker_notebook_instance",
|
||||
"aws_sagemaker_endpoint_configuration",
|
||||
"aws_sagemaker_model"],
|
||||
"SecretsManager": ["aws_secretsmanager_secret",
|
||||
"aws_secretsmanager_secret_rotation"],
|
||||
"ServiceCatalog": ["aws_servicecatalog_portfolio"],
|
||||
"SES": ["aws_ses_configuration_set",
|
||||
"aws_ses_domain_identity"],
|
||||
"SNS": ["aws_sns_topic", "aws_sns_topic_policy"],
|
||||
"SQS": ["aws_sqs_queue"],
|
||||
"SSM": ["aws_ssm_document",
|
||||
"aws_ssm_parameter",
|
||||
"aws_ssm_association"],
|
||||
"StepFunctions": ["aws_sfn_state_machine"],
|
||||
"Transfer": ["aws_transfer_server", "aws_transfer_user"],
|
||||
"WAF": ["aws_wafv2_web_acl", "aws_waf_web_acl",
|
||||
"aws_wafv2_web_acl_association"],
|
||||
"WorkSpaces": ["aws_workspaces_directory", "aws_workspaces_workspace"],
|
||||
}
|
||||
|
||||
|
||||
def control_id_to_types(control_id: str) -> list[str]:
|
||||
"""Given e.g. 'FSBP S3.5' or 'S3.5', return likely terraform types.
|
||||
|
||||
Strips the leading 'FSBP '/'CIS ' source prefix, then matches the
|
||||
service prefix (the segment before the first dot).
|
||||
"""
|
||||
s = control_id.strip()
|
||||
for source in ("FSBP", "CIS"):
|
||||
prefix = f"{source} "
|
||||
if s.startswith(prefix):
|
||||
s = s[len(prefix):]
|
||||
break
|
||||
service, _, _ = s.partition(".")
|
||||
return list(PREFIX_TO_TERRAFORM_TYPES.get(service, []))
|
||||
@@ -0,0 +1,56 @@
|
||||
"""Schema and serialization for AWS controls cache files."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass, asdict
|
||||
from datetime import datetime
|
||||
from typing import Literal
|
||||
|
||||
|
||||
Severity = Literal["critical", "high", "medium", "low", "informational"]
|
||||
ControlSource = Literal["fsbp", "cis", "aws-bp"]
|
||||
|
||||
|
||||
@dataclass
|
||||
class Control:
|
||||
control_id: str
|
||||
title: str
|
||||
severity: Severity
|
||||
resource_types: list[str]
|
||||
requirement: str
|
||||
source_url: str
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
@dataclass
|
||||
class ControlsFile:
|
||||
source: ControlSource
|
||||
fetched_at: datetime
|
||||
controls: list[Control]
|
||||
|
||||
def _by_resource_type(self) -> dict[str, list[str]]:
|
||||
"""Map terraform resource type -> list of control_ids.
|
||||
|
||||
Full control records live in `controls`. Looking up details by ID
|
||||
from `controls` keeps the index small.
|
||||
"""
|
||||
index: dict[str, list[str]] = {}
|
||||
for c in self.controls:
|
||||
for rt in c.resource_types:
|
||||
index.setdefault(rt, []).append(c.control_id)
|
||||
for rt in index:
|
||||
index[rt].sort()
|
||||
return dict(sorted(index.items()))
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {
|
||||
"source": self.source,
|
||||
"fetched_at": self.fetched_at.isoformat(),
|
||||
"controls": [c.to_dict() for c in self.controls],
|
||||
"by_resource_type": self._by_resource_type(),
|
||||
}
|
||||
|
||||
def to_json(self, indent: int = 2) -> str:
|
||||
return json.dumps(self.to_dict(), indent=indent, sort_keys=False)
|
||||
@@ -0,0 +1,97 @@
|
||||
"""Git diff scanning: changed files, dirs, and per-file added-line ranges."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import subprocess
|
||||
from pathlib import PurePosixPath
|
||||
|
||||
|
||||
_HCL_SUFFIXES = (".tf", ".tf.json", ".hcl")
|
||||
_HUNK_RE = re.compile(r"^@@ -\d+(?:,\d+)? \+(\d+)(?:,(\d+))? @@")
|
||||
|
||||
|
||||
def _run_git_diff(repo: str, base: str, head: str) -> str:
|
||||
result = subprocess.run(
|
||||
[
|
||||
"git", "-C", repo,
|
||||
"-c", "diff.noprefix=false",
|
||||
"-c", "color.ui=never",
|
||||
"diff", "--no-ext-diff", "--unified=0", f"{base}...{head}"
|
||||
],
|
||||
capture_output=True, text=True, check=False,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
raise RuntimeError(f"git diff failed: {result.stderr.strip()}")
|
||||
return result.stdout
|
||||
|
||||
|
||||
def _iter_file_blocks(diff_text: str):
|
||||
"""Yield (path, block_text) for each file section in a unified diff."""
|
||||
current_path: str | None = None
|
||||
buf: list[str] = []
|
||||
for line in diff_text.splitlines():
|
||||
if line.startswith("diff --git "):
|
||||
if current_path is not None:
|
||||
yield current_path, "\n".join(buf)
|
||||
current_path = None
|
||||
buf = []
|
||||
elif line.startswith("+++ b/"):
|
||||
current_path = line[len("+++ b/"):]
|
||||
if current_path is not None:
|
||||
buf.append(line)
|
||||
if current_path is not None:
|
||||
yield current_path, "\n".join(buf)
|
||||
|
||||
|
||||
def changed_files(repo: str, base: str, head: str) -> list[str]:
|
||||
out: list[str] = []
|
||||
for path, _ in _iter_file_blocks(_run_git_diff(repo, base, head)):
|
||||
if path.endswith(_HCL_SUFFIXES):
|
||||
out.append(path)
|
||||
return out
|
||||
|
||||
|
||||
def is_terragrunt_file(path: str) -> bool:
|
||||
return PurePosixPath(path).name == "terragrunt.hcl"
|
||||
|
||||
|
||||
def changed_dirs(repo: str, base: str, head: str) -> set[str]:
|
||||
return {str(PurePosixPath(p).parent) for p in changed_files(repo, base, head)}
|
||||
|
||||
|
||||
def changed_line_ranges(
|
||||
repo: str, base: str, head: str, path: str,
|
||||
) -> list[tuple[int, int]]:
|
||||
"""Return inclusive (start, end) line ranges of *added* lines in `path`."""
|
||||
ranges: list[tuple[int, int]] = []
|
||||
for fpath, block in _iter_file_blocks(_run_git_diff(repo, base, head)):
|
||||
if fpath != path:
|
||||
continue
|
||||
cur: int | None = None
|
||||
run_start: int | None = None
|
||||
run_end: int | None = None
|
||||
for line in block.splitlines():
|
||||
m = _HUNK_RE.match(line)
|
||||
if m:
|
||||
if run_start is not None:
|
||||
ranges.append((run_start, run_end)) # type: ignore[arg-type]
|
||||
run_start = run_end = None
|
||||
cur = int(m.group(1))
|
||||
continue
|
||||
if cur is None:
|
||||
continue
|
||||
if line.startswith("+") and not line.startswith("+++"):
|
||||
if run_start is None:
|
||||
run_start = cur
|
||||
run_end = cur
|
||||
cur += 1
|
||||
elif line.startswith("-") and not line.startswith("---"):
|
||||
continue
|
||||
else:
|
||||
if run_start is not None:
|
||||
ranges.append((run_start, run_end)) # type: ignore[arg-type]
|
||||
run_start = run_end = None
|
||||
cur += 1
|
||||
if run_start is not None:
|
||||
ranges.append((run_start, run_end)) # type: ignore[arg-type]
|
||||
return ranges
|
||||
@@ -0,0 +1,64 @@
|
||||
"""Locate `resource` blocks in .tf files and intersect with changed lines."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
_HEADER_RE = re.compile(
|
||||
r'^\s*resource\s+"(?P<type>[^"]+)"\s+"(?P<name>[^"]+)"\s*\{'
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ResourceBlock:
|
||||
type: str
|
||||
name: str
|
||||
start: int
|
||||
end: int
|
||||
|
||||
|
||||
def find_resource_blocks(path: Path) -> list[ResourceBlock]:
|
||||
"""Locate top-level `resource` blocks by tracking brace depth."""
|
||||
text = path.read_text()
|
||||
lines = text.splitlines()
|
||||
blocks: list[ResourceBlock] = []
|
||||
i = 0
|
||||
while i < len(lines):
|
||||
line = lines[i]
|
||||
m = _HEADER_RE.match(line)
|
||||
if not m:
|
||||
i += 1
|
||||
continue
|
||||
start = i + 1
|
||||
depth = line.count("{") - line.count("}")
|
||||
j = i
|
||||
while depth > 0 and j + 1 < len(lines):
|
||||
j += 1
|
||||
depth += lines[j].count("{") - lines[j].count("}")
|
||||
blocks.append(ResourceBlock(
|
||||
type=m.group("type"),
|
||||
name=m.group("name"),
|
||||
start=start,
|
||||
end=j + 1,
|
||||
))
|
||||
i = j + 1
|
||||
return blocks
|
||||
|
||||
|
||||
def _intersects(a: tuple[int, int], b: tuple[int, int]) -> bool:
|
||||
return not (a[1] < b[0] or b[1] < a[0])
|
||||
|
||||
|
||||
def touched_resources(
|
||||
path: Path, added_ranges: list[tuple[int, int]],
|
||||
) -> list[ResourceBlock]:
|
||||
if not added_ranges:
|
||||
return []
|
||||
blocks = find_resource_blocks(path)
|
||||
out: list[ResourceBlock] = []
|
||||
for blk in blocks:
|
||||
if any(_intersects((blk.start, blk.end), r) for r in added_ranges):
|
||||
out.append(blk)
|
||||
return out
|
||||
@@ -0,0 +1,84 @@
|
||||
"""Append one subagent_run row per agent after the fan-out completes.
|
||||
|
||||
The orchestrator calls this once after all subagents return. It scans
|
||||
<output-dir>/findings-<agent>.json for each agent listed in the
|
||||
--usage-json payload, counts findings, and writes a subagent_run row to
|
||||
runs.jsonl using token / duration metadata supplied by the orchestrator.
|
||||
|
||||
Usage:
|
||||
|
||||
python scripts/log-run.py \\
|
||||
--output-dir <OUTPUT> --run-id <hex> --repo <path> --mode <local|ref> \\
|
||||
--usage-json - <<JSON
|
||||
{
|
||||
"walkthrough-reviewer": {"model":"sonnet","input_tokens":1234,"output_tokens":567,"duration_ms":4500},
|
||||
"aws-bp-reviewer": {"model":"sonnet","input_tokens":2345,"output_tokens":678,"duration_ms":5200}
|
||||
}
|
||||
JSON
|
||||
|
||||
`--log-path` defaults to ~/.claude/cache/audit-terraform/runs.jsonl.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from scripts.telemetry import append_subagent_run
|
||||
|
||||
_DEFAULT_LOG = Path.home() / ".claude/cache/audit-terraform/runs.jsonl"
|
||||
|
||||
|
||||
def _count_findings(path: Path) -> int:
|
||||
if not path.exists():
|
||||
return 0
|
||||
try:
|
||||
data = json.loads(path.read_text(encoding="utf-8"))
|
||||
except json.JSONDecodeError:
|
||||
return 0
|
||||
if isinstance(data, dict):
|
||||
findings = data.get("findings")
|
||||
if isinstance(findings, list):
|
||||
return len(findings)
|
||||
elif isinstance(data, list):
|
||||
return len(data)
|
||||
return 0
|
||||
|
||||
|
||||
def main(argv: list[str]) -> int:
|
||||
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||
p.add_argument("--output-dir", required=True, type=Path)
|
||||
p.add_argument("--run-id", required=True)
|
||||
p.add_argument("--repo", required=True)
|
||||
p.add_argument("--mode", required=True, choices=["local", "ref"])
|
||||
p.add_argument("--log-path", type=Path, default=_DEFAULT_LOG)
|
||||
p.add_argument("--usage-json", required=True,
|
||||
help="Path to JSON file, or '-' for stdin.")
|
||||
args = p.parse_args(argv)
|
||||
|
||||
raw = sys.stdin.read() if args.usage_json == "-" else Path(args.usage_json).read_text(encoding="utf-8")
|
||||
usage = json.loads(raw)
|
||||
if not isinstance(usage, dict):
|
||||
print("usage-json must be a JSON object keyed by agent name", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
for agent, meta in usage.items():
|
||||
if not isinstance(meta, dict):
|
||||
print(f"skipping {agent}: usage entry not an object", file=sys.stderr)
|
||||
continue
|
||||
findings_path = args.output_dir / f"findings-{agent}.json"
|
||||
append_subagent_run(
|
||||
args.log_path,
|
||||
run_id=args.run_id, repo=args.repo, mode=args.mode, agent=agent,
|
||||
model=str(meta.get("model", "?")),
|
||||
input_tokens=int(meta.get("input_tokens", 0)),
|
||||
output_tokens=int(meta.get("output_tokens", 0)),
|
||||
duration_ms=int(meta.get("duration_ms", 0)),
|
||||
finding_count=_count_findings(findings_path),
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main(sys.argv[1:]))
|
||||
@@ -0,0 +1,182 @@
|
||||
"""Manifest dataclasses for collect-changes.py output.
|
||||
|
||||
The manifest is the contract between the script and the review subagents.
|
||||
Schema mirrors DESIGN.md.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass, field, asdict
|
||||
from typing import Literal
|
||||
|
||||
|
||||
Mode = Literal["local", "ref"]
|
||||
Tool = Literal["terragrunt", "tofu"]
|
||||
Action = Literal["create", "update", "delete", "replace", "read", "no-op", "unknown"]
|
||||
Source = Literal["plan", "diff", "both"]
|
||||
|
||||
|
||||
@dataclass
|
||||
class InitResult:
|
||||
ok: bool
|
||||
stdout_tail: str
|
||||
stderr_tail: str
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
@dataclass
|
||||
class PlanResult:
|
||||
ok: bool
|
||||
stdout_path: str
|
||||
exit_code: int
|
||||
summary: str
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
@dataclass
|
||||
class PlanUnit:
|
||||
plan_dir: str
|
||||
tool: Tool
|
||||
init: InitResult
|
||||
plan: PlanResult
|
||||
triggered_by: list[str]
|
||||
terragrunt_changed: bool = False
|
||||
changed_files: list[str] = field(default_factory=list)
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {
|
||||
"plan_dir": self.plan_dir,
|
||||
"tool": self.tool,
|
||||
"init": self.init.to_dict(),
|
||||
"plan": self.plan.to_dict(),
|
||||
"triggered_by": list(self.triggered_by),
|
||||
"terragrunt_changed": self.terragrunt_changed,
|
||||
"changed_files": list(self.changed_files),
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class TrivyFinding:
|
||||
check_id: str
|
||||
title: str
|
||||
severity: str
|
||||
message: str
|
||||
file: str
|
||||
start_line: int = 0
|
||||
end_line: int = 0
|
||||
resource_type: str = ""
|
||||
source: str = "trivy"
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
@dataclass
|
||||
class TflintFinding:
|
||||
rule: str
|
||||
severity: str
|
||||
message: str
|
||||
file: str
|
||||
start_line: int = 0
|
||||
end_line: int = 0
|
||||
link: str = ""
|
||||
source: str = "tflint"
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
@dataclass
|
||||
class CatalogInstance:
|
||||
plan_dir: str
|
||||
address_at_plan: str
|
||||
action: Action
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
@dataclass
|
||||
class CatalogEntry:
|
||||
source_dir: str
|
||||
local_address: str
|
||||
type: str
|
||||
source: Source
|
||||
instances: list[CatalogInstance] = field(default_factory=list)
|
||||
block_header: str = ""
|
||||
evidence_line: str = ""
|
||||
key_attributes: dict[str, str | bool | int | list[str]] = field(default_factory=dict)
|
||||
review_context: dict[str, object] = field(default_factory=dict)
|
||||
block_file: str = ""
|
||||
block_start: int = 0
|
||||
block_end: int = 0
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {
|
||||
"source_dir": self.source_dir,
|
||||
"local_address": self.local_address,
|
||||
"type": self.type,
|
||||
"source": self.source,
|
||||
"instances": [i.to_dict() for i in self.instances],
|
||||
"block_header": self.block_header,
|
||||
"evidence_line": self.evidence_line,
|
||||
"key_attributes": dict(self.key_attributes),
|
||||
"review_context": dict(self.review_context),
|
||||
"block_file": self.block_file,
|
||||
"block_start": self.block_start,
|
||||
"block_end": self.block_end,
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class ModuleGraphEntry:
|
||||
callsites: list[str]
|
||||
sibling_modules_at_callsites: list[str] = field(default_factory=list)
|
||||
callsite_local_names: dict[str, dict[str, str]] = field(default_factory=dict)
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {
|
||||
"callsites": list(self.callsites),
|
||||
"sibling_modules_at_callsites": list(self.sibling_modules_at_callsites),
|
||||
"callsite_local_names": {
|
||||
callsite: dict(local_names)
|
||||
for callsite, local_names in self.callsite_local_names.items()
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class Manifest:
|
||||
base_ref: str
|
||||
head_ref: str
|
||||
mode: Mode
|
||||
default_branch: str
|
||||
changed_source_dirs: list[str]
|
||||
plan_units: list[PlanUnit]
|
||||
catalog: list[CatalogEntry]
|
||||
trivy_findings: list[TrivyFinding]
|
||||
module_graph: dict[str, ModuleGraphEntry]
|
||||
errors: list[str]
|
||||
tflint_findings: list[TflintFinding] = field(default_factory=list)
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {
|
||||
"base_ref": self.base_ref,
|
||||
"head_ref": self.head_ref,
|
||||
"mode": self.mode,
|
||||
"default_branch": self.default_branch,
|
||||
"changed_source_dirs": list(self.changed_source_dirs),
|
||||
"plan_units": [p.to_dict() for p in self.plan_units],
|
||||
"catalog": [c.to_dict() for c in self.catalog],
|
||||
"trivy_findings": [f.to_dict() for f in self.trivy_findings],
|
||||
"tflint_findings": [f.to_dict() for f in self.tflint_findings],
|
||||
"module_graph": {k: v.to_dict() for k, v in self.module_graph.items()},
|
||||
"errors": list(self.errors),
|
||||
}
|
||||
|
||||
def to_json(self, indent: int = 2) -> str:
|
||||
return json.dumps(self.to_dict(), indent=indent, sort_keys=False)
|
||||
@@ -0,0 +1,90 @@
|
||||
"""Repo-wide module callsite graph.
|
||||
|
||||
For each local-source module dir, list callsite dirs and sibling modules
|
||||
(other modules instantiated alongside it at any callsite).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path, PurePosixPath
|
||||
|
||||
import hcl2
|
||||
|
||||
from scripts.manifest import ModuleGraphEntry
|
||||
|
||||
|
||||
def _strip_hcl_quotes(s):
|
||||
if isinstance(s, str) and len(s) >= 2 and s[0] == '"' and s[-1] == '"':
|
||||
return s[1:-1]
|
||||
return s
|
||||
|
||||
|
||||
def _iter_module_blocks(tf_file: Path):
|
||||
try:
|
||||
with tf_file.open() as fh:
|
||||
parsed = hcl2.load(fh)
|
||||
except Exception:
|
||||
return
|
||||
for block in parsed.get("module", []):
|
||||
if not isinstance(block, dict):
|
||||
continue
|
||||
for name, body in block.items():
|
||||
if isinstance(body, dict):
|
||||
src = body.get("source")
|
||||
if isinstance(src, list):
|
||||
src = src[0] if src else None
|
||||
src = _strip_hcl_quotes(src)
|
||||
yield _strip_hcl_quotes(name), src
|
||||
|
||||
|
||||
def _is_local_source(src: str | None) -> bool:
|
||||
if not isinstance(src, str):
|
||||
return False
|
||||
return src.startswith(("./", "../"))
|
||||
|
||||
|
||||
def _resolve_local(callsite_dir: PurePosixPath, source: str) -> str:
|
||||
combined = (callsite_dir / source).as_posix()
|
||||
parts: list[str] = []
|
||||
for part in combined.split("/"):
|
||||
if part in ("", "."):
|
||||
continue
|
||||
if part == "..":
|
||||
if parts:
|
||||
parts.pop()
|
||||
continue
|
||||
parts.append(part)
|
||||
return "/".join(parts)
|
||||
|
||||
|
||||
def build_module_graph(repo_root: Path | str) -> dict[str, ModuleGraphEntry]:
|
||||
root = Path(repo_root)
|
||||
raw: dict[str, list[str]] = {}
|
||||
callsite_to_modules: dict[str, dict[str, str]] = {}
|
||||
|
||||
for tf in root.rglob("*.tf"):
|
||||
rel_dir = tf.parent.relative_to(root).as_posix()
|
||||
callsite_dir = PurePosixPath(rel_dir)
|
||||
for name, src in _iter_module_blocks(tf):
|
||||
if not _is_local_source(src):
|
||||
continue
|
||||
target = _resolve_local(callsite_dir, src) # type: ignore[arg-type]
|
||||
raw.setdefault(target, []).append(rel_dir)
|
||||
callsite_to_modules.setdefault(rel_dir, {})[name] = target
|
||||
|
||||
out: dict[str, ModuleGraphEntry] = {}
|
||||
for mod_dir, callsites in raw.items():
|
||||
unique_callsites = sorted(set(callsites))
|
||||
siblings: set[str] = set()
|
||||
for cs in unique_callsites:
|
||||
for other in callsite_to_modules.get(cs, {}).values():
|
||||
if other != mod_dir:
|
||||
siblings.add(other)
|
||||
out[mod_dir] = ModuleGraphEntry(
|
||||
callsites=unique_callsites,
|
||||
sibling_modules_at_callsites=sorted(siblings),
|
||||
callsite_local_names={
|
||||
cs: dict(callsite_to_modules.get(cs, {}))
|
||||
for cs in unique_callsites
|
||||
},
|
||||
)
|
||||
return out
|
||||
@@ -0,0 +1,59 @@
|
||||
"""Parse terraform/tofu/terragrunt plan output for resource actions."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ParsedResource:
|
||||
address: str
|
||||
type: str
|
||||
action: str
|
||||
|
||||
|
||||
_HEADER_RE = re.compile(
|
||||
r"^\s*#\s+(?P<addr>\S+)\s+(?:"
|
||||
r"(?P<create>will be created)|"
|
||||
r"(?P<update>will be updated in-place)|"
|
||||
r"(?P<delete>will be destroyed)|"
|
||||
r"(?P<replace>must be replaced|will be replaced)|"
|
||||
r"(?P<read>will be read during apply)"
|
||||
r")\s*$"
|
||||
)
|
||||
|
||||
|
||||
def _action_from_match(m: re.Match) -> str:
|
||||
for k in ("create", "update", "delete", "replace", "read"):
|
||||
if m.group(k):
|
||||
return k
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _type_from_address(address: str) -> str:
|
||||
"""Strip module prefixes and any index suffix; return the resource type."""
|
||||
parts = address.split(".")
|
||||
i = 0
|
||||
while i + 1 < len(parts) and parts[i].startswith("module"):
|
||||
i += 2
|
||||
if i < len(parts) and parts[i] == "data":
|
||||
i += 1
|
||||
if i >= len(parts):
|
||||
return ""
|
||||
type_token = parts[i]
|
||||
return re.sub(r"\[.*\]$", "", type_token)
|
||||
|
||||
|
||||
def parse_plan_resources(plan_stdout: str) -> list[ParsedResource]:
|
||||
out: list[ParsedResource] = []
|
||||
for line in plan_stdout.splitlines():
|
||||
m = _HEADER_RE.match(line)
|
||||
if not m:
|
||||
continue
|
||||
addr = m.group("addr")
|
||||
out.append(ParsedResource(
|
||||
address=addr,
|
||||
type=_type_from_address(addr),
|
||||
action=_action_from_match(m),
|
||||
))
|
||||
return out
|
||||
@@ -0,0 +1,64 @@
|
||||
"""Detect tool, run init + plan, capture output."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from typing import Literal
|
||||
|
||||
from scripts.manifest import InitResult, PlanResult
|
||||
|
||||
|
||||
Tool = Literal["terragrunt", "tofu"]
|
||||
|
||||
_TAIL_BYTES = 4096
|
||||
_SUMMARY_RE = re.compile(
|
||||
r"Plan:\s+(\d+\s+to add,\s*\d+\s+to change,\s*\d+\s+to destroy)\."
|
||||
)
|
||||
|
||||
|
||||
def detect_tool(plan_dir: Path | str) -> Tool:
|
||||
p = Path(plan_dir)
|
||||
if (p / "terragrunt.hcl").is_file():
|
||||
return "terragrunt"
|
||||
return "tofu"
|
||||
|
||||
|
||||
def _tail(s: str, n: int = _TAIL_BYTES) -> str:
|
||||
return s[-n:] if len(s) > n else s
|
||||
|
||||
|
||||
def extract_summary(plan_stdout: str) -> str:
|
||||
m = _SUMMARY_RE.search(plan_stdout)
|
||||
if m:
|
||||
return re.sub(r"\s+", " ", m.group(1)).strip()
|
||||
if "No changes." in plan_stdout or "no changes." in plan_stdout.lower():
|
||||
return "no changes"
|
||||
return "unknown"
|
||||
|
||||
|
||||
def run_init(plan_dir: Path | str, tool: Tool) -> InitResult:
|
||||
args = [tool, "init", "-input=false", "-no-color"]
|
||||
result = subprocess.run(
|
||||
args, cwd=str(plan_dir), capture_output=True, text=True, check=False,
|
||||
)
|
||||
return InitResult(
|
||||
ok=(result.returncode == 0),
|
||||
stdout_tail=_tail(result.stdout),
|
||||
stderr_tail=_tail(result.stderr),
|
||||
)
|
||||
|
||||
|
||||
def run_plan(plan_dir: Path | str, tool: Tool, stdout_path: Path) -> PlanResult:
|
||||
args = [tool, "plan", "-input=false", "-no-color"]
|
||||
result = subprocess.run(
|
||||
args, cwd=str(plan_dir), capture_output=True, text=True, check=False,
|
||||
)
|
||||
stdout_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
stdout_path.write_text(result.stdout)
|
||||
return PlanResult(
|
||||
ok=(result.returncode == 0),
|
||||
stdout_path=str(stdout_path),
|
||||
exit_code=result.returncode,
|
||||
summary=extract_summary(result.stdout),
|
||||
)
|
||||
@@ -0,0 +1,44 @@
|
||||
"""Classify a directory as a plan unit, a reusable module, or unknown."""
|
||||
from __future__ import annotations
|
||||
|
||||
from enum import Enum
|
||||
from pathlib import Path
|
||||
|
||||
import hcl2
|
||||
|
||||
|
||||
class DirKind(str, Enum):
|
||||
PLAN_UNIT = "plan_unit"
|
||||
MODULE = "module"
|
||||
UNKNOWN = "unknown"
|
||||
|
||||
|
||||
def _has_backend_block(tf_path: Path) -> bool:
|
||||
try:
|
||||
with tf_path.open() as fh:
|
||||
parsed = hcl2.load(fh)
|
||||
except Exception:
|
||||
return False
|
||||
for block in parsed.get("terraform", []):
|
||||
if isinstance(block, dict) and "backend" in block:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def classify_dir(repo_root: Path | str, rel_dir: str) -> DirKind:
|
||||
root = Path(repo_root)
|
||||
full = root / rel_dir
|
||||
if not full.is_dir():
|
||||
raise FileNotFoundError(full)
|
||||
|
||||
if (full / "terragrunt.hcl").is_file():
|
||||
return DirKind.PLAN_UNIT
|
||||
|
||||
tf_files = sorted(full.glob("*.tf")) + sorted(full.glob("*.tf.json"))
|
||||
if not tf_files:
|
||||
return DirKind.UNKNOWN
|
||||
|
||||
for tf in tf_files:
|
||||
if tf.suffix == ".tf" and _has_backend_block(tf):
|
||||
return DirKind.PLAN_UNIT
|
||||
return DirKind.MODULE
|
||||
@@ -0,0 +1,146 @@
|
||||
"""Compute reference sets for the consistency reviewer."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from collections import Counter, defaultdict
|
||||
from pathlib import Path
|
||||
|
||||
from scripts.manifest import ModuleGraphEntry
|
||||
|
||||
|
||||
def _peer_dirs_at_depth(repo_root: Path, dir_path: str) -> list[str]:
|
||||
parts = dir_path.split("/")
|
||||
depth = len(parts)
|
||||
parent = repo_root.joinpath(*parts[:-1]) if depth > 1 else repo_root
|
||||
if not parent.is_dir():
|
||||
return []
|
||||
out: list[str] = []
|
||||
for p in sorted(parent.iterdir()):
|
||||
if not p.is_dir():
|
||||
continue
|
||||
rel = p.relative_to(repo_root).as_posix()
|
||||
if rel != dir_path:
|
||||
out.append(rel)
|
||||
return out
|
||||
|
||||
|
||||
def _terragrunt_refs(repo_root: Path, dir_path: str) -> list[str]:
|
||||
"""Region peers + same-component cross-env (layout: live/<env>/<region>/<component>)."""
|
||||
parts = dir_path.split("/")
|
||||
if len(parts) != 4:
|
||||
return _peer_dirs_at_depth(repo_root, dir_path)
|
||||
live, env, region, component = parts
|
||||
refs: set[str] = set()
|
||||
region_dir = repo_root / live / env / region
|
||||
if region_dir.is_dir():
|
||||
for p in region_dir.iterdir():
|
||||
if p.is_dir() and (p / "terragrunt.hcl").is_file():
|
||||
rel = p.relative_to(repo_root).as_posix()
|
||||
if rel != dir_path:
|
||||
refs.add(rel)
|
||||
live_dir = repo_root / live
|
||||
if live_dir.is_dir():
|
||||
for env_dir in live_dir.iterdir():
|
||||
candidate = env_dir / region / component
|
||||
if candidate.is_dir() and (candidate / "terragrunt.hcl").is_file():
|
||||
rel = candidate.relative_to(repo_root).as_posix()
|
||||
if rel != dir_path:
|
||||
refs.add(rel)
|
||||
env_live_dir = repo_root / live / env
|
||||
if env_live_dir.is_dir():
|
||||
for region_dir2 in env_live_dir.iterdir():
|
||||
candidate = region_dir2 / component
|
||||
if candidate.is_dir() and (candidate / "terragrunt.hcl").is_file():
|
||||
rel = candidate.relative_to(repo_root).as_posix()
|
||||
if rel != dir_path:
|
||||
refs.add(rel)
|
||||
return sorted(refs)
|
||||
|
||||
|
||||
def compute_reference_sets(
|
||||
repo_root: Path | str,
|
||||
changed_dirs: set[str],
|
||||
module_graph: dict[str, ModuleGraphEntry],
|
||||
) -> dict[str, list[str]]:
|
||||
root = Path(repo_root)
|
||||
out: dict[str, list[str]] = {}
|
||||
for d in sorted(changed_dirs):
|
||||
if d.startswith("modules/"):
|
||||
entry = module_graph.get(d)
|
||||
out[d] = list(entry.sibling_modules_at_callsites) if entry else []
|
||||
continue
|
||||
full = root / d
|
||||
if (full / "terragrunt.hcl").is_file():
|
||||
out[d] = _terragrunt_refs(root, d)
|
||||
continue
|
||||
out[d] = _peer_dirs_at_depth(root, d)
|
||||
return out
|
||||
|
||||
|
||||
_ATTR_RE = re.compile(r"^\s*(?P<key>[A-Za-z0-9_]+)\s*=")
|
||||
_MODULE_RE = re.compile(r'^\s*module\s+"(?P<name>[^"]+)"\s*\{')
|
||||
|
||||
|
||||
def _iter_tf_lines(dir_path: Path) -> list[str]:
|
||||
lines: list[str] = []
|
||||
for tf in sorted(dir_path.glob("*.tf")):
|
||||
lines.extend(tf.read_text().splitlines())
|
||||
return lines
|
||||
|
||||
|
||||
def _dir_signature(dir_path: Path) -> tuple[set[str], set[str]]:
|
||||
attrs: set[str] = set()
|
||||
modules: set[str] = set()
|
||||
for line in _iter_tf_lines(dir_path):
|
||||
attr_match = _ATTR_RE.match(line)
|
||||
if attr_match:
|
||||
attrs.add(attr_match.group("key"))
|
||||
module_match = _MODULE_RE.match(line)
|
||||
if module_match:
|
||||
modules.add(module_match.group("name"))
|
||||
return attrs, modules
|
||||
|
||||
|
||||
def compute_consistency_norms(
|
||||
repo_root: Path | str,
|
||||
reference_sets: dict[str, list[str]],
|
||||
) -> dict[str, dict[str, list[dict[str, object]]]]:
|
||||
root = Path(repo_root)
|
||||
out: dict[str, dict[str, list[dict[str, object]]]] = {}
|
||||
for changed_dir, refs in sorted(reference_sets.items()):
|
||||
attr_support: dict[str, list[str]] = defaultdict(list)
|
||||
module_support: dict[str, list[str]] = defaultdict(list)
|
||||
naming_tokens: Counter[str] = Counter()
|
||||
usable_refs: list[str] = []
|
||||
for ref in refs:
|
||||
ref_path = root / ref
|
||||
if not ref_path.is_dir():
|
||||
continue
|
||||
attrs, modules = _dir_signature(ref_path)
|
||||
if not attrs and not modules:
|
||||
continue
|
||||
usable_refs.append(ref)
|
||||
for attr in attrs:
|
||||
attr_support[attr].append(ref)
|
||||
for module in modules:
|
||||
module_support[module].append(ref)
|
||||
naming_tokens.update(Path(ref).parts[-1:])
|
||||
|
||||
out[changed_dir] = {
|
||||
"attribute_norms": [
|
||||
{"attribute": attr, "peer_dirs": sorted(peer_dirs)}
|
||||
for attr, peer_dirs in sorted(attr_support.items())
|
||||
if len(peer_dirs) >= 2
|
||||
],
|
||||
"module_wrapper_norms": [
|
||||
{"module": module, "peer_dirs": sorted(peer_dirs)}
|
||||
for module, peer_dirs in sorted(module_support.items())
|
||||
if len(peer_dirs) >= 2
|
||||
],
|
||||
"naming_norms": [
|
||||
{"token": token, "peer_dirs": sorted(usable_refs)}
|
||||
for token, count in sorted(naming_tokens.items())
|
||||
if count >= 2
|
||||
],
|
||||
}
|
||||
return out
|
||||
@@ -0,0 +1,95 @@
|
||||
"""refresh-controls.py — populate data/controls/{fsbp,cis,meta}.json."""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
import requests
|
||||
|
||||
_HERE = Path(__file__).resolve().parent
|
||||
if str(_HERE.parent) not in sys.path:
|
||||
sys.path.insert(0, str(_HERE.parent))
|
||||
|
||||
from scripts.controls_schema import Control, ControlsFile
|
||||
from scripts.scrape_cis import CIS_URL, parse_cis_page
|
||||
from scripts.scrape_fsbp import FSBP_INDEX_URL, parse_fsbp_index
|
||||
|
||||
|
||||
_TIMEOUT = 30
|
||||
_USER_AGENT = "audit-terraform-controls-refresh/1.0"
|
||||
|
||||
|
||||
def _fetch(url: str) -> str:
|
||||
r = requests.get(url, headers={"User-Agent": _USER_AGENT}, timeout=_TIMEOUT)
|
||||
r.raise_for_status()
|
||||
return r.text
|
||||
|
||||
|
||||
def _now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
def _build_fsbp() -> ControlsFile:
|
||||
html = _fetch(FSBP_INDEX_URL)
|
||||
rows = parse_fsbp_index(html, base_url=FSBP_INDEX_URL)
|
||||
controls = [
|
||||
Control(
|
||||
control_id=f"FSBP {r.control_id}",
|
||||
title=r.title,
|
||||
severity=r.severity,
|
||||
resource_types=r.resource_types,
|
||||
requirement=r.requirement or r.title,
|
||||
source_url=r.detail_url,
|
||||
)
|
||||
for r in rows
|
||||
]
|
||||
return ControlsFile(source="fsbp", fetched_at=_now(), controls=controls)
|
||||
|
||||
|
||||
def _build_cis() -> ControlsFile:
|
||||
html = _fetch(CIS_URL)
|
||||
rows = parse_cis_page(html, source_url=CIS_URL)
|
||||
controls = [
|
||||
Control(
|
||||
control_id=r.control_id,
|
||||
title=r.title,
|
||||
severity=r.severity,
|
||||
resource_types=r.resource_types,
|
||||
requirement=r.requirement,
|
||||
source_url=r.source_url,
|
||||
)
|
||||
for r in rows
|
||||
]
|
||||
return ControlsFile(source="cis", fetched_at=_now(), controls=controls)
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--output-dir", default=str(_HERE.parent / "data" / "controls"))
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
out_dir = Path(args.output_dir).resolve()
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
fsbp = _build_fsbp()
|
||||
cis = _build_cis()
|
||||
|
||||
(out_dir / "fsbp.json").write_text(fsbp.to_json())
|
||||
(out_dir / "cis.json").write_text(cis.to_json())
|
||||
|
||||
meta = {
|
||||
"fsbp": {"url": FSBP_INDEX_URL, "fetched_at": fsbp.fetched_at.isoformat()},
|
||||
"cis": {"url": CIS_URL, "fetched_at": cis.fetched_at.isoformat()},
|
||||
}
|
||||
(out_dir / "meta.json").write_text(json.dumps(meta, indent=2))
|
||||
|
||||
print(f"Wrote {len(fsbp.controls)} FSBP controls, {len(cis.controls)} CIS controls "
|
||||
f"to {out_dir}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,41 @@
|
||||
"""Map changed dirs to plan units, preserving the direct trigger directories."""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from scripts.manifest import ModuleGraphEntry
|
||||
from scripts.plan_unit import classify_dir, DirKind
|
||||
|
||||
|
||||
def resolve_plan_units(
|
||||
repo_root: Path | str,
|
||||
changed_dirs: set[str],
|
||||
module_graph: dict[str, ModuleGraphEntry],
|
||||
) -> tuple[dict[str, list[str]], list[str]]:
|
||||
"""Return (plan_units, orphan_modules).
|
||||
|
||||
plan_units maps plan_dir -> sorted-unique list of changed dirs that triggered it.
|
||||
Direct Terragrunt or root-module changes remain in the trigger set so later
|
||||
manifest assembly can attach per-plan-unit change context.
|
||||
orphan_modules is a list of changed module dirs with zero callsites.
|
||||
"""
|
||||
plan_units: dict[str, set[str]] = {}
|
||||
orphans: list[str] = []
|
||||
|
||||
for d in sorted(changed_dirs):
|
||||
kind = classify_dir(repo_root, d)
|
||||
if kind is DirKind.PLAN_UNIT:
|
||||
plan_units.setdefault(d, set()).add(d)
|
||||
elif kind is DirKind.MODULE:
|
||||
entry = module_graph.get(d)
|
||||
callsites = list(entry.callsites) if entry else []
|
||||
if not callsites:
|
||||
orphans.append(d)
|
||||
continue
|
||||
for cs in callsites:
|
||||
plan_units.setdefault(cs, set()).add(d)
|
||||
|
||||
return (
|
||||
{k: sorted(v) for k, v in sorted(plan_units.items())},
|
||||
sorted(orphans),
|
||||
)
|
||||
@@ -0,0 +1,70 @@
|
||||
"""Aggregate runs.jsonl into precision + token-cost stats."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def compute_stats(log_path: Path) -> dict:
|
||||
if not log_path.exists():
|
||||
return {"by_agent": {}, "by_rule": {}, "runs": 0}
|
||||
|
||||
by_agent: dict[str, dict] = {}
|
||||
by_rule: dict[str, dict] = {}
|
||||
runs: set[str] = set()
|
||||
|
||||
for line in log_path.read_text(encoding="utf-8").splitlines():
|
||||
if not line.strip():
|
||||
continue
|
||||
rec = json.loads(line)
|
||||
agent = rec.get("agent", "?")
|
||||
|
||||
if rec["kind"] == "subagent_run":
|
||||
runs.add(rec["run_id"])
|
||||
a = by_agent.setdefault(agent, _empty_agent())
|
||||
a["tokens"] += rec.get("input_tokens", 0) + rec.get("output_tokens", 0)
|
||||
a["duration_ms"] += rec.get("duration_ms", 0)
|
||||
a["runs"] += 1
|
||||
|
||||
elif rec["kind"] == "verdict":
|
||||
verdict = rec["verdict"]
|
||||
a = by_agent.setdefault(agent, _empty_agent())
|
||||
a["total"] += 1
|
||||
a[verdict] = a.get(verdict, 0) + 1
|
||||
|
||||
rule_key = f"{agent}/{rec['rule_id']}"
|
||||
r = by_rule.setdefault(rule_key, _empty_rule())
|
||||
r["total"] += 1
|
||||
r[verdict] = r.get(verdict, 0) + 1
|
||||
|
||||
for a in by_agent.values():
|
||||
a["precision"] = a["kept"] / a["total"] if a["total"] else 0.0
|
||||
a["tokens_per_kept"] = a["tokens"] / a["kept"] if a["kept"] else float("inf")
|
||||
|
||||
for r in by_rule.values():
|
||||
r["precision"] = r["kept"] / r["total"] if r["total"] else 0.0
|
||||
|
||||
return {"by_agent": by_agent, "by_rule": by_rule, "runs": len(runs)}
|
||||
|
||||
|
||||
def _empty_agent() -> dict:
|
||||
return {
|
||||
"tokens": 0, "duration_ms": 0, "runs": 0,
|
||||
"total": 0, "kept": 0, "dismissed": 0, "false_positive": 0,
|
||||
}
|
||||
|
||||
|
||||
def _empty_rule() -> dict:
|
||||
return {"total": 0, "kept": 0, "dismissed": 0, "false_positive": 0}
|
||||
|
||||
|
||||
def main(argv: list[str]) -> int:
|
||||
log = Path.home() / ".claude/cache/audit-terraform/runs.jsonl"
|
||||
stats = compute_stats(log)
|
||||
print(json.dumps(stats, indent=2))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main(sys.argv[1:]))
|
||||
@@ -0,0 +1,117 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import shutil
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from scripts.manifest import TflintFinding, TrivyFinding
|
||||
|
||||
|
||||
def _normalize_trivy_findings(payload: dict[str, Any]) -> list[TrivyFinding]:
|
||||
findings: list[TrivyFinding] = []
|
||||
for result in payload.get("Results", []):
|
||||
for misconf in result.get("Misconfigurations", []):
|
||||
cause = misconf.get("CauseMetadata") or {}
|
||||
resource = cause.get("Resource") or ""
|
||||
resource_type = ""
|
||||
if isinstance(resource, str) and "." in resource:
|
||||
resource_type = resource.split(".", 1)[0]
|
||||
findings.append(TrivyFinding(
|
||||
check_id=(
|
||||
misconf.get("AVDID")
|
||||
or misconf.get("ID")
|
||||
or misconf.get("Query")
|
||||
or ""
|
||||
),
|
||||
title=misconf.get("Title") or misconf.get("ID") or "",
|
||||
severity=str(misconf.get("Severity") or "UNKNOWN").lower(),
|
||||
message=misconf.get("Message") or misconf.get("Description") or "",
|
||||
file=result.get("Target") or "",
|
||||
start_line=int(cause.get("StartLine") or 0),
|
||||
end_line=int(cause.get("EndLine") or 0),
|
||||
resource_type=resource_type,
|
||||
))
|
||||
return findings
|
||||
|
||||
|
||||
def _run_trivy_config(repo: Path, changed_files_in_repo: list[str]) -> tuple[dict[str, Any], list[TrivyFinding], str | None]:
|
||||
changed_dirs = sorted({
|
||||
str(Path(path).parent)
|
||||
for path in changed_files_in_repo
|
||||
})
|
||||
scan_paths = [str((repo / rel).resolve()) for rel in changed_dirs] or [str(repo)]
|
||||
result = subprocess.run(
|
||||
["trivy", "config", "--quiet", "--format", "json", *scan_paths],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
stderr = result.stderr.strip() or result.stdout.strip()
|
||||
return {}, [], f"trivy config failed: {stderr}"
|
||||
try:
|
||||
payload = json.loads(result.stdout) if result.stdout.strip() else {}
|
||||
except json.JSONDecodeError as exc:
|
||||
return {}, [], f"trivy config output was not valid JSON: {exc}"
|
||||
return payload, _normalize_trivy_findings(payload), None
|
||||
|
||||
|
||||
def _normalize_tflint_findings(payload: dict, *, scanned_dir: str) -> list[TflintFinding]:
|
||||
findings: list[TflintFinding] = []
|
||||
for issue in payload.get("issues", []) or []:
|
||||
rule = issue.get("rule", {}) or {}
|
||||
rng = issue.get("range", {}) or {}
|
||||
start = rng.get("start", {}) or {}
|
||||
end = rng.get("end", {}) or {}
|
||||
rel_file = rng.get("filename") or ""
|
||||
if "/" in rel_file:
|
||||
full = rel_file
|
||||
elif rel_file:
|
||||
full = (Path(scanned_dir) / rel_file).as_posix()
|
||||
else:
|
||||
full = ""
|
||||
findings.append(TflintFinding(
|
||||
rule=rule.get("name") or "",
|
||||
severity=str(rule.get("severity") or "warning").lower(),
|
||||
message=issue.get("message") or "",
|
||||
file=full,
|
||||
start_line=int(start.get("line") or 0),
|
||||
end_line=int(end.get("line") or 0),
|
||||
link=rule.get("link") or "",
|
||||
))
|
||||
return findings
|
||||
|
||||
|
||||
def _run_tflint(repo: Path, changed_files_in_repo: list[str]) -> tuple[dict, list[TflintFinding], str | None]:
|
||||
if shutil.which("tflint") is None:
|
||||
return {}, [], None
|
||||
changed_dirs = sorted({
|
||||
str(Path(p).parent.as_posix()) for p in changed_files_in_repo
|
||||
if p.endswith((".tf", ".tf.json"))
|
||||
})
|
||||
if not changed_dirs:
|
||||
return {}, [], None
|
||||
aggregated: list[TflintFinding] = []
|
||||
raw_payloads: dict[str, dict] = {}
|
||||
errs: list[str] = []
|
||||
for rel in changed_dirs:
|
||||
target = (repo / rel).resolve()
|
||||
if not target.is_dir():
|
||||
continue
|
||||
proc = subprocess.run(
|
||||
["tflint", "--format", "json", "--chdir", str(target)],
|
||||
capture_output=True, text=True, check=False,
|
||||
)
|
||||
if proc.returncode not in (0, 2):
|
||||
errs.append(f"tflint failed in {rel}: {proc.stderr.strip() or proc.stdout.strip()}")
|
||||
continue
|
||||
try:
|
||||
payload = json.loads(proc.stdout) if proc.stdout.strip() else {}
|
||||
except json.JSONDecodeError as exc:
|
||||
errs.append(f"tflint output in {rel} was not valid JSON: {exc}")
|
||||
continue
|
||||
raw_payloads[rel] = payload
|
||||
aggregated.extend(_normalize_tflint_findings(payload, scanned_dir=rel))
|
||||
return {"by_dir": raw_payloads}, aggregated, ("; ".join(errs) if errs else None)
|
||||
@@ -0,0 +1,84 @@
|
||||
"""Scrape the CIS AWS Foundations Benchmark page.
|
||||
|
||||
The page is a mapping table:
|
||||
| Control ID and title (FSBP-style) | CIS v5.0.0 | CIS v3.0.0 | CIS v1.4.0 | CIS v1.2.0 |
|
||||
|
||||
We emit one CISRow per unified control. The `requirement` field summarises
|
||||
which CIS versions reference it.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
from scripts.control_resource_map import control_id_to_types
|
||||
|
||||
|
||||
CIS_URL = (
|
||||
"https://docs.aws.amazon.com/securityhub/latest/userguide/"
|
||||
"cis-aws-foundations-benchmark.html"
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class CISRow:
|
||||
control_id: str
|
||||
title: str
|
||||
severity: str
|
||||
requirement: str
|
||||
source_url: str
|
||||
resource_types: list[str] = field(default_factory=list)
|
||||
|
||||
|
||||
_TITLE_RE = re.compile(r"^\s*\[(?P<id>[A-Za-z][A-Za-z0-9]*\.\d+)\]\s*(?P<title>.+)$")
|
||||
_VERSION_RE = re.compile(r"CIS\s+v(?P<ver>[\d.]+)", re.IGNORECASE)
|
||||
|
||||
|
||||
def _extract_versions(headers: list[str]) -> list[str]:
|
||||
"""From header cells, pull out version strings like '5.0.0' for each
|
||||
column. Non-version columns yield empty string."""
|
||||
versions: list[str] = []
|
||||
for h in headers:
|
||||
m = _VERSION_RE.search(h)
|
||||
versions.append(m.group("ver") if m else "")
|
||||
return versions
|
||||
|
||||
|
||||
def parse_cis_page(html: str, source_url: str) -> list[CISRow]:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
rows: list[CISRow] = []
|
||||
for table in soup.find_all("table"):
|
||||
header_cells = table.find("tr").find_all(["th", "td"]) if table.find("tr") else []
|
||||
header_texts = [c.get_text(strip=True) for c in header_cells]
|
||||
versions = _extract_versions(header_texts)
|
||||
if not any(versions):
|
||||
continue
|
||||
for tr in table.find_all("tr")[1:]:
|
||||
cells = tr.find_all("td")
|
||||
if len(cells) < 2:
|
||||
continue
|
||||
title_text = cells[0].get_text(strip=True)
|
||||
m = _TITLE_RE.match(title_text)
|
||||
if not m:
|
||||
continue
|
||||
unified_id = m.group("id")
|
||||
title = m.group("title").strip()
|
||||
version_refs: list[str] = []
|
||||
for ver, cell in zip(versions[1:], cells[1:]):
|
||||
if not ver:
|
||||
continue
|
||||
num = cell.get_text(strip=True)
|
||||
if num:
|
||||
version_refs.append(f"v{ver} §{num}")
|
||||
requirement = "CIS " + ", ".join(version_refs) if version_refs else ""
|
||||
rows.append(CISRow(
|
||||
control_id=f"CIS {unified_id}",
|
||||
title=title,
|
||||
severity="medium",
|
||||
requirement=requirement,
|
||||
source_url=source_url,
|
||||
resource_types=control_id_to_types(f"CIS {unified_id}"),
|
||||
))
|
||||
return rows
|
||||
@@ -0,0 +1,61 @@
|
||||
"""Scrape AWS Security Hub FSBP controls index page into structured rows.
|
||||
|
||||
The index page (https://docs.aws.amazon.com/securityhub/latest/userguide/fsbp-standard.html)
|
||||
lists controls as <p><a href="./<svc>-controls.html#<id>">[<Service>.<Number>] <Title></a></p>.
|
||||
This module only parses the index — severity defaults to "medium" because the
|
||||
index doesn't surface severity. Detail-page enrichment is a future task.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from urllib.parse import urljoin
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
from scripts.control_resource_map import control_id_to_types
|
||||
|
||||
|
||||
FSBP_INDEX_URL = (
|
||||
"https://docs.aws.amazon.com/securityhub/latest/userguide/"
|
||||
"fsbp-standard.html"
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class FSBPRow:
|
||||
control_id: str
|
||||
title: str
|
||||
severity: str
|
||||
detail_url: str
|
||||
resource_types: list[str] = field(default_factory=list)
|
||||
requirement: str = ""
|
||||
|
||||
|
||||
_ANCHOR_RE = re.compile(r"^\s*\[(?P<id>[A-Za-z][A-Za-z0-9]*\.\d+)\]\s*(?P<title>.+)$")
|
||||
|
||||
|
||||
def parse_fsbp_index(html: str, base_url: str) -> list[FSBPRow]:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
rows: list[FSBPRow] = []
|
||||
seen: set[str] = set()
|
||||
for a in soup.find_all("a"):
|
||||
text = a.get_text(strip=True)
|
||||
m = _ANCHOR_RE.match(text)
|
||||
if not m:
|
||||
continue
|
||||
control_id = m.group("id")
|
||||
if control_id in seen:
|
||||
continue
|
||||
seen.add(control_id)
|
||||
title = m.group("title").strip()
|
||||
href = a.get("href", "")
|
||||
detail_url = urljoin(base_url, href) if href else ""
|
||||
rows.append(FSBPRow(
|
||||
control_id=control_id,
|
||||
title=title,
|
||||
severity="medium",
|
||||
detail_url=detail_url,
|
||||
resource_types=control_id_to_types(control_id),
|
||||
))
|
||||
return rows
|
||||
@@ -0,0 +1,64 @@
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
def _slice_for_agent(manifest_dict: dict, agent: str) -> dict:
|
||||
"""Return a subset of the manifest appropriate to the named agent.
|
||||
|
||||
- fsbp/cis/aws-bp: AWS-filtered catalog + plan_units summary + trivy + tflint + errors
|
||||
- consistency: full catalog + full plan_units + trivy + tflint + errors
|
||||
- tf-hygiene: full catalog + full plan_units + tflint + changed_source_dirs + errors (no trivy)
|
||||
- walkthrough: full catalog + plan_units summary + changed_source_dirs + errors (no findings)
|
||||
"""
|
||||
base = {
|
||||
"base_ref": manifest_dict["base_ref"],
|
||||
"head_ref": manifest_dict["head_ref"],
|
||||
"mode": manifest_dict["mode"],
|
||||
"default_branch": manifest_dict["default_branch"],
|
||||
"errors": list(manifest_dict["errors"]),
|
||||
}
|
||||
catalog = manifest_dict["catalog"]
|
||||
if agent in ("fsbp", "cis", "aws-bp"):
|
||||
return {
|
||||
**base,
|
||||
"catalog": [e for e in catalog if e["type"].startswith("aws_")],
|
||||
"trivy_findings": list(manifest_dict.get("trivy_findings", [])),
|
||||
"tflint_findings": list(manifest_dict.get("tflint_findings", [])),
|
||||
"plan_units": [
|
||||
{"plan_dir": pu["plan_dir"], "tool": pu["tool"],
|
||||
"plan_ok": pu["plan"]["ok"], "summary": pu["plan"]["summary"],
|
||||
"terragrunt_changed": pu["terragrunt_changed"],
|
||||
"changed_files": list(pu["changed_files"])}
|
||||
for pu in manifest_dict["plan_units"]
|
||||
],
|
||||
}
|
||||
if agent == "consistency":
|
||||
return {
|
||||
**base,
|
||||
"changed_source_dirs": list(manifest_dict["changed_source_dirs"]),
|
||||
"catalog": catalog,
|
||||
"trivy_findings": list(manifest_dict.get("trivy_findings", [])),
|
||||
"tflint_findings": list(manifest_dict.get("tflint_findings", [])),
|
||||
"plan_units": [dict(pu) for pu in manifest_dict["plan_units"]],
|
||||
}
|
||||
if agent == "walkthrough":
|
||||
return {
|
||||
**base,
|
||||
"changed_source_dirs": list(manifest_dict["changed_source_dirs"]),
|
||||
"catalog": catalog,
|
||||
"plan_units": [
|
||||
{"plan_dir": pu["plan_dir"], "tool": pu["tool"],
|
||||
"plan_ok": pu["plan"]["ok"], "summary": pu["plan"]["summary"],
|
||||
"terragrunt_changed": pu["terragrunt_changed"],
|
||||
"changed_files": list(pu["changed_files"])}
|
||||
for pu in manifest_dict["plan_units"]
|
||||
],
|
||||
}
|
||||
if agent == "tf-hygiene":
|
||||
return {
|
||||
**base,
|
||||
"changed_source_dirs": list(manifest_dict["changed_source_dirs"]),
|
||||
"catalog": catalog,
|
||||
"tflint_findings": list(manifest_dict.get("tflint_findings", [])),
|
||||
"plan_units": [dict(pu) for pu in manifest_dict["plan_units"]],
|
||||
}
|
||||
return manifest_dict
|
||||
@@ -0,0 +1,220 @@
|
||||
"""Locate a specific resource block in a directory's .tf files."""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
import re
|
||||
|
||||
from scripts.hcl_diff import find_resource_blocks
|
||||
|
||||
|
||||
_KEY_ATTRIBUTES = {
|
||||
"kms_key_id",
|
||||
"kms_key_arn",
|
||||
"bucket_key_enabled",
|
||||
"publicly_accessible",
|
||||
"acl",
|
||||
"versioning",
|
||||
"tags",
|
||||
"deletion_protection",
|
||||
}
|
||||
_VAR_REF_RE = re.compile(r"\bvar\.([A-Za-z0-9_]+)")
|
||||
_LOCAL_REF_RE = re.compile(r"\blocal\.([A-Za-z0-9_]+)")
|
||||
_DATA_REF_RE = re.compile(r"\bdata\.aws_iam_policy_document\.([A-Za-z0-9_]+)")
|
||||
_SG_REF_TEMPLATE = 'security_group_id = aws_security_group.{name}.id'
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BlockLocation:
|
||||
file: Path
|
||||
start_line: int
|
||||
end_line: int
|
||||
text: str
|
||||
header: str
|
||||
evidence_line: str
|
||||
key_attributes: dict[str, str | bool | int | list[str]]
|
||||
review_context: dict[str, object]
|
||||
|
||||
|
||||
def _parse_value(raw: str) -> str | bool | int | list[str]:
|
||||
value = raw.strip().rstrip(",")
|
||||
if value.lower() in {"true", "false"}:
|
||||
return value.lower() == "true"
|
||||
if re.fullmatch(r"-?\d+", value):
|
||||
return int(value)
|
||||
if value.startswith('"') and value.endswith('"'):
|
||||
return value[1:-1]
|
||||
if value.startswith("[") and value.endswith("]"):
|
||||
inner = value[1:-1].strip()
|
||||
if not inner:
|
||||
return []
|
||||
parts = [part.strip().strip('"') for part in inner.split(",")]
|
||||
return [part for part in parts if part]
|
||||
return value
|
||||
|
||||
|
||||
def _extract_key_attributes(lines: list[str]) -> dict[str, str | bool | int | list[str]]:
|
||||
out: dict[str, str | bool | int | list[str]] = {}
|
||||
for line in lines:
|
||||
stripped = line.strip()
|
||||
if "=" not in stripped or stripped.startswith("#"):
|
||||
continue
|
||||
key, _, raw_value = stripped.partition("=")
|
||||
key = key.strip()
|
||||
if key not in _KEY_ATTRIBUTES:
|
||||
continue
|
||||
out[key] = _parse_value(raw_value)
|
||||
return out
|
||||
|
||||
|
||||
def _find_named_block(lines: list[str], header_re: re.Pattern[str], name: str) -> list[str]:
|
||||
i = 0
|
||||
while i < len(lines):
|
||||
line = lines[i]
|
||||
match = header_re.match(line)
|
||||
if not match or match.group("name") != name:
|
||||
i += 1
|
||||
continue
|
||||
depth = line.count("{") - line.count("}")
|
||||
j = i
|
||||
while depth > 0 and j + 1 < len(lines):
|
||||
j += 1
|
||||
depth += lines[j].count("{") - lines[j].count("}")
|
||||
return lines[i:j + 1]
|
||||
return []
|
||||
|
||||
|
||||
def _collect_variable_defaults(root: Path, names: set[str]) -> dict[str, str | bool | int | list[str]]:
|
||||
out: dict[str, str | bool | int | list[str]] = {}
|
||||
header_re = re.compile(r'^\s*variable\s+"(?P<name>[^"]+)"\s*\{')
|
||||
for tf in sorted(root.glob("*.tf")):
|
||||
lines = tf.read_text().splitlines()
|
||||
for name in names:
|
||||
if name in out:
|
||||
continue
|
||||
block = _find_named_block(lines, header_re, name)
|
||||
for line in block[1:]:
|
||||
stripped = line.strip()
|
||||
if stripped.startswith("default"):
|
||||
_, _, raw = stripped.partition("=")
|
||||
out[name] = _parse_value(raw)
|
||||
break
|
||||
return out
|
||||
|
||||
|
||||
def _collect_locals(root: Path, names: set[str]) -> dict[str, str | bool | int | list[str]]:
|
||||
out: dict[str, str | bool | int | list[str]] = {}
|
||||
header_re = re.compile(r"^\s*locals\s*\{")
|
||||
for tf in sorted(root.glob("*.tf")):
|
||||
lines = tf.read_text().splitlines()
|
||||
i = 0
|
||||
while i < len(lines):
|
||||
line = lines[i]
|
||||
if not header_re.match(line):
|
||||
i += 1
|
||||
continue
|
||||
depth = line.count("{") - line.count("}")
|
||||
j = i
|
||||
while depth > 0 and j + 1 < len(lines):
|
||||
j += 1
|
||||
depth += lines[j].count("{") - lines[j].count("}")
|
||||
for block_line in lines[i + 1:j]:
|
||||
stripped = block_line.strip()
|
||||
if "=" not in stripped:
|
||||
continue
|
||||
key, _, raw = stripped.partition("=")
|
||||
key = key.strip()
|
||||
if key in names and key not in out:
|
||||
out[key] = _parse_value(raw)
|
||||
i = j + 1
|
||||
return out
|
||||
|
||||
|
||||
def _collect_related_policy_docs(root: Path, names: set[str]) -> list[str]:
|
||||
out: list[str] = []
|
||||
header_re = re.compile(
|
||||
r'^\s*data\s+"aws_iam_policy_document"\s+"(?P<name>[^"]+)"\s*\{'
|
||||
)
|
||||
for tf in sorted(root.glob("*.tf")):
|
||||
lines = tf.read_text().splitlines()
|
||||
for name in names:
|
||||
block = _find_named_block(lines, header_re, name)
|
||||
if block:
|
||||
out.append(block[0].strip())
|
||||
return out
|
||||
|
||||
|
||||
def _collect_related_sg_rules(root: Path, name: str) -> list[str]:
|
||||
out: list[str] = []
|
||||
target = _SG_REF_TEMPLATE.format(name=name)
|
||||
header_re = re.compile(
|
||||
r'^\s*resource\s+"aws_security_group_rule"\s+"(?P<name>[^"]+)"\s*\{'
|
||||
)
|
||||
for tf in sorted(root.glob("*.tf")):
|
||||
lines = tf.read_text().splitlines()
|
||||
i = 0
|
||||
while i < len(lines):
|
||||
line = lines[i]
|
||||
if not header_re.match(line):
|
||||
i += 1
|
||||
continue
|
||||
depth = line.count("{") - line.count("}")
|
||||
j = i
|
||||
while depth > 0 and j + 1 < len(lines):
|
||||
j += 1
|
||||
depth += lines[j].count("{") - lines[j].count("}")
|
||||
block = lines[i:j + 1]
|
||||
if any(target in block_line for block_line in block[1:]):
|
||||
out.append(block[0].strip())
|
||||
i = j + 1
|
||||
return out
|
||||
|
||||
|
||||
def _build_review_context(root: Path, rtype: str, rname: str, block_lines: list[str]) -> dict[str, object]:
|
||||
joined = "\n".join(block_lines)
|
||||
variable_names = set(_VAR_REF_RE.findall(joined))
|
||||
local_names = set(_LOCAL_REF_RE.findall(joined))
|
||||
policy_doc_names = set(_DATA_REF_RE.findall(joined))
|
||||
related_blocks: list[str] = []
|
||||
related_blocks.extend(_collect_related_policy_docs(root, policy_doc_names))
|
||||
if rtype == "aws_security_group":
|
||||
related_blocks.extend(_collect_related_sg_rules(root, rname))
|
||||
return {
|
||||
"variables": _collect_variable_defaults(root, variable_names),
|
||||
"locals": _collect_locals(root, local_names),
|
||||
"related_blocks": related_blocks,
|
||||
}
|
||||
|
||||
|
||||
def find_block(source_dir: Path | str, local_address: str) -> BlockLocation | None:
|
||||
"""Search .tf files in `source_dir` for a resource matching `local_address`.
|
||||
|
||||
`local_address` is `<type>.<name>` (e.g. `aws_iam_role.svc`). Returns the
|
||||
first match found. Returns None if no match.
|
||||
"""
|
||||
if "." not in local_address:
|
||||
return None
|
||||
rtype, _, rname = local_address.partition(".")
|
||||
root = Path(source_dir)
|
||||
for tf in sorted(root.glob("*.tf")):
|
||||
for blk in find_resource_blocks(tf):
|
||||
if blk.type == rtype and blk.name == rname:
|
||||
lines = tf.read_text().splitlines()
|
||||
block_lines = lines[blk.start - 1: blk.end]
|
||||
header = block_lines[0] if block_lines else ""
|
||||
evidence_line = header
|
||||
for line in block_lines[1:]:
|
||||
if line.strip():
|
||||
evidence_line = line.strip()
|
||||
break
|
||||
return BlockLocation(
|
||||
file=tf,
|
||||
start_line=blk.start,
|
||||
end_line=blk.end,
|
||||
text="\n".join(block_lines),
|
||||
header=header,
|
||||
evidence_line=evidence_line,
|
||||
key_attributes=_extract_key_attributes(block_lines[1:]),
|
||||
review_context=_build_review_context(root, rtype, rname, block_lines),
|
||||
)
|
||||
return None
|
||||
@@ -0,0 +1,53 @@
|
||||
"""Append-only JSONL telemetry for audit-terraform runs."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def _now() -> str:
|
||||
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
||||
|
||||
|
||||
def _append(log_path: Path, record: dict) -> None:
|
||||
log_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with log_path.open("a", encoding="utf-8") as f:
|
||||
f.write(json.dumps(record) + "\n")
|
||||
|
||||
|
||||
def append_subagent_run(
|
||||
log_path: Path, *, run_id: str, repo: str, mode: str, agent: str,
|
||||
model: str, input_tokens: int, output_tokens: int,
|
||||
duration_ms: int, finding_count: int,
|
||||
) -> None:
|
||||
_append(log_path, {
|
||||
"kind": "subagent_run", "ts": _now(),
|
||||
"run_id": run_id, "repo": repo, "mode": mode, "agent": agent,
|
||||
"model": model,
|
||||
"input_tokens": input_tokens, "output_tokens": output_tokens,
|
||||
"duration_ms": duration_ms, "finding_count": finding_count,
|
||||
})
|
||||
|
||||
|
||||
def append_verdict(
|
||||
log_path: Path, *, run_id: str, agent: str, rule_id: str,
|
||||
file: str, line: int, verdict: str, notes: str = "",
|
||||
) -> None:
|
||||
if verdict not in {"kept", "dismissed", "false_positive"}:
|
||||
raise ValueError(f"invalid verdict: {verdict!r}")
|
||||
_append(log_path, {
|
||||
"kind": "verdict", "ts": _now(),
|
||||
"run_id": run_id, "agent": agent, "rule_id": rule_id,
|
||||
"file": file, "line": line, "verdict": verdict, "notes": notes,
|
||||
})
|
||||
|
||||
|
||||
def read_runs(log_path: Path) -> list[dict]:
|
||||
if not log_path.exists():
|
||||
return []
|
||||
return [
|
||||
json.loads(line)
|
||||
for line in log_path.read_text(encoding="utf-8").splitlines()
|
||||
if line.strip()
|
||||
]
|
||||
Reference in New Issue
Block a user