Move the code and terraform audits into the reviews plugin

Copy the standalone code-review and terraform-review skills into
plugins/reviews as audit-code and audit-terraform. The rename separates the
automated, linter-driven audits from the guided review-pr walkthrough that
already lived here.

Resolve bundled script paths through ${SKILL_DIR}, exported in a new step 0.
CLAUDE_PLUGIN_ROOT is not set in the Bash tool environment, so the obvious
substitution would have expanded to nothing and broken every collection
script invocation.

Replace the PLAN and DESIGN docs with READMEs written from the current
SKILL.md and scripts. The old docs had drifted badly: they named semgrep
where the code calls opengrep, scoped five review agents where there are
now eight, and predated Lua, PowerShell, and GitHub Actions support.

Add CONSISTENCY_NORMS to the audit-terraform agent inputs. The collection
script writes consistency_norms.json and the agent prompt declares it, but
SKILL.md never listed it, leaving the variable unsubstituted.

Drop the --ingest-verdicts instruction from both skills. review_stats.py
parses no arguments, so the ref-mode verdict template it told users to feed
back could never be read.

Point audit-terraform's smoke test at README.md and resolve its fixture
paths relative to the test file rather than an absolute home directory.

Tests: 197 passing (audit-code), 106 passing (audit-terraform).
This commit is contained in:
2026-07-21 11:11:05 -05:00
parent 600c1fef86
commit f5934181ec
179 changed files with 20779 additions and 3 deletions
@@ -0,0 +1,132 @@
"""Reconcile plan-detected and diff-detected resources into a catalog.
Schema: one entry per (source_dir, local_address). Each entry carries a list
of CatalogInstance (plan_dir, address_at_plan, action) so a module reviewed
at N callsites yields one entry, not N.
"""
from __future__ import annotations
from dataclasses import dataclass
from scripts.manifest import CatalogEntry, CatalogInstance, ModuleGraphEntry
@dataclass(frozen=True)
class PlanHit:
address: str
type: str
action: str
@dataclass(frozen=True)
class DiffHit:
source_dir: str
local_address: str
type: str
def _strip_module_prefix(address: str) -> tuple[list[str], str]:
parts = address.split(".")
locals_: list[str] = []
while len(parts) >= 2 and parts[0] == "module":
locals_.append(parts[1])
parts = parts[2:]
return locals_, ".".join(parts)
def _plan_source_dir(plan_dir: str, address: str,
module_graph: dict[str, ModuleGraphEntry]) -> str:
module_locals, _ = _strip_module_prefix(address)
if not module_locals:
return plan_dir
current_dir = plan_dir
for module_local in module_locals:
next_dir = None
has_explicit_mapping = False
for entry in module_graph.values():
local_names = entry.callsite_local_names.get(current_dir)
if local_names is None:
continue
has_explicit_mapping = True
if module_local in local_names:
next_dir = local_names[module_local]
break
if next_dir is None:
if has_explicit_mapping:
return current_dir
unique_candidates = sorted(
mod_dir for mod_dir, entry in module_graph.items()
if current_dir in entry.callsites
)
if len(unique_candidates) != 1:
return current_dir
next_dir = unique_candidates[0]
current_dir = next_dir
return current_dir
def build_catalog(
plan_hits: dict[str, list[PlanHit]],
diff_hits: list[DiffHit],
module_graph: dict[str, ModuleGraphEntry],
) -> list[CatalogEntry]:
by_key: dict[tuple[str, str], CatalogEntry] = {}
def _get(source_dir: str, local_addr: str, type_: str) -> CatalogEntry:
k = (source_dir, local_addr)
if k not in by_key:
by_key[k] = CatalogEntry(
source_dir=source_dir,
local_address=local_addr,
type=type_,
source="plan",
instances=[],
)
return by_key[k]
for plan_dir, hits in plan_hits.items():
for h in hits:
source_dir = _plan_source_dir(plan_dir, h.address, module_graph)
_, local_addr = _strip_module_prefix(h.address)
entry = _get(source_dir, local_addr, h.type)
entry.instances.append(CatalogInstance(
plan_dir=plan_dir, address_at_plan=h.address, action=h.action,
))
for d in diff_hits:
k = (d.source_dir, d.local_address)
if k in by_key:
existing = by_key[k]
by_key[k] = CatalogEntry(
source_dir=existing.source_dir,
local_address=existing.local_address,
type=existing.type,
source="both",
instances=list(existing.instances),
)
continue
graph_entry = module_graph.get(d.source_dir)
if graph_entry and graph_entry.callsites:
instances = [
CatalogInstance(
plan_dir=cs,
address_at_plan=f"module.<{d.source_dir}>.{d.local_address}",
action="unknown",
)
for cs in graph_entry.callsites
]
else:
instances = [CatalogInstance(
plan_dir=d.source_dir,
address_at_plan=d.local_address,
action="unknown",
)]
by_key[k] = CatalogEntry(
source_dir=d.source_dir,
local_address=d.local_address,
type=d.type,
source="diff",
instances=instances,
)
return sorted(by_key.values(), key=lambda e: (e.source_dir, e.local_address))
@@ -0,0 +1,309 @@
"""collect-changes.py — produce a audit-terraform manifest.json."""
from __future__ import annotations
import argparse
import concurrent.futures as cf
import json
import subprocess
import sys
from pathlib import Path
_HERE = Path(__file__).resolve().parent
if str(_HERE.parent) not in sys.path:
sys.path.insert(0, str(_HERE.parent))
from scripts.catalog import build_catalog, DiffHit, PlanHit
from scripts.git_diff import (
changed_files,
changed_line_ranges,
is_terragrunt_file,
)
from scripts.hcl_diff import touched_resources
from scripts.manifest import Manifest, PlanResult, PlanUnit
from scripts.module_graph import build_module_graph
from scripts.scanners import _run_trivy_config, _run_tflint
from scripts.slicing import _slice_for_agent
from scripts.plan_output import parse_plan_resources
from scripts.plan_runner import detect_tool, run_init, run_plan
from scripts.reference_set import compute_consistency_norms, compute_reference_sets
from scripts.resolve_plan_units import resolve_plan_units
from scripts.source_lookup import find_block
_PLAN_CONCURRENCY = 8
def _resolve_default_branch(repo: Path) -> str:
try:
r = subprocess.run(
["git", "-C", str(repo), "symbolic-ref", "refs/remotes/origin/HEAD"],
capture_output=True, text=True, check=False,
)
if r.returncode == 0:
return r.stdout.strip().rsplit("/", 1)[-1]
except FileNotFoundError:
pass
for candidate in ("main", "master"):
r = subprocess.run(
["git", "-C", str(repo), "rev-parse", f"origin/{candidate}"],
capture_output=True, text=True, check=False,
)
if r.returncode == 0:
return candidate
return "main"
def _resolve_base_ref(repo: Path, branch: str) -> str:
"""Return the diff base for `branch`, fetching origin first so the
comparison is against the canonical remote tip rather than a stale local
branch. Falls back to the local branch name if origin isn't reachable.
"""
try:
subprocess.run(
["git", "-C", str(repo), "fetch", "--quiet", "--no-tags",
"origin", branch],
capture_output=True, text=True, check=False, timeout=60,
)
except (FileNotFoundError, subprocess.TimeoutExpired):
pass
check = subprocess.run(
["git", "-C", str(repo), "rev-parse", "--verify", "--quiet",
f"origin/{branch}"],
capture_output=True, text=True, check=False,
)
return f"origin/{branch}" if check.returncode == 0 else branch
def _safe_plan_filename(plan_dir: str) -> str:
"""Map a plan_dir like '.' or 'live/prod/app' to a stable, filename-safe stem.
'.' becomes 'root' (otherwise we'd produce '..txt' on disk).
"""
cleaned = plan_dir.replace("/", "_").strip(".")
return cleaned or "root"
def _run_one(
repo: Path,
plan_dir: str,
triggered_by: list[str],
changed_files_for_unit: list[str],
out_dir: Path,
) -> tuple[PlanUnit, list[PlanHit], str | None]:
full = repo / plan_dir
tool = detect_tool(full)
terragrunt_changed = any(
is_terragrunt_file(changed_file)
for changed_file in changed_files_for_unit
)
init_res = run_init(full, tool)
if not init_res.ok:
plan_res = PlanResult(ok=False, stdout_path="", exit_code=-1,
summary="init-failed")
err = f"init failed in {plan_dir}: {init_res.stderr_tail.strip()}"
return (
PlanUnit(plan_dir=plan_dir, tool=tool, init=init_res,
plan=plan_res, triggered_by=triggered_by,
terragrunt_changed=terragrunt_changed,
changed_files=changed_files_for_unit),
[], err,
)
stdout_path = out_dir / "plans" / f"{_safe_plan_filename(plan_dir)}.txt"
plan_res = run_plan(full, tool, stdout_path)
if not plan_res.ok:
captured = Path(plan_res.stdout_path).read_text()[-1000:].strip()
err = f"plan failed in {plan_dir} (exit {plan_res.exit_code}): {captured}"
return (
PlanUnit(plan_dir=plan_dir, tool=tool, init=init_res,
plan=plan_res, triggered_by=triggered_by,
terragrunt_changed=terragrunt_changed,
changed_files=changed_files_for_unit),
[], err,
)
parsed = parse_plan_resources(Path(plan_res.stdout_path).read_text())
hits = [
PlanHit(address=p.address, type=p.type, action=p.action) for p in parsed
]
return (
PlanUnit(plan_dir=plan_dir, tool=tool, init=init_res,
plan=plan_res, triggered_by=triggered_by,
terragrunt_changed=terragrunt_changed,
changed_files=changed_files_for_unit),
hits, None,
)
def _changed_files_for_plan_unit(
all_changed_files: list[str],
triggered_by: list[str],
) -> list[str]:
triggered_dirs = set(triggered_by)
return sorted(
changed_file
for changed_file in all_changed_files
if str(Path(changed_file).parent.as_posix()) in triggered_dirs
)
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--repo", required=True)
parser.add_argument("--base", default=None)
parser.add_argument("--head", default="HEAD")
parser.add_argument("--output-dir", required=True)
parser.add_argument("--mode", choices=("local", "ref"), required=True)
args = parser.parse_args(argv)
repo = Path(args.repo).resolve()
out_dir = Path(args.output_dir).resolve()
out_dir.mkdir(parents=True, exist_ok=True)
default_branch = _resolve_default_branch(repo)
base = args.base or _resolve_base_ref(repo, default_branch)
errors: list[str] = []
try:
files = changed_files(str(repo), base, args.head)
dirs = {str(Path(path).parent.as_posix()) or "." for path in files}
terragrunt_files = {path for path in files if is_terragrunt_file(path)}
except RuntimeError as e:
manifest = Manifest(
base_ref=base, head_ref=args.head, mode=args.mode,
default_branch=default_branch,
changed_source_dirs=[], plan_units=[], catalog=[],
trivy_findings=[],
module_graph={}, errors=[str(e)],
)
(out_dir / "manifest.json").write_text(manifest.to_json())
return 1
if not files:
manifest = Manifest(
base_ref=base, head_ref=args.head, mode=args.mode,
default_branch=default_branch,
changed_source_dirs=[], plan_units=[], catalog=[],
trivy_findings=[],
module_graph={}, errors=["no terraform/hcl files changed"],
)
(out_dir / "manifest.json").write_text(manifest.to_json())
return 0
module_graph = build_module_graph(repo)
plan_units_map, orphan_modules = resolve_plan_units(repo, dirs, module_graph)
for orphan in orphan_modules:
errors.append(f"module has no callsites; diff-only review: {orphan}")
plan_hits: dict[str, list[PlanHit]] = {}
plan_unit_records: list[PlanUnit] = []
if plan_units_map:
with cf.ThreadPoolExecutor(max_workers=_PLAN_CONCURRENCY) as ex:
futures = {
ex.submit(
_run_one,
repo,
pd,
tb,
_changed_files_for_plan_unit(files, tb),
out_dir,
): pd
for pd, tb in plan_units_map.items()
}
for fut in cf.as_completed(futures):
record, hits, err = fut.result()
plan_unit_records.append(record)
if err:
errors.append(err)
else:
plan_hits[record.plan_dir] = hits
diff_hits: list[DiffHit] = []
for f in files:
if not f.endswith(".tf"):
continue
try:
ranges = changed_line_ranges(str(repo), base, args.head, f)
except RuntimeError as e:
errors.append(f"diff scan failed for {f}: {e}")
continue
if not ranges:
continue
blocks = touched_resources(repo / f, ranges)
src_dir = str(Path(f).parent.as_posix()) or "."
for b in blocks:
diff_hits.append(DiffHit(
source_dir=src_dir,
local_address=f"{b.type}.{b.name}",
type=b.type,
))
if terragrunt_files:
changed_plan_dirs = {pu.plan_dir for pu in plan_unit_records}
for terragrunt_file in sorted(terragrunt_files):
src_dir = str(Path(terragrunt_file).parent.as_posix()) or "."
if src_dir not in changed_plan_dirs:
errors.append(
f"terragrunt change has no planned unit context: {terragrunt_file}"
)
trivy_payload, trivy_findings, trivy_error = _run_trivy_config(repo, files)
(out_dir / "trivy-findings.json").write_text(json.dumps(trivy_payload, indent=2))
if trivy_error:
errors.append(trivy_error)
tflint_payload, tflint_findings, tflint_error = _run_tflint(repo, files)
(out_dir / "tflint-findings.json").write_text(json.dumps(tflint_payload, indent=2))
if tflint_error:
errors.append(tflint_error)
catalog = build_catalog(plan_hits, diff_hits, module_graph)
for entry in catalog:
loc = find_block(repo / entry.source_dir, entry.local_address)
if loc is None:
continue
entry.block_header = loc.header
entry.evidence_line = loc.evidence_line
entry.key_attributes = loc.key_attributes
entry.review_context = loc.review_context
entry.block_file = loc.file.name
entry.block_start = loc.start_line
entry.block_end = loc.end_line
ref_sets = compute_reference_sets(repo, dirs, module_graph)
(out_dir / "reference_sets.json").write_text(
json.dumps(ref_sets, indent=2)
)
consistency_norms = compute_consistency_norms(repo, ref_sets)
(out_dir / "consistency_norms.json").write_text(
json.dumps(consistency_norms, indent=2)
)
manifest = Manifest(
base_ref=base,
head_ref=args.head,
mode=args.mode,
default_branch=default_branch,
changed_source_dirs=sorted(dirs),
plan_units=sorted(plan_unit_records, key=lambda p: p.plan_dir),
catalog=catalog,
trivy_findings=trivy_findings,
tflint_findings=tflint_findings,
module_graph=module_graph,
errors=errors,
)
(out_dir / "manifest.json").write_text(manifest.to_json())
manifest_dict = manifest.to_dict()
for agent in ("fsbp", "cis", "aws-bp", "consistency", "tf-hygiene", "walkthrough"):
sliced = _slice_for_agent(manifest_dict, agent)
(out_dir / f"manifest-{agent}.json").write_text(
json.dumps(sliced, indent=2)
)
return 1 if any(not pu.plan.ok for pu in plan_unit_records) else 0
if __name__ == "__main__":
sys.exit(main())
@@ -0,0 +1,147 @@
"""Map Security Hub control-ID prefixes to terraform resource types.
This table is hand-curated and conservative — when a control could apply to
multiple terraform resource types, list them all. Update as new control
families ship.
"""
from __future__ import annotations
PREFIX_TO_TERRAFORM_TYPES: dict[str, list[str]] = {
"Account": ["aws_account_alternate_contact"],
"ACM": ["aws_acm_certificate"],
"APIGateway": ["aws_api_gateway_rest_api", "aws_apigatewayv2_api",
"aws_api_gateway_stage", "aws_api_gateway_method"],
"AppSync": ["aws_appsync_graphql_api"],
"Athena": ["aws_athena_workgroup"],
"AutoScaling": ["aws_autoscaling_group", "aws_launch_template",
"aws_launch_configuration"],
"Autoscaling": ["aws_autoscaling_group", "aws_launch_template",
"aws_launch_configuration"],
"Backup": ["aws_backup_plan", "aws_backup_vault"],
"BedrockAgentCore": [],
"CloudFormation": ["aws_cloudformation_stack",
"aws_cloudformation_stack_set"],
"CloudFront": ["aws_cloudfront_distribution"],
"CloudTrail": ["aws_cloudtrail"],
"CodeBuild": ["aws_codebuild_project"],
"Cognito": ["aws_cognito_user_pool",
"aws_cognito_identity_pool"],
"Config": ["aws_config_configuration_recorder",
"aws_config_delivery_channel",
"aws_config_config_rule"],
"Connect": ["aws_connect_instance"],
"DataFirehose": ["aws_kinesis_firehose_delivery_stream"],
"DataSync": ["aws_datasync_task"],
"DMS": ["aws_dms_endpoint",
"aws_dms_replication_instance"],
"DocumentDB": ["aws_docdb_cluster", "aws_docdb_cluster_instance"],
"DynamoDB": ["aws_dynamodb_table"],
"EC2": ["aws_instance", "aws_security_group",
"aws_security_group_rule", "aws_vpc", "aws_subnet",
"aws_network_acl", "aws_ebs_volume",
"aws_ebs_default_kms_key", "aws_eip", "aws_nat_gateway",
"aws_internet_gateway", "aws_route_table",
"aws_vpc_endpoint", "aws_flow_log",
"aws_default_security_group"],
"ECR": ["aws_ecr_repository",
"aws_ecr_repository_policy"],
"ECS": ["aws_ecs_cluster", "aws_ecs_service",
"aws_ecs_task_definition"],
"EFS": ["aws_efs_file_system",
"aws_efs_file_system_policy",
"aws_efs_access_point"],
"EKS": ["aws_eks_cluster", "aws_eks_node_group"],
"ELB": ["aws_lb", "aws_alb", "aws_elb",
"aws_lb_listener", "aws_alb_listener",
"aws_lb_target_group"],
"ElastiCache": ["aws_elasticache_cluster",
"aws_elasticache_replication_group"],
"ElasticBeanstalk": ["aws_elastic_beanstalk_environment"],
"ElasticSearch": ["aws_elasticsearch_domain",
"aws_opensearch_domain"],
"EMR": ["aws_emr_cluster"],
"ES": ["aws_elasticsearch_domain"],
"EventBridge": ["aws_cloudwatch_event_rule",
"aws_cloudwatch_event_bus"],
"FSx": ["aws_fsx_lustre_file_system",
"aws_fsx_windows_file_system",
"aws_fsx_openzfs_file_system",
"aws_fsx_ontap_file_system"],
"Glue": ["aws_glue_catalog_database",
"aws_glue_crawler", "aws_glue_job"],
"GuardDuty": ["aws_guardduty_detector"],
"IAM": ["aws_iam_user", "aws_iam_role", "aws_iam_policy",
"aws_iam_user_policy", "aws_iam_role_policy",
"aws_iam_access_key",
"aws_iam_account_password_policy",
"aws_iam_group", "aws_iam_user_policy_attachment",
"aws_iam_role_policy_attachment"],
"Inspector": ["aws_inspector2_enabler"],
"Kinesis": ["aws_kinesis_stream"],
"KMS": ["aws_kms_key", "aws_kms_alias"],
"Lambda": ["aws_lambda_function",
"aws_lambda_permission",
"aws_lambda_function_url"],
"Macie": ["aws_macie2_account"],
"MQ": ["aws_mq_broker"],
"MSK": ["aws_msk_cluster"],
"Neptune": ["aws_neptune_cluster",
"aws_neptune_cluster_instance"],
"NetworkFirewall": ["aws_networkfirewall_firewall",
"aws_networkfirewall_firewall_policy",
"aws_networkfirewall_rule_group"],
"Opensearch": ["aws_opensearch_domain"],
"PCA": ["aws_acmpca_certificate_authority"],
"RDS": ["aws_db_instance", "aws_rds_cluster",
"aws_db_subnet_group", "aws_db_parameter_group",
"aws_rds_cluster_instance",
"aws_db_snapshot",
"aws_db_event_subscription"],
"Redshift": ["aws_redshift_cluster",
"aws_redshift_parameter_group"],
"RedshiftServerless": ["aws_redshiftserverless_namespace",
"aws_redshiftserverless_workgroup"],
"Route53": ["aws_route53_zone"],
"S3": ["aws_s3_bucket", "aws_s3_bucket_policy",
"aws_s3_bucket_public_access_block",
"aws_s3_bucket_versioning",
"aws_s3_bucket_server_side_encryption_configuration",
"aws_s3_bucket_logging",
"aws_s3_bucket_lifecycle_configuration",
"aws_s3_bucket_acl"],
"SageMaker": ["aws_sagemaker_notebook_instance",
"aws_sagemaker_endpoint_configuration",
"aws_sagemaker_model"],
"SecretsManager": ["aws_secretsmanager_secret",
"aws_secretsmanager_secret_rotation"],
"ServiceCatalog": ["aws_servicecatalog_portfolio"],
"SES": ["aws_ses_configuration_set",
"aws_ses_domain_identity"],
"SNS": ["aws_sns_topic", "aws_sns_topic_policy"],
"SQS": ["aws_sqs_queue"],
"SSM": ["aws_ssm_document",
"aws_ssm_parameter",
"aws_ssm_association"],
"StepFunctions": ["aws_sfn_state_machine"],
"Transfer": ["aws_transfer_server", "aws_transfer_user"],
"WAF": ["aws_wafv2_web_acl", "aws_waf_web_acl",
"aws_wafv2_web_acl_association"],
"WorkSpaces": ["aws_workspaces_directory", "aws_workspaces_workspace"],
}
def control_id_to_types(control_id: str) -> list[str]:
"""Given e.g. 'FSBP S3.5' or 'S3.5', return likely terraform types.
Strips the leading 'FSBP '/'CIS ' source prefix, then matches the
service prefix (the segment before the first dot).
"""
s = control_id.strip()
for source in ("FSBP", "CIS"):
prefix = f"{source} "
if s.startswith(prefix):
s = s[len(prefix):]
break
service, _, _ = s.partition(".")
return list(PREFIX_TO_TERRAFORM_TYPES.get(service, []))
@@ -0,0 +1,56 @@
"""Schema and serialization for AWS controls cache files."""
from __future__ import annotations
import json
from dataclasses import dataclass, asdict
from datetime import datetime
from typing import Literal
Severity = Literal["critical", "high", "medium", "low", "informational"]
ControlSource = Literal["fsbp", "cis", "aws-bp"]
@dataclass
class Control:
control_id: str
title: str
severity: Severity
resource_types: list[str]
requirement: str
source_url: str
def to_dict(self) -> dict:
return asdict(self)
@dataclass
class ControlsFile:
source: ControlSource
fetched_at: datetime
controls: list[Control]
def _by_resource_type(self) -> dict[str, list[str]]:
"""Map terraform resource type -> list of control_ids.
Full control records live in `controls`. Looking up details by ID
from `controls` keeps the index small.
"""
index: dict[str, list[str]] = {}
for c in self.controls:
for rt in c.resource_types:
index.setdefault(rt, []).append(c.control_id)
for rt in index:
index[rt].sort()
return dict(sorted(index.items()))
def to_dict(self) -> dict:
return {
"source": self.source,
"fetched_at": self.fetched_at.isoformat(),
"controls": [c.to_dict() for c in self.controls],
"by_resource_type": self._by_resource_type(),
}
def to_json(self, indent: int = 2) -> str:
return json.dumps(self.to_dict(), indent=indent, sort_keys=False)
@@ -0,0 +1,97 @@
"""Git diff scanning: changed files, dirs, and per-file added-line ranges."""
from __future__ import annotations
import re
import subprocess
from pathlib import PurePosixPath
_HCL_SUFFIXES = (".tf", ".tf.json", ".hcl")
_HUNK_RE = re.compile(r"^@@ -\d+(?:,\d+)? \+(\d+)(?:,(\d+))? @@")
def _run_git_diff(repo: str, base: str, head: str) -> str:
result = subprocess.run(
[
"git", "-C", repo,
"-c", "diff.noprefix=false",
"-c", "color.ui=never",
"diff", "--no-ext-diff", "--unified=0", f"{base}...{head}"
],
capture_output=True, text=True, check=False,
)
if result.returncode != 0:
raise RuntimeError(f"git diff failed: {result.stderr.strip()}")
return result.stdout
def _iter_file_blocks(diff_text: str):
"""Yield (path, block_text) for each file section in a unified diff."""
current_path: str | None = None
buf: list[str] = []
for line in diff_text.splitlines():
if line.startswith("diff --git "):
if current_path is not None:
yield current_path, "\n".join(buf)
current_path = None
buf = []
elif line.startswith("+++ b/"):
current_path = line[len("+++ b/"):]
if current_path is not None:
buf.append(line)
if current_path is not None:
yield current_path, "\n".join(buf)
def changed_files(repo: str, base: str, head: str) -> list[str]:
out: list[str] = []
for path, _ in _iter_file_blocks(_run_git_diff(repo, base, head)):
if path.endswith(_HCL_SUFFIXES):
out.append(path)
return out
def is_terragrunt_file(path: str) -> bool:
return PurePosixPath(path).name == "terragrunt.hcl"
def changed_dirs(repo: str, base: str, head: str) -> set[str]:
return {str(PurePosixPath(p).parent) for p in changed_files(repo, base, head)}
def changed_line_ranges(
repo: str, base: str, head: str, path: str,
) -> list[tuple[int, int]]:
"""Return inclusive (start, end) line ranges of *added* lines in `path`."""
ranges: list[tuple[int, int]] = []
for fpath, block in _iter_file_blocks(_run_git_diff(repo, base, head)):
if fpath != path:
continue
cur: int | None = None
run_start: int | None = None
run_end: int | None = None
for line in block.splitlines():
m = _HUNK_RE.match(line)
if m:
if run_start is not None:
ranges.append((run_start, run_end)) # type: ignore[arg-type]
run_start = run_end = None
cur = int(m.group(1))
continue
if cur is None:
continue
if line.startswith("+") and not line.startswith("+++"):
if run_start is None:
run_start = cur
run_end = cur
cur += 1
elif line.startswith("-") and not line.startswith("---"):
continue
else:
if run_start is not None:
ranges.append((run_start, run_end)) # type: ignore[arg-type]
run_start = run_end = None
cur += 1
if run_start is not None:
ranges.append((run_start, run_end)) # type: ignore[arg-type]
return ranges
@@ -0,0 +1,64 @@
"""Locate `resource` blocks in .tf files and intersect with changed lines."""
from __future__ import annotations
import re
from dataclasses import dataclass
from pathlib import Path
_HEADER_RE = re.compile(
r'^\s*resource\s+"(?P<type>[^"]+)"\s+"(?P<name>[^"]+)"\s*\{'
)
@dataclass(frozen=True)
class ResourceBlock:
type: str
name: str
start: int
end: int
def find_resource_blocks(path: Path) -> list[ResourceBlock]:
"""Locate top-level `resource` blocks by tracking brace depth."""
text = path.read_text()
lines = text.splitlines()
blocks: list[ResourceBlock] = []
i = 0
while i < len(lines):
line = lines[i]
m = _HEADER_RE.match(line)
if not m:
i += 1
continue
start = i + 1
depth = line.count("{") - line.count("}")
j = i
while depth > 0 and j + 1 < len(lines):
j += 1
depth += lines[j].count("{") - lines[j].count("}")
blocks.append(ResourceBlock(
type=m.group("type"),
name=m.group("name"),
start=start,
end=j + 1,
))
i = j + 1
return blocks
def _intersects(a: tuple[int, int], b: tuple[int, int]) -> bool:
return not (a[1] < b[0] or b[1] < a[0])
def touched_resources(
path: Path, added_ranges: list[tuple[int, int]],
) -> list[ResourceBlock]:
if not added_ranges:
return []
blocks = find_resource_blocks(path)
out: list[ResourceBlock] = []
for blk in blocks:
if any(_intersects((blk.start, blk.end), r) for r in added_ranges):
out.append(blk)
return out
@@ -0,0 +1,84 @@
"""Append one subagent_run row per agent after the fan-out completes.
The orchestrator calls this once after all subagents return. It scans
<output-dir>/findings-<agent>.json for each agent listed in the
--usage-json payload, counts findings, and writes a subagent_run row to
runs.jsonl using token / duration metadata supplied by the orchestrator.
Usage:
python scripts/log-run.py \\
--output-dir <OUTPUT> --run-id <hex> --repo <path> --mode <local|ref> \\
--usage-json - <<JSON
{
"walkthrough-reviewer": {"model":"sonnet","input_tokens":1234,"output_tokens":567,"duration_ms":4500},
"aws-bp-reviewer": {"model":"sonnet","input_tokens":2345,"output_tokens":678,"duration_ms":5200}
}
JSON
`--log-path` defaults to ~/.claude/cache/audit-terraform/runs.jsonl.
"""
from __future__ import annotations
import argparse
import json
import sys
from pathlib import Path
from scripts.telemetry import append_subagent_run
_DEFAULT_LOG = Path.home() / ".claude/cache/audit-terraform/runs.jsonl"
def _count_findings(path: Path) -> int:
if not path.exists():
return 0
try:
data = json.loads(path.read_text(encoding="utf-8"))
except json.JSONDecodeError:
return 0
if isinstance(data, dict):
findings = data.get("findings")
if isinstance(findings, list):
return len(findings)
elif isinstance(data, list):
return len(data)
return 0
def main(argv: list[str]) -> int:
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--output-dir", required=True, type=Path)
p.add_argument("--run-id", required=True)
p.add_argument("--repo", required=True)
p.add_argument("--mode", required=True, choices=["local", "ref"])
p.add_argument("--log-path", type=Path, default=_DEFAULT_LOG)
p.add_argument("--usage-json", required=True,
help="Path to JSON file, or '-' for stdin.")
args = p.parse_args(argv)
raw = sys.stdin.read() if args.usage_json == "-" else Path(args.usage_json).read_text(encoding="utf-8")
usage = json.loads(raw)
if not isinstance(usage, dict):
print("usage-json must be a JSON object keyed by agent name", file=sys.stderr)
return 2
for agent, meta in usage.items():
if not isinstance(meta, dict):
print(f"skipping {agent}: usage entry not an object", file=sys.stderr)
continue
findings_path = args.output_dir / f"findings-{agent}.json"
append_subagent_run(
args.log_path,
run_id=args.run_id, repo=args.repo, mode=args.mode, agent=agent,
model=str(meta.get("model", "?")),
input_tokens=int(meta.get("input_tokens", 0)),
output_tokens=int(meta.get("output_tokens", 0)),
duration_ms=int(meta.get("duration_ms", 0)),
finding_count=_count_findings(findings_path),
)
return 0
if __name__ == "__main__":
raise SystemExit(main(sys.argv[1:]))
@@ -0,0 +1,182 @@
"""Manifest dataclasses for collect-changes.py output.
The manifest is the contract between the script and the review subagents.
Schema mirrors DESIGN.md.
"""
from __future__ import annotations
import json
from dataclasses import dataclass, field, asdict
from typing import Literal
Mode = Literal["local", "ref"]
Tool = Literal["terragrunt", "tofu"]
Action = Literal["create", "update", "delete", "replace", "read", "no-op", "unknown"]
Source = Literal["plan", "diff", "both"]
@dataclass
class InitResult:
ok: bool
stdout_tail: str
stderr_tail: str
def to_dict(self) -> dict:
return asdict(self)
@dataclass
class PlanResult:
ok: bool
stdout_path: str
exit_code: int
summary: str
def to_dict(self) -> dict:
return asdict(self)
@dataclass
class PlanUnit:
plan_dir: str
tool: Tool
init: InitResult
plan: PlanResult
triggered_by: list[str]
terragrunt_changed: bool = False
changed_files: list[str] = field(default_factory=list)
def to_dict(self) -> dict:
return {
"plan_dir": self.plan_dir,
"tool": self.tool,
"init": self.init.to_dict(),
"plan": self.plan.to_dict(),
"triggered_by": list(self.triggered_by),
"terragrunt_changed": self.terragrunt_changed,
"changed_files": list(self.changed_files),
}
@dataclass
class TrivyFinding:
check_id: str
title: str
severity: str
message: str
file: str
start_line: int = 0
end_line: int = 0
resource_type: str = ""
source: str = "trivy"
def to_dict(self) -> dict:
return asdict(self)
@dataclass
class TflintFinding:
rule: str
severity: str
message: str
file: str
start_line: int = 0
end_line: int = 0
link: str = ""
source: str = "tflint"
def to_dict(self) -> dict:
return asdict(self)
@dataclass
class CatalogInstance:
plan_dir: str
address_at_plan: str
action: Action
def to_dict(self) -> dict:
return asdict(self)
@dataclass
class CatalogEntry:
source_dir: str
local_address: str
type: str
source: Source
instances: list[CatalogInstance] = field(default_factory=list)
block_header: str = ""
evidence_line: str = ""
key_attributes: dict[str, str | bool | int | list[str]] = field(default_factory=dict)
review_context: dict[str, object] = field(default_factory=dict)
block_file: str = ""
block_start: int = 0
block_end: int = 0
def to_dict(self) -> dict:
return {
"source_dir": self.source_dir,
"local_address": self.local_address,
"type": self.type,
"source": self.source,
"instances": [i.to_dict() for i in self.instances],
"block_header": self.block_header,
"evidence_line": self.evidence_line,
"key_attributes": dict(self.key_attributes),
"review_context": dict(self.review_context),
"block_file": self.block_file,
"block_start": self.block_start,
"block_end": self.block_end,
}
@dataclass
class ModuleGraphEntry:
callsites: list[str]
sibling_modules_at_callsites: list[str] = field(default_factory=list)
callsite_local_names: dict[str, dict[str, str]] = field(default_factory=dict)
def to_dict(self) -> dict:
return {
"callsites": list(self.callsites),
"sibling_modules_at_callsites": list(self.sibling_modules_at_callsites),
"callsite_local_names": {
callsite: dict(local_names)
for callsite, local_names in self.callsite_local_names.items()
},
}
@dataclass
class Manifest:
base_ref: str
head_ref: str
mode: Mode
default_branch: str
changed_source_dirs: list[str]
plan_units: list[PlanUnit]
catalog: list[CatalogEntry]
trivy_findings: list[TrivyFinding]
module_graph: dict[str, ModuleGraphEntry]
errors: list[str]
tflint_findings: list[TflintFinding] = field(default_factory=list)
def to_dict(self) -> dict:
return {
"base_ref": self.base_ref,
"head_ref": self.head_ref,
"mode": self.mode,
"default_branch": self.default_branch,
"changed_source_dirs": list(self.changed_source_dirs),
"plan_units": [p.to_dict() for p in self.plan_units],
"catalog": [c.to_dict() for c in self.catalog],
"trivy_findings": [f.to_dict() for f in self.trivy_findings],
"tflint_findings": [f.to_dict() for f in self.tflint_findings],
"module_graph": {k: v.to_dict() for k, v in self.module_graph.items()},
"errors": list(self.errors),
}
def to_json(self, indent: int = 2) -> str:
return json.dumps(self.to_dict(), indent=indent, sort_keys=False)
@@ -0,0 +1,90 @@
"""Repo-wide module callsite graph.
For each local-source module dir, list callsite dirs and sibling modules
(other modules instantiated alongside it at any callsite).
"""
from __future__ import annotations
from pathlib import Path, PurePosixPath
import hcl2
from scripts.manifest import ModuleGraphEntry
def _strip_hcl_quotes(s):
if isinstance(s, str) and len(s) >= 2 and s[0] == '"' and s[-1] == '"':
return s[1:-1]
return s
def _iter_module_blocks(tf_file: Path):
try:
with tf_file.open() as fh:
parsed = hcl2.load(fh)
except Exception:
return
for block in parsed.get("module", []):
if not isinstance(block, dict):
continue
for name, body in block.items():
if isinstance(body, dict):
src = body.get("source")
if isinstance(src, list):
src = src[0] if src else None
src = _strip_hcl_quotes(src)
yield _strip_hcl_quotes(name), src
def _is_local_source(src: str | None) -> bool:
if not isinstance(src, str):
return False
return src.startswith(("./", "../"))
def _resolve_local(callsite_dir: PurePosixPath, source: str) -> str:
combined = (callsite_dir / source).as_posix()
parts: list[str] = []
for part in combined.split("/"):
if part in ("", "."):
continue
if part == "..":
if parts:
parts.pop()
continue
parts.append(part)
return "/".join(parts)
def build_module_graph(repo_root: Path | str) -> dict[str, ModuleGraphEntry]:
root = Path(repo_root)
raw: dict[str, list[str]] = {}
callsite_to_modules: dict[str, dict[str, str]] = {}
for tf in root.rglob("*.tf"):
rel_dir = tf.parent.relative_to(root).as_posix()
callsite_dir = PurePosixPath(rel_dir)
for name, src in _iter_module_blocks(tf):
if not _is_local_source(src):
continue
target = _resolve_local(callsite_dir, src) # type: ignore[arg-type]
raw.setdefault(target, []).append(rel_dir)
callsite_to_modules.setdefault(rel_dir, {})[name] = target
out: dict[str, ModuleGraphEntry] = {}
for mod_dir, callsites in raw.items():
unique_callsites = sorted(set(callsites))
siblings: set[str] = set()
for cs in unique_callsites:
for other in callsite_to_modules.get(cs, {}).values():
if other != mod_dir:
siblings.add(other)
out[mod_dir] = ModuleGraphEntry(
callsites=unique_callsites,
sibling_modules_at_callsites=sorted(siblings),
callsite_local_names={
cs: dict(callsite_to_modules.get(cs, {}))
for cs in unique_callsites
},
)
return out
@@ -0,0 +1,59 @@
"""Parse terraform/tofu/terragrunt plan output for resource actions."""
from __future__ import annotations
import re
from dataclasses import dataclass
@dataclass(frozen=True)
class ParsedResource:
address: str
type: str
action: str
_HEADER_RE = re.compile(
r"^\s*#\s+(?P<addr>\S+)\s+(?:"
r"(?P<create>will be created)|"
r"(?P<update>will be updated in-place)|"
r"(?P<delete>will be destroyed)|"
r"(?P<replace>must be replaced|will be replaced)|"
r"(?P<read>will be read during apply)"
r")\s*$"
)
def _action_from_match(m: re.Match) -> str:
for k in ("create", "update", "delete", "replace", "read"):
if m.group(k):
return k
return "unknown"
def _type_from_address(address: str) -> str:
"""Strip module prefixes and any index suffix; return the resource type."""
parts = address.split(".")
i = 0
while i + 1 < len(parts) and parts[i].startswith("module"):
i += 2
if i < len(parts) and parts[i] == "data":
i += 1
if i >= len(parts):
return ""
type_token = parts[i]
return re.sub(r"\[.*\]$", "", type_token)
def parse_plan_resources(plan_stdout: str) -> list[ParsedResource]:
out: list[ParsedResource] = []
for line in plan_stdout.splitlines():
m = _HEADER_RE.match(line)
if not m:
continue
addr = m.group("addr")
out.append(ParsedResource(
address=addr,
type=_type_from_address(addr),
action=_action_from_match(m),
))
return out
@@ -0,0 +1,64 @@
"""Detect tool, run init + plan, capture output."""
from __future__ import annotations
import re
import subprocess
from pathlib import Path
from typing import Literal
from scripts.manifest import InitResult, PlanResult
Tool = Literal["terragrunt", "tofu"]
_TAIL_BYTES = 4096
_SUMMARY_RE = re.compile(
r"Plan:\s+(\d+\s+to add,\s*\d+\s+to change,\s*\d+\s+to destroy)\."
)
def detect_tool(plan_dir: Path | str) -> Tool:
p = Path(plan_dir)
if (p / "terragrunt.hcl").is_file():
return "terragrunt"
return "tofu"
def _tail(s: str, n: int = _TAIL_BYTES) -> str:
return s[-n:] if len(s) > n else s
def extract_summary(plan_stdout: str) -> str:
m = _SUMMARY_RE.search(plan_stdout)
if m:
return re.sub(r"\s+", " ", m.group(1)).strip()
if "No changes." in plan_stdout or "no changes." in plan_stdout.lower():
return "no changes"
return "unknown"
def run_init(plan_dir: Path | str, tool: Tool) -> InitResult:
args = [tool, "init", "-input=false", "-no-color"]
result = subprocess.run(
args, cwd=str(plan_dir), capture_output=True, text=True, check=False,
)
return InitResult(
ok=(result.returncode == 0),
stdout_tail=_tail(result.stdout),
stderr_tail=_tail(result.stderr),
)
def run_plan(plan_dir: Path | str, tool: Tool, stdout_path: Path) -> PlanResult:
args = [tool, "plan", "-input=false", "-no-color"]
result = subprocess.run(
args, cwd=str(plan_dir), capture_output=True, text=True, check=False,
)
stdout_path.parent.mkdir(parents=True, exist_ok=True)
stdout_path.write_text(result.stdout)
return PlanResult(
ok=(result.returncode == 0),
stdout_path=str(stdout_path),
exit_code=result.returncode,
summary=extract_summary(result.stdout),
)
@@ -0,0 +1,44 @@
"""Classify a directory as a plan unit, a reusable module, or unknown."""
from __future__ import annotations
from enum import Enum
from pathlib import Path
import hcl2
class DirKind(str, Enum):
PLAN_UNIT = "plan_unit"
MODULE = "module"
UNKNOWN = "unknown"
def _has_backend_block(tf_path: Path) -> bool:
try:
with tf_path.open() as fh:
parsed = hcl2.load(fh)
except Exception:
return False
for block in parsed.get("terraform", []):
if isinstance(block, dict) and "backend" in block:
return True
return False
def classify_dir(repo_root: Path | str, rel_dir: str) -> DirKind:
root = Path(repo_root)
full = root / rel_dir
if not full.is_dir():
raise FileNotFoundError(full)
if (full / "terragrunt.hcl").is_file():
return DirKind.PLAN_UNIT
tf_files = sorted(full.glob("*.tf")) + sorted(full.glob("*.tf.json"))
if not tf_files:
return DirKind.UNKNOWN
for tf in tf_files:
if tf.suffix == ".tf" and _has_backend_block(tf):
return DirKind.PLAN_UNIT
return DirKind.MODULE
@@ -0,0 +1,146 @@
"""Compute reference sets for the consistency reviewer."""
from __future__ import annotations
import re
from collections import Counter, defaultdict
from pathlib import Path
from scripts.manifest import ModuleGraphEntry
def _peer_dirs_at_depth(repo_root: Path, dir_path: str) -> list[str]:
parts = dir_path.split("/")
depth = len(parts)
parent = repo_root.joinpath(*parts[:-1]) if depth > 1 else repo_root
if not parent.is_dir():
return []
out: list[str] = []
for p in sorted(parent.iterdir()):
if not p.is_dir():
continue
rel = p.relative_to(repo_root).as_posix()
if rel != dir_path:
out.append(rel)
return out
def _terragrunt_refs(repo_root: Path, dir_path: str) -> list[str]:
"""Region peers + same-component cross-env (layout: live/<env>/<region>/<component>)."""
parts = dir_path.split("/")
if len(parts) != 4:
return _peer_dirs_at_depth(repo_root, dir_path)
live, env, region, component = parts
refs: set[str] = set()
region_dir = repo_root / live / env / region
if region_dir.is_dir():
for p in region_dir.iterdir():
if p.is_dir() and (p / "terragrunt.hcl").is_file():
rel = p.relative_to(repo_root).as_posix()
if rel != dir_path:
refs.add(rel)
live_dir = repo_root / live
if live_dir.is_dir():
for env_dir in live_dir.iterdir():
candidate = env_dir / region / component
if candidate.is_dir() and (candidate / "terragrunt.hcl").is_file():
rel = candidate.relative_to(repo_root).as_posix()
if rel != dir_path:
refs.add(rel)
env_live_dir = repo_root / live / env
if env_live_dir.is_dir():
for region_dir2 in env_live_dir.iterdir():
candidate = region_dir2 / component
if candidate.is_dir() and (candidate / "terragrunt.hcl").is_file():
rel = candidate.relative_to(repo_root).as_posix()
if rel != dir_path:
refs.add(rel)
return sorted(refs)
def compute_reference_sets(
repo_root: Path | str,
changed_dirs: set[str],
module_graph: dict[str, ModuleGraphEntry],
) -> dict[str, list[str]]:
root = Path(repo_root)
out: dict[str, list[str]] = {}
for d in sorted(changed_dirs):
if d.startswith("modules/"):
entry = module_graph.get(d)
out[d] = list(entry.sibling_modules_at_callsites) if entry else []
continue
full = root / d
if (full / "terragrunt.hcl").is_file():
out[d] = _terragrunt_refs(root, d)
continue
out[d] = _peer_dirs_at_depth(root, d)
return out
_ATTR_RE = re.compile(r"^\s*(?P<key>[A-Za-z0-9_]+)\s*=")
_MODULE_RE = re.compile(r'^\s*module\s+"(?P<name>[^"]+)"\s*\{')
def _iter_tf_lines(dir_path: Path) -> list[str]:
lines: list[str] = []
for tf in sorted(dir_path.glob("*.tf")):
lines.extend(tf.read_text().splitlines())
return lines
def _dir_signature(dir_path: Path) -> tuple[set[str], set[str]]:
attrs: set[str] = set()
modules: set[str] = set()
for line in _iter_tf_lines(dir_path):
attr_match = _ATTR_RE.match(line)
if attr_match:
attrs.add(attr_match.group("key"))
module_match = _MODULE_RE.match(line)
if module_match:
modules.add(module_match.group("name"))
return attrs, modules
def compute_consistency_norms(
repo_root: Path | str,
reference_sets: dict[str, list[str]],
) -> dict[str, dict[str, list[dict[str, object]]]]:
root = Path(repo_root)
out: dict[str, dict[str, list[dict[str, object]]]] = {}
for changed_dir, refs in sorted(reference_sets.items()):
attr_support: dict[str, list[str]] = defaultdict(list)
module_support: dict[str, list[str]] = defaultdict(list)
naming_tokens: Counter[str] = Counter()
usable_refs: list[str] = []
for ref in refs:
ref_path = root / ref
if not ref_path.is_dir():
continue
attrs, modules = _dir_signature(ref_path)
if not attrs and not modules:
continue
usable_refs.append(ref)
for attr in attrs:
attr_support[attr].append(ref)
for module in modules:
module_support[module].append(ref)
naming_tokens.update(Path(ref).parts[-1:])
out[changed_dir] = {
"attribute_norms": [
{"attribute": attr, "peer_dirs": sorted(peer_dirs)}
for attr, peer_dirs in sorted(attr_support.items())
if len(peer_dirs) >= 2
],
"module_wrapper_norms": [
{"module": module, "peer_dirs": sorted(peer_dirs)}
for module, peer_dirs in sorted(module_support.items())
if len(peer_dirs) >= 2
],
"naming_norms": [
{"token": token, "peer_dirs": sorted(usable_refs)}
for token, count in sorted(naming_tokens.items())
if count >= 2
],
}
return out
@@ -0,0 +1,95 @@
"""refresh-controls.py — populate data/controls/{fsbp,cis,meta}.json."""
from __future__ import annotations
import argparse
import json
import sys
from datetime import datetime, timezone
from pathlib import Path
import requests
_HERE = Path(__file__).resolve().parent
if str(_HERE.parent) not in sys.path:
sys.path.insert(0, str(_HERE.parent))
from scripts.controls_schema import Control, ControlsFile
from scripts.scrape_cis import CIS_URL, parse_cis_page
from scripts.scrape_fsbp import FSBP_INDEX_URL, parse_fsbp_index
_TIMEOUT = 30
_USER_AGENT = "audit-terraform-controls-refresh/1.0"
def _fetch(url: str) -> str:
r = requests.get(url, headers={"User-Agent": _USER_AGENT}, timeout=_TIMEOUT)
r.raise_for_status()
return r.text
def _now() -> datetime:
return datetime.now(timezone.utc)
def _build_fsbp() -> ControlsFile:
html = _fetch(FSBP_INDEX_URL)
rows = parse_fsbp_index(html, base_url=FSBP_INDEX_URL)
controls = [
Control(
control_id=f"FSBP {r.control_id}",
title=r.title,
severity=r.severity,
resource_types=r.resource_types,
requirement=r.requirement or r.title,
source_url=r.detail_url,
)
for r in rows
]
return ControlsFile(source="fsbp", fetched_at=_now(), controls=controls)
def _build_cis() -> ControlsFile:
html = _fetch(CIS_URL)
rows = parse_cis_page(html, source_url=CIS_URL)
controls = [
Control(
control_id=r.control_id,
title=r.title,
severity=r.severity,
resource_types=r.resource_types,
requirement=r.requirement,
source_url=r.source_url,
)
for r in rows
]
return ControlsFile(source="cis", fetched_at=_now(), controls=controls)
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--output-dir", default=str(_HERE.parent / "data" / "controls"))
args = parser.parse_args(argv)
out_dir = Path(args.output_dir).resolve()
out_dir.mkdir(parents=True, exist_ok=True)
fsbp = _build_fsbp()
cis = _build_cis()
(out_dir / "fsbp.json").write_text(fsbp.to_json())
(out_dir / "cis.json").write_text(cis.to_json())
meta = {
"fsbp": {"url": FSBP_INDEX_URL, "fetched_at": fsbp.fetched_at.isoformat()},
"cis": {"url": CIS_URL, "fetched_at": cis.fetched_at.isoformat()},
}
(out_dir / "meta.json").write_text(json.dumps(meta, indent=2))
print(f"Wrote {len(fsbp.controls)} FSBP controls, {len(cis.controls)} CIS controls "
f"to {out_dir}")
return 0
if __name__ == "__main__":
sys.exit(main())
@@ -0,0 +1,41 @@
"""Map changed dirs to plan units, preserving the direct trigger directories."""
from __future__ import annotations
from pathlib import Path
from scripts.manifest import ModuleGraphEntry
from scripts.plan_unit import classify_dir, DirKind
def resolve_plan_units(
repo_root: Path | str,
changed_dirs: set[str],
module_graph: dict[str, ModuleGraphEntry],
) -> tuple[dict[str, list[str]], list[str]]:
"""Return (plan_units, orphan_modules).
plan_units maps plan_dir -> sorted-unique list of changed dirs that triggered it.
Direct Terragrunt or root-module changes remain in the trigger set so later
manifest assembly can attach per-plan-unit change context.
orphan_modules is a list of changed module dirs with zero callsites.
"""
plan_units: dict[str, set[str]] = {}
orphans: list[str] = []
for d in sorted(changed_dirs):
kind = classify_dir(repo_root, d)
if kind is DirKind.PLAN_UNIT:
plan_units.setdefault(d, set()).add(d)
elif kind is DirKind.MODULE:
entry = module_graph.get(d)
callsites = list(entry.callsites) if entry else []
if not callsites:
orphans.append(d)
continue
for cs in callsites:
plan_units.setdefault(cs, set()).add(d)
return (
{k: sorted(v) for k, v in sorted(plan_units.items())},
sorted(orphans),
)
@@ -0,0 +1,70 @@
"""Aggregate runs.jsonl into precision + token-cost stats."""
from __future__ import annotations
import json
import sys
from pathlib import Path
def compute_stats(log_path: Path) -> dict:
if not log_path.exists():
return {"by_agent": {}, "by_rule": {}, "runs": 0}
by_agent: dict[str, dict] = {}
by_rule: dict[str, dict] = {}
runs: set[str] = set()
for line in log_path.read_text(encoding="utf-8").splitlines():
if not line.strip():
continue
rec = json.loads(line)
agent = rec.get("agent", "?")
if rec["kind"] == "subagent_run":
runs.add(rec["run_id"])
a = by_agent.setdefault(agent, _empty_agent())
a["tokens"] += rec.get("input_tokens", 0) + rec.get("output_tokens", 0)
a["duration_ms"] += rec.get("duration_ms", 0)
a["runs"] += 1
elif rec["kind"] == "verdict":
verdict = rec["verdict"]
a = by_agent.setdefault(agent, _empty_agent())
a["total"] += 1
a[verdict] = a.get(verdict, 0) + 1
rule_key = f"{agent}/{rec['rule_id']}"
r = by_rule.setdefault(rule_key, _empty_rule())
r["total"] += 1
r[verdict] = r.get(verdict, 0) + 1
for a in by_agent.values():
a["precision"] = a["kept"] / a["total"] if a["total"] else 0.0
a["tokens_per_kept"] = a["tokens"] / a["kept"] if a["kept"] else float("inf")
for r in by_rule.values():
r["precision"] = r["kept"] / r["total"] if r["total"] else 0.0
return {"by_agent": by_agent, "by_rule": by_rule, "runs": len(runs)}
def _empty_agent() -> dict:
return {
"tokens": 0, "duration_ms": 0, "runs": 0,
"total": 0, "kept": 0, "dismissed": 0, "false_positive": 0,
}
def _empty_rule() -> dict:
return {"total": 0, "kept": 0, "dismissed": 0, "false_positive": 0}
def main(argv: list[str]) -> int:
log = Path.home() / ".claude/cache/audit-terraform/runs.jsonl"
stats = compute_stats(log)
print(json.dumps(stats, indent=2))
return 0
if __name__ == "__main__":
raise SystemExit(main(sys.argv[1:]))
@@ -0,0 +1,117 @@
from __future__ import annotations
import json
import shutil
import subprocess
from pathlib import Path
from typing import Any
from scripts.manifest import TflintFinding, TrivyFinding
def _normalize_trivy_findings(payload: dict[str, Any]) -> list[TrivyFinding]:
findings: list[TrivyFinding] = []
for result in payload.get("Results", []):
for misconf in result.get("Misconfigurations", []):
cause = misconf.get("CauseMetadata") or {}
resource = cause.get("Resource") or ""
resource_type = ""
if isinstance(resource, str) and "." in resource:
resource_type = resource.split(".", 1)[0]
findings.append(TrivyFinding(
check_id=(
misconf.get("AVDID")
or misconf.get("ID")
or misconf.get("Query")
or ""
),
title=misconf.get("Title") or misconf.get("ID") or "",
severity=str(misconf.get("Severity") or "UNKNOWN").lower(),
message=misconf.get("Message") or misconf.get("Description") or "",
file=result.get("Target") or "",
start_line=int(cause.get("StartLine") or 0),
end_line=int(cause.get("EndLine") or 0),
resource_type=resource_type,
))
return findings
def _run_trivy_config(repo: Path, changed_files_in_repo: list[str]) -> tuple[dict[str, Any], list[TrivyFinding], str | None]:
changed_dirs = sorted({
str(Path(path).parent)
for path in changed_files_in_repo
})
scan_paths = [str((repo / rel).resolve()) for rel in changed_dirs] or [str(repo)]
result = subprocess.run(
["trivy", "config", "--quiet", "--format", "json", *scan_paths],
capture_output=True,
text=True,
check=False,
)
if result.returncode != 0:
stderr = result.stderr.strip() or result.stdout.strip()
return {}, [], f"trivy config failed: {stderr}"
try:
payload = json.loads(result.stdout) if result.stdout.strip() else {}
except json.JSONDecodeError as exc:
return {}, [], f"trivy config output was not valid JSON: {exc}"
return payload, _normalize_trivy_findings(payload), None
def _normalize_tflint_findings(payload: dict, *, scanned_dir: str) -> list[TflintFinding]:
findings: list[TflintFinding] = []
for issue in payload.get("issues", []) or []:
rule = issue.get("rule", {}) or {}
rng = issue.get("range", {}) or {}
start = rng.get("start", {}) or {}
end = rng.get("end", {}) or {}
rel_file = rng.get("filename") or ""
if "/" in rel_file:
full = rel_file
elif rel_file:
full = (Path(scanned_dir) / rel_file).as_posix()
else:
full = ""
findings.append(TflintFinding(
rule=rule.get("name") or "",
severity=str(rule.get("severity") or "warning").lower(),
message=issue.get("message") or "",
file=full,
start_line=int(start.get("line") or 0),
end_line=int(end.get("line") or 0),
link=rule.get("link") or "",
))
return findings
def _run_tflint(repo: Path, changed_files_in_repo: list[str]) -> tuple[dict, list[TflintFinding], str | None]:
if shutil.which("tflint") is None:
return {}, [], None
changed_dirs = sorted({
str(Path(p).parent.as_posix()) for p in changed_files_in_repo
if p.endswith((".tf", ".tf.json"))
})
if not changed_dirs:
return {}, [], None
aggregated: list[TflintFinding] = []
raw_payloads: dict[str, dict] = {}
errs: list[str] = []
for rel in changed_dirs:
target = (repo / rel).resolve()
if not target.is_dir():
continue
proc = subprocess.run(
["tflint", "--format", "json", "--chdir", str(target)],
capture_output=True, text=True, check=False,
)
if proc.returncode not in (0, 2):
errs.append(f"tflint failed in {rel}: {proc.stderr.strip() or proc.stdout.strip()}")
continue
try:
payload = json.loads(proc.stdout) if proc.stdout.strip() else {}
except json.JSONDecodeError as exc:
errs.append(f"tflint output in {rel} was not valid JSON: {exc}")
continue
raw_payloads[rel] = payload
aggregated.extend(_normalize_tflint_findings(payload, scanned_dir=rel))
return {"by_dir": raw_payloads}, aggregated, ("; ".join(errs) if errs else None)
@@ -0,0 +1,84 @@
"""Scrape the CIS AWS Foundations Benchmark page.
The page is a mapping table:
| Control ID and title (FSBP-style) | CIS v5.0.0 | CIS v3.0.0 | CIS v1.4.0 | CIS v1.2.0 |
We emit one CISRow per unified control. The `requirement` field summarises
which CIS versions reference it.
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from bs4 import BeautifulSoup
from scripts.control_resource_map import control_id_to_types
CIS_URL = (
"https://docs.aws.amazon.com/securityhub/latest/userguide/"
"cis-aws-foundations-benchmark.html"
)
@dataclass
class CISRow:
control_id: str
title: str
severity: str
requirement: str
source_url: str
resource_types: list[str] = field(default_factory=list)
_TITLE_RE = re.compile(r"^\s*\[(?P<id>[A-Za-z][A-Za-z0-9]*\.\d+)\]\s*(?P<title>.+)$")
_VERSION_RE = re.compile(r"CIS\s+v(?P<ver>[\d.]+)", re.IGNORECASE)
def _extract_versions(headers: list[str]) -> list[str]:
"""From header cells, pull out version strings like '5.0.0' for each
column. Non-version columns yield empty string."""
versions: list[str] = []
for h in headers:
m = _VERSION_RE.search(h)
versions.append(m.group("ver") if m else "")
return versions
def parse_cis_page(html: str, source_url: str) -> list[CISRow]:
soup = BeautifulSoup(html, "html.parser")
rows: list[CISRow] = []
for table in soup.find_all("table"):
header_cells = table.find("tr").find_all(["th", "td"]) if table.find("tr") else []
header_texts = [c.get_text(strip=True) for c in header_cells]
versions = _extract_versions(header_texts)
if not any(versions):
continue
for tr in table.find_all("tr")[1:]:
cells = tr.find_all("td")
if len(cells) < 2:
continue
title_text = cells[0].get_text(strip=True)
m = _TITLE_RE.match(title_text)
if not m:
continue
unified_id = m.group("id")
title = m.group("title").strip()
version_refs: list[str] = []
for ver, cell in zip(versions[1:], cells[1:]):
if not ver:
continue
num = cell.get_text(strip=True)
if num:
version_refs.append(f"v{ver} §{num}")
requirement = "CIS " + ", ".join(version_refs) if version_refs else ""
rows.append(CISRow(
control_id=f"CIS {unified_id}",
title=title,
severity="medium",
requirement=requirement,
source_url=source_url,
resource_types=control_id_to_types(f"CIS {unified_id}"),
))
return rows
@@ -0,0 +1,61 @@
"""Scrape AWS Security Hub FSBP controls index page into structured rows.
The index page (https://docs.aws.amazon.com/securityhub/latest/userguide/fsbp-standard.html)
lists controls as <p><a href="./<svc>-controls.html#<id>">[<Service>.<Number>] <Title></a></p>.
This module only parses the index — severity defaults to "medium" because the
index doesn't surface severity. Detail-page enrichment is a future task.
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from urllib.parse import urljoin
from bs4 import BeautifulSoup
from scripts.control_resource_map import control_id_to_types
FSBP_INDEX_URL = (
"https://docs.aws.amazon.com/securityhub/latest/userguide/"
"fsbp-standard.html"
)
@dataclass
class FSBPRow:
control_id: str
title: str
severity: str
detail_url: str
resource_types: list[str] = field(default_factory=list)
requirement: str = ""
_ANCHOR_RE = re.compile(r"^\s*\[(?P<id>[A-Za-z][A-Za-z0-9]*\.\d+)\]\s*(?P<title>.+)$")
def parse_fsbp_index(html: str, base_url: str) -> list[FSBPRow]:
soup = BeautifulSoup(html, "html.parser")
rows: list[FSBPRow] = []
seen: set[str] = set()
for a in soup.find_all("a"):
text = a.get_text(strip=True)
m = _ANCHOR_RE.match(text)
if not m:
continue
control_id = m.group("id")
if control_id in seen:
continue
seen.add(control_id)
title = m.group("title").strip()
href = a.get("href", "")
detail_url = urljoin(base_url, href) if href else ""
rows.append(FSBPRow(
control_id=control_id,
title=title,
severity="medium",
detail_url=detail_url,
resource_types=control_id_to_types(control_id),
))
return rows
@@ -0,0 +1,64 @@
from __future__ import annotations
def _slice_for_agent(manifest_dict: dict, agent: str) -> dict:
"""Return a subset of the manifest appropriate to the named agent.
- fsbp/cis/aws-bp: AWS-filtered catalog + plan_units summary + trivy + tflint + errors
- consistency: full catalog + full plan_units + trivy + tflint + errors
- tf-hygiene: full catalog + full plan_units + tflint + changed_source_dirs + errors (no trivy)
- walkthrough: full catalog + plan_units summary + changed_source_dirs + errors (no findings)
"""
base = {
"base_ref": manifest_dict["base_ref"],
"head_ref": manifest_dict["head_ref"],
"mode": manifest_dict["mode"],
"default_branch": manifest_dict["default_branch"],
"errors": list(manifest_dict["errors"]),
}
catalog = manifest_dict["catalog"]
if agent in ("fsbp", "cis", "aws-bp"):
return {
**base,
"catalog": [e for e in catalog if e["type"].startswith("aws_")],
"trivy_findings": list(manifest_dict.get("trivy_findings", [])),
"tflint_findings": list(manifest_dict.get("tflint_findings", [])),
"plan_units": [
{"plan_dir": pu["plan_dir"], "tool": pu["tool"],
"plan_ok": pu["plan"]["ok"], "summary": pu["plan"]["summary"],
"terragrunt_changed": pu["terragrunt_changed"],
"changed_files": list(pu["changed_files"])}
for pu in manifest_dict["plan_units"]
],
}
if agent == "consistency":
return {
**base,
"changed_source_dirs": list(manifest_dict["changed_source_dirs"]),
"catalog": catalog,
"trivy_findings": list(manifest_dict.get("trivy_findings", [])),
"tflint_findings": list(manifest_dict.get("tflint_findings", [])),
"plan_units": [dict(pu) for pu in manifest_dict["plan_units"]],
}
if agent == "walkthrough":
return {
**base,
"changed_source_dirs": list(manifest_dict["changed_source_dirs"]),
"catalog": catalog,
"plan_units": [
{"plan_dir": pu["plan_dir"], "tool": pu["tool"],
"plan_ok": pu["plan"]["ok"], "summary": pu["plan"]["summary"],
"terragrunt_changed": pu["terragrunt_changed"],
"changed_files": list(pu["changed_files"])}
for pu in manifest_dict["plan_units"]
],
}
if agent == "tf-hygiene":
return {
**base,
"changed_source_dirs": list(manifest_dict["changed_source_dirs"]),
"catalog": catalog,
"tflint_findings": list(manifest_dict.get("tflint_findings", [])),
"plan_units": [dict(pu) for pu in manifest_dict["plan_units"]],
}
return manifest_dict
@@ -0,0 +1,220 @@
"""Locate a specific resource block in a directory's .tf files."""
from __future__ import annotations
from dataclasses import dataclass
from pathlib import Path
import re
from scripts.hcl_diff import find_resource_blocks
_KEY_ATTRIBUTES = {
"kms_key_id",
"kms_key_arn",
"bucket_key_enabled",
"publicly_accessible",
"acl",
"versioning",
"tags",
"deletion_protection",
}
_VAR_REF_RE = re.compile(r"\bvar\.([A-Za-z0-9_]+)")
_LOCAL_REF_RE = re.compile(r"\blocal\.([A-Za-z0-9_]+)")
_DATA_REF_RE = re.compile(r"\bdata\.aws_iam_policy_document\.([A-Za-z0-9_]+)")
_SG_REF_TEMPLATE = 'security_group_id = aws_security_group.{name}.id'
@dataclass(frozen=True)
class BlockLocation:
file: Path
start_line: int
end_line: int
text: str
header: str
evidence_line: str
key_attributes: dict[str, str | bool | int | list[str]]
review_context: dict[str, object]
def _parse_value(raw: str) -> str | bool | int | list[str]:
value = raw.strip().rstrip(",")
if value.lower() in {"true", "false"}:
return value.lower() == "true"
if re.fullmatch(r"-?\d+", value):
return int(value)
if value.startswith('"') and value.endswith('"'):
return value[1:-1]
if value.startswith("[") and value.endswith("]"):
inner = value[1:-1].strip()
if not inner:
return []
parts = [part.strip().strip('"') for part in inner.split(",")]
return [part for part in parts if part]
return value
def _extract_key_attributes(lines: list[str]) -> dict[str, str | bool | int | list[str]]:
out: dict[str, str | bool | int | list[str]] = {}
for line in lines:
stripped = line.strip()
if "=" not in stripped or stripped.startswith("#"):
continue
key, _, raw_value = stripped.partition("=")
key = key.strip()
if key not in _KEY_ATTRIBUTES:
continue
out[key] = _parse_value(raw_value)
return out
def _find_named_block(lines: list[str], header_re: re.Pattern[str], name: str) -> list[str]:
i = 0
while i < len(lines):
line = lines[i]
match = header_re.match(line)
if not match or match.group("name") != name:
i += 1
continue
depth = line.count("{") - line.count("}")
j = i
while depth > 0 and j + 1 < len(lines):
j += 1
depth += lines[j].count("{") - lines[j].count("}")
return lines[i:j + 1]
return []
def _collect_variable_defaults(root: Path, names: set[str]) -> dict[str, str | bool | int | list[str]]:
out: dict[str, str | bool | int | list[str]] = {}
header_re = re.compile(r'^\s*variable\s+"(?P<name>[^"]+)"\s*\{')
for tf in sorted(root.glob("*.tf")):
lines = tf.read_text().splitlines()
for name in names:
if name in out:
continue
block = _find_named_block(lines, header_re, name)
for line in block[1:]:
stripped = line.strip()
if stripped.startswith("default"):
_, _, raw = stripped.partition("=")
out[name] = _parse_value(raw)
break
return out
def _collect_locals(root: Path, names: set[str]) -> dict[str, str | bool | int | list[str]]:
out: dict[str, str | bool | int | list[str]] = {}
header_re = re.compile(r"^\s*locals\s*\{")
for tf in sorted(root.glob("*.tf")):
lines = tf.read_text().splitlines()
i = 0
while i < len(lines):
line = lines[i]
if not header_re.match(line):
i += 1
continue
depth = line.count("{") - line.count("}")
j = i
while depth > 0 and j + 1 < len(lines):
j += 1
depth += lines[j].count("{") - lines[j].count("}")
for block_line in lines[i + 1:j]:
stripped = block_line.strip()
if "=" not in stripped:
continue
key, _, raw = stripped.partition("=")
key = key.strip()
if key in names and key not in out:
out[key] = _parse_value(raw)
i = j + 1
return out
def _collect_related_policy_docs(root: Path, names: set[str]) -> list[str]:
out: list[str] = []
header_re = re.compile(
r'^\s*data\s+"aws_iam_policy_document"\s+"(?P<name>[^"]+)"\s*\{'
)
for tf in sorted(root.glob("*.tf")):
lines = tf.read_text().splitlines()
for name in names:
block = _find_named_block(lines, header_re, name)
if block:
out.append(block[0].strip())
return out
def _collect_related_sg_rules(root: Path, name: str) -> list[str]:
out: list[str] = []
target = _SG_REF_TEMPLATE.format(name=name)
header_re = re.compile(
r'^\s*resource\s+"aws_security_group_rule"\s+"(?P<name>[^"]+)"\s*\{'
)
for tf in sorted(root.glob("*.tf")):
lines = tf.read_text().splitlines()
i = 0
while i < len(lines):
line = lines[i]
if not header_re.match(line):
i += 1
continue
depth = line.count("{") - line.count("}")
j = i
while depth > 0 and j + 1 < len(lines):
j += 1
depth += lines[j].count("{") - lines[j].count("}")
block = lines[i:j + 1]
if any(target in block_line for block_line in block[1:]):
out.append(block[0].strip())
i = j + 1
return out
def _build_review_context(root: Path, rtype: str, rname: str, block_lines: list[str]) -> dict[str, object]:
joined = "\n".join(block_lines)
variable_names = set(_VAR_REF_RE.findall(joined))
local_names = set(_LOCAL_REF_RE.findall(joined))
policy_doc_names = set(_DATA_REF_RE.findall(joined))
related_blocks: list[str] = []
related_blocks.extend(_collect_related_policy_docs(root, policy_doc_names))
if rtype == "aws_security_group":
related_blocks.extend(_collect_related_sg_rules(root, rname))
return {
"variables": _collect_variable_defaults(root, variable_names),
"locals": _collect_locals(root, local_names),
"related_blocks": related_blocks,
}
def find_block(source_dir: Path | str, local_address: str) -> BlockLocation | None:
"""Search .tf files in `source_dir` for a resource matching `local_address`.
`local_address` is `<type>.<name>` (e.g. `aws_iam_role.svc`). Returns the
first match found. Returns None if no match.
"""
if "." not in local_address:
return None
rtype, _, rname = local_address.partition(".")
root = Path(source_dir)
for tf in sorted(root.glob("*.tf")):
for blk in find_resource_blocks(tf):
if blk.type == rtype and blk.name == rname:
lines = tf.read_text().splitlines()
block_lines = lines[blk.start - 1: blk.end]
header = block_lines[0] if block_lines else ""
evidence_line = header
for line in block_lines[1:]:
if line.strip():
evidence_line = line.strip()
break
return BlockLocation(
file=tf,
start_line=blk.start,
end_line=blk.end,
text="\n".join(block_lines),
header=header,
evidence_line=evidence_line,
key_attributes=_extract_key_attributes(block_lines[1:]),
review_context=_build_review_context(root, rtype, rname, block_lines),
)
return None
@@ -0,0 +1,53 @@
"""Append-only JSONL telemetry for audit-terraform runs."""
from __future__ import annotations
import json
from datetime import datetime, timezone
from pathlib import Path
def _now() -> str:
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
def _append(log_path: Path, record: dict) -> None:
log_path.parent.mkdir(parents=True, exist_ok=True)
with log_path.open("a", encoding="utf-8") as f:
f.write(json.dumps(record) + "\n")
def append_subagent_run(
log_path: Path, *, run_id: str, repo: str, mode: str, agent: str,
model: str, input_tokens: int, output_tokens: int,
duration_ms: int, finding_count: int,
) -> None:
_append(log_path, {
"kind": "subagent_run", "ts": _now(),
"run_id": run_id, "repo": repo, "mode": mode, "agent": agent,
"model": model,
"input_tokens": input_tokens, "output_tokens": output_tokens,
"duration_ms": duration_ms, "finding_count": finding_count,
})
def append_verdict(
log_path: Path, *, run_id: str, agent: str, rule_id: str,
file: str, line: int, verdict: str, notes: str = "",
) -> None:
if verdict not in {"kept", "dismissed", "false_positive"}:
raise ValueError(f"invalid verdict: {verdict!r}")
_append(log_path, {
"kind": "verdict", "ts": _now(),
"run_id": run_id, "agent": agent, "rule_id": rule_id,
"file": file, "line": line, "verdict": verdict, "notes": notes,
})
def read_runs(log_path: Path) -> list[dict]:
if not log_path.exists():
return []
return [
json.loads(line)
for line in log_path.read_text(encoding="utf-8").splitlines()
if line.strip()
]