Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
52 changes: 51 additions & 1 deletion cloudsec/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,8 @@
parse_baseline_csv, summarize_comparison, write_golden_baseline)
from .demo import get_demo_snapshots
from .frameworks import framework_check_ids
from .importers.prowler import build_scan_result as build_prowler_scan_result
from .importers.prowler import load_prowler_csv, top_failing_checks
from .models import Finding, ScanResult
from .output.csv_io import (parse_cases_csv, write_comparison_csv,
write_findings_csv, write_json, write_review_csv)
Expand Down Expand Up @@ -275,6 +277,41 @@ def _finalize(results: Dict[str, ScanResult], out: str,
write_dashboard(os.path.join(out, "dashboard.html"), html)


def cmd_import_prowler(a: argparse.Namespace) -> int:
"""Import a Prowler CSV export and render it through CloudGuard's own
findings CSV / JSON result / dashboard writers - no live cloud
credentials required, since Prowler already did the collecting."""
if not a.csv or not os.path.exists(a.csv):
print("ERROR: --csv <prowler_export.csv> required.")
return 2
rows = load_prowler_csv(a.csv, include_muted=a.include_muted)
if not rows:
print("ERROR: no rows parsed - check the file and delimiter (Prowler uses ';').")
return 2

result = build_prowler_scan_result(rows, cloud=a.cloud)
print(SEP)
print(f" Prowler import ({os.path.basename(a.csv)}) -> {len(rows)} row(s), "
f"{len(result.findings)} kept")
_print_result(result)
print(SEP)
print(" Top failing checks (CRITICAL/HIGH always shown in full; "
"MEDIUM/LOW filled in by resource count):")
for cid, title, sev, n in top_failing_checks(result):
print(f" [{sev:<8}] {cid:<40} {n:>3} resource(s) - {title}")

out = _out_dir(a.output)
write_findings_csv(os.path.join(out, f"findings_{a.cloud}.csv"), result)
write_json(os.path.join(out, f"result_{a.cloud}.json"), result.to_dict())
if not a.no_html:
html = build_dashboard_html({a.cloud: result},
title="Cloud Configuration Review - Prowler Import")
write_dashboard(os.path.join(out, "dashboard.html"), html)
print(SEP)
print(f" Output written to: {out}")
return 0


def cmd_save_baseline(a: argparse.Namespace) -> int:
"""Freeze a trusted scan as a golden baseline for drift detection."""
try:
Expand Down Expand Up @@ -710,6 +747,19 @@ def build_parser() -> argparse.ArgumentParser:
c.add_argument("--output", default=None)
c.set_defaults(func=cmd_compare)

ip = sub.add_parser(
"import-prowler",
help="Import a Prowler CSV export and render it via CloudGuard's dashboard/CSV/JSON")
ip.add_argument("--csv", required=True, help="Path to the Prowler CSV export ("
"native Prowler export format, ';'-delimited)")
ip.add_argument("--cloud", choices=list(CLOUDS), default="azure",
help="Cloud label to tag the imported findings with (default: azure)")
ip.add_argument("--output", default="reports")
ip.add_argument("--include-muted", action="store_true",
help="Include rows Prowler marked MUTED (excluded by default)")
ip.add_argument("--no-html", action="store_true")
ip.set_defaults(func=cmd_import_prowler)

sb = sub.add_parser("save-baseline", help="Freeze a trusted scan as a golden baseline CSV")
sb.add_argument("--scan", required=True, help="scan JSON file or result directory")
sb.add_argument("--output", default=None, help="baseline CSV path (default: golden_baseline.csv)")
Expand Down Expand Up @@ -759,4 +809,4 @@ def main(argv: Optional[List[str]] = None) -> int:


if __name__ == "__main__":
sys.exit(main())
sys.exit(main())
10 changes: 7 additions & 3 deletions cloudsec/frameworks.py
Original file line number Diff line number Diff line change
Expand Up @@ -282,7 +282,11 @@ def frameworks_for(check) -> Dict[str, str]:
explicit = EXPLICIT_MAP.get(cid)
if explicit:
return dict(explicit)
if hasattr(check, "title"):
# NB: don't use hasattr(check, "title") to detect "is this a Check
# object" - str also has a .title() method (str.title()), so a plain
# check_id string would be misdetected as a Check and crash on
# check.service/check.category below.
if not isinstance(check, str):
hay = " ".join([check.title or "", check.service or "", check.category or ""]).lower()
else:
hay = str(check).lower()
Expand All @@ -305,7 +309,7 @@ def lead_frameworks(check) -> Set[str]:
lead is what makes ``--frameworks`` filtering meaningful: a network
exposure check leads in PCI DSS / NIST, an MFA check in SOC 2 / HIPAA.
"""
if hasattr(check, "title"):
if not isinstance(check, str): # see note in frameworks_for() above
hay = " ".join([check.title or "", check.service or "", check.category or ""]).lower()
else:
hay = str(check).lower()
Expand All @@ -331,4 +335,4 @@ def framework_check_ids(cloud: str, frameworks: List[str]) -> Set[str]:
elif FRAMEWORK_ALIASES.get(fw) in lead_frameworks(ch):
out.add(ch.id)
break
return out
return out
Empty file added cloudsec/importers/__init__.py
Empty file.
142 changes: 142 additions & 0 deletions cloudsec/importers/prowler.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,142 @@
"""Import a Prowler CSV export into a CloudGuard ScanResult.

Prowler's native CSV export uses ';' as the delimiter and a much wider,
differently-named column set than CloudGuard's own findings CSV. This
module maps that format onto CloudGuard's ``ScanResult``/``Finding``
models so the *same* dashboard/CSV/JSON writers used by ``scan``/``demo``
can render a Prowler run without a live cloud credential.

Only PASS/FAIL rows contribute to the risk score and findings table;
Prowler statuses outside that pair are kept but recorded as NOT_ASSESSED
and reported back to the caller. MUTED rows are excluded by default.

Note: check_id/check_title/severity/service come straight from Prowler.
These are Prowler's own identifiers, not CloudGuard catalog IDs, so the
CIS Benchmark / compliance-framework dashboard panels (which key off
CloudGuard's own check registry) stay empty for these findings - every
other dashboard feature (risk score, severity/service breakdown,
findings table, CSV/JSON export) works the same as a native scan.
"""
from __future__ import annotations

import csv
import re
from collections import Counter, defaultdict
from typing import Any, Dict, List, Tuple

from ..models import Finding, ScanResult, Severity, Status

SEVERITY_MAP = {
"critical": Severity.CRITICAL,
"high": Severity.HIGH,
"medium": Severity.MEDIUM,
"low": Severity.LOW,
"informational": Severity.INFO,
"info": Severity.INFO,
}

STATUS_MAP = {
"PASS": Status.PASS,
"FAIL": Status.FAIL,
}

CIS_RE = re.compile(r"CIS-[\d.]+:[\d.]+")


def _cis_ref(compliance: str) -> str:
if not compliance:
return ""
hits = sorted(set(CIS_RE.findall(compliance)))
return "; ".join(hits[:3])


def _category(categories: str, service: str) -> str:
if categories:
return categories.split(",")[0].strip()
return service.title() if service else "Uncategorized"


def load_prowler_csv(path: str, include_muted: bool = False) -> List[Dict[str, str]]:
"""Read a Prowler CSV export (';'-delimited) into raw dict rows."""
with open(path, encoding="utf-8-sig", newline="") as fh:
reader = csv.DictReader(fh, delimiter=";")
rows = list(reader)
if not include_muted:
rows = [r for r in rows if (r.get("MUTED") or "").strip().lower() != "true"]
return rows


def build_scan_result(rows: List[Dict[str, str]], cloud: str = "azure") -> ScanResult:
"""Turn Prowler CSV rows into a CloudGuard ScanResult."""
findings: List[Finding] = []
unmapped_status: Counter = Counter()
seen_finding_uids = set()

account_id = rows[0].get("ACCOUNT_UID", "") if rows else ""
account_name = rows[0].get("ACCOUNT_NAME", "") if rows else ""
ts = rows[0].get("TIMESTAMP", "") if rows else ""

for r in rows:
uid = r.get("FINDING_UID", "")
if uid and uid in seen_finding_uids:
continue # defensive de-dup: skip exact repeat rows
if uid:
seen_finding_uids.add(uid)

raw_status = (r.get("STATUS") or "").strip().upper()
status = STATUS_MAP.get(raw_status)
if status is None:
unmapped_status[raw_status or "(blank)"] += 1
status = Status.NOT_ASSESSED

raw_sev = (r.get("SEVERITY") or "").strip().lower()
severity = SEVERITY_MAP.get(raw_sev, Severity.MEDIUM)

service = (r.get("SERVICE_NAME") or "general").strip()
resource = (r.get("RESOURCE_UID") or r.get("RESOURCE_NAME") or "").strip()

findings.append(Finding(
check_id=r.get("CHECK_ID", ""),
check_title=r.get("CHECK_TITLE", ""),
cloud=cloud,
service=service.upper() if len(service) <= 4 else service.title(),
category=_category(r.get("CATEGORIES", ""), service),
severity=severity,
status=status,
resource=resource,
detail=r.get("STATUS_EXTENDED", ""),
remediation=r.get("REMEDIATION_RECOMMENDATION_TEXT", ""),
cis=_cis_ref(r.get("COMPLIANCE", "")) or None,
))

check_ids = {f.check_id for f in findings}
result = ScanResult(
cloud=cloud,
account_id=account_id,
account_name=account_name,
timestamp=ts,
principal="prowler-import",
auth_mode="prowler:import",
regions=sorted({r.get("REGION", "") for r in rows if r.get("REGION")}),
findings=findings,
checks_total=len(check_ids),
checks_executed=len(check_ids),
)
if unmapped_status:
result.errors.append({
"service": "prowler_import",
"error": f"Rows with unrecognized STATUS values (kept as NOT_ASSESSED): "
f"{dict(unmapped_status)}",
})
return result


def top_failing_checks(result: ScanResult, limit: int = 15) -> List[Tuple[str, str, str, int]]:
"""(check_id, title, severity, affected_resource_count), busiest first."""
by_check: Dict[Tuple[str, str, str], List[str]] = defaultdict(list)
for f in result.findings:
if f.status != Status.FAIL:
continue
by_check[(f.check_id, f.check_title, f.severity.value)].append(f.resource)
ranked = sorted(by_check.items(), key=lambda kv: -len(kv[1]))
return [(cid, title, sev, len(res)) for (cid, title, sev), res in ranked[:limit]]