diff --git a/cloudsec/cli.py b/cloudsec/cli.py index a43cf30..5023540 100644 --- a/cloudsec/cli.py +++ b/cloudsec/cli.py @@ -13,6 +13,8 @@ parse_baseline_csv, summarize_comparison, write_golden_baseline) from .demo import get_demo_snapshots from .frameworks import framework_check_ids +from .importers.prowler import build_scan_result as build_prowler_scan_result +from .importers.prowler import load_prowler_csv, top_failing_checks from .models import Finding, ScanResult from .output.csv_io import (parse_cases_csv, write_comparison_csv, write_findings_csv, write_json, write_review_csv) @@ -275,6 +277,41 @@ def _finalize(results: Dict[str, ScanResult], out: str, write_dashboard(os.path.join(out, "dashboard.html"), html) +def cmd_import_prowler(a: argparse.Namespace) -> int: + """Import a Prowler CSV export and render it through CloudGuard's own + findings CSV / JSON result / dashboard writers - no live cloud + credentials required, since Prowler already did the collecting.""" + if not a.csv or not os.path.exists(a.csv): + print("ERROR: --csv required.") + return 2 + rows = load_prowler_csv(a.csv, include_muted=a.include_muted) + if not rows: + print("ERROR: no rows parsed - check the file and delimiter (Prowler uses ';').") + return 2 + + result = build_prowler_scan_result(rows, cloud=a.cloud) + print(SEP) + print(f" Prowler import ({os.path.basename(a.csv)}) -> {len(rows)} row(s), " + f"{len(result.findings)} kept") + _print_result(result) + print(SEP) + print(" Top failing checks (CRITICAL/HIGH always shown in full; " + "MEDIUM/LOW filled in by resource count):") + for cid, title, sev, n in top_failing_checks(result): + print(f" [{sev:<8}] {cid:<40} {n:>3} resource(s) - {title}") + + out = _out_dir(a.output) + write_findings_csv(os.path.join(out, f"findings_{a.cloud}.csv"), result) + write_json(os.path.join(out, f"result_{a.cloud}.json"), result.to_dict()) + if not a.no_html: + html = build_dashboard_html({a.cloud: result}, + title="Cloud Configuration Review - Prowler Import") + write_dashboard(os.path.join(out, "dashboard.html"), html) + print(SEP) + print(f" Output written to: {out}") + return 0 + + def cmd_save_baseline(a: argparse.Namespace) -> int: """Freeze a trusted scan as a golden baseline for drift detection.""" try: @@ -710,6 +747,19 @@ def build_parser() -> argparse.ArgumentParser: c.add_argument("--output", default=None) c.set_defaults(func=cmd_compare) + ip = sub.add_parser( + "import-prowler", + help="Import a Prowler CSV export and render it via CloudGuard's dashboard/CSV/JSON") + ip.add_argument("--csv", required=True, help="Path to the Prowler CSV export (" + "native Prowler export format, ';'-delimited)") + ip.add_argument("--cloud", choices=list(CLOUDS), default="azure", + help="Cloud label to tag the imported findings with (default: azure)") + ip.add_argument("--output", default="reports") + ip.add_argument("--include-muted", action="store_true", + help="Include rows Prowler marked MUTED (excluded by default)") + ip.add_argument("--no-html", action="store_true") + ip.set_defaults(func=cmd_import_prowler) + sb = sub.add_parser("save-baseline", help="Freeze a trusted scan as a golden baseline CSV") sb.add_argument("--scan", required=True, help="scan JSON file or result directory") sb.add_argument("--output", default=None, help="baseline CSV path (default: golden_baseline.csv)") @@ -759,4 +809,4 @@ def main(argv: Optional[List[str]] = None) -> int: if __name__ == "__main__": - sys.exit(main()) + sys.exit(main()) \ No newline at end of file diff --git a/cloudsec/frameworks.py b/cloudsec/frameworks.py index b7361e3..abde261 100644 --- a/cloudsec/frameworks.py +++ b/cloudsec/frameworks.py @@ -282,7 +282,11 @@ def frameworks_for(check) -> Dict[str, str]: explicit = EXPLICIT_MAP.get(cid) if explicit: return dict(explicit) - if hasattr(check, "title"): + # NB: don't use hasattr(check, "title") to detect "is this a Check + # object" - str also has a .title() method (str.title()), so a plain + # check_id string would be misdetected as a Check and crash on + # check.service/check.category below. + if not isinstance(check, str): hay = " ".join([check.title or "", check.service or "", check.category or ""]).lower() else: hay = str(check).lower() @@ -305,7 +309,7 @@ def lead_frameworks(check) -> Set[str]: lead is what makes ``--frameworks`` filtering meaningful: a network exposure check leads in PCI DSS / NIST, an MFA check in SOC 2 / HIPAA. """ - if hasattr(check, "title"): + if not isinstance(check, str): # see note in frameworks_for() above hay = " ".join([check.title or "", check.service or "", check.category or ""]).lower() else: hay = str(check).lower() @@ -331,4 +335,4 @@ def framework_check_ids(cloud: str, frameworks: List[str]) -> Set[str]: elif FRAMEWORK_ALIASES.get(fw) in lead_frameworks(ch): out.add(ch.id) break - return out + return out \ No newline at end of file diff --git a/cloudsec/importers/__init__.py b/cloudsec/importers/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/cloudsec/importers/prowler.py b/cloudsec/importers/prowler.py new file mode 100644 index 0000000..c5da772 --- /dev/null +++ b/cloudsec/importers/prowler.py @@ -0,0 +1,142 @@ +"""Import a Prowler CSV export into a CloudGuard ScanResult. + +Prowler's native CSV export uses ';' as the delimiter and a much wider, +differently-named column set than CloudGuard's own findings CSV. This +module maps that format onto CloudGuard's ``ScanResult``/``Finding`` +models so the *same* dashboard/CSV/JSON writers used by ``scan``/``demo`` +can render a Prowler run without a live cloud credential. + +Only PASS/FAIL rows contribute to the risk score and findings table; +Prowler statuses outside that pair are kept but recorded as NOT_ASSESSED +and reported back to the caller. MUTED rows are excluded by default. + +Note: check_id/check_title/severity/service come straight from Prowler. +These are Prowler's own identifiers, not CloudGuard catalog IDs, so the +CIS Benchmark / compliance-framework dashboard panels (which key off +CloudGuard's own check registry) stay empty for these findings - every +other dashboard feature (risk score, severity/service breakdown, +findings table, CSV/JSON export) works the same as a native scan. +""" +from __future__ import annotations + +import csv +import re +from collections import Counter, defaultdict +from typing import Any, Dict, List, Tuple + +from ..models import Finding, ScanResult, Severity, Status + +SEVERITY_MAP = { + "critical": Severity.CRITICAL, + "high": Severity.HIGH, + "medium": Severity.MEDIUM, + "low": Severity.LOW, + "informational": Severity.INFO, + "info": Severity.INFO, +} + +STATUS_MAP = { + "PASS": Status.PASS, + "FAIL": Status.FAIL, +} + +CIS_RE = re.compile(r"CIS-[\d.]+:[\d.]+") + + +def _cis_ref(compliance: str) -> str: + if not compliance: + return "" + hits = sorted(set(CIS_RE.findall(compliance))) + return "; ".join(hits[:3]) + + +def _category(categories: str, service: str) -> str: + if categories: + return categories.split(",")[0].strip() + return service.title() if service else "Uncategorized" + + +def load_prowler_csv(path: str, include_muted: bool = False) -> List[Dict[str, str]]: + """Read a Prowler CSV export (';'-delimited) into raw dict rows.""" + with open(path, encoding="utf-8-sig", newline="") as fh: + reader = csv.DictReader(fh, delimiter=";") + rows = list(reader) + if not include_muted: + rows = [r for r in rows if (r.get("MUTED") or "").strip().lower() != "true"] + return rows + + +def build_scan_result(rows: List[Dict[str, str]], cloud: str = "azure") -> ScanResult: + """Turn Prowler CSV rows into a CloudGuard ScanResult.""" + findings: List[Finding] = [] + unmapped_status: Counter = Counter() + seen_finding_uids = set() + + account_id = rows[0].get("ACCOUNT_UID", "") if rows else "" + account_name = rows[0].get("ACCOUNT_NAME", "") if rows else "" + ts = rows[0].get("TIMESTAMP", "") if rows else "" + + for r in rows: + uid = r.get("FINDING_UID", "") + if uid and uid in seen_finding_uids: + continue # defensive de-dup: skip exact repeat rows + if uid: + seen_finding_uids.add(uid) + + raw_status = (r.get("STATUS") or "").strip().upper() + status = STATUS_MAP.get(raw_status) + if status is None: + unmapped_status[raw_status or "(blank)"] += 1 + status = Status.NOT_ASSESSED + + raw_sev = (r.get("SEVERITY") or "").strip().lower() + severity = SEVERITY_MAP.get(raw_sev, Severity.MEDIUM) + + service = (r.get("SERVICE_NAME") or "general").strip() + resource = (r.get("RESOURCE_UID") or r.get("RESOURCE_NAME") or "").strip() + + findings.append(Finding( + check_id=r.get("CHECK_ID", ""), + check_title=r.get("CHECK_TITLE", ""), + cloud=cloud, + service=service.upper() if len(service) <= 4 else service.title(), + category=_category(r.get("CATEGORIES", ""), service), + severity=severity, + status=status, + resource=resource, + detail=r.get("STATUS_EXTENDED", ""), + remediation=r.get("REMEDIATION_RECOMMENDATION_TEXT", ""), + cis=_cis_ref(r.get("COMPLIANCE", "")) or None, + )) + + check_ids = {f.check_id for f in findings} + result = ScanResult( + cloud=cloud, + account_id=account_id, + account_name=account_name, + timestamp=ts, + principal="prowler-import", + auth_mode="prowler:import", + regions=sorted({r.get("REGION", "") for r in rows if r.get("REGION")}), + findings=findings, + checks_total=len(check_ids), + checks_executed=len(check_ids), + ) + if unmapped_status: + result.errors.append({ + "service": "prowler_import", + "error": f"Rows with unrecognized STATUS values (kept as NOT_ASSESSED): " + f"{dict(unmapped_status)}", + }) + return result + + +def top_failing_checks(result: ScanResult, limit: int = 15) -> List[Tuple[str, str, str, int]]: + """(check_id, title, severity, affected_resource_count), busiest first.""" + by_check: Dict[Tuple[str, str, str], List[str]] = defaultdict(list) + for f in result.findings: + if f.status != Status.FAIL: + continue + by_check[(f.check_id, f.check_title, f.severity.value)].append(f.resource) + ranked = sorted(by_check.items(), key=lambda kv: -len(kv[1])) + return [(cid, title, sev, len(res)) for (cid, title, sev), res in ranked[:limit]]