diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 798b85a..d3a5b64 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -80,6 +80,8 @@ jobs: SBOM_FILE: clickhouse-server-ubi9-${{ matrix.architecture }}.spdx.json GRYPE_SARIF: grype-${{ matrix.architecture }}.sarif GRYPE_ALL: grype-all-${{ matrix.architecture }}.json + SCAP_SCANNER_IMAGE: localhost/datopsis-openscap:0.1.82-${{ matrix.architecture }} + SCAP_RESULTS_DIR: scap-results-${{ matrix.architecture }} steps: - name: Check out repository uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 @@ -134,6 +136,30 @@ jobs: IMAGE: ${{ env.TEST_IMAGE }} run: bash tests/tls-rehearsal.sh + - name: Build pinned OpenSCAP tool image + id: build-scap + continue-on-error: true + uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7.3.0 + with: + context: . + file: Containerfile.scap + platforms: ${{ matrix.platform }} + load: true + push: false + tags: ${{ env.SCAP_SCANNER_IMAGE }} + cache-from: type=gha,scope=scap-${{ matrix.architecture }} + cache-to: type=gha,mode=max,scope=scap-${{ matrix.architecture }} + + - name: Run SCAP discovery scan + id: scan-scap + if: ${{ steps.build-scap.outcome == 'success' }} + continue-on-error: true + env: + ARCHITECTURE: ${{ matrix.architecture }} + CONTAINER_RUNTIME: docker + IMAGE: ${{ env.TEST_IMAGE }} + run: bash scripts/scap-scan.sh + - name: Scan image uses: aquasecurity/trivy-action@ed142fd0673e97e23eac54620cfb913e5ce36c25 # v0.36.0 with: @@ -194,6 +220,7 @@ jobs: ${{ env.SBOM_FILE }} ${{ env.GRYPE_SARIF }} ${{ env.GRYPE_ALL }} + ${{ env.SCAP_RESULTS_DIR }}/ if-no-files-found: warn retention-days: 14 @@ -204,6 +231,15 @@ jobs: sarif_file: ${{ env.GRYPE_SARIF }} category: grype-image-${{ matrix.architecture }} + - name: Require successful SCAP evaluation + if: ${{ always() }} + env: + SCAP_BUILD_OUTCOME: ${{ steps.build-scap.outcome }} + SCAP_SCAN_OUTCOME: ${{ steps.scan-scap.outcome }} + run: | + test "${SCAP_BUILD_OUTCOME}" = success + test "${SCAP_SCAN_OUTCOME}" = success + image-result: name: image if: ${{ always() }} diff --git a/Containerfile.scap b/Containerfile.scap new file mode 100644 index 0000000..8b8a74a --- /dev/null +++ b/Containerfile.scap @@ -0,0 +1,45 @@ +# syntax=docker/dockerfile:1.7 + +ARG UBI_MINIMAL_IMAGE="registry.access.redhat.com/ubi9/ubi-minimal:9.8@sha256:7fbeae18dc9476399f565e68255f602a3374ea8614ba3d14843565131a13ff93" + +FROM ${UBI_MINIMAL_IMAGE} + +ARG OPENSCAP_NEVRA="1.3.14-1.el9_8" +ARG SSG_VERSION="0.1.82" +ARG SSG_ARCHIVE_SHA256="765e84bdce7f9055f9b9c2dd0ee2b713d4255f8eec94eac6d35ea4973c28919c" +ARG SSG_RHEL9_DATASTREAM_SHA256="92204daafbf4f38011671ef034fae4cffb48f708516186710346a9ec702a1f8f" + +# This image is a CI tool, not a runtime layer of the ClickHouse image. The +# OpenSCAP RPM version and upstream content are deliberately pinned so a +# repository or content update cannot silently change compliance evidence. +# hadolint ignore=DL3041 +RUN microdnf install -y \ + "openscap-scanner-${OPENSCAP_NEVRA}" \ + tar \ + unzip \ + && mkdir -p /opt/scap \ + && archive="/tmp/scap-security-guide-${SSG_VERSION}.zip" \ + && curl --fail --location --proto '=https' --tlsv1.2 \ + --retry 5 --retry-delay 2 --retry-all-errors \ + --output "${archive}" \ + "https://github.com/ComplianceAsCode/content/releases/download/v${SSG_VERSION}/scap-security-guide-${SSG_VERSION}.zip" \ + && printf '%s %s\n' "${SSG_ARCHIVE_SHA256}" "${archive}" | sha256sum --check --strict \ + && unzip -j "${archive}" \ + "scap-security-guide-${SSG_VERSION}/ssg-rhel9-ds.xml" \ + -d /opt/scap \ + && printf '%s %s\n' \ + "${SSG_RHEL9_DATASTREAM_SHA256}" \ + /opt/scap/ssg-rhel9-ds.xml | sha256sum --check --strict \ + && rm -f "${archive}" \ + && microdnf clean all \ + && rm -rf /var/cache/yum /var/log/dnf* /var/log/yum.* + +COPY --chmod=0755 scripts/scap-container.sh /usr/local/bin/scap-container + +LABEL org.opencontainers.image.title="Datopsis OpenSCAP offline scanner" \ + org.opencontainers.image.description="Pinned, isolated CI scanner for clickhouse-server-ubi9 exported filesystems" \ + org.opencontainers.image.source="https://github.com/datopsis/clickhouse-server-ubi9" + +USER 65534:65534 + +ENTRYPOINT ["/usr/local/bin/scap-container"] diff --git a/docs/CI.md b/docs/CI.md index dbac45b..37ffafb 100644 --- a/docs/CI.md +++ b/docs/CI.md @@ -6,7 +6,7 @@ This repository treats the built image as the primary deliverable. CI therefore | Workflow | Triggers | Purpose | | --- | --- | --- | -| `CI` | Pull requests, pushes to `main`, weekly schedule, manual dispatch | Lint and workflow audit, followed by native AMD64 and ARM64 image builds, smoke tests, Trivy scans, Syft SBOMs, and Grype scans. | +| `CI` | Pull requests, pushes to `main`, weekly schedule, manual dispatch | Lint and workflow audit, followed by native AMD64 and ARM64 image builds, smoke tests, isolated SCAP discovery, Trivy scans, Syft SBOMs, and Grype scans. | | `CodeQL` | Workflow changes, weekly schedule, manual dispatch | Static analysis of GitHub Actions with the security-extended query suite. | | `OpenSSF Scorecard` | Pushes to `main`, ruleset changes, weekly schedule, manual dispatch | Supply-chain posture analysis, SARIF upload, and public Scorecard publication. | | `Release image` | Tags matching `v*` | Tag/input/changelog validation, multi-architecture publish, digest scans, evidence generation, keyless signing, and GitHub release creation. | @@ -24,6 +24,7 @@ Workflow-level permissions default to read-only. Write scopes are applied only t | Runtime behavior | Native GitHub-hosted AMD64 and ARM64 runners, Buildx, and `tests/smoke.sh` | Each architecture's exact test image starts and stops correctly under production-oriented restrictions and supports documented initialization/authentication behavior. Runner and loaded-image assertions prevent emulation or a mislabeled image from being treated as native evidence. | Hosted-runner tests do not replace OpenShift qualification or application-specific performance testing. | | Image vulnerabilities | Trivy image scan | No fixed high/critical findings according to Trivy's current databases and vendor severity selection. | `ignore-unfixed` intentionally leaves unfixed risk for human release review. | | Independent inventory and scan | Syft plus Grype | SPDX inventory of the tested image and a second vulnerability matcher/database; fixed high/critical findings block. | Overlap is intentional, but scanner agreement is not proof of absence. | +| Filesystem compliance discovery | Pinned OpenSCAP engine and ComplianceAsCode RHEL 9 STIG profile | Complete architecture-specific inventory of upstream rule results against a root-owner-preserving export; scanner errors block. | STIG is a broad discovery source, not wholesale adoption. Findings are non-blocking until applicability is reviewed and a container-specific tailoring is approved. | | Supply-chain posture | OpenSSF Scorecard | Repository and build-pipeline practice signals published independently. | Historical and popularity signals improve only through genuine project operation. | | Release integrity | BuildKit attestations, Cosign, GHCR, and GitHub Releases | Digest-bound multi-architecture artifact, SBOM/provenance evidence, keyless signature, and durable release assets. | The tag workflow publishes before post-build scans; a failed candidate must be quarantined or removed. | @@ -57,7 +58,7 @@ Audit these settings before each release and after organization policy changes. For every required run, verify the event and head SHA first. Then review the following evidence: - `lint`: every hook and the release-tag test ran, Zizmor audited every workflow, and there are no warnings or annotations hidden behind a successful wrapper. -- `image (amd64)` and `image (arm64)`: the native runner assertion, loaded-image architecture assertion, configuration scan count, ClickHouse version printed by the smoke suite, Trivy target/OS/package count and result count, SBOM package count, and both Grype's blocking fixed-findings result and full finding inventory. The aggregate `image` job is only the merge gate; inspect the two jobs that produced the evidence. +- `image (amd64)` and `image (arm64)`: the native runner assertion, loaded-image architecture assertion, configuration scan count, ClickHouse version printed by the smoke suite, SCAP execution outcome and result counts, Trivy target/OS/package count and result count, SBOM package count, and both Grype's blocking fixed-findings result and full finding inventory. The aggregate `image` job is only the merge gate; inspect the two jobs that produced the evidence. - `CodeQL` and Scorecard: analysis covered the intended files, SARIF processing completed, and the Security tab has no new open alert. A successful upload is not the same as zero findings. - skipped steps: PR SARIF publication is intentionally skipped to avoid permission failures from untrusted forks; it runs on `main`. A skipped build, smoke test, or scanner is not acceptable. - warnings: Trivy may use another vendor's severity when Red Hat data is absent. Grype's `only-fixed` option can ignore real but currently unfixable findings. Review both against Red Hat and ClickHouse advisories before a release. @@ -70,11 +71,12 @@ Each native image matrix job runs these controls in order. AMD64 uses `ubuntu-24 1. **Trivy configuration scan** checks the `Containerfile`, Compose configuration, and repository infrastructure configuration for high and critical misconfigurations. 2. **Build and runtime tests** exercise startup, authentication, initialization, persistence, shutdown, read-only operation, dropped capabilities, arbitrary UIDs, chained CA-issued HTTPS/native TLS, public/private outbound trust, disconnected isolation, negative certificate cases, renewal, and rollback. -3. **Trivy image scan** blocks fixed high and critical operating-system or application vulnerabilities and reports its detected OS and package count for review. -4. **Complete SPDX inventory** uses Syft to inventory the tested filesystem and RPM database, then `scripts/augment-spdx.py` declares the three pinned ClickHouse TGZ components that have no RPM metadata. The script takes their version and channel from `Containerfile`, records Apache-2.0 licensing and package identifiers, and fails instead of duplicating a component Syft already found. -5. **Blocking Grype SBOM scan** scans that exact SPDX document and blocks fixed high and critical vulnerabilities. -6. **Full Grype inventory** performs a non-blocking scan of the same SBOM without filtering unfixed matches and retains `grype-all.json`. Non-blocking means “record for triage,” not “accepted risk.” -7. **Artifact and SARIF publication** retains the inventory and results for investigation and publishes fixed Grype findings from non-PR runs to GitHub code scanning. +3. **OpenSCAP discovery** builds a pinned-input UBI scanner, exports but never executes the stopped target, preserves filesystem ownership inside an isolated tmpfs, and evaluates the pinned RHEL 9 STIG profile without network or an engine socket. The profile is an analysis source rather than wholesale control adoption. Findings remain report-only; execution errors block. +4. **Trivy image scan** blocks fixed high and critical operating-system or application vulnerabilities and reports its detected OS and package count for review. +5. **Complete SPDX inventory** uses Syft to inventory the tested filesystem and RPM database, then `scripts/augment-spdx.py` declares the three pinned ClickHouse TGZ components that have no RPM metadata. The script takes their version and channel from `Containerfile`, records Apache-2.0 licensing and package identifiers, and fails instead of duplicating a component Syft already found. +6. **Blocking Grype SBOM scan** scans that exact SPDX document and blocks fixed high and critical vulnerabilities. +7. **Full Grype inventory** performs a non-blocking scan of the same SBOM without filtering unfixed matches and retains `grype-all.json`. Non-blocking means “record for triage,” not “accepted risk.” +8. **Artifact and SARIF publication** retains the inventory and results for investigation and publishes fixed Grype findings from non-PR runs to GitHub code scanning. Trivy and Grype deliberately overlap. They use different databases and matching logic, so a clean result from one does not replace the other. Both gates ignore vulnerabilities without an upstream fix; unfixed findings still require periodic review before release. Scanner disagreements should be investigated against the vendor advisory and documented if accepted. @@ -85,6 +87,7 @@ Trivy and Grype deliberately overlap. They use different databases and matching | `clickhouse-server-ubi9-.spdx.json` | CI artifact `image-security--` | 14 days | Package inventory for the exact native AMD64 or ARM64 test image. | | `grype-.sarif` | Same architecture-specific CI artifact and GitHub code scanning on non-PR runs | 14 days for the downloadable artifact | Machine-readable findings and architecture-specific review evidence. | | `grype-all-.json` | Architecture-specific CI artifact | 14 days | Complete point-in-time inventory including unfixed Low and Medium matches for human triage. The release workflow separately retains `grype-all.json` for 30 days. | +| `scap-results-/` | Architecture-specific CI artifact | 14 days | Discovery ARF/XCCDF/HTML, full JSON rule inventory, exit code, data-stream hash, scanner version, and RPM versions for the exact target/scanner image IDs. | | `image.spdx.json` | Tag-run artifact and GitHub release asset | 30-day Actions copy; release asset retained with the release | Downloadable inventory for the published digest. | | Release `grype.sarif` | Tag-run artifact and GitHub code scanning | 30 days for the downloadable artifact | Point-in-time scan evidence; not attached to the release because vulnerability data ages rapidly. | | BuildKit SBOM/provenance and complete SPDX attestation | OCI registry attestations; downloaded together as `image.intoto.jsonl` | Lifetime of the package/release | Registry-native build evidence plus the keyless, digest-bound copy of `image.spdx.json`. | @@ -110,6 +113,10 @@ CONTAINER_RUNTIME=podman \ IMAGE="clickhouse-server-ubi9:test-${ARCHITECTURE}" bash tests/smoke.sh ``` +Build and run the isolated SCAP discovery scanner with the Podman procedure in +[SCAP.md](SCAP.md). It intentionally creates a second tooling image and does +not alter the ClickHouse deliverable. + Running an ARM64 image under emulation on an AMD64 workstation can help diagnose portable build failures, but it does not reproduce the native ARM64 qualification. GitHub's `ubuntu-24.04-arm` runner supplies that evidence using Docker Engine and Buildx. To reproduce both jobs faithfully, run the procedure once on each native architecture and retain separate results. Podman and Docker exercise the same image contract but remain distinct runtime implementations, so first-release evidence records both the native CI results and the separately tested Podman version. For the scanner examples below, keep using the architecture-specific image name: diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index b5fb87f..f72f67e 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -102,22 +102,22 @@ This repository owns image-specific behavior, basic usage, and minimal platform **Profile discovery and tailoring** -- [ ] Pin a UBI 9 OpenSCAP scanner image by digest and pin the OpenSCAP and ComplianceAsCode content versions. Record the RHEL 9 data-stream SHA-256 and reject an unexpected stream. -- [ ] Run the upstream RHEL 9 Standard profile in report-only discovery mode against the exported image filesystem. Inventory every pass, failure, error, not-applicable, and not-checked result without claiming host or deployment compliance. +- [x] Build the scanner from the same digest-pinned UBI 9 base, pin OpenSCAP `1.3.14-1.el9_8` and ComplianceAsCode `0.1.82`, verify the release archive, and verify/record the RHEL 9 data-stream SHA-256. Retain the produced scanner image ID for every run; use a manifest digest if the tool image is later published for reuse. +- [x] Add upstream RHEL 9 STIG-profile report-only discovery against an ownership-preserving exported image filesystem; ComplianceAsCode `0.1.82` does not contain a RHEL 9 Standard profile. Treat STIG as an analysis source rather than wholesale adoption, inventory every result, and keep evaluation errors blocking without claiming host or deployment compliance. - [ ] Create a reviewed XCCDF tailoring profile containing only rules that are applicable to and controlled by this image. Commit a rule-rationale matrix and document every host/platform exclusion. - [ ] Exclude kernel, boot-loader, partition, mount-layout, systemd, audit, host-networking, sysctl, SELinux-mode, and FIPS-mode controls unless the image later gains direct ownership of one. Do not use automatic remediation. **Safe CI integration** -- [ ] Export the stopped, already-tested image's merged filesystem into an ephemeral directory and mount that directory read-only into the scanner. Never execute target-image content to prepare the scan. -- [ ] Run `oscap-chroot` in a digest-pinned scanner with no Docker/Podman socket, no host namespace, no workflow secrets, and no evaluation-time network. Prove the minimum chroot-related capability; do not use `--privileged`, Podman-in-Podman, or broad host mounts. -- [ ] Generate architecture-specific ARF XML, XCCDF XML, and HTML reports containing the image digest, architecture, scanner/content versions, data-stream hash, and tailoring hash. Retain them with the other image-security evidence. +- [x] Export the stopped, already-tested image without executing it, mount the archive read-only, and extract as namespaced root into the scanner's disposable tmpfs so numeric ownership evidence is preserved. +- [x] Configure OpenSCAP offline mode directly with `OSCAP_PROBE_ROOT` because UBI AppStream does not ship the `oscap-chroot` wrapper. Run with no Docker/Podman socket, host namespace, workflow secrets, or evaluation-time network; use a read-only scanner root, `no-new-privileges`, drop all capabilities, and add only `CHOWN`, `FOWNER`, `DAC_OVERRIDE`, and `SYS_CHROOT` for metadata preservation, restrictive-file inspection/results output, and offline probes. +- [x] Generate and retain architecture-specific ARF XML, XCCDF XML, HTML, full JSON rule inventory, target/scanner image IDs, architecture, scanner/content versions, data-stream hash, and exit code with the other image-security evidence. Add the tailoring hash when the reviewed tailoring exists. - [ ] Run report-only on native AMD64 and ARM64 for at least three scheduled or `main` executions. Evaluation errors fail immediately; selected-rule findings become blocking only after the baseline is stable and reviewed. - [ ] Cross-check one exact image digest with `oscap-podman` on a disposable RHEL 9 host. Reconcile platform/applicability differences before enforcement; do not grant routine hosted CI root or engine access merely to match that command. **Exit evidence** -- [ ] Retain the discovery report, final tailoring, rule-rationale/exclusion review, three stable two-architecture runs, capability inspection, and `oscap-chroot` versus `oscap-podman` comparison. +- [ ] Retain the discovery report, final tailoring, rule-rationale/exclusion review, three stable two-architecture runs, capability inspection, and `OSCAP_PROBE_ROOT` versus `oscap-podman` comparison. - [ ] State precisely that the result covers selected image-filesystem controls and is not CIS/STIG certification of the host, OpenShift cluster, or production deployment. #### 6. Security-control provenance, STIG analysis, and SCTM export diff --git a/docs/SCAP.md b/docs/SCAP.md index e9c2444..a016e9a 100644 --- a/docs/SCAP.md +++ b/docs/SCAP.md @@ -7,13 +7,42 @@ audit daemon, host networking, or machine-wide security policy. Those controls belong to the container host or deployment platform and are outside this image's control. -The project will therefore establish a measured baseline first, then maintain -a tailored UBI 9 Micro container profile containing only applicable, +The project is therefore establishing a measured baseline first, then will +maintain a tailored UBI 9 Micro container profile containing only applicable, image-owned rules. Passing that profile means the inspected image filesystem meets the documented rules; it is not a claim that the image, host, OpenShift cluster, or complete ClickHouse deployment is CIS- or STIG-certified. -## Recommended CI architecture +## Implemented discovery baseline + +The discovery implementation pins these inputs. An update requires changing +the values in `Containerfile.scap`, recalculating both hashes, reviewing the +content changes, and rerunning both native architecture jobs. + +| Input | Pin | Verification | +| --- | --- | --- | +| Scanner base | UBI Minimal 9.8 manifest digest `sha256:7fbeae18dc9476399f565e68255f602a3374ea8614ba3d14843565131a13ff93` | BuildKit resolves the digest for the native runner architecture. | +| OpenSCAP engine | UBI AppStream RPM `openscap-scanner-1.3.14-1.el9_8` | The build fails if that exact NEVRA cannot be installed; the complete resulting RPM inventory and `oscap --version` are retained so dependency drift is visible. | +| ComplianceAsCode content | Release `0.1.82` ZIP | Archive SHA-256 `765e84bdce7f9055f9b9c2dd0ee2b713d4255f8eec94eac6d35ea4973c28919c` is checked before extraction. | +| RHEL 9 data stream | `ssg-rhel9-ds.xml` from release `0.1.82` | Data-stream SHA-256 `92204daafbf4f38011671ef034fae4cffb48f708516186710346a9ec702a1f8f` is checked at build and recorded at evaluation. | +| Discovery profile | `xccdf_org.ssgproject.content_profile_stig` | ComplianceAsCode 0.1.82 does not contain a RHEL 9 Standard profile. STIG is used as a broad discovery source because this project must review DISA-derived objectives; the inventory records every result without adopting the profile wholesale. | + +Red Hat's public UBI 9 AppStream repositories provide `openscap-scanner` for +both x86_64 and aarch64, but do not provide `openscap-utils`. Consequently, +the implementation invokes the documented `OSCAP_PROBE_ROOT` offline mode +directly instead of relying on the `oscap-chroot` convenience wrapper from +`openscap-utils`. The underlying OpenSCAP offline evaluation mechanism is the +same. CI records the locally built scanner image ID; if the project later +publishes the tool image for reuse, consumers must use its manifest digest, +not its mutable tag. + +`Containerfile.scap` and the scanner scripts are CI tooling only. They do not +add OpenSCAP, Python, `unzip`, a package manager, or any scanner content to the +released ClickHouse image. The scanner image defaults to unprivileged UID/GID +65534; only the reviewed wrapper overrides it to namespaced UID 0 while adding +the explicitly bounded mounts, capabilities, network isolation, and limits. + +## CI security architecture Do not run Podman inside Podman and do not mount a Docker or Podman socket into the scanner. Nested container engines commonly require additional namespace, @@ -25,35 +54,104 @@ Use this flow instead: 1. Build and smoke-test the architecture-specific image in the existing native CI job. -2. Create a stopped container and export its merged filesystem into an - ephemeral staging directory. Exporting does not execute image content. -3. Run a digest-pinned UBI 9 OpenSCAP tool image with the exported filesystem - mounted read-only and a separate results directory mounted read-write. -4. Run `oscap-chroot` against the read-only root and a pinned RHEL 9 data - stream. Give the scanner no engine socket, host namespaces, secrets, or - network access during evaluation. Grant only the minimum chroot-related - capability proven necessary by the qualification test. +2. Create a stopped container and export its merged filesystem to an ephemeral + tar archive. `create` and `export` do not execute image content. +3. Run the locally built, pinned-input UBI 9 OpenSCAP tool image with that tar + archive mounted read-only and a separate results directory mounted + read-write. Extract inside the scanner's disposable tmpfs as namespaced UID + 0 so numeric owners and modes are preserved; unprivileged host extraction + would corrupt ownership evidence. +4. Evaluate the extracted tree with `OSCAP_PROBE_ROOT` and the pinned RHEL 9 + data stream. Give the scanner no engine socket, host namespace, workflow + secret, or evaluation-time network. Drop all capabilities, then add only + `CAP_CHOWN` and `CAP_FOWNER` to preserve exported metadata, + `CAP_DAC_OVERRIDE` to read restrictive target files and write the + host-owned results mount, and `CAP_SYS_CHROOT` for offline probes. Enable + `no-new-privileges`, make the scanner root filesystem read-only, and bound + memory and process count. Limit writable storage to the extracted-root and + OpenSCAP temporary tmpfs mounts plus the results directory. 5. Upload the ARF XML, XCCDF results, HTML report, tool/content versions, data stream hash, image digest, architecture, and tailoring hash as CI evidence. 6. Delete the exported root filesystem and stopped container after the job. -The scanner process may run as UID 0 *inside its isolated scanner container* so +The scanner process runs as UID 0 *inside its isolated scanner container* so it can inspect the mounted tree. That is different from granting the workflow a privileged container, host root, an engine socket, or broad host mounts. The target filesystem remains read-only and the ClickHouse image is never started as root. -The implementation must first prove whether `CAP_SYS_CHROOT` alone is needed. -If the selected runner cannot operate with that narrow allowance, stop and -review the design rather than adding `--privileged` or broad capabilities. +The native CI jobs prove that the scan operates with only those four named +capabilities after `--cap-drop all`. If either runner cannot operate with that +narrow allowance, the job fails; do not add `--privileged` or broader +capabilities to make it pass. + +The scanner build and evaluation steps temporarily continue so later security +checks and artifact upload still run. A final step requires both outcomes to +be successful, so this sequencing preserves evidence without weakening the +required `image` check. + +## Result semantics and evidence + +Discovery is non-blocking only for ordinary `fail`, `notapplicable`, and +`notchecked` rule results. Failure to build or run the scanner, malformed or +empty XCCDF, and any `error`, `unknown`, or missing result fail the architecture +job. This distinction prevents "report-only" from hiding a broken scan. + +Each `image-security--` artifact includes: + +- `results.arf.xml`, the complete Asset Reporting Format evidence; +- `results.xccdf.xml`, the machine-readable evaluation results; +- `report.html`, the reviewer-oriented report; +- `summary.json`, a deterministic count and complete rule/result inventory; +- the OpenSCAP version, exact installed RPMs, data-stream hash, and OpenSCAP + exit code. + +`summary.json` also binds the evidence to the target image ID, scanner image +ID, architecture, profile, and data-stream SHA-256. It deliberately does not +label an upstream discovery-profile result as container, host, CIS, or STIG +certification. + +## Initial native discovery result + +[GitHub Actions run 34151979084](https://github.com/datopsis/clickhouse-server-ubi9/actions/runs/34151979084) +qualified commit `c099690` on both native architectures. The retained AMD64 +and ARM64 inventories contained the same 1,540 rule IDs and results: + +| Result | AMD64 | ARM64 | +| --- | ---: | ---: | +| `pass` | 59 | 59 | +| `fail` | 7 | 7 | +| `notapplicable` | 410 | 410 | +| `notchecked` | 1 | 1 | +| `notselected` | 1,063 | 1,063 | +| `error`, `unknown`, or missing | 0 | 0 | + +OpenSCAP returned its documented noncompliance status 2 on both runners. The +seven discovery failures were: + +- `accounts_umask_etc_bashrc`; +- `accounts_umask_etc_profile`; +- `configure_crypto_policy`; +- `file_groupownership_system_commands_dirs`; +- `file_ownership_binary_dirs`; +- `network_configure_name_resolution`; +- `package_crypto-policies_installed`. + +`security_patches_up_to_date` was `notchecked`. None of these results is an +adopted container control yet. The next package must inspect the associated +OVAL logic, distinguish image-owned behavior from absent host facilities, +compare the result with the current Containerfile/SBOM, and document its +applicability decision and rationale before selecting or excluding the rule. +In particular, a profile `pass` is not sufficient evidence that a rule is +applicable to a minimal container. ## Profile-development method -The first implementation pull request should: +The upstream RHEL 9 STIG profile is a discovery input, not a statement that +every rule is applicable or inherited. The discovery implementation provides +the pinned scan and inventory. The next +profile-development pull request must: -- pin the OpenSCAP engine, ComplianceAsCode content, and scanner-image digest; -- verify the downloaded or packaged RHEL 9 data stream and record its SHA-256; -- run the upstream RHEL 9 Standard profile in discovery/report-only mode; - classify every result as applicable, not applicable, inherited from the platform, pass, fail, error, or not checked; - commit an XCCDF tailoring file with a Datopsis-specific profile identifier; @@ -87,27 +185,70 @@ SCAP integration should be incremental: 4. **Drift review:** re-run discovery whenever the UBI major version, OpenSCAP engine, ComplianceAsCode data stream, or tailoring changes. -Before enforcement, compare one image digest's `oscap-chroot` result with +Before enforcement, compare one image digest's `OSCAP_PROBE_ROOT` result with `oscap-podman` on a disposable RHEL 9 host. Investigate differences in platform facts, applicability, and rule results. This is a qualification cross-check, not a reason to give routine GitHub-hosted CI root access. -## Local Red Hat reproduction +## Local discovery with Podman + +Run this only on a native Linux AMD64 or ARM64 host with Podman, Bash, and +Python 3. Rootless Podman is sufficient if the host supports user namespaces +and delegated cgroups. The scan wrapper grants the four documented +filesystem/chroot capabilities only inside Podman's user namespace; it does +not require host root or start the ClickHouse image as root. + +```console +git clone https://github.com/datopsis/clickhouse-server-ubi9.git +cd clickhouse-server-ubi9 + +ARCHITECTURE=amd64 # use arm64 on an ARM64 host +IMAGE="localhost/clickhouse-server-ubi9:test-${ARCHITECTURE}" +SCANNER_IMAGE="localhost/datopsis-openscap:0.1.82-${ARCHITECTURE}" + +podman build --format docker --platform "linux/${ARCHITECTURE}" \ + --file Containerfile --tag "${IMAGE}" . +podman build --format docker --platform "linux/${ARCHITECTURE}" \ + --file Containerfile.scap --tag "${SCANNER_IMAGE}" . + +CONTAINER_RUNTIME=podman \ +IMAGE="${IMAGE}" \ +SCAP_SCANNER_IMAGE="${SCANNER_IMAGE}" \ +ARCHITECTURE="${ARCHITECTURE}" \ +SCAP_RESULTS_DIR="scap-results-${ARCHITECTURE}" \ + bash scripts/scap-scan.sh +``` + +Review `summary.json` first, then the HTML report and XML evidence. Confirm +`rootfs.tar` was removed. Do not commit the generated results. If rootless +Podman reports that memory limits are unsupported, qualify the host's cgroup +configuration; do not remove the CI limit without a documented risk review. + +For a disconnected scan, build both images and run the wrapper while connected +once, save them with `podman save`, transfer the repository plus image archive +through the approved media process, load with `podman load`, and run the same +wrapper. Evaluation already uses `--network none`; no content fetch occurs. +Record and verify SHA-256 hashes for the repository commit/archive and saved +images at both sides of the transfer. + +## RHEL `oscap-podman` qualification cross-check -On a disposable RHEL 9 test host with the OpenSCAP container tooling installed, -the reference scan remains: +On a disposable RHEL 9 test host with `openscap-utils`, `openscap-scanner`, and +`scap-security-guide` installed, use `oscap-podman` only for the planned +one-digest comparison. First load the exact CI-qualified target digest and copy +the reviewed tailoring file to the host. Then run: ```console -sudo oscap-podman xccdf eval \ +sudo oscap-podman xccdf eval \ --profile \ - --tailoring-file datopsis-ubi9-micro-tailoring.xml \ + --tailoring-file security/scap/datopsis-ubi9-micro-tailoring.xml \ --results-arf results.arf.xml \ --report report.html \ /usr/share/xml/scap/ssg/content/ssg-rhel9-ds.xml ``` -`oscap-podman` requires root because it integrates with local container storage. -Use only a dedicated test host containing no unrelated workloads or secrets. +`oscap-podman` requires host root because it integrates with local container +storage. Use only a dedicated test host containing no unrelated workloads or secrets. Record the exact image digest, host version, Podman/OpenSCAP versions, content package version, tailoring hash, command, and sanitized results. diff --git a/docs/diagrams/assurance-pipeline.svg b/docs/diagrams/assurance-pipeline.svg index 7135f42..04609e5 100644 --- a/docs/diagrams/assurance-pipeline.svg +++ b/docs/diagrams/assurance-pipeline.svg @@ -4,7 +4,7 @@ Build, release, and assurance pipeline Pinned inputsUBI digestsClickHouse archives + SHA-512 Native buildsAMD64 runnerARM64 runnerArchitecture assertion - Qualification gatesSmoke + rootless storageConnected/disconnected TLSTrivy + Grype + CodeQLTailored SCAP (planned) + Qualification gatesSmoke + rootless storageConnected/disconnected TLSTrivy + Grype + CodeQLSCAP discovery reports Digest-bound evidenceSPDX SBOM + vulnerability inventoryTest logs + SCAP ARF/XCCDFControl matrix + source hashes Human reviewApplicability and residual riskSecond-maintainer release review Release workflowMulti-arch manifestCosign + provenanceImmutable tag diff --git a/scripts/scap-container.sh b/scripts/scap-container.sh new file mode 100644 index 0000000..2597fe3 --- /dev/null +++ b/scripts/scap-container.sh @@ -0,0 +1,59 @@ +#!/usr/bin/env bash +set -Eeuo pipefail + +readonly input_tar="/input/rootfs.tar" +readonly scan_root="/scan-root" +readonly results_dir="/results" +readonly data_stream="/opt/scap/ssg-rhel9-ds.xml" +readonly profile="xccdf_org.ssgproject.content_profile_stig" + +if [[ ! -r "${input_tar}" ]]; then + echo "SCAP input is not readable: ${input_tar}" >&2 + exit 1 +fi +if [[ ! -d "${results_dir}" || ! -w "${results_dir}" ]]; then + echo "SCAP results directory is not writable: ${results_dir}" >&2 + exit 1 +fi +if [[ ! -d "${scan_root}" || ! -w "${scan_root}" ]]; then + echo "SCAP extraction directory must be a writable tmpfs: ${scan_root}" >&2 + exit 1 +fi + +# Docker/Podman creates this archive from a stopped container. Extracting here +# as namespaced root preserves the numeric owners OpenSCAP evaluates. Target +# content is data only: nothing from the exported filesystem is executed. +tar --extract --file "${input_tar}" --directory "${scan_root}" --same-owner + +sha256sum "${data_stream}" > "${results_dir}/datastream.sha256" +oscap --version > "${results_dir}/openscap-version.txt" +rpm --query openscap openscap-scanner > "${results_dir}/openscap-packages.txt" +rpm --query --all --queryformat '%{NAME}-%{EPOCHNUM}:%{VERSION}-%{RELEASE}.%{ARCH}\n' \ + | sort > "${results_dir}/scanner-packages.txt" + +# OSCAP_PROBE_ROOT is OpenSCAP's documented offline-filesystem mechanism and +# is what the oscap-chroot convenience wrapper configures. UBI AppStream ships +# openscap-scanner but not the openscap-utils package containing that wrapper. +export OSCAP_PROBE_ROOT="${scan_root}" + +set +e +oscap xccdf eval \ + --profile "${profile}" \ + --results-arf "${results_dir}/results.arf.xml" \ + --results "${results_dir}/results.xccdf.xml" \ + --report "${results_dir}/report.html" \ + "${data_stream}" +oscap_status=$? +set -e + +printf '%s\n' "${oscap_status}" > "${results_dir}/oscap-exit-code.txt" +case "${oscap_status}" in + 0 | 2) + # OpenSCAP uses 2 for a completed evaluation with noncompliant rules. + # Findings are report-only during baseline discovery. + ;; + *) + echo "OpenSCAP evaluation failed with exit code ${oscap_status}" >&2 + exit "${oscap_status}" + ;; +esac diff --git a/scripts/scap-scan.sh b/scripts/scap-scan.sh new file mode 100644 index 0000000..75d6c61 --- /dev/null +++ b/scripts/scap-scan.sh @@ -0,0 +1,82 @@ +#!/usr/bin/env bash +set -Eeuo pipefail + +runtime="${CONTAINER_RUNTIME:-podman}" +target_image="${IMAGE:-ghcr.io/datopsis/clickhouse-server-ubi9:test}" +scanner_image="${SCAP_SCANNER_IMAGE:-localhost/datopsis-openscap:0.1.82}" +architecture="${ARCHITECTURE:-$(uname -m)}" +results_dir="${SCAP_RESULTS_DIR:-scap-results-${architecture}}" + +case "${runtime}" in + podman | docker) ;; + *) + echo "Unsupported CONTAINER_RUNTIME '${runtime}'; use podman or docker" >&2 + exit 1 + ;; +esac + +command -v "${runtime}" >/dev/null 2>&1 || { + echo "Container runtime not found: ${runtime}" >&2 + exit 1 +} +command -v python3 >/dev/null 2>&1 || { + echo "python3 is required to create the SCAP result inventory" >&2 + exit 1 +} + +mkdir -p "${results_dir}" +results_dir="$(cd "${results_dir}" && pwd -P)" +rootfs_tar="${results_dir}/rootfs.tar" +container_id="" + +cleanup() { + if [[ -n "${container_id}" ]]; then + "${runtime}" rm --force "${container_id}" >/dev/null 2>&1 || true + fi + rm -f -- "${rootfs_tar}" +} +trap cleanup EXIT + +container_id="$("${runtime}" create "${target_image}")" +"${runtime}" export --output "${rootfs_tar}" "${container_id}" +"${runtime}" rm "${container_id}" >/dev/null +container_id="" + +target_image_id="$("${runtime}" image inspect --format '{{.Id}}' "${target_image}")" +scanner_image_id="$("${runtime}" image inspect --format '{{.Id}}' "${scanner_image}")" + +volume_label="" +if [[ "${runtime}" == "podman" ]]; then + volume_label=",Z" +fi + +"${runtime}" run --rm \ + --network none \ + --read-only \ + --user 0:0 \ + --cap-drop all \ + --cap-add chown \ + --cap-add dac_override \ + --cap-add fowner \ + --cap-add sys_chroot \ + --security-opt no-new-privileges \ + --pids-limit 256 \ + --memory 4g \ + --tmpfs /scan-root:rw,nosuid,nodev,size=3g \ + --tmpfs /tmp:rw,nosuid,nodev,size=512m \ + --volume "${rootfs_tar}:/input/rootfs.tar:ro${volume_label}" \ + --volume "${results_dir}:/results:rw${volume_label}" \ + "${scanner_image}" + +python3 scripts/scap-summary.py \ + --results "${results_dir}/results.xccdf.xml" \ + --output "${results_dir}/summary.json" \ + --architecture "${architecture}" \ + --profile xccdf_org.ssgproject.content_profile_stig \ + --target-image "${target_image}" \ + --target-image-id "${target_image_id}" \ + --scanner-image "${scanner_image}" \ + --scanner-image-id "${scanner_image_id}" \ + --datastream-sha256 92204daafbf4f38011671ef034fae4cffb48f708516186710346a9ec702a1f8f + +echo "SCAP discovery evidence written to ${results_dir}" diff --git a/scripts/scap-summary.py b/scripts/scap-summary.py new file mode 100644 index 0000000..f4aa2db --- /dev/null +++ b/scripts/scap-summary.py @@ -0,0 +1,92 @@ +#!/usr/bin/env python3 +"""Create a deterministic inventory from an OpenSCAP XCCDF result file.""" + +from __future__ import annotations + +import argparse +import json +import sys +import xml.etree.ElementTree as ET +from collections import Counter +from pathlib import Path + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser() + parser.add_argument("--results", required=True, type=Path) + parser.add_argument("--output", required=True, type=Path) + parser.add_argument("--architecture", required=True) + parser.add_argument("--profile", required=True) + parser.add_argument("--target-image", required=True) + parser.add_argument("--target-image-id", required=True) + parser.add_argument("--scanner-image", required=True) + parser.add_argument("--scanner-image-id", required=True) + parser.add_argument("--datastream-sha256", required=True) + return parser.parse_args() + + +def local_name(tag: str) -> str: + return tag.rsplit("}", 1)[-1] + + +def inventory_root(root: ET.Element) -> list[dict[str, str]]: + rules: list[dict[str, str]] = [] + for element in root.iter(): + if local_name(element.tag) != "rule-result": + continue + result = next( + ( + (child.text or "").strip() + for child in element + if local_name(child.tag) == "result" + ), + "missing", + ) + rules.append({"id": element.attrib.get("idref", ""), "result": result}) + if not rules: + raise ValueError("XCCDF results contain no rule-result elements") + return sorted(rules, key=lambda item: item["id"]) + + +def inventory(path: Path) -> list[dict[str, str]]: + return inventory_root(ET.parse(path).getroot()) + + +def has_operational_errors(counts: Counter[str]) -> bool: + return bool(counts["error"] or counts["unknown"] or counts["missing"]) + + +def main() -> int: + args = parse_args() + try: + rules = inventory(args.results) + except (ET.ParseError, OSError, ValueError) as error: + print(f"Unable to inventory SCAP results: {error}", file=sys.stderr) + return 1 + + counts = Counter(rule["result"] for rule in rules) + document = { + "schema_version": 1, + "mode": "discovery-report-only", + "architecture": args.architecture, + "profile": args.profile, + "target": {"reference": args.target_image, "image_id": args.target_image_id}, + "scanner": { + "reference": args.scanner_image, + "image_id": args.scanner_image_id, + }, + "datastream_sha256": args.datastream_sha256, + "counts": dict(sorted(counts.items())), + "rules": rules, + } + args.output.write_text(json.dumps(document, indent=2) + "\n", encoding="utf-8") + + print("SCAP discovery counts: " + ", ".join(f"{k}={v}" for k, v in sorted(counts.items()))) + if has_operational_errors(counts): + print("SCAP produced an error, unknown, or missing result", file=sys.stderr) + return 1 + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_scap_summary.py b/tests/test_scap_summary.py new file mode 100644 index 0000000..739d718 --- /dev/null +++ b/tests/test_scap_summary.py @@ -0,0 +1,46 @@ +import unittest +from collections import Counter +from importlib.util import module_from_spec, spec_from_file_location +from pathlib import Path +from xml.etree import ElementTree as ET + + +ROOT = Path(__file__).resolve().parents[1] +SCRIPT = ROOT / "scripts" / "scap-summary.py" +SPEC = spec_from_file_location("scap_summary", SCRIPT) +assert SPEC is not None and SPEC.loader is not None +SCAP_SUMMARY = module_from_spec(SPEC) +SPEC.loader.exec_module(SCAP_SUMMARY) + + +class ScapSummaryTests(unittest.TestCase): + def inventory(self, xml: str) -> list[dict[str, str]]: + return SCAP_SUMMARY.inventory_root(ET.fromstring(xml)) + + def test_inventories_and_sorts_results(self) -> None: + rules = self.inventory( + """ + + fail + pass + notapplicable + """ + ) + self.assertEqual( + Counter(rule["result"] for rule in rules), + {"fail": 1, "notapplicable": 1, "pass": 1}, + ) + self.assertEqual([rule["id"] for rule in rules], ["rule_a", "rule_b", "rule_c"]) + + def test_errors_make_discovery_operationally_fail(self) -> None: + self.assertTrue(SCAP_SUMMARY.has_operational_errors(Counter({"error": 1}))) + self.assertTrue(SCAP_SUMMARY.has_operational_errors(Counter({"unknown": 1}))) + self.assertFalse(SCAP_SUMMARY.has_operational_errors(Counter({"fail": 3}))) + + def test_empty_result_is_rejected(self) -> None: + with self.assertRaisesRegex(ValueError, "no rule-result"): + self.inventory("") + + +if __name__ == "__main__": + unittest.main()