diff --git a/Makefile b/Makefile new file mode 100644 index 00000000..a152c836 --- /dev/null +++ b/Makefile @@ -0,0 +1,36 @@ +PYTHON_VERSION = 3.10 +CUDA_VERSION = 12.2 + +PYTHON_INC = /usr/include/python$(PYTHON_VERSION) +TORCH_INC = /usr/local/lib/python$(PYTHON_VERSION)/dist-packages/torch/include +TORCH_PYBIND = $(TORCH_INC)/pybind11 +OPENCV_INC = /usr/local/include/opencv4 +CUDA_INC = /usr/local/cuda-$(CUDA_VERSION)/targets/aarch64-linux/include +DEEPSTREAM_INC = /opt/nvidia/deepstream/deepstream/sources/includes +GLIB_INC = /usr/include/glib-2.0 +GLIB_LIB_INC = /usr/lib/aarch64-linux-gnu/glib-2.0/include +GST_INC = /usr/include/gstreamer-1.0 +PYBIND11_INC = /usr/local/lib/python$(PYTHON_VERSION)/dist-packages/pybind11/include + +CUDA_LIB = /usr/local/cuda-$(CUDA_VERSION)/targets/aarch64-linux/lib +DEEPSTREAM_LIB = /opt/nvidia/deepstream/deepstream/lib + +LIBS = -lpython$(PYTHON_VERSION) \ + -lopencv_core -lopencv_imgproc -lopencv_cudaimgproc -lopencv_cudaarithm -lopencv_cudaoptflow \ + -lcuda -lcudart -lEGL -lnvbufsurface -lnvds_meta \ + -lgstreamer-1.0 -lgobject-2.0 -lglib-2.0 + +CXXFLAGS = -O3 -Wall -shared -std=c++17 -fPIC +INCLUDES = -I$(PYTHON_INC) -I$(TORCH_INC) -I$(TORCH_PYBIND) -I$(PYBIND11_INC) \ + -I$(OPENCV_INC) -I$(CUDA_INC) -I$(DEEPSTREAM_INC) \ + -I$(GLIB_INC) -I$(GLIB_LIB_INC) -I$(GST_INC) +LDFLAGS = -L$(CUDA_LIB) -L$(DEEPSTREAM_LIB) + +TARGET = src/gpu_probe$(shell python3-config --extension-suffix) +SRC = src/gpu_probe.cpp src/gpu_frame_change_detector.cpp + +all: + g++ $(CXXFLAGS) $(INCLUDES) -o $(TARGET) $(SRC) $(LDFLAGS) $(LIBS) + +clean: + rm -f $(TARGET) diff --git a/docs/frame_comparison_results.md b/docs/frame_comparison_results.md index f5674ca7..9ecfb907 100644 --- a/docs/frame_comparison_results.md +++ b/docs/frame_comparison_results.md @@ -76,7 +76,9 @@ Decision: ⚙️ PROCESS ## Decision Logic -Frames are skipped when minimal change is detected. +### CPU Implementation + +Frames are skipped when minimal change is detected: ```python if mse_val < 50 and ssim_val > 0.995 and flow_val < 0.1: @@ -85,18 +87,60 @@ else: process = True ``` +### GPU Implementation (C++/CUDA) + +The GPU implementation uses adjusted thresholds due to precision differences: + +```cpp +bool is_static = (mse_val < 10.0 && ssim_val > 0.995 && flow_val < 0.05); +return {!is_static, metrics}; +``` + +**Key Differences:** +- **MSE threshold:** `50` (CPU) → `10.0` (GPU) +- **SSIM threshold:** `0.995` (both, but different precision) +- **Flow threshold:** `0.1` (CPU) → `0.05` (GPU) --- -## Metric Interpretation +## Implementation Differences: CPU vs GPU + +### Precision & Accuracy + +| Aspect | CPU Implementation | GPU Implementation | +|--------|-------------------|-------------------| +| **Math Precision** | `float64` (double precision) | `float32` (single precision) | +| **Optical Flow API** | `cv2.calcOpticalFlowFarneback()` | `cv::cuda::FarnebackOpticalFlow::create()->calc()` | +| **Flow Accuracy** | Higher precision, more accurate | Lower precision, slightly less accurate | +| **Computation** | Sequential on CPU cores | Parallel on GPU | -| Metric | Typical Range | Interpretation | -| ------------ | ------------------------------------------------------------------------------------- | -------------- | -| **MSE** | 0–50 → no change
50–1000 → moderate change
>1000 → significant scene change | | -| **SSIM** | >0.995 → identical
0.90–0.995 → moderate change
<0.90 → major visual difference | | -| **FLOW** | <0.1 → static
0.5–2.0 → small motion
>2.0 → strong/global motion | | -| **MOTION %** | <5% → static
5–30% → partial motion
>30% → large motion area | | +### Behavioral Changes + +Due to the precision and algorithm differences in GPU implementation: + +1. **Lower MSE threshold (10.0 vs 50):** + - GPU's `float32` arithmetic produces slightly different MSE values + - More conservative threshold compensates for reduced precision + - Prevents false negatives (missing actual changes) + +2. **Lower Flow threshold (0.05 vs 0.1):** + - GPU optical flow (`cv::cuda::FarnebackOpticalFlow`) is less accurate than CPU version + - Threshold adjusted to account for measurement noise + - Reduces false positives (detecting motion where there is none) + +3. **Simplified method selection:** + - GPU focuses on MSE + SSIM + Optical Flow only + - Histogram correlation, Motion Mask, and ORB matching excluded for performance + - These methods don't provide significant GPU acceleration benefits + +--- + +## Metric Interpretation -These thresholds were derived empirically from the RTSP domain test videos and can be fine-tuned for other environments or resolutions. +| Metric | CPU Typical Range | GPU Typical Range | Interpretation | +| ------------ | ----------------- | ----------------- | -------------- | +| **MSE** | 0–50 → no change
50–1000 → moderate
>1000 → major | 0–10 → no change
10–800 → moderate
>800 → major | GPU values tend to be slightly lower due to float32 | +| **SSIM** | >0.995 → identical
0.90–0.995 → moderate
<0.90 → major | >0.995 → identical
0.92–0.995 → moderate
<0.92 → major | Similar ranges, GPU slightly less precise | +| **FLOW** | <0.1 → static
0.5–2.0 → small motion
>2.0 → strong | <0.05 → static
0.3–1.5 → small motion
>1.5 → strong | GPU flow values are generally lower | --- diff --git a/src/gpu_frame_change_detector.cpp b/src/gpu_frame_change_detector.cpp new file mode 100644 index 00000000..56c97365 --- /dev/null +++ b/src/gpu_frame_change_detector.cpp @@ -0,0 +1,124 @@ +#include "gpu_frame_change_detector.hpp" +#include +#include +#include +#include +#include + +std::unique_ptr GPUFrameChangeDetector::instance = nullptr; + +GPUFrameChangeDetector::GPUFrameChangeDetector() + : mse_thresh(500.0), ssim_thresh(0.99), flow_thresh(2.0), initialized(true) {} + +GPUFrameChangeDetector* GPUFrameChangeDetector::getInstance() { + if (!instance) { + instance = std::unique_ptr(new GPUFrameChangeDetector()); + } + return instance.get(); +} + +static cv::Scalar mean_gpu(const cv::cuda::GpuMat& mat) { + cv::Scalar sum_val = cv::cuda::sum(mat); + double total_elements = static_cast(mat.rows * mat.cols); + return sum_val * (1.0 / total_elements); +} + +cv::cuda::GpuMat GPUFrameChangeDetector::preprocess_gpu(const cv::cuda::GpuMat& frame) { + return frame; +} + +double GPUFrameChangeDetector::mse_gpu(const cv::cuda::GpuMat& imgA, const cv::cuda::GpuMat& imgB) { + cv::cuda::GpuMat bgrA, bgrB; + cv::cuda::cvtColor(imgA, bgrA, cv::COLOR_RGBA2BGR); + cv::cuda::cvtColor(imgB, bgrB, cv::COLOR_RGBA2BGR); + + cv::cuda::GpuMat diff, sqr; + cv::cuda::absdiff(bgrA, bgrB, diff); + cv::cuda::multiply(diff, diff, sqr); + cv::Scalar mean_val = mean_gpu(sqr); + return mean_val[0]; +} + +double GPUFrameChangeDetector::simple_ssim_gpu(const cv::cuda::GpuMat& imgA, const cv::cuda::GpuMat& imgB) { + cv::cuda::GpuMat grayA, grayB; + cv::cuda::cvtColor(imgA, grayA, cv::COLOR_RGBA2GRAY); + cv::cuda::cvtColor(imgB, grayB, cv::COLOR_RGBA2GRAY); + + cv::Scalar mu_x = mean_gpu(grayA); + cv::Scalar mu_y = mean_gpu(grayB); + + cv::cuda::GpuMat diffX, diffY; + cv::cuda::subtract(grayA, mu_x, diffX); + cv::cuda::subtract(grayB, mu_y, diffY); + + cv::cuda::GpuMat diffX2, diffY2, diffXY; + cv::cuda::multiply(diffX, diffX, diffX2); + cv::cuda::multiply(diffY, diffY, diffY2); + cv::cuda::multiply(diffX, diffY, diffXY); + + double sigma_x = std::sqrt(mean_gpu(diffX2)[0]); + double sigma_y = std::sqrt(mean_gpu(diffY2)[0]); + double sigma_xy = mean_gpu(diffXY)[0]; + + double C1 = 6.5025, C2 = 58.5225; + double ssim = ((2 * mu_x[0] * mu_y[0] + C1) * (2 * sigma_xy + C2)) / + ((mu_x[0]*mu_x[0] + mu_y[0]*mu_y[0] + C1) * + (sigma_x*sigma_x + sigma_y*sigma_y + C2)); + + return ssim; +} + +double GPUFrameChangeDetector::optical_flow_gpu(const cv::cuda::GpuMat& imgA, const cv::cuda::GpuMat& imgB) { + cv::cuda::GpuMat bgrA, bgrB, grayA, grayB; + + cv::cuda::cvtColor(imgA, bgrA, cv::COLOR_RGBA2BGR); + cv::cuda::cvtColor(imgB, bgrB, cv::COLOR_RGBA2BGR); + cv::cuda::cvtColor(bgrA, grayA, cv::COLOR_BGR2GRAY); + cv::cuda::cvtColor(bgrB, grayB, cv::COLOR_BGR2GRAY); + + auto flow = cv::cuda::FarnebackOpticalFlow::create(3, 0.5, false, 15, 3, 5, 1.2, 0); + + cv::cuda::GpuMat flow_result; + flow->calc(grayA, grayB, flow_result); + + std::vector flow_xy; + cv::cuda::split(flow_result, flow_xy); + + cv::cuda::GpuMat mag, angle; + cv::cuda::cartToPolar(flow_xy[0], flow_xy[1], mag, angle, true); + + cv::cuda::GpuMat active_mask; + cv::cuda::threshold(mag, active_mask, 0.5, 1.0, cv::THRESH_BINARY); + + cv::cuda::GpuMat active_pixels; + mag.copyTo(active_pixels, active_mask); + + cv::Scalar sum_active = cv::cuda::sum(active_pixels); + cv::Scalar count_active = cv::cuda::sum(active_mask); + + if (count_active[0] > 0) { + return sum_active[0] / count_active[0]; + } else { + return 0.0; + } +} + +std::pair> +GPUFrameChangeDetector::should_process_gpu_direct(const cv::cuda::GpuMat& gpu_frame) { + if (prev_frame_gpu.empty()) { + prev_frame_gpu = preprocess_gpu(gpu_frame); + return {true, {{"MSE", 0.0}, {"SSIM", 1.0}, {"FLOW", 0.0}}}; + } + + cv::cuda::GpuMat processed_curr = preprocess_gpu(gpu_frame); + + double mse_val = mse_gpu(prev_frame_gpu, processed_curr); + double ssim_val = simple_ssim_gpu(prev_frame_gpu, processed_curr); + double flow_val = optical_flow_gpu(prev_frame_gpu, processed_curr); + + bool is_static = (mse_val < 10.0 && ssim_val > 0.995 && flow_val < 0.05); + + prev_frame_gpu = processed_curr; + + return {!is_static, {{"MSE", mse_val}, {"SSIM", ssim_val}, {"FLOW", flow_val}}}; +} diff --git a/src/gpu_frame_change_detector.hpp b/src/gpu_frame_change_detector.hpp new file mode 100644 index 00000000..0def34a1 --- /dev/null +++ b/src/gpu_frame_change_detector.hpp @@ -0,0 +1,36 @@ +#pragma once +#include +#include +#include +#include + +class GPUFrameChangeDetector { +public: + // Singleton access + static GPUFrameChangeDetector* getInstance(); + + // Main logic: decide if frame should be processed + std::pair> + should_process_gpu_direct(const cv::cuda::GpuMat& gpu_frame); + +private: + // Constructor (private for singleton) + GPUFrameChangeDetector(); + + // Internal processing functions + cv::cuda::GpuMat preprocess_gpu(const cv::cuda::GpuMat& frame); + double mse_gpu(const cv::cuda::GpuMat& imgA, const cv::cuda::GpuMat& imgB); + double simple_ssim_gpu(const cv::cuda::GpuMat& imgA, const cv::cuda::GpuMat& imgB); + double optical_flow_gpu(const cv::cuda::GpuMat& imgA, const cv::cuda::GpuMat& imgB); + + // Internal state + static std::unique_ptr instance; + cv::cuda::GpuMat prev_frame_gpu; + + // Thresholds + const double mse_thresh; + const double ssim_thresh; + const double flow_thresh; + + bool initialized; +}; diff --git a/src/gpu_frames_skipping_pipeline.py b/src/gpu_frames_skipping_pipeline.py new file mode 100644 index 00000000..757d0479 --- /dev/null +++ b/src/gpu_frames_skipping_pipeline.py @@ -0,0 +1,140 @@ +import os +import gi +import cv2 +import pyds +import ctypes +import gpu_probe +from typing import Any, Dict, Optional + +gi.require_version("Gst", "1.0") +from gi.repository import GLib, Gst +from gpu_frame_change_detector import GPUFrameChangeDetector + +# Configurations +RTSP_URL: str = "rtsp://127.0.0.1:8554/test" +CONFIG_FILE: str = "/workspace/src/configs/resnet18.txt" +OUTPUT_DIR: str = "/workspace/output/frames/" +os.makedirs(OUTPUT_DIR, exist_ok=True) + +stats: Dict[str, int] = {"total": 0, "skipped": 0, "processed": 0} + +def jetson_frame_skip_probe(pad: Gst.Pad, info: Gst.PadProbeInfo, u_data: Optional[Any]) -> Gst.PadProbeReturn: + buffer_ptr: int = hash(info.get_buffer()) + batch_id: int = 0 + should_process: bool = gpu_probe.frame_probe(buffer_ptr, batch_id) + if should_process: + stats["processed"] += 1 + print(f"✅ PROCESSING frame {stats['total']}") + return Gst.PadProbeReturn.OK + else: + stats["skipped"] += 1 + print(f"⏭️ SKIPPING frame {stats['total']}") + return Gst.PadProbeReturn.DROP + +def build_jetson_pipeline() -> Gst.Pipeline: + """Build Jetson-compatible pipeline""" + pipeline: Gst.Pipeline = Gst.Pipeline.new("jetson-frame-skipping-pipeline") + + # Elements + rtspsrc: Gst.Element = Gst.ElementFactory.make("rtspsrc", "source") + depay: Gst.Element = Gst.ElementFactory.make("rtph264depay", "depay") + parse: Gst.Element = Gst.ElementFactory.make("h264parse", "parse") + decode: Gst.Element = Gst.ElementFactory.make("decodebin", "decode") + convert: Gst.Element = Gst.ElementFactory.make("videoconvert", "convert") + nvvideoconvert: Gst.Element = Gst.ElementFactory.make("nvvideoconvert", "nvvideoconvert") + capsfilter: Gst.Element = Gst.ElementFactory.make("capsfilter", "capsfilter") + streammux: Gst.Element = Gst.ElementFactory.make("nvstreammux", "streammux") + nvinfer: Gst.Element = Gst.ElementFactory.make("nvinfer", "nvinfer") + nvosd: Gst.Element = Gst.ElementFactory.make("nvdsosd", "osd") + nvvideoconvert2: Gst.Element = Gst.ElementFactory.make("nvvideoconvert", "nvvideoconvert2") + jpegenc: Gst.Element = Gst.ElementFactory.make("jpegenc", "jpegenc") + sink: Gst.Element = Gst.ElementFactory.make("multifilesink", "sink") + + for e in [rtspsrc, depay, parse, decode, convert, nvvideoconvert, + capsfilter, streammux, nvinfer, nvosd, nvvideoconvert2, jpegenc, sink]: + pipeline.add(e) + + # Properties + rtspsrc.set_property("location", RTSP_URL) + streammux.set_property("batch-size", 1) + streammux.set_property("width", 640) + streammux.set_property("height", 480) + nvinfer.set_property("config-file-path", CONFIG_FILE) + sink.set_property("location", os.path.join(OUTPUT_DIR, "jetson_frame_%05d.jpg")) + + caps: Gst.Caps = Gst.Caps.from_string("video/x-raw(memory:NVMM), format=RGBA") + capsfilter.set_property("caps", caps) + + # Links + depay.link(parse) + parse.link(decode) + convert.link(nvvideoconvert) + nvvideoconvert.link(capsfilter) + streammux.link(nvinfer) + nvinfer.link(nvosd) + nvosd.link(nvvideoconvert2) + nvvideoconvert2.link(jpegenc) + jpegenc.link(sink) + + # Dynamic pads + def on_pad_added_rtspsrc(src: Gst.Element, pad: Gst.Pad) -> None: + sinkpad: Gst.Pad = depay.get_static_pad("sink") + if not sinkpad.is_linked(): + pad.link(sinkpad) + + def on_pad_added_decode(src: Gst.Element, pad: Gst.Pad) -> None: + sinkpad: Gst.Pad = convert.get_static_pad("sink") + if not sinkpad.is_linked(): + pad.link(sinkpad) + + rtspsrc.connect("pad-added", on_pad_added_rtspsrc) + decode.connect("pad-added", on_pad_added_decode) + + srcpad: Gst.Pad = capsfilter.get_static_pad("src") + sinkpad: Gst.Pad = streammux.get_request_pad("sink_0") + srcpad.link(sinkpad) + + # Add probe + streammux_src_pad: Gst.Pad = streammux.get_static_pad("src") + streammux_src_pad.add_probe(Gst.PadProbeType.BUFFER, jetson_frame_skip_probe, None) + + return pipeline + +def run_jetson_pipeline() -> None: + """Run Jetson pipeline""" + Gst.init(None) + pipeline: Gst.Pipeline = build_jetson_pipeline() + + loop: GLib.MainLoop = GLib.MainLoop() + bus: Gst.Bus = pipeline.get_bus() + bus.add_signal_watch() + + def on_message(bus: Gst.Bus, msg: Gst.Message) -> None: + if msg.type == Gst.MessageType.ERROR: + err, _ = msg.parse_error() + print("ERROR:", err) + loop.quit() + elif msg.type == Gst.MessageType.EOS: + print("End of stream.") + loop.quit() + + bus.connect("message", on_message) + pipeline.set_state(Gst.State.PLAYING) + print(f"🔥 Jetson Frame Skipping Pipeline Started! Output: {OUTPUT_DIR}") + + try: + loop.run() + except KeyboardInterrupt: + pass + finally: + pipeline.set_state(Gst.State.NULL) + if stats["total"] > 0: + skip_ratio: float = stats["skipped"] / stats["total"] + print(f"\n🔥 Jetson Final Stats:") + print(f" Total: {stats['total']}") + print(f" Processed: {stats['processed']}") + print(f" Skipped: {stats['skipped']}") + print(f" Skip Ratio: {skip_ratio:.1%}") + +if __name__ == "__main__": + run_jetson_pipeline() diff --git a/src/gpu_probe.cpp b/src/gpu_probe.cpp new file mode 100644 index 00000000..38578bb9 --- /dev/null +++ b/src/gpu_probe.cpp @@ -0,0 +1,103 @@ +#include +#include +#include +#include +#include "nvbufsurface.h" +#include "gstnvdsmeta.h" +#include "gpu_frame_change_detector.hpp" +#include +#include + +namespace py = pybind11; + +bool frame_probe(uintptr_t buffer_ptr, int batch_id) +{ + static int frame_count = 0; + frame_count++; + + GstBuffer *buffer = nullptr; + GstMapInfo map_info = {0}; + NvBufSurface *surface = nullptr; + CUgraphicsResource cudaResource = 0; + bool buffer_mapped = false; + bool egl_mapped = false; + bool cuda_registered = false; + + try { + buffer = reinterpret_cast(buffer_ptr); + if (!buffer) return true; + + if (!gst_buffer_map(buffer, &map_info, GST_MAP_READ)) { + return true; + } + buffer_mapped = true; + + surface = (NvBufSurface *)map_info.data; + if (!surface) goto cleanup; + + guint index = batch_id; + if (index >= surface->numFilled) goto cleanup; + + if (NvBufSurfaceMapEglImage(surface, index) != 0) goto cleanup; + egl_mapped = true; + + EGLImageKHR eglImage = surface->surfaceList[index].mappedAddr.eglImage; + if (eglImage == EGL_NO_IMAGE_KHR) goto cleanup; + + CUresult status = cuGraphicsEGLRegisterImage(&cudaResource, eglImage, CU_GRAPHICS_MAP_RESOURCE_FLAGS_NONE); + if (status != CUDA_SUCCESS) goto cleanup; + cuda_registered = true; + + CUeglFrame eglFrame; + status = cuGraphicsResourceGetMappedEglFrame(&eglFrame, cudaResource, 0, 0); + if (status != CUDA_SUCCESS) goto cleanup; + + void *dev_ptr = eglFrame.frame.pPitch[0]; + if (!dev_ptr) goto cleanup; + + int width = surface->surfaceList[index].width; + int height = surface->surfaceList[index].height; + size_t pitch = eglFrame.pitch; + + if (width <= 0 || height <= 0 || pitch <= 0) goto cleanup; + + cv::cuda::GpuMat gpuMat(height, width, CV_8UC4, dev_ptr, pitch); + cv::cuda::GpuMat safe_copy; + gpuMat.copyTo(safe_copy); + + GPUFrameChangeDetector *detector = GPUFrameChangeDetector::getInstance(); + auto [should_process, metrics] = detector->should_process_gpu_direct(safe_copy); + + safe_copy.release(); + gpuMat.release(); + + if (cuda_registered) cuGraphicsUnregisterResource(cudaResource); + if (egl_mapped) NvBufSurfaceUnMapEglImage(surface, index); + if (buffer_mapped) gst_buffer_unmap(buffer, &map_info); + + std::cout << "Frame " << frame_count + << " MSE=" << metrics["MSE"] + << " SSIM=" << metrics["SSIM"] + << " FLOW=" << metrics["FLOW"] + << " -> " << (should_process ? "PROCESS" : "SKIP") << std::endl; + + return should_process; + + } catch (...) { + std::cout << "Exception in frame " << frame_count << std::endl; + } + +cleanup: + if (cuda_registered) cuGraphicsUnregisterResource(cudaResource); + if (egl_mapped && surface) NvBufSurfaceUnMapEglImage(surface, batch_id); + if (buffer_mapped && buffer) gst_buffer_unmap(buffer, &map_info); + + return true; +} + +PYBIND11_MODULE(gpu_probe, m) +{ + m.def("frame_probe", &frame_probe, + "Run DeepStream frame probe on GPU", + py::arg("buffer_ptr"), py::arg("batch_id")); +} diff --git a/src/gpu_probe.cpython-310-aarch64-linux-gnu.so b/src/gpu_probe.cpython-310-aarch64-linux-gnu.so new file mode 100755 index 00000000..bc14f0b3 Binary files /dev/null and b/src/gpu_probe.cpython-310-aarch64-linux-gnu.so differ