diff --git a/Makefile b/Makefile
new file mode 100644
index 00000000..a152c836
--- /dev/null
+++ b/Makefile
@@ -0,0 +1,36 @@
+PYTHON_VERSION = 3.10
+CUDA_VERSION = 12.2
+
+PYTHON_INC = /usr/include/python$(PYTHON_VERSION)
+TORCH_INC = /usr/local/lib/python$(PYTHON_VERSION)/dist-packages/torch/include
+TORCH_PYBIND = $(TORCH_INC)/pybind11
+OPENCV_INC = /usr/local/include/opencv4
+CUDA_INC = /usr/local/cuda-$(CUDA_VERSION)/targets/aarch64-linux/include
+DEEPSTREAM_INC = /opt/nvidia/deepstream/deepstream/sources/includes
+GLIB_INC = /usr/include/glib-2.0
+GLIB_LIB_INC = /usr/lib/aarch64-linux-gnu/glib-2.0/include
+GST_INC = /usr/include/gstreamer-1.0
+PYBIND11_INC = /usr/local/lib/python$(PYTHON_VERSION)/dist-packages/pybind11/include
+
+CUDA_LIB = /usr/local/cuda-$(CUDA_VERSION)/targets/aarch64-linux/lib
+DEEPSTREAM_LIB = /opt/nvidia/deepstream/deepstream/lib
+
+LIBS = -lpython$(PYTHON_VERSION) \
+ -lopencv_core -lopencv_imgproc -lopencv_cudaimgproc -lopencv_cudaarithm -lopencv_cudaoptflow \
+ -lcuda -lcudart -lEGL -lnvbufsurface -lnvds_meta \
+ -lgstreamer-1.0 -lgobject-2.0 -lglib-2.0
+
+CXXFLAGS = -O3 -Wall -shared -std=c++17 -fPIC
+INCLUDES = -I$(PYTHON_INC) -I$(TORCH_INC) -I$(TORCH_PYBIND) -I$(PYBIND11_INC) \
+ -I$(OPENCV_INC) -I$(CUDA_INC) -I$(DEEPSTREAM_INC) \
+ -I$(GLIB_INC) -I$(GLIB_LIB_INC) -I$(GST_INC)
+LDFLAGS = -L$(CUDA_LIB) -L$(DEEPSTREAM_LIB)
+
+TARGET = src/gpu_probe$(shell python3-config --extension-suffix)
+SRC = src/gpu_probe.cpp src/gpu_frame_change_detector.cpp
+
+all:
+ g++ $(CXXFLAGS) $(INCLUDES) -o $(TARGET) $(SRC) $(LDFLAGS) $(LIBS)
+
+clean:
+ rm -f $(TARGET)
diff --git a/docs/frame_comparison_results.md b/docs/frame_comparison_results.md
index f5674ca7..9ecfb907 100644
--- a/docs/frame_comparison_results.md
+++ b/docs/frame_comparison_results.md
@@ -76,7 +76,9 @@ Decision: ⚙️ PROCESS
## Decision Logic
-Frames are skipped when minimal change is detected.
+### CPU Implementation
+
+Frames are skipped when minimal change is detected:
```python
if mse_val < 50 and ssim_val > 0.995 and flow_val < 0.1:
@@ -85,18 +87,60 @@ else:
process = True
```
+### GPU Implementation (C++/CUDA)
+
+The GPU implementation uses adjusted thresholds due to precision differences:
+
+```cpp
+bool is_static = (mse_val < 10.0 && ssim_val > 0.995 && flow_val < 0.05);
+return {!is_static, metrics};
+```
+
+**Key Differences:**
+- **MSE threshold:** `50` (CPU) → `10.0` (GPU)
+- **SSIM threshold:** `0.995` (both, but different precision)
+- **Flow threshold:** `0.1` (CPU) → `0.05` (GPU)
---
-## Metric Interpretation
+## Implementation Differences: CPU vs GPU
+
+### Precision & Accuracy
+
+| Aspect | CPU Implementation | GPU Implementation |
+|--------|-------------------|-------------------|
+| **Math Precision** | `float64` (double precision) | `float32` (single precision) |
+| **Optical Flow API** | `cv2.calcOpticalFlowFarneback()` | `cv::cuda::FarnebackOpticalFlow::create()->calc()` |
+| **Flow Accuracy** | Higher precision, more accurate | Lower precision, slightly less accurate |
+| **Computation** | Sequential on CPU cores | Parallel on GPU |
-| Metric | Typical Range | Interpretation |
-| ------------ | ------------------------------------------------------------------------------------- | -------------- |
-| **MSE** | 0–50 → no change
50–1000 → moderate change
>1000 → significant scene change | |
-| **SSIM** | >0.995 → identical
0.90–0.995 → moderate change
<0.90 → major visual difference | |
-| **FLOW** | <0.1 → static
0.5–2.0 → small motion
>2.0 → strong/global motion | |
-| **MOTION %** | <5% → static
5–30% → partial motion
>30% → large motion area | |
+### Behavioral Changes
+
+Due to the precision and algorithm differences in GPU implementation:
+
+1. **Lower MSE threshold (10.0 vs 50):**
+ - GPU's `float32` arithmetic produces slightly different MSE values
+ - More conservative threshold compensates for reduced precision
+ - Prevents false negatives (missing actual changes)
+
+2. **Lower Flow threshold (0.05 vs 0.1):**
+ - GPU optical flow (`cv::cuda::FarnebackOpticalFlow`) is less accurate than CPU version
+ - Threshold adjusted to account for measurement noise
+ - Reduces false positives (detecting motion where there is none)
+
+3. **Simplified method selection:**
+ - GPU focuses on MSE + SSIM + Optical Flow only
+ - Histogram correlation, Motion Mask, and ORB matching excluded for performance
+ - These methods don't provide significant GPU acceleration benefits
+
+---
+
+## Metric Interpretation
-These thresholds were derived empirically from the RTSP domain test videos and can be fine-tuned for other environments or resolutions.
+| Metric | CPU Typical Range | GPU Typical Range | Interpretation |
+| ------------ | ----------------- | ----------------- | -------------- |
+| **MSE** | 0–50 → no change
50–1000 → moderate
>1000 → major | 0–10 → no change
10–800 → moderate
>800 → major | GPU values tend to be slightly lower due to float32 |
+| **SSIM** | >0.995 → identical
0.90–0.995 → moderate
<0.90 → major | >0.995 → identical
0.92–0.995 → moderate
<0.92 → major | Similar ranges, GPU slightly less precise |
+| **FLOW** | <0.1 → static
0.5–2.0 → small motion
>2.0 → strong | <0.05 → static
0.3–1.5 → small motion
>1.5 → strong | GPU flow values are generally lower |
---
diff --git a/src/gpu_frame_change_detector.cpp b/src/gpu_frame_change_detector.cpp
new file mode 100644
index 00000000..56c97365
--- /dev/null
+++ b/src/gpu_frame_change_detector.cpp
@@ -0,0 +1,124 @@
+#include "gpu_frame_change_detector.hpp"
+#include
+#include
+#include
+#include
+#include
+
+std::unique_ptr GPUFrameChangeDetector::instance = nullptr;
+
+GPUFrameChangeDetector::GPUFrameChangeDetector()
+ : mse_thresh(500.0), ssim_thresh(0.99), flow_thresh(2.0), initialized(true) {}
+
+GPUFrameChangeDetector* GPUFrameChangeDetector::getInstance() {
+ if (!instance) {
+ instance = std::unique_ptr(new GPUFrameChangeDetector());
+ }
+ return instance.get();
+}
+
+static cv::Scalar mean_gpu(const cv::cuda::GpuMat& mat) {
+ cv::Scalar sum_val = cv::cuda::sum(mat);
+ double total_elements = static_cast(mat.rows * mat.cols);
+ return sum_val * (1.0 / total_elements);
+}
+
+cv::cuda::GpuMat GPUFrameChangeDetector::preprocess_gpu(const cv::cuda::GpuMat& frame) {
+ return frame;
+}
+
+double GPUFrameChangeDetector::mse_gpu(const cv::cuda::GpuMat& imgA, const cv::cuda::GpuMat& imgB) {
+ cv::cuda::GpuMat bgrA, bgrB;
+ cv::cuda::cvtColor(imgA, bgrA, cv::COLOR_RGBA2BGR);
+ cv::cuda::cvtColor(imgB, bgrB, cv::COLOR_RGBA2BGR);
+
+ cv::cuda::GpuMat diff, sqr;
+ cv::cuda::absdiff(bgrA, bgrB, diff);
+ cv::cuda::multiply(diff, diff, sqr);
+ cv::Scalar mean_val = mean_gpu(sqr);
+ return mean_val[0];
+}
+
+double GPUFrameChangeDetector::simple_ssim_gpu(const cv::cuda::GpuMat& imgA, const cv::cuda::GpuMat& imgB) {
+ cv::cuda::GpuMat grayA, grayB;
+ cv::cuda::cvtColor(imgA, grayA, cv::COLOR_RGBA2GRAY);
+ cv::cuda::cvtColor(imgB, grayB, cv::COLOR_RGBA2GRAY);
+
+ cv::Scalar mu_x = mean_gpu(grayA);
+ cv::Scalar mu_y = mean_gpu(grayB);
+
+ cv::cuda::GpuMat diffX, diffY;
+ cv::cuda::subtract(grayA, mu_x, diffX);
+ cv::cuda::subtract(grayB, mu_y, diffY);
+
+ cv::cuda::GpuMat diffX2, diffY2, diffXY;
+ cv::cuda::multiply(diffX, diffX, diffX2);
+ cv::cuda::multiply(diffY, diffY, diffY2);
+ cv::cuda::multiply(diffX, diffY, diffXY);
+
+ double sigma_x = std::sqrt(mean_gpu(diffX2)[0]);
+ double sigma_y = std::sqrt(mean_gpu(diffY2)[0]);
+ double sigma_xy = mean_gpu(diffXY)[0];
+
+ double C1 = 6.5025, C2 = 58.5225;
+ double ssim = ((2 * mu_x[0] * mu_y[0] + C1) * (2 * sigma_xy + C2)) /
+ ((mu_x[0]*mu_x[0] + mu_y[0]*mu_y[0] + C1) *
+ (sigma_x*sigma_x + sigma_y*sigma_y + C2));
+
+ return ssim;
+}
+
+double GPUFrameChangeDetector::optical_flow_gpu(const cv::cuda::GpuMat& imgA, const cv::cuda::GpuMat& imgB) {
+ cv::cuda::GpuMat bgrA, bgrB, grayA, grayB;
+
+ cv::cuda::cvtColor(imgA, bgrA, cv::COLOR_RGBA2BGR);
+ cv::cuda::cvtColor(imgB, bgrB, cv::COLOR_RGBA2BGR);
+ cv::cuda::cvtColor(bgrA, grayA, cv::COLOR_BGR2GRAY);
+ cv::cuda::cvtColor(bgrB, grayB, cv::COLOR_BGR2GRAY);
+
+ auto flow = cv::cuda::FarnebackOpticalFlow::create(3, 0.5, false, 15, 3, 5, 1.2, 0);
+
+ cv::cuda::GpuMat flow_result;
+ flow->calc(grayA, grayB, flow_result);
+
+ std::vector flow_xy;
+ cv::cuda::split(flow_result, flow_xy);
+
+ cv::cuda::GpuMat mag, angle;
+ cv::cuda::cartToPolar(flow_xy[0], flow_xy[1], mag, angle, true);
+
+ cv::cuda::GpuMat active_mask;
+ cv::cuda::threshold(mag, active_mask, 0.5, 1.0, cv::THRESH_BINARY);
+
+ cv::cuda::GpuMat active_pixels;
+ mag.copyTo(active_pixels, active_mask);
+
+ cv::Scalar sum_active = cv::cuda::sum(active_pixels);
+ cv::Scalar count_active = cv::cuda::sum(active_mask);
+
+ if (count_active[0] > 0) {
+ return sum_active[0] / count_active[0];
+ } else {
+ return 0.0;
+ }
+}
+
+std::pair>
+GPUFrameChangeDetector::should_process_gpu_direct(const cv::cuda::GpuMat& gpu_frame) {
+ if (prev_frame_gpu.empty()) {
+ prev_frame_gpu = preprocess_gpu(gpu_frame);
+ return {true, {{"MSE", 0.0}, {"SSIM", 1.0}, {"FLOW", 0.0}}};
+ }
+
+ cv::cuda::GpuMat processed_curr = preprocess_gpu(gpu_frame);
+
+ double mse_val = mse_gpu(prev_frame_gpu, processed_curr);
+ double ssim_val = simple_ssim_gpu(prev_frame_gpu, processed_curr);
+ double flow_val = optical_flow_gpu(prev_frame_gpu, processed_curr);
+
+ bool is_static = (mse_val < 10.0 && ssim_val > 0.995 && flow_val < 0.05);
+
+ prev_frame_gpu = processed_curr;
+
+ return {!is_static, {{"MSE", mse_val}, {"SSIM", ssim_val}, {"FLOW", flow_val}}};
+}
diff --git a/src/gpu_frame_change_detector.hpp b/src/gpu_frame_change_detector.hpp
new file mode 100644
index 00000000..0def34a1
--- /dev/null
+++ b/src/gpu_frame_change_detector.hpp
@@ -0,0 +1,36 @@
+#pragma once
+#include
+#include