Repository navigation
Expand file tree
/
Copy pathplot_feature_analysis.py
More file actions
108 lines (89 loc) · 3.78 KB
/
Copy pathplot_feature_analysis.py
File metadata and controls
108 lines (89 loc) · 3.78 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
import os
import pandas as pd
import numpy as np
import matplotlib.pyplot as plt
import seaborn as sns
from ml_pipeline import extract_features_from_scenario
def load_and_aggregate_data():
"""
Loads scenarios and aggregates them into a single DataFrame for analysis.
Uses ds1, ds2, and ds3 to get a robust mix of normal and anomalous data.
"""
print("Loading data for feature analysis...")
dfs = []
# Load clean dataset
print("Loading cleanStatic...")
df_clean = extract_features_from_scenario('data/cleanStatic', spoofing_start_sec=9999, is_clean_scenario=True)
if df_clean is not None:
dfs.append(df_clean)
# Load spoofed datasets
for ds in ['ds1', 'ds2', 'ds3']:
print(f"Loading {ds}...")
df_spoofed = extract_features_from_scenario(f'data/{ds}', spoofing_start_sec=100, is_clean_scenario=False)
if df_spoofed is not None:
dfs.append(df_spoofed)
if not dfs:
print("Failed to load any data.")
return None
full_df = pd.concat(dfs, ignore_index=True)
# Map binary labels to descriptive strings for plotting
full_df['Class'] = full_df['Label'].map({0: 'Normal (Authentic)', 1: 'Anomalous (Spoofed)'})
return full_df
def plot_feature_distributions(df, save_dir='plots/analysis'):
"""
Generates KDE density plots comparing Normal vs Anomalous distributions
for key engineered features.
"""
os.makedirs(save_dir, exist_ok=True)
# Define the features we want to analyze visually
features_to_plot = [
('Mean_CN0', 'Mean C/N_0 (dB-Hz)'),
('Mean_AGC', 'Estimated AGC Proxy (dB)'),
('Std_CN0', '10-sec Rolling Std of C/N_0'),
('CUSUM_Neg_AGC', 'CUSUM Negative Drift (AGC)')
]
# Set the aesthetic style of the plots
sns.set_theme(style="whitegrid")
# Generate individual KDE plots
for feature_col, feature_name in features_to_plot:
plt.figure(figsize=(10, 6))
# KDE Plot for distribution comparison
sns.kdeplot(data=df, x=feature_col, hue='Class', fill=True, common_norm=False,
palette={'Normal (Authentic)': '#2ecc71', 'Anomalous (Spoofed)': '#e74c3c'},
alpha=0.5, linewidth=2)
plt.title(f'Distribution Analysis: {feature_name}', fontsize=16, pad=15)
plt.xlabel(feature_name, fontsize=14)
plt.ylabel('Density', fontsize=14)
save_path = os.path.join(save_dir, f'dist_{feature_col}.png')
plt.tight_layout()
plt.savefig(save_path, dpi=300)
plt.close()
print(f"Saved distribution plot: {save_path}")
def plot_boxplots(df, save_dir='plots/analysis'):
"""
Generates boxplots to show median, quartiles, and outliers.
"""
os.makedirs(save_dir, exist_ok=True)
features = ['Mean_CN0', 'Mean_AGC', 'Std_CN0', 'Std_AGC']
plt.figure(figsize=(14, 10))
sns.set_theme(style="whitegrid")
for i, feature in enumerate(features, 1):
plt.subplot(2, 2, i)
sns.boxplot(data=df, x='Class', y=feature,
palette={'Normal (Authentic)': '#2ecc71', 'Anomalous (Spoofed)': '#e74c3c'})
plt.title(f'Boxplot: {feature}')
plt.xlabel('')
plt.ylabel(feature)
save_path = os.path.join(save_dir, 'feature_boxplots.png')
plt.tight_layout()
plt.savefig(save_path, dpi=300)
plt.close()
print(f"Saved combined boxplots: {save_path}")
if __name__ == '__main__':
df = load_and_aggregate_data()
if df is not None:
print("\nGenerating Distribution Plots...")
plot_feature_distributions(df)
print("\nGenerating Box Plots...")
plot_boxplots(df)
print("\nFeature analysis complete! Check the 'plots/analysis' directory.")