Repository navigation
Expand file tree
/
Copy pathVisualization.py
More file actions
136 lines (102 loc) · 4 KB
/
Copy pathVisualization.py
File metadata and controls
136 lines (102 loc) · 4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
# Import necessary libraries
import pandas as pd
import matplotlib.pyplot as plt
import seaborn as sns
from sklearn.preprocessing import LabelEncoder
from google.colab import drive
import os
drive.mount('/content/drive')
folder_path = '/content/drive/My Drive/CSE-CIC-IDS2018/'
parquet_files = [
'Web2-Friday-23-02-2018_TrafficForML_CICFlowMeter.parquet',
'Web1-Thursday-22-02-2018_TrafficForML_CICFlowMeter.parquet',
'Infil2-Thursday-01-03-2018_TrafficForML_CICFlowMeter.parquet',
'Infil1-Wednesday-28-02-2018_TrafficForML_CICFlowMeter.parquet',
'DoS2-Friday-16-02-2018_TrafficForML_CICFlowMeter.parquet',
'DoS1-Thursday-15-02-2018_TrafficForML_CICFlowMeter.parquet',
'DDoS2-Wednesday-21-02-2018_TrafficForML_CICFlowMeter.parquet',
'DDoS1-Tuesday-20-02-2018_TrafficForML_CICFlowMeter.parquet',
'Bruteforce-Wednesday-14-02-2018_TrafficForML_CICFlowMeter.parquet',
'Botnet-Friday-02-03-2018_TrafficForML_CICFlowMeter.parquet'
]
dataframes = [pd.read_parquet(os.path.join(folder_path, file)) for file in parquet_files]
complete_data = pd.concat(dataframes, ignore_index=True)
print("First few rows of the dataset:")
print(complete_data.head())
print("\nDescriptive Statistics:")
print(complete_data.describe())
print("\nColumn Names:")
print(complete_data.columns)
sns.set(style='whitegrid')
# Check the data types of each column
print(complete_data.dtypes)
plt.figure(figsize=(10, 6))
sns.histplot(complete_data['Flow Duration'], bins=30, kde=True)
plt.title('Distribution of Flow Duration', fontsize=16)
plt.xlabel('Flow Duration (in microseconds)', fontsize=12)
plt.ylabel('Frequency', fontsize=12)
plt.show()
plt.figure(figsize=(10, 6))
sns.countplot(data=complete_data, x='Label', order=complete_data['Label'].value_counts().index, palette='viridis')
plt.title('Distribution of Attack Types', fontsize=16)
plt.xlabel('Attack Type', fontsize=12)
plt.ylabel('Count', fontsize=12)
plt.xticks(rotation=45)
plt.show()
print("Data Types of Each Column:")
print(complete_data.dtypes)
label_encoder = LabelEncoder()
complete_data['Label'] = label_encoder.fit_transform(complete_data['Label'])
print("Unique Values After Encoding:")
print(complete_data['Label'].unique())
label_mapping = {
0: 'Benign',
1: 'DoS',
2: 'DDoS',
3: 'Port Scan',
4: 'Brute Force',
5: 'Botnet',
6: 'Infiltration',
7: 'Web Attack',
8: 'SQL Injection',
9: 'XSS',
10: 'Malicious File',
11: 'DDOS Attack',
12: 'Password Guessing',
13: 'Information Theft',
14: 'Other Attacks'
}
complete_data['Label'] = complete_data['Label'].map(label_mapping)
plt.figure(figsize=(12, 6))
sns.countplot(data=complete_data, x='Label', palette='viridis')
plt.title('Distribution of Attack Types', fontsize=20)
plt.xlabel('Attack Type', fontsize=14)
plt.ylabel('Count', fontsize=14)
plt.xticks(rotation=45)
plt.show()
plt.figure(figsize=(14, 8))
sns.boxplot(x='Label', y='Flow Duration', data=complete_data, palette='Set2')
plt.title('Flow Duration by Attack Type', fontsize=20)
plt.xlabel('Attack Type', fontsize=14)
plt.ylabel('Flow Duration (in microseconds)', fontsize=14)
plt.xticks(rotation=45)
plt.show()
sample_data = complete_data.sample(frac=0.1) # Adjust the fraction as needed
plt.figure(figsize=(12, 12))
sns.pairplot(sample_data[['Flow Duration', 'Total Fwd Packets', 'Flow Bytes/s', 'Label']],
hue='Label', diag_kind='kde', height=2.5)
plt.suptitle('Pairplot of Selected Features (Sample)', y=1.02)
plt.show()
numeric_data = complete_data.select_dtypes(include='number')
top_features = numeric_data.corr().nlargest(10, 'Flow Duration').index
plt.figure(figsize=(10, 8))
sns.heatmap(numeric_data[top_features].corr(), annot=True, cmap='coolwarm', fmt=".2f", square=True)
plt.title('Top Feature Correlations', fontsize=20)
plt.show()
plt.figure(figsize=(14, 6))
plt.plot(complete_data['Flow Duration'], label='Flow Duration', color='b')
plt.title('Flow Duration Over Time', fontsize=20)
plt.xlabel('Index', fontsize=14)
plt.ylabel('Flow Duration', fontsize=14)
plt.legend()
plt.show()