Files
FaceDetectionScrath/process_data.py
T
2025-03-30 23:18:41 +02:00

85 lines
2.6 KiB
Python

import datetime
import os
import cv2
import joblib
import numpy as np
from skimage.feature import hog
from sklearn.svm import SVC
from sklearn.model_selection import train_test_split, GridSearchCV
from sklearn.metrics import classification_report, accuracy_score
# Parameters for HOG
hog_params = {
'orientations': 9,
'pixels_per_cell': (8, 8),
'cells_per_block': (2, 2),
'block_norm': 'L2-Hys'
}
# Path to your dataset (ensure it is extracted and organized into subfolders by class)
dataset_dir = 'natural_images' # update with your actual path
# Prepare lists for features and labels
features = []
labels = []
# Iterate over each class folder in the dataset
for label in os.listdir(dataset_dir):
label_folder = os.path.join(dataset_dir, label)
if not os.path.isdir(label_folder):
continue
# Process each image in the folder
for file in os.listdir(label_folder):
file_path = os.path.join(label_folder, file)
# Read the image in grayscale
image = cv2.imread(file_path, cv2.IMREAD_GRAYSCALE)
if image is None:
continue
# Resize the image to a fixed size (optional, adjust based on your dataset)
image = cv2.resize(image, (64, 64))
# Extract HOG features from the image
hog_features = hog(image, **hog_params)
features.append(hog_features)
labels.append(label)
# Convert lists to NumPy arrays
features = np.array(features)
labels = np.array(labels)
# Split the data into training and testing sets
X_train, X_test, y_train, y_test = train_test_split(features, labels, test_size=0.2, random_state=42)
param_grid = [
# {'kernel': ['linear'], 'C': [0.1, 1, 10, 100]},
{'kernel': ['poly'], 'C': [0.1, 1, 10, 100], 'degree': [2, 3, 4], 'gamma': ['scale', 'auto']},
{'kernel': ['rbf'], 'C': [0.1, 1, 10, 50, 100], 'gamma': ['scale', 'auto']},
#{'kernel': ['sigmoid'], 'C': [0.1, 1, 10, 100], 'gamma': ['scale', 'auto']}
]
svm_model = SVC(probability=True)
grid_search = GridSearchCV(svm_model, param_grid, cv=5, n_jobs=-1, verbose=3)
grid_search.fit(X_train, y_train)
print("Best parameters found:", grid_search.best_params_)
print("Best cross-validation score:", grid_search.best_score_)
# Use the best estimator to predict on the test set
best_model = grid_search.best_estimator_
y_pred = best_model.predict(X_test)
print("Test Accuracy:", accuracy_score(y_test, y_pred))
print(classification_report(y_test, y_pred))
# Evaluate the model on the test set
y_pred = grid_search.predict(X_test)
print("Accuracy:", accuracy_score(y_test, y_pred))
# Save the trained model for later use
joblib.dump(grid_search, f'svm_natural_images_model{datetime.UTC}.pkl')