| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261 |
- import torch
- import torch.nn as nn
- import torch.optim as optim
- import os
- import pandas as pd
- import numpy as np
- from sklearn.model_selection import train_test_split
- from utils.training_interface import TrainerInterface
- from utils.file_utils import list_files_in_directory
- import time
- import json
- # Global variables for progress tracking
- current_progress = 0
- current_loss = 0.0
- current_accuracy = 0.0
- total_epochs = 0
- total_batches = 0
- class CustomDataset(torch.utils.data.Dataset):
- def __init__(self, features, labels):
- self.features = features
- self.labels = labels
- def __len__(self):
- return len(self.features)
- def __getitem__(self, idx):
- return self.features[idx], self.labels[idx]
- class PyTorchTrainer(TrainerInterface):
- def __init__(self, learning_rate=0.001):
- self.model_path = None
- self.training_directories = []
- self.sample_directory = None
- self.background_directory = None
- self.dropout_rate = 0.5
- self.epochs = 3000
- self.batch_size = 32
- self.learning_rate = learning_rate
- self.model = None
- self.optimizer = None
- self.criterion = nn.CrossEntropyLoss()
- self.current_epoch = 0
- self.stop_training = False
- self.training_status = "Not started"
- self.train_loader = None
- self.val_loader = None
- def set_model_path(self, model_path):
- self.model_path = model_path
- self.model = self.load_model()
- self.optimizer = optim.Adam(self.model.parameters(), lr=self.learning_rate)
- def set_training_directories(self, directories):
- self.training_directories = directories
- def set_sample_directory(self, directory):
- self.sample_directory = directory
- def set_background_directory(self, directory):
- self.background_directory = directory
- def set_dropout_rate(self, rate):
- self.dropout_rate = rate
- def set_epochs(self, epochs):
- self.epochs = epochs
- def set_batch_size(self, batch_size):
- self.batch_size = batch_size
- def get_training_data(self):
- return self.training_directories
- def get_model_path(self):
- return self.model_path
- def get_epochs(self):
- return self.epochs
- def get_batch_size(self):
- return self.batch_size
- def get_dropout_rate(self):
- return self.dropout_rate
- def get_training_directories(self):
- return self.training_directories
- def get_sample_directory(self):
- return self.sample_directory
- def get_background_directory(self):
- return self.background_directory
- def set_training_data(self, features, labels):
- X_train, X_val, y_train, y_val = train_test_split(features, labels, test_size=0.2, random_state=42)
- self.train_loader = torch.utils.data.DataLoader(CustomDataset(X_train, y_train), batch_size=self.batch_size, shuffle=True)
- self.val_loader = torch.utils.data.DataLoader(CustomDataset(X_val, y_val), batch_size=self.batch_size, shuffle=False)
- def preprocess_training_data(self):
- import librosa
- data_path_dict = {}
- data_path_dict[0] = list_files_in_directory(self.background_directory, extension='.wav')
- for idx, directory in enumerate(self.training_directories):
- data_path_dict[idx+1] = list_files_in_directory(directory, extension='.wav')
- all_data = []
- total_files = sum(len(files) for files in data_path_dict.values())
- processed_files = 0
- for class_label, list_of_files in data_path_dict.items():
- for single_file in list_of_files:
- try:
- audio, sample_rate = librosa.load(single_file)
- mfcc = librosa.feature.mfcc(y=audio, sr=sample_rate, n_mfcc=40)
- mfcc_processed = np.mean(mfcc.T, axis=0)
- all_data.append([mfcc_processed, class_label])
- processed_files += 1
- global current_progress
- current_progress = (processed_files / total_files) * 100
- except Exception as e:
- print(f"Exception: {str(e)}")
- df = pd.DataFrame(all_data, columns=["feature", "class_label"])
- df.to_pickle(self.model_path + "audio_data.csv")
- def load_model(self):
- num_classes = len(self.training_directories) + 1 # Assuming one extra for background or unknown
- model = nn.Sequential(
- nn.Linear(40, 128),
- nn.ReLU(),
- nn.Dropout(self.dropout_rate),
- nn.Linear(128, num_classes), # Update output layer to match num_classes
- nn.Softmax(dim=1)
- )
- model_path = self.model_path + "hotword_model.pth"
- if os.path.exists(model_path):
- state_dict = torch.load(model_path)
- try:
- model.load_state_dict(state_dict)
- except RuntimeError as e:
- print(f"Error loading model state dict: {e}")
- # Handle dimension mismatch by reinitializing the last layer
- model = nn.Sequential(
- nn.Linear(40, 128),
- nn.ReLU(),
- nn.Dropout(self.dropout_rate),
- nn.Linear(128, num_classes),
- nn.Softmax(dim=1)
- )
- print("Reinitialized the final layer to match the number of classes.")
- return model
- def start(self):
- self.training_status = "Training"
- self.stop_training = False
- self.run_training()
- def start_new(self):
- self.training_status = "Training"
- self.stop_training = False
- self.model = self.load_model()
- self.current_epoch = 0
- self.run_training()
- def pause(self):
- self.training_status = "Paused"
- self.stop_training = True
- def resume(self):
- self.training_status = "Training"
- self.stop_training = False
- self.run_training()
- def stop(self):
- self.training_status = "Stopped"
- self.stop_training = True
- def run_training(self):
- global current_progress, current_loss, current_accuracy, total_epochs, total_batches
- self.preprocess_training_data()
- start_time = time.time()
- df = pd.read_pickle(self.model_path + "audio_data.csv")
- X = df["feature"].values
- X = np.concatenate(X, axis=0).reshape(len(X), 40)
- y = np.array(df["class_label"].tolist())
- self.set_training_data(X, y)
- total_batches = len(self.train_loader)
- total_epochs = self.epochs
-
- training_info = {
- "trainer_script": "tensorflow_trainer",
- "training_time": 0,
- "model_size": 0,
- "epochs": self.epochs,
- "dropout_rate": self.dropout_rate,
- "batch_size": self.batch_size,
- "total_training_data": len(X),
- "training_directories": self.training_directories,
- "accuracy": [],
- "loss": [],
- "test_accuracy": 0
- }
- for epoch in range(self.current_epoch, self.epochs):
- if self.stop_training:
- break
- self.current_epoch = epoch
- self.model.train()
- for batch_idx, (batch_X, batch_y) in enumerate(self.train_loader):
- global current_loss, current_accuracy
- self.optimizer.zero_grad()
- outputs = self.model(batch_X)
- loss = self.criterion(outputs, batch_y.long()) # Convert labels to Long type
- loss.backward()
- self.optimizer.step()
- # Update global progress variables
- current_loss = loss.item()
- current_accuracy = (outputs.argmax(dim=1) == batch_y).float().mean().item()
- current_progress = (epoch * len(self.train_loader) + batch_idx + 1) / (self.epochs * len(self.train_loader)) * 100
- training_info["accuracy"].append(current_accuracy)
- training_info["loss"].append(current_loss)
-
- print(f"Epoch {epoch + 1}/{self.epochs}, Loss: {current_loss}, Accuracy: {current_accuracy}")
- epoch_time = time.time() - start_time
- training_info["training_time"] = epoch_time
-
- if not self.stop_training:
- torch.save(self.model.state_dict(), self.model_path + "hotword_model.pth")
- self.evaluate_model(X, y)
- self.save_training_info(training_info)
- self.training_status = "Finished"
- else:
- self.training_status = "Paused"
- def evaluate_model(self, X_test, y_test):
- self.model.eval()
- with torch.no_grad():
- X_test_tensor = torch.FloatTensor(X_test)
- y_test_tensor = torch.LongTensor(y_test)
- outputs = self.model(X_test_tensor)
- _, predicted = torch.max(outputs, 1)
- accuracy = (predicted == y_test_tensor).sum().item() / len(y_test_tensor)
- print(f"Test Accuracy: {accuracy * 100:.2f}%")
-
- def save_training_info(self, info):
- with open(self.model_path + "info_pytorch_trainer.json", "w") as f:
- json.dump(info, f)
- def get_training_info(self):
- return {
- "status": self.training_status,
- "current_epoch": self.current_epoch,
- "total_epochs": self.epochs,
- "loss": current_loss,
- "accuracy": current_accuracy
- }
|