{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-08T08:15:41.547328Z","iopub.execute_input":"2023-03-08T08:15:41.547797Z","iopub.status.idle":"2023-03-08T08:15:44.98513Z","shell.execute_reply.started":"2023-03-08T08:15:41.547755Z","shell.execute_reply":"2023-03-08T08:15:44.983612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"/kaggle/input/ml-olympiad-dialectrecognition/batch_1","metadata":{"execution":{"iopub.status.busy":"2023-03-08T08:16:30.224023Z","iopub.execute_input":"2023-03-08T08:16:30.225181Z","iopub.status.idle":"2023-03-08T08:16:30.248313Z","shell.execute_reply.started":"2023-03-08T08:16:30.225136Z","shell.execute_reply":"2023-03-08T08:16:30.246582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import necessary libraries\n","metadata":{}},{"cell_type":"code","source":"# Import necessary libraries\nimport pandas as pd\nimport numpy as np\nimport librosa\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import f1_score, confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2023-03-08T07:37:52.354358Z","iopub.execute_input":"2023-03-08T07:37:52.354819Z","iopub.status.idle":"2023-03-08T07:37:52.889367Z","shell.execute_reply.started":"2023-03-08T07:37:52.354763Z","shell.execute_reply":"2023-03-08T07:37:52.88793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Set file paths\n","metadata":{}},{"cell_type":"code","source":"# Set file paths\nTRAIN_FILE = '/kaggle/input/ml-olympiad-dialectrecognition/train.csv'\nTEST_FILE = '/kaggle/input/ml-olympiad-dialectrecognition/test.csv'\nOUTPUT_FILE = '/kaggle/working/predictions.csv'","metadata":{"execution":{"iopub.status.busy":"2023-03-08T07:37:54.968071Z","iopub.execute_input":"2023-03-08T07:37:54.968494Z","iopub.status.idle":"2023-03-08T07:37:54.974738Z","shell.execute_reply.started":"2023-03-08T07:37:54.968459Z","shell.execute_reply":"2023-03-08T07:37:54.973029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"# Load data\ntrain_df = pd.read_csv(TRAIN_FILE)# Load training data\ntest_df = pd.read_csv(TEST_FILE)# Load testning data\ntrain_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-03-08T07:38:19.024088Z","iopub.execute_input":"2023-03-08T07:38:19.024537Z","iopub.status.idle":"2023-03-08T07:38:20.387366Z","shell.execute_reply.started":"2023-03-08T07:38:19.0245Z","shell.execute_reply":"2023-03-08T07:38:20.386204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA (Exploratory Data Analysis)\n","metadata":{}},{"cell_type":"code","source":"# EDA\n# Show the number of recordings for each dialect in the training set\nprint(train_df['SpeakerDialect'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-03-08T06:23:44.69179Z","iopub.execute_input":"2023-03-08T06:23:44.692213Z","iopub.status.idle":"2023-03-08T06:23:44.72163Z","shell.execute_reply.started":"2023-03-08T06:23:44.692176Z","shell.execute_reply":"2023-03-08T06:23:44.720391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature extraction","metadata":{}},{"cell_type":"code","source":"import librosa\nimport numpy as np\nimport pandas as pd\n\ntrain_df = pd.read_csv('/kaggle/input/ml-olympiad-dialectrecognition/train.csv')\n\ndef extract_features(file_path):\n    try:\n        signal, sample_rate = librosa.load(file_path, sr=16000)\n        mfccs = librosa.feature.mfcc(signal, sr=sample_rate, n_mfcc=13)\n        return np.mean(mfccs.T, axis=0)\n    except Exception as e:\n        print(f\"Error encountered while parsing file: {file_path}\")\n        return None\n\ntrain_features = []\nfor fname in train_df['SegmentID'].values:\n    file_path = '/kaggle/input/ml-olympiad-dialectrecognition/train/' + fname + '.wav'\n    features = extract_features(file_path)\n    if features is not None:\n        train_features.append(features)\n\ntrain_labels = train_df['SpeakerDialect'].values\n","metadata":{"execution":{"iopub.status.busy":"2023-03-08T08:38:05.006494Z","iopub.execute_input":"2023-03-08T08:38:05.007037Z","iopub.status.idle":"2023-03-08T08:38:31.455137Z","shell.execute_reply.started":"2023-03-08T08:38:05.006991Z","shell.execute_reply":"2023-03-08T08:38:31.453545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Feature extraction\ndef extract_features(file_path):\n    signal, sample_rate = librosa.load(file_path, sr=16000)# Load the audio file and its sampling rate\n    mfccs = librosa.feature.mfcc(signal, sr=sample_rate, n_mfcc=13)# Extract Mel-frequency cepstral coefficients (MFCCs)\n    return np.mean(mfccs.T, axis=0) # Return the average of the MFCCs of the signal\n\ntrain_features = [extract_features('/kaggle/input/ml-olympiad-dialectrecognition/train/' + fname + '.wav')\n                  for fname in train_df['SegmentID'].values]# Extract features (MFCCs) from each training recording\ntrain_labels = train_df['SpeakerDialect'].values# Get the corresponding dialect labels for each training recording","metadata":{"execution":{"iopub.status.busy":"2023-03-08T06:30:10.251596Z","iopub.execute_input":"2023-03-08T06:30:10.252355Z","iopub.status.idle":"2023-03-08T06:30:10.305802Z","shell.execute_reply.started":"2023-03-08T06:30:10.2523Z","shell.execute_reply":"2023-03-08T06:30:10.30415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model training (SVM)\n","metadata":{}},{"cell_type":"code","source":"\n# Model training\nsvm = SVC(kernel='linear')# Create a Support Vector Machine (SVM) classifier with a linear kernel\nsvm.fit(train_features, train_labels)# Train the SVM classifier on the training set\n","metadata":{"execution":{"iopub.status.busy":"2023-03-08T08:42:33.156372Z","iopub.execute_input":"2023-03-08T08:42:33.158437Z","iopub.status.idle":"2023-03-08T08:42:33.206013Z","shell.execute_reply.started":"2023-03-08T08:42:33.158377Z","shell.execute_reply":"2023-03-08T08:42:33.203936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model evaluation\n","metadata":{}},{"cell_type":"code","source":"# Model evaluation\ntest_features = [extract_features('/kaggle/input/ml-olympiad-dialectrecognition/test/' + fname + '.wav')\n                 for fname in test_df['SegmentID'].values]# Extract features (MFCCs) from each testing recording\ntest_labels = [1, 2, 3, 4]# Set the ground truth dialect labels for the testing set (Note: this is for the example dataset only)\npred_labels = svm.predict(test_features)# Predict the dialect labels for the testing set using the trained SVM classifier\nf_score = f1_score(test_labels, pred_labels, average='macro')# Calculate the F-score (macro) of the SVM classifier\nconf_mat = confusion_matrix(test_labels, pred_labels)# Calculate the confusion matrix of the SVM classifier\n\nprint('F-Score (Macro):', f_score)# Print the F-score (macro) of the SVM classifier\nprint('Confusion Matrix:\\n', conf_mat)# Print the confusion matrix of the SVM classifier\n","metadata":{"execution":{"iopub.status.busy":"2023-03-08T06:30:26.895096Z","iopub.execute_input":"2023-03-08T06:30:26.89555Z","iopub.status.idle":"2023-03-08T06:30:26.953486Z","shell.execute_reply.started":"2023-03-08T06:30:26.895513Z","shell.execute_reply":"2023-03-08T06:30:26.951413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Output writing\n# Create a dataframe for the predicted dialect labels of the testing set\noutput_df = pd.DataFrame({'SegmentID': test_df['SegmentID'], 'SpeakerDialect': pred_labels})\n# Write the predicted dialect labels of the testing set to a CSV file\noutput_df.to_csv(OUTPUT_FILE, index=False)","metadata":{"execution":{"iopub.status.busy":"2023-03-08T06:30:32.684538Z","iopub.execute_input":"2023-03-08T06:30:32.685441Z","iopub.status.idle":"2023-03-08T06:30:32.703037Z","shell.execute_reply.started":"2023-03-08T06:30:32.685396Z","shell.execute_reply":"2023-03-08T06:30:32.701621Z"},"trusted":true},"execution_count":null,"outputs":[]}]}