{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Speaker sound dialect classification using Keras model","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n! pip install librosa\nimport librosa\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-11T19:49:06.504129Z","iopub.execute_input":"2023-03-11T19:49:06.504432Z","iopub.status.idle":"2023-03-11T19:49:43.040685Z","shell.execute_reply.started":"2023-03-11T19:49:06.504403Z","shell.execute_reply":"2023-03-11T19:49:43.039222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Extract features from audio","metadata":{}},{"cell_type":"code","source":"def extract_features(file_name):\n   \n    audio, sample_rate = librosa.load(file_name) \n    mfccs = librosa.feature.mfcc(y=audio, sr=sample_rate, n_mfcc=40)\n    mfccsscaled = np.mean(mfccs.T,axis=0)\n    return mfccsscaled","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:49:43.047151Z","iopub.execute_input":"2023-03-11T19:49:43.049731Z","iopub.status.idle":"2023-03-11T19:49:43.058435Z","shell.execute_reply.started":"2023-03-11T19:49:43.049673Z","shell.execute_reply":"2023-03-11T19:49:43.057208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Get SegmentId from audio dataset","metadata":{}},{"cell_type":"code","source":"for dirname, _, filenames in os.walk('/kaggle/input/sada-sound-dataset/dataset'):\n    file_names_list = [filename.replace('.wav','') for filename in filenames]\n    ","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:49:43.060238Z","iopub.execute_input":"2023-03-11T19:49:43.061174Z","iopub.status.idle":"2023-03-11T19:49:52.371830Z","shell.execute_reply.started":"2023-03-11T19:49:43.061130Z","shell.execute_reply":"2023-03-11T19:49:52.370646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read csv file, extract rows existing in dataset\nmetadata = pd.read_csv('/kaggle/input/ml-olympiad-dialectrecognition/train.csv')\nmetadata = metadata[metadata.SegmentID.isin(file_names_list)]\nmetadata.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:49:52.374651Z","iopub.execute_input":"2023-03-11T19:49:52.375058Z","iopub.status.idle":"2023-03-11T19:49:54.394976Z","shell.execute_reply.started":"2023-03-11T19:49:52.375018Z","shell.execute_reply":"2023-03-11T19:49:54.393945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# list to store features\nfeatures = []\n\n# Iterate through each sound file and extract the features \nfor index, row in metadata.iterrows():\n    \n    file_name = '/kaggle/input/sada-sound-dataset/dataset/'+str(row[\"SegmentID\"]+'.wav')\n    \n    class_label = row[\"SpeakerDialect\"]\n    data = extract_features(file_name)\n    segment_Id = row[\"SegmentID\"]\n    \n    features.append([data, class_label, segment_Id])","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:49:54.396621Z","iopub.execute_input":"2023-03-11T19:49:54.397037Z","iopub.status.idle":"2023-03-11T20:01:05.933914Z","shell.execute_reply.started":"2023-03-11T19:49:54.396997Z","shell.execute_reply":"2023-03-11T20:01:05.932100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert into a Panda dataframe \nfeaturesdf = pd.DataFrame(features, columns=['feature','class_label','segment_Id'])\n\nprint('Finished feature extraction from ', len(featuresdf), ' files')","metadata":{"execution":{"iopub.status.busy":"2023-03-11T20:02:22.134335Z","iopub.execute_input":"2023-03-11T20:02:22.134952Z","iopub.status.idle":"2023-03-11T20:02:22.152443Z","shell.execute_reply.started":"2023-03-11T20:02:22.134904Z","shell.execute_reply":"2023-03-11T20:02:22.149900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove nan file \nfeaturesdf = featuresdf[~ featuresdf.feature.isna()]","metadata":{"execution":{"iopub.status.busy":"2023-03-11T20:02:23.918258Z","iopub.execute_input":"2023-03-11T20:02:23.919215Z","iopub.status.idle":"2023-03-11T20:02:23.929967Z","shell.execute_reply.started":"2023-03-11T20:02:23.919169Z","shell.execute_reply":"2023-03-11T20:02:23.928837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"featuresdf.to_csv('train_features.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T20:02:24.732575Z","iopub.execute_input":"2023-03-11T20:02:24.733323Z","iopub.status.idle":"2023-03-11T20:02:32.513361Z","shell.execute_reply.started":"2023-03-11T20:02:24.733285Z","shell.execute_reply":"2023-03-11T20:02:32.512253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"featuresdf[~featuresdf.feature.isna()].head(5)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T20:02:32.515954Z","iopub.execute_input":"2023-03-11T20:02:32.516357Z","iopub.status.idle":"2023-03-11T20:02:32.537161Z","shell.execute_reply.started":"2023-03-11T20:02:32.516318Z","shell.execute_reply":"2023-03-11T20:02:32.536097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prepare data","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfrom keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split ","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:15:57.165373Z","iopub.execute_input":"2023-03-11T19:15:57.165997Z","iopub.status.idle":"2023-03-11T19:16:06.386474Z","shell.execute_reply.started":"2023-03-11T19:15:57.165949Z","shell.execute_reply":"2023-03-11T19:16:06.384770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Convert features and corresponding classification labels into numpy arrays\nX = np.array(featuresdf.feature.tolist())\ny = np.array(featuresdf.class_label.tolist())\n\n# Encode the classification labels\nle = LabelEncoder()\ny_label = to_categorical(le.fit_transform(y)) \n\n# split the dataset \nx_train, x_test, y_train, y_test = train_test_split(X, y_label, test_size=0.2, random_state = 42)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:16:06.389142Z","iopub.execute_input":"2023-03-11T19:16:06.390848Z","iopub.status.idle":"2023-03-11T19:16:06.400687Z","shell.execute_reply.started":"2023-03-11T19:16:06.390801Z","shell.execute_reply":"2023-03-11T19:16:06.399239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Build model ","metadata":{}},{"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Activation\nfrom sklearn import metrics \nimport tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:18:38.131961Z","iopub.execute_input":"2023-03-11T19:18:38.132459Z","iopub.status.idle":"2023-03-11T19:18:38.138670Z","shell.execute_reply.started":"2023-03-11T19:18:38.132417Z","shell.execute_reply":"2023-03-11T19:18:38.137426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build model \nmodel=Sequential()\n# First layer\nmodel.add(Dense(100,input_shape=(40,)))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\n# Second layer\nmodel.add(Dense(200))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\n# Third layer\nmodel.add(Dense(100))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\n# Final layer\nmodel.add(Dense(4))\nmodel.add(Activation('softmax'))","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:18:41.681561Z","iopub.execute_input":"2023-03-11T19:18:41.681966Z","iopub.status.idle":"2023-03-11T19:18:41.951828Z","shell.execute_reply.started":"2023-03-11T19:18:41.681933Z","shell.execute_reply":"2023-03-11T19:18:41.950471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss='categorical_crossentropy'\n              ,metrics=[tf.keras.metrics.Precision()]\n              ,optimizer='adam')\n\n# Display model architecture summary \nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:19:12.785962Z","iopub.execute_input":"2023-03-11T19:19:12.786428Z","iopub.status.idle":"2023-03-11T19:19:13.298807Z","shell.execute_reply.started":"2023-03-11T19:19:12.786386Z","shell.execute_reply":"2023-03-11T19:19:13.297535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_epochs = 100\nhistory = model.fit(x_train, \n                    y_train, \n                    validation_data=(x_test, y_test), \n                    epochs=num_epochs, \n                    batch_size=32)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:19:26.957031Z","iopub.execute_input":"2023-03-11T19:19:26.957512Z","iopub.status.idle":"2023-03-11T19:19:27.626139Z","shell.execute_reply.started":"2023-03-11T19:19:26.957468Z","shell.execute_reply":"2023-03-11T19:19:27.624156Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate pre-training accuracy \nscore = model.evaluate(x_test, y_test, verbose=1)\naccuracy = 100*score[1]\n\nprint(\"Pre-training accuracy: %.4f%%\" % accuracy) ","metadata":{"execution":{"iopub.status.busy":"2023-02-27T12:54:13.955104Z","iopub.execute_input":"2023-02-27T12:54:13.956924Z","iopub.status.idle":"2023-02-27T12:54:14.472211Z","shell.execute_reply.started":"2023-02-27T12:54:13.956845Z","shell.execute_reply":"2023-02-27T12:54:14.470348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test and predict result","metadata":{}},{"cell_type":"code","source":"metadata_test = pd.read_csv('/kaggle/input/ml-olympiad-dialectrecognition/test.csv')\n\n\nfeatures_test = []\n\n# Iterate through each sound file and extract the features \nfor index, row in metadata_test.iterrows():\n    \n    file_name = '/kaggle/input/sada-sound-dataset/test/'+str(row[\"SegmentID\"]+'.wav')\n    \n    data = extract_features(file_name)\n    segment_Id = row[\"SegmentID\"]\n    \n    features_test.append([data, segment_Id])","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:20:47.262595Z","iopub.execute_input":"2023-03-11T19:20:47.263755Z","iopub.status.idle":"2023-03-11T19:20:47.521549Z","shell.execute_reply.started":"2023-03-11T19:20:47.263691Z","shell.execute_reply":"2023-03-11T19:20:47.519455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert into a Panda dataframe \nfeatures_df_test = pd.DataFrame(features_test, columns=['feature','segment_Id'])\n\nprint('Finished feature extraction from ', len(features_df_test), ' files')","metadata":{"execution":{"iopub.status.busy":"2023-02-27T12:57:55.978599Z","iopub.execute_input":"2023-02-27T12:57:55.979207Z","iopub.status.idle":"2023-02-27T12:57:55.992126Z","shell.execute_reply.started":"2023-02-27T12:57:55.979148Z","shell.execute_reply":"2023-02-27T12:57:55.990060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_df_test[~features_df_test.feature.isna()]","metadata":{"execution":{"iopub.status.busy":"2023-02-27T12:58:20.507129Z","iopub.execute_input":"2023-02-27T12:58:20.508475Z","iopub.status.idle":"2023-02-27T12:58:20.532481Z","shell.execute_reply.started":"2023-02-27T12:58:20.508392Z","shell.execute_reply":"2023-02-27T12:58:20.530839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_df_test.to_csv('test_features.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:00:33.192398Z","iopub.execute_input":"2023-02-27T10:00:33.193554Z","iopub.status.idle":"2023-02-27T10:00:34.067692Z","shell.execute_reply.started":"2023-02-27T10:00:33.193481Z","shell.execute_reply":"2023-02-27T10:00:34.066305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict results\nX_test_ = np.array(features_df_test.feature.tolist())\nx_predict=model.predict(X_test_) \npredicted_label=np.argmax(x_predict,axis=1)\nprediction_class = le.inverse_transform(predicted_label) \nprint(prediction_class)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T11:19:55.374582Z","iopub.execute_input":"2023-02-27T11:19:55.375055Z","iopub.status.idle":"2023-02-27T11:19:55.658554Z","shell.execute_reply.started":"2023-02-27T11:19:55.375012Z","shell.execute_reply":"2023-02-27T11:19:55.657616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = metadata_test","metadata":{"execution":{"iopub.status.busy":"2023-02-27T11:20:00.840406Z","iopub.execute_input":"2023-02-27T11:20:00.841175Z","iopub.status.idle":"2023-02-27T11:20:00.846842Z","shell.execute_reply.started":"2023-02-27T11:20:00.841134Z","shell.execute_reply":"2023-02-27T11:20:00.845275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['pred'] = prediction_class","metadata":{"execution":{"iopub.status.busy":"2023-02-27T11:20:01.377734Z","iopub.execute_input":"2023-02-27T11:20:01.378494Z","iopub.status.idle":"2023-02-27T11:20:01.385505Z","shell.execute_reply.started":"2023-02-27T11:20:01.378447Z","shell.execute_reply":"2023-02-27T11:20:01.384119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SpeakerDialect_dic = {'Najdi':1, 'Hijazi':2, 'Khaliji':3, 'ModernStandardArabic':4}","metadata":{"execution":{"iopub.status.busy":"2023-02-27T11:12:02.650653Z","iopub.execute_input":"2023-02-27T11:12:02.651520Z","iopub.status.idle":"2023-02-27T11:12:02.656312Z","shell.execute_reply.started":"2023-02-27T11:12:02.651476Z","shell.execute_reply":"2023-02-27T11:12:02.655405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['SpeakerDialect'] = test_df['pred'].map(SpeakerDialect_dic)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-27T11:12:03.019367Z","iopub.execute_input":"2023-02-27T11:12:03.020361Z","iopub.status.idle":"2023-02-27T11:12:03.025892Z","shell.execute_reply.started":"2023-02-27T11:12:03.020321Z","shell.execute_reply":"2023-02-27T11:12:03.025022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[['SegmentID','SpeakerDialect']]","metadata":{"execution":{"iopub.status.busy":"2023-02-27T11:12:03.466306Z","iopub.execute_input":"2023-02-27T11:12:03.467387Z","iopub.status.idle":"2023-02-27T11:12:03.481528Z","shell.execute_reply.started":"2023-02-27T11:12:03.467343Z","shell.execute_reply":"2023-02-27T11:12:03.480292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[['SegmentID','SpeakerDialect']].to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T11:12:05.330694Z","iopub.execute_input":"2023-02-27T11:12:05.331122Z","iopub.status.idle":"2023-02-27T11:12:05.344357Z","shell.execute_reply.started":"2023-02-27T11:12:05.331084Z","shell.execute_reply":"2023-02-27T11:12:05.343105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:21:52.711382Z","iopub.execute_input":"2023-03-11T19:21:52.711782Z","iopub.status.idle":"2023-03-11T19:21:52.740929Z","shell.execute_reply.started":"2023-03-11T19:21:52.711740Z","shell.execute_reply":"2023-03-11T19:21:52.739740Z"},"trusted":true},"execution_count":null,"outputs":[]}]}