{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport plotly.express as px\nimport matplotlib.pyplot as plt\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n'''\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n'''\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-19T13:52:03.680484Z","iopub.execute_input":"2023-02-19T13:52:03.680905Z","iopub.status.idle":"2023-02-19T13:52:04.995682Z","shell.execute_reply.started":"2023-02-19T13:52:03.680873Z","shell.execute_reply":"2023-02-19T13:52:04.994049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read data","metadata":{}},{"cell_type":"code","source":"import pandas as pd\ntrain_df = pd.read_csv('/kaggle/input/ml-olympiad-dialectrecognition/train.csv')\ntrain_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:04.998195Z","iopub.execute_input":"2023-02-19T13:52:04.998698Z","iopub.status.idle":"2023-02-19T13:52:07.098599Z","shell.execute_reply.started":"2023-02-19T13:52:04.998649Z","shell.execute_reply":"2023-02-19T13:52:07.097046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:07.099846Z","iopub.execute_input":"2023-02-19T13:52:07.100644Z","iopub.status.idle":"2023-02-19T13:52:07.107802Z","shell.execute_reply.started":"2023-02-19T13:52:07.100609Z","shell.execute_reply":"2023-02-19T13:52:07.106889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(train_df, \n             x=train_df.SpeakerDialect.value_counts().index, \n             y=train_df.SpeakerDialect.value_counts().values,\n             template=  \"simple_white\"  ,\n             title= \"Distribuation of speakers based on Dialect\",\n             color_discrete_sequence = px.colors.sequential.Teal_r,  \n             labels={'y':\"Count of speakers\", 'x':\"Dialect\"},\n             text=['{}'.format(p) for p in train_df.SpeakerDialect.value_counts().values]\n            )\n\nfig.update_layout(coloraxis_showscale=False)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:07.110355Z","iopub.execute_input":"2023-02-19T13:52:07.111032Z","iopub.status.idle":"2023-02-19T13:52:08.335596Z","shell.execute_reply.started":"2023-02-19T13:52:07.110977Z","shell.execute_reply":"2023-02-19T13:52:08.333762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(train_df, \n             x=train_df.SpeakerGender.value_counts().index, \n             y=train_df.SpeakerGender.value_counts().values,\n             template=  \"simple_white\"  ,\n             title= \"Distribuation of speakers based on gender\",\n             color_discrete_sequence = px.colors.sequential.Teal_r,  \n             labels={'y':\"Count of speakers\", 'x':\"Gender\"},\n             text=['{}'.format(p) for p in train_df.SpeakerGender.value_counts().values]\n            )\n\nfig.update_layout(coloraxis_showscale=False)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:08.337561Z","iopub.execute_input":"2023-02-19T13:52:08.338272Z","iopub.status.idle":"2023-02-19T13:52:08.432683Z","shell.execute_reply.started":"2023-02-19T13:52:08.338223Z","shell.execute_reply":"2023-02-19T13:52:08.431468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(train_df, \n             x=train_df.SpeakerAge.value_counts().index, \n             y=train_df.SpeakerAge.value_counts().values,\n             template=  \"simple_white\"  ,\n             title= \"Distribuation of speakers age\",\n             color_discrete_sequence = px.colors.sequential.Teal_r,  \n             labels={'y':\"Count of speakers\", 'x':\"Age\"},\n             text=['{}'.format(p) for p in train_df.SpeakerAge.value_counts().values]\n            )\n\nfig.update_layout(coloraxis_showscale=False)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:08.434276Z","iopub.execute_input":"2023-02-19T13:52:08.437369Z","iopub.status.idle":"2023-02-19T13:52:08.527871Z","shell.execute_reply.started":"2023-02-19T13:52:08.437316Z","shell.execute_reply":"2023-02-19T13:52:08.526539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(train_df, \n             x=train_df.SegmentLength,\n             template=  \"simple_white\"  ,\n             title= \"Distribuation of segment Length\",\n             # color_discrete_sequence = px.colors.sequential.Teal_r,  \n             labels={'count':\"Count of segmenst\", 'SegmentLength':\"Segment Length\"},\n             #text=['{}'.format(p) for p in train_df.SegmentLength.value_counts().values]\n            )\n\nfig.update_layout(coloraxis_showscale=False)\n#fig.update_layout(bargap=0.6)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:08.529842Z","iopub.execute_input":"2023-02-19T13:52:08.530283Z","iopub.status.idle":"2023-02-19T13:52:08.683947Z","shell.execute_reply.started":"2023-02-19T13:52:08.530241Z","shell.execute_reply":"2023-02-19T13:52:08.682903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(train_df, \n             x=train_df.Environment, \n             template=  \"simple_white\"  ,\n             title= \"Coutn of records for each show\",\n             color_discrete_sequence = px.colors.sequential.Teal_r,  \n             labels={'y':\"Count of speakers\", 'x':\"Dialect\"},\n             #text=['{}'.format(p) for p in train_df.ShowName.value_counts().values]\n            )\n\nfig.update_layout(coloraxis_showscale=False)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:08.685010Z","iopub.execute_input":"2023-02-19T13:52:08.685464Z","iopub.status.idle":"2023-02-19T13:52:09.360836Z","shell.execute_reply.started":"2023-02-19T13:52:08.685411Z","shell.execute_reply":"2023-02-19T13:52:09.359044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare data ","metadata":{}},{"cell_type":"code","source":"import librosa\nimport IPython.display as ipd\nfrom pydub import AudioSegment","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:09.363436Z","iopub.execute_input":"2023-02-19T13:52:09.363893Z","iopub.status.idle":"2023-02-19T13:52:11.299327Z","shell.execute_reply.started":"2023-02-19T13:52:09.363857Z","shell.execute_reply":"2023-02-19T13:52:11.297932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepath = \"/kaggle/input/ml-olympiad-dialectrecognition/batch_4/6k_v_SBA_330_3.wav\"\ny,sr = librosa.load(filepath)\nipd.Audio(filepath, rate=sr)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:11.303116Z","iopub.execute_input":"2023-02-19T13:52:11.303507Z","iopub.status.idle":"2023-02-19T13:52:19.102803Z","shell.execute_reply.started":"2023-02-19T13:52:11.303472Z","shell.execute_reply":"2023-02-19T13:52:19.100595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Take equal number of samples from each class","metadata":{}},{"cell_type":"code","source":"Samplesize = 3950  # number of samples that you want       \ndf = train_df.groupby('SpeakerDialect', as_index=False).apply(lambda array: array.loc[np.random.choice(array.index, Samplesize, False),:])\ndf.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:19.105617Z","iopub.execute_input":"2023-02-19T13:52:19.107143Z","iopub.status.idle":"2023-02-19T13:52:19.267812Z","shell.execute_reply.started":"2023-02-19T13:52:19.107062Z","shell.execute_reply":"2023-02-19T13:52:19.266942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_index = df.index.droplevel()\nsample_index","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:19.269114Z","iopub.execute_input":"2023-02-19T13:52:19.269786Z","iopub.status.idle":"2023-02-19T13:52:19.278508Z","shell.execute_reply.started":"2023-02-19T13:52:19.269748Z","shell.execute_reply":"2023-02-19T13:52:19.277442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.sample(n=5)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:19.280196Z","iopub.execute_input":"2023-02-19T13:52:19.280880Z","iopub.status.idle":"2023-02-19T13:52:19.309942Z","shell.execute_reply.started":"2023-02-19T13:52:19.280819Z","shell.execute_reply":"2023-02-19T13:52:19.308879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[:4]","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:19.311754Z","iopub.execute_input":"2023-02-19T13:52:19.312506Z","iopub.status.idle":"2023-02-19T13:52:19.332685Z","shell.execute_reply.started":"2023-02-19T13:52:19.312467Z","shell.execute_reply":"2023-02-19T13:52:19.331576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Preprocessing text + Build model","metadata":{}},{"cell_type":"code","source":"# Machine Learning packages\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer # to create Bag of words\nfrom sklearn.model_selection import train_test_split  # for splitting data\nfrom sklearn.linear_model import LogisticRegression # to bulid classifier model\nfrom sklearn.preprocessing import LabelEncoder # to convert classes to number \nfrom sklearn.metrics import accuracy_score, classification_report, plot_confusion_matrix # to calculate accuracy and classification report\nfrom sklearn.naive_bayes import GaussianNB # to bulid classifier model\nfrom sklearn.svm import SVC\n\n# NLP libraries\nimport re # for preprocessing text\nimport string # for preprocessing text\nimport nltk # for processing texts\nfrom nltk.stem import PorterStemmer\nfrom nltk.corpus import stopwords # list of stop words\nnltk.download('punkt')\nnltk.download('stopwords')\nnltk.download('wordnet')","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:19.334318Z","iopub.execute_input":"2023-02-19T13:52:19.335518Z","iopub.status.idle":"2023-02-19T13:52:20.242192Z","shell.execute_reply.started":"2023-02-19T13:52:19.335469Z","shell.execute_reply":"2023-02-19T13:52:20.241026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"arabic_punctuations = '''`÷×؛<>_()*&^%][ـ،/:\"؟.,'{}~¦+|!”…“–ـ''' # define arabic punctuations\n\ndef clean_text(text):\n  '''\n  DESCRIPTION:\n  This function to clean text \n  INPUT: \n  text: string\n  OUTPUT: \n  text: string after clean it\n  ''' \n  text = re.sub(\"[a-zA-Z]\", \" \", text) # remove english letters\n  text = re.sub('\\n', ' ', text) # remove \\n from text\n  text = re.sub(r'\\d+', '', text) #remove number\n  text = re.sub(r'http\\S+', '', text) # remove links\n  #text = text.translate(str.maketrans('','', arabic_punctuations)) # remove punctuation\n  # text = ' '.join([word for word in text.split() if word not in stopwords.words(\"arabic\")]) # remove stop word\n  text = re.sub(' +', ' ',text) # remove extra space\n  text = text.strip() #remove whitespaces\n\n\n  return text","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:20.244118Z","iopub.execute_input":"2023-02-19T13:52:20.244954Z","iopub.status.idle":"2023-02-19T13:52:20.253821Z","shell.execute_reply.started":"2023-02-19T13:52:20.244909Z","shell.execute_reply":"2023-02-19T13:52:20.252707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The cleaning function applied in all rows\ntrain_df['Cleaned_Text'] = train_df.iloc[sample_index]['GroundTruthText'].apply(clean_text)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:20.255384Z","iopub.execute_input":"2023-02-19T13:52:20.256614Z","iopub.status.idle":"2023-02-19T13:52:20.481636Z","shell.execute_reply.started":"2023-02-19T13:52:20.256555Z","shell.execute_reply":"2023-02-19T13:52:20.480609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[train_df['ProcessedText'].isna()].head()","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:20.483452Z","iopub.execute_input":"2023-02-19T13:52:20.483874Z","iopub.status.idle":"2023-02-19T13:52:20.548483Z","shell.execute_reply.started":"2023-02-19T13:52:20.483828Z","shell.execute_reply":"2023-02-19T13:52:20.547358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_features = 200 # maximum number of features \ncount_vector = TfidfVectorizer()  # create Count Vectorizer\nX = count_vector.fit_transform(train_df.iloc[sample_index]['Cleaned_Text']).toarray() # fit the CountVectorizer using reviews data\nX","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:20.549960Z","iopub.execute_input":"2023-02-19T13:52:20.550413Z","iopub.status.idle":"2023-02-19T13:52:21.657711Z","shell.execute_reply.started":"2023-02-19T13:52:20.550370Z","shell.execute_reply":"2023-02-19T13:52:21.656465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Split the dataset into independent and dependent dataset\n\ny=np.array(train_df.iloc[sample_index]['SpeakerDialect'].tolist())\n### Label Encoding -> Label Encoder\nfrom tensorflow.keras.utils import to_categorical\nfrom sklearn.preprocessing import LabelEncoder\nlabelencoder=LabelEncoder()\ny=to_categorical(labelencoder.fit_transform(y))\n### Train Test Split\n\nX_train, X_test, y_train, y_test = train_test_split(X,y, test_size =0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:53:10.089969Z","iopub.execute_input":"2023-02-19T13:53:10.090417Z","iopub.status.idle":"2023-02-19T13:53:11.263330Z","shell.execute_reply.started":"2023-02-19T13:53:10.090382Z","shell.execute_reply":"2023-02-19T13:53:11.262104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:53:11.265393Z","iopub.execute_input":"2023-02-19T13:53:11.266324Z","iopub.status.idle":"2023-02-19T13:53:11.271020Z","shell.execute_reply.started":"2023-02-19T13:53:11.266262Z","shell.execute_reply":"2023-02-19T13:53:11.269925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define Logistic Regression\n#model =  LogisticRegression()\n\n# train model\n#model.fit(X_train, y_train) ","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:53:11.768226Z","iopub.execute_input":"2023-02-19T13:53:11.769380Z","iopub.status.idle":"2023-02-19T13:53:11.774509Z","shell.execute_reply.started":"2023-02-19T13:53:11.769317Z","shell.execute_reply":"2023-02-19T13:53:11.773405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predicting the Train set results \n#print('Train model accuracy: ',accuracy_score(y_train, model.predict(X_train)))","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:53:12.288785Z","iopub.execute_input":"2023-02-19T13:53:12.290161Z","iopub.status.idle":"2023-02-19T13:53:12.294103Z","shell.execute_reply.started":"2023-02-19T13:53:12.290116Z","shell.execute_reply":"2023-02-19T13:53:12.293220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predicting the Test set results \n#y_pred = model.predict(X_test) \n#print('Test model accuracy: ',accuracy_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:53:12.627069Z","iopub.execute_input":"2023-02-19T13:53:12.627893Z","iopub.status.idle":"2023-02-19T13:53:12.633053Z","shell.execute_reply.started":"2023-02-19T13:53:12.627849Z","shell.execute_reply":"2023-02-19T13:53:12.631656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense,Dropout,Activation,Flatten\nfrom tensorflow.keras.optimizers import Adam\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:53:13.219225Z","iopub.execute_input":"2023-02-19T13:53:13.219663Z","iopub.status.idle":"2023-02-19T13:53:13.226272Z","shell.execute_reply.started":"2023-02-19T13:53:13.219630Z","shell.execute_reply":"2023-02-19T13:53:13.225085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_labels= 4","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:53:13.666307Z","iopub.execute_input":"2023-02-19T13:53:13.667159Z","iopub.status.idle":"2023-02-19T13:53:13.671927Z","shell.execute_reply.started":"2023-02-19T13:53:13.667114Z","shell.execute_reply":"2023-02-19T13:53:13.670814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmodel=Sequential()\n###first laye\nmodel.add(Dense(100))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\n###second layer\nmodel.add(Dense(200))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\n###third layer\nmodel.add(Dense(100))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\n###final layer\nmodel.add(Dense(num_labels))\nmodel.add(Activation('softmax'))","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:53:14.497642Z","iopub.execute_input":"2023-02-19T13:53:14.498076Z","iopub.status.idle":"2023-02-19T13:53:14.521083Z","shell.execute_reply.started":"2023-02-19T13:53:14.498039Z","shell.execute_reply":"2023-02-19T13:53:14.519863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss='categorical_crossentropy',metrics=['accuracy'],optimizer='Adam')\n\n","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:53:15.067158Z","iopub.execute_input":"2023-02-19T13:53:15.067621Z","iopub.status.idle":"2023-02-19T13:53:15.078441Z","shell.execute_reply.started":"2023-02-19T13:53:15.067583Z","shell.execute_reply":"2023-02-19T13:53:15.077229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint\nfrom datetime import datetime \nnum_epochs = 30\nnum_batch_size = 64\ncheckpointer = ModelCheckpoint(filepath='./audio_classification.hdf5', \n                               verbose=1, save_best_only=True)\nstart = datetime.now()\nmodel.fit(X_train, y_train, batch_size=num_batch_size, epochs=num_epochs, callbacks=[checkpointer], verbose=1)\nduration = datetime.now() - start\nprint(\"Training completed in time: \", duration)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:54:29.649090Z","iopub.execute_input":"2023-02-19T13:54:29.649626Z","iopub.status.idle":"2023-02-19T13:56:27.312919Z","shell.execute_reply.started":"2023-02-19T13:54:29.649578Z","shell.execute_reply":"2023-02-19T13:56:27.311705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_accuracy=model.evaluate(X_test,y_test,verbose=0)\nprint(test_accuracy[1])","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:56:27.315431Z","iopub.execute_input":"2023-02-19T13:56:27.315824Z","iopub.status.idle":"2023-02-19T13:56:28.550872Z","shell.execute_reply.started":"2023-02-19T13:56:27.315792Z","shell.execute_reply":"2023-02-19T13:56:28.549529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/ml-olympiad-dialectrecognition/test.csv')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:56:28.552646Z","iopub.execute_input":"2023-02-19T13:56:28.553476Z","iopub.status.idle":"2023-02-19T13:56:28.604856Z","shell.execute_reply.started":"2023-02-19T13:56:28.553429Z","shell.execute_reply":"2023-02-19T13:56:28.603986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['Cleaned_Text'] = test_df['GroundTruthText'].apply(clean_text)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:56:28.606884Z","iopub.execute_input":"2023-02-19T13:56:28.607949Z","iopub.status.idle":"2023-02-19T13:56:28.640854Z","shell.execute_reply.started":"2023-02-19T13:56:28.607899Z","shell.execute_reply":"2023-02-19T13:56:28.639929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert to number\ntest_vector = count_vector.transform(test_df['Cleaned_Text'])\ntest_vector = test_vector.toarray()","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:56:28.642102Z","iopub.execute_input":"2023-02-19T13:56:28.643048Z","iopub.status.idle":"2023-02-19T13:56:28.754176Z","shell.execute_reply.started":"2023-02-19T13:56:28.643012Z","shell.execute_reply":"2023-02-19T13:56:28.753016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_predict = model.predict(test_vector)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:56:46.609212Z","iopub.execute_input":"2023-02-19T13:56:46.609645Z","iopub.status.idle":"2023-02-19T13:56:47.273365Z","shell.execute_reply.started":"2023-02-19T13:56:46.609611Z","shell.execute_reply":"2023-02-19T13:56:47.271997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_label=np.argmax(x_predict,axis=1)\nprint(predicted_label)\nprediction_class = labelencoder.inverse_transform(predicted_label) \nprint(prediction_class)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:56:47.275765Z","iopub.execute_input":"2023-02-19T13:56:47.276508Z","iopub.status.idle":"2023-02-19T13:56:47.284970Z","shell.execute_reply.started":"2023-02-19T13:56:47.276460Z","shell.execute_reply":"2023-02-19T13:56:47.283747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## encodeing predict class\n## text_predict_class = model.predict(test_vector)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:56:50.042033Z","iopub.execute_input":"2023-02-19T13:56:50.042530Z","iopub.status.idle":"2023-02-19T13:56:50.048032Z","shell.execute_reply.started":"2023-02-19T13:56:50.042491Z","shell.execute_reply":"2023-02-19T13:56:50.046748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['SpeakerDialect'] = prediction_class","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:56:50.642656Z","iopub.execute_input":"2023-02-19T13:56:50.643312Z","iopub.status.idle":"2023-02-19T13:56:50.648816Z","shell.execute_reply.started":"2023-02-19T13:56:50.643255Z","shell.execute_reply":"2023-02-19T13:56:50.647774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#list(test_df.SegmentID[:50])","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:56:51.097688Z","iopub.execute_input":"2023-02-19T13:56:51.098975Z","iopub.status.idle":"2023-02-19T13:56:51.103500Z","shell.execute_reply.started":"2023-02-19T13:56:51.098929Z","shell.execute_reply":"2023-02-19T13:56:51.102366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_class","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:56:52.232219Z","iopub.execute_input":"2023-02-19T13:56:52.232619Z","iopub.status.idle":"2023-02-19T13:56:52.240809Z","shell.execute_reply.started":"2023-02-19T13:56:52.232586Z","shell.execute_reply":"2023-02-19T13:56:52.239656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[:50]","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:56:56.232871Z","iopub.execute_input":"2023-02-19T13:56:56.233915Z","iopub.status.idle":"2023-02-19T13:56:56.288309Z","shell.execute_reply.started":"2023-02-19T13:56:56.233875Z","shell.execute_reply":"2023-02-19T13:56:56.287029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepath = \"/kaggle/input/ml-olympiad-dialectrecognition/batch_4/6k_v_SBA_330_3.wav\"\ny,sr = librosa.load(filepath)\nipd.Audio(filepath, rate=sr)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:54:21.429251Z","iopub.status.idle":"2023-02-19T13:54:21.430238Z","shell.execute_reply.started":"2023-02-19T13:54:21.429913Z","shell.execute_reply":"2023-02-19T13:54:21.429943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SpeakerDialect_dic = {'Najdi':1, 'Hijazi':2, 'Khaliji':3, 'ModernStandardArabic':4}","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:00.751224Z","iopub.status.idle":"2023-02-19T13:52:00.751714Z","shell.execute_reply.started":"2023-02-19T13:52:00.751457Z","shell.execute_reply":"2023-02-19T13:52:00.751476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['SpeakerDialect_number'] = test_df['SpeakerDialect'].map(SpeakerDialect_dic)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:00.752768Z","iopub.status.idle":"2023-02-19T13:52:00.753142Z","shell.execute_reply.started":"2023-02-19T13:52:00.752955Z","shell.execute_reply":"2023-02-19T13:52:00.752974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = test_df[['SegmentID','SpeakerDialect_number']].rename(columns={\"SpeakerDialect_number\": \"SpeakerDialect\"})\n\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:00.754311Z","iopub.status.idle":"2023-02-19T13:52:00.754691Z","shell.execute_reply.started":"2023-02-19T13:52:00.754494Z","shell.execute_reply":"2023-02-19T13:52:00.754511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('submission.csv')\nsubmission\n","metadata":{"execution":{"iopub.status.busy":"2023-02-19T13:52:00.755971Z","iopub.status.idle":"2023-02-19T13:52:00.756392Z","shell.execute_reply.started":"2023-02-19T13:52:00.756159Z","shell.execute_reply":"2023-02-19T13:52:00.756185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}