{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-06T00:23:41.609337Z","iopub.execute_input":"2023-03-06T00:23:41.609849Z","iopub.status.idle":"2023-03-06T00:23:44.701128Z","shell.execute_reply.started":"2023-03-06T00:23:41.609792Z","shell.execute_reply":"2023-03-06T00:23:44.699772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/traindata/train.csv\", sep=\",\")\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2023-03-06T00:23:44.703897Z","iopub.execute_input":"2023-03-06T00:23:44.704502Z","iopub.status.idle":"2023-03-06T00:23:46.720272Z","shell.execute_reply.started":"2023-03-06T00:23:44.704451Z","shell.execute_reply":"2023-03-06T00:23:46.719139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_dict = {\"ProcessedText\": train_data['ProcessedText'], \"SpeakerDialect\":train_data['SpeakerDialect'] }\ntrain_data_new = pd.DataFrame.from_dict(train_data_dict)\ntrain_data_new","metadata":{"execution":{"iopub.status.busy":"2023-03-06T00:23:46.721577Z","iopub.execute_input":"2023-03-06T00:23:46.721918Z","iopub.status.idle":"2023-03-06T00:23:46.742540Z","shell.execute_reply.started":"2023-03-06T00:23:46.721887Z","shell.execute_reply":"2023-03-06T00:23:46.741188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_data_new['SpeakerDialect'].value_counts())\n","metadata":{"execution":{"iopub.status.busy":"2023-03-06T00:23:46.746017Z","iopub.execute_input":"2023-03-06T00:23:46.746546Z","iopub.status.idle":"2023-03-06T00:23:46.767938Z","shell.execute_reply.started":"2023-03-06T00:23:46.746484Z","shell.execute_reply":"2023-03-06T00:23:46.766553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nfrom pyarabic.araby import strip_tashkeel, strip_tatweel\n\ndef clean_arabic_text (text):\n    \n    text = str(text)\n    # Remove non-Arabic characters\n    text = re.sub(\"[^\\u0600-\\u06FF\\u0750-\\u077F\\u08A0-\\u08FF\\uFB50-\\uFDCF\\uFDF0-\\uFDFF\\uFE70-\\uFEFF]+\", \" \", text)\n    # Remove tatweel (elongation) characters\n    text = strip_tatweel(text)\n    # Remove diacritical marks (tashkeel) characters\n    text = strip_tashkeel(text)\n    # Normalize spaces\n    text = re.sub(r'\\s+', ' ', text).strip()\n    return text\n","metadata":{"execution":{"iopub.status.busy":"2023-03-06T00:23:46.769561Z","iopub.execute_input":"2023-03-06T00:23:46.770068Z","iopub.status.idle":"2023-03-06T00:23:46.796433Z","shell.execute_reply.started":"2023-03-06T00:23:46.770021Z","shell.execute_reply":"2023-03-06T00:23:46.795208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_new[\"ProcessedText\"] = train_data_new[\"ProcessedText\"].apply(lambda x:   clean_arabic_text(x))\ntrain_data_new","metadata":{"execution":{"iopub.status.busy":"2023-03-06T00:23:46.797987Z","iopub.execute_input":"2023-03-06T00:23:46.798470Z","iopub.status.idle":"2023-03-06T00:23:50.470614Z","shell.execute_reply.started":"2023-03-06T00:23:46.798421Z","shell.execute_reply":"2023-03-06T00:23:50.469337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the mapping from string to numerical values\nclass_mapping = {\n    'Najdi': 1,\n    'Hijazi': 2,\n    'Khaliji': 3,\n    'ModernStandardArabic': 4\n}\n\n# Map the target column from string to numerical values\ntrain_data_new['target_numerical'] = train_data_new['SpeakerDialect'].map(class_mapping)\n\n# Print the new column in the dataset\ntrain_data_new","metadata":{"execution":{"iopub.status.busy":"2023-03-06T00:23:50.473903Z","iopub.execute_input":"2023-03-06T00:23:50.474516Z","iopub.status.idle":"2023-03-06T00:23:50.502870Z","shell.execute_reply.started":"2023-03-06T00:23:50.474478Z","shell.execute_reply":"2023-03-06T00:23:50.501574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ncolors = ['#FF5733', '#FFC300', '#DAF7A6', '#C70039']\ntrain_data_new['SpeakerDialect'].value_counts().plot(kind='barh' , color=colors);\nplt.title('Speaker Dialect counts')\nplt.xlabel(\"Speaker Dialect\", labelpad=14)\n#plt.ylabel(\"Count\", labelpad=14)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T00:23:50.504911Z","iopub.execute_input":"2023-03-06T00:23:50.505301Z","iopub.status.idle":"2023-03-06T00:23:50.849072Z","shell.execute_reply.started":"2023-03-06T00:23:50.505265Z","shell.execute_reply":"2023-03-06T00:23:50.847782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ncolors= ['#FF5733', '#FFC300', '#DAF7A6', '#C70039']\n\ntrain_data_new['SpeakerDialect'].value_counts().plot(kind='pie' , autopct='%1.0f%%', colors=colors);\nplt.title('Speaker Dialect counts in Piechart')\nplt.xlabel(\"Speaker Dialect\", labelpad=14)\nplt.ylabel(\"Count\", labelpad=14)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T00:23:50.850610Z","iopub.execute_input":"2023-03-06T00:23:50.850989Z","iopub.status.idle":"2023-03-06T00:23:51.067713Z","shell.execute_reply.started":"2023-03-06T00:23:50.850953Z","shell.execute_reply":"2023-03-06T00:23:51.065744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"\nfrom sklearn.feature_extraction.text import TfidfVectorizer\ncorpus = train_data_new['ProcessedText'].values.astype('U')\ntfidf = TfidfVectorizer(max_features = 10000) \ntdidf_tensor = tfidf.fit_transform(corpus)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-06T00:23:51.074542Z","iopub.execute_input":"2023-03-06T00:23:51.076491Z","iopub.status.idle":"2023-03-06T00:23:54.094681Z","shell.execute_reply.started":"2023-03-06T00:23:51.076407Z","shell.execute_reply":"2023-03-06T00:23:54.093271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdidf_tensor.shape\nprint('{} Number of text has {} words'.format(tdidf_tensor.shape[0], tdidf_tensor.shape[1]))","metadata":{"execution":{"iopub.status.busy":"2023-03-06T00:23:54.096133Z","iopub.execute_input":"2023-03-06T00:23:54.096492Z","iopub.status.idle":"2023-03-06T00:23:54.102712Z","shell.execute_reply.started":"2023-03-06T00:23:54.096458Z","shell.execute_reply":"2023-03-06T00:23:54.101402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Splitting the data into trainig and testing\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, Y_train, Y_test = train_test_split(tdidf_tensor, train_data_new['target_numerical'], test_size=0.3, random_state=5)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T00:23:54.104401Z","iopub.execute_input":"2023-03-06T00:23:54.105453Z","iopub.status.idle":"2023-03-06T00:23:54.150903Z","shell.execute_reply.started":"2023-03-06T00:23:54.105411Z","shell.execute_reply":"2023-03-06T00:23:54.149631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import svm,metrics\n\n#Create a svm Classifier\nclf = svm.SVC(kernel='linear') # Linear Kernel\n\n#Train the model using the training sets\nclf.fit(X_train, Y_train)\n\n#Predict the response for test dataset\npredicted = clf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T00:23:54.152705Z","iopub.execute_input":"2023-03-06T00:23:54.153591Z","iopub.status.idle":"2023-03-06T01:08:01.377998Z","shell.execute_reply.started":"2023-03-06T00:23:54.153546Z","shell.execute_reply":"2023-03-06T01:08:01.376648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, classification_report\naccuracy_score = metrics.accuracy_score(predicted, Y_test)\nprint(\"Accuracuy Score: \",accuracy_score)\nprint(classification_report(Y_test, predicted, digits=5))","metadata":{"execution":{"iopub.status.busy":"2023-03-06T01:09:35.709665Z","iopub.execute_input":"2023-03-06T01:09:35.710444Z","iopub.status.idle":"2023-03-06T01:09:35.780573Z","shell.execute_reply.started":"2023-03-06T01:09:35.710403Z","shell.execute_reply":"2023-03-06T01:09:35.779066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(\"/kaggle/input/ml-olympiad-dialectrecognition/test.csv\", sep=\",\")\ntest_data","metadata":{"execution":{"iopub.status.busy":"2023-03-06T01:15:54.442973Z","iopub.execute_input":"2023-03-06T01:15:54.443768Z","iopub.status.idle":"2023-03-06T01:15:54.498169Z","shell.execute_reply.started":"2023-03-06T01:15:54.443725Z","shell.execute_reply":"2023-03-06T01:15:54.496865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_dict = {\"SegmentID\":test_data['SegmentID'],\"ProcessedText\": test_data['ProcessedText']}\ntest_data_new = pd.DataFrame.from_dict(test_data_dict)\ntest_data_new","metadata":{"execution":{"iopub.status.busy":"2023-03-06T01:16:52.323933Z","iopub.execute_input":"2023-03-06T01:16:52.324702Z","iopub.status.idle":"2023-03-06T01:16:52.340067Z","shell.execute_reply.started":"2023-03-06T01:16:52.324657Z","shell.execute_reply":"2023-03-06T01:16:52.338716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_new[\"ProcessedText\"] = test_data_new[\"ProcessedText\"].apply(lambda x:   clean_arabic_text(x))\ntest_data_new","metadata":{"execution":{"iopub.status.busy":"2023-03-06T01:16:55.711420Z","iopub.execute_input":"2023-03-06T01:16:55.712632Z","iopub.status.idle":"2023-03-06T01:16:55.779380Z","shell.execute_reply.started":"2023-03-06T01:16:55.712584Z","shell.execute_reply":"2023-03-06T01:16:55.778171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a new column in the dataframe to store the predicted values\ntest_data_new['SpeakerDialect'] = pd.Series(predicted)\ntest_data_new\n\nSubmission={'SegmentID' : test_data_new['SegmentID'] , \"SpeakerDialect\":test_data_new['SpeakerDialect'] }\nSubmission = pd.DataFrame.from_dict(Submission)\nSubmission","metadata":{"execution":{"iopub.status.busy":"2023-03-06T01:28:02.990590Z","iopub.execute_input":"2023-03-06T01:28:02.991045Z","iopub.status.idle":"2023-03-06T01:28:03.009367Z","shell.execute_reply.started":"2023-03-06T01:28:02.991010Z","shell.execute_reply":"2023-03-06T01:28:03.008161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nresult_file = \"submission.csv\"\nSubmission.to_csv(result_file, sep= \",\", index = False)\nSubmission","metadata":{"execution":{"iopub.status.busy":"2023-03-06T01:28:14.295819Z","iopub.execute_input":"2023-03-06T01:28:14.296228Z","iopub.status.idle":"2023-03-06T01:28:14.319504Z","shell.execute_reply.started":"2023-03-06T01:28:14.296193Z","shell.execute_reply":"2023-03-06T01:28:14.318358Z"},"trusted":true},"execution_count":null,"outputs":[]}]}