{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Loading Packages and Data**","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-18T10:48:38.608209Z","iopub.execute_input":"2022-03-18T10:48:38.608556Z","iopub.status.idle":"2022-03-18T10:48:38.624029Z","shell.execute_reply.started":"2022-03-18T10:48:38.608518Z","shell.execute_reply":"2022-03-18T10:48:38.622961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport math\nfrom wordcloud import WordCloud # for words statistics","metadata":{"execution":{"iopub.status.busy":"2022-03-18T10:48:38.632986Z","iopub.execute_input":"2022-03-18T10:48:38.633319Z","iopub.status.idle":"2022-03-18T10:48:38.638040Z","shell.execute_reply.started":"2022-03-18T10:48:38.633283Z","shell.execute_reply":"2022-03-18T10:48:38.637316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training data\ntrain_data = pd.read_csv(\"../input/quora-insincere-questions-classification/train.csv\")\n# Testing data\ntest_data = pd.read_csv(\"../input/quora-insincere-questions-classification/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-03-18T10:48:38.642048Z","iopub.execute_input":"2022-03-18T10:48:38.642331Z","iopub.status.idle":"2022-03-18T10:48:43.059752Z","shell.execute_reply.started":"2022-03-18T10:48:38.642301Z","shell.execute_reply":"2022-03-18T10:48:43.058589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data Exploration**","metadata":{}},{"cell_type":"code","source":"# Show some information \ntrain_data.info()\ntest_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-03-18T10:48:43.062781Z","iopub.execute_input":"2022-03-18T10:48:43.063008Z","iopub.status.idle":"2022-03-18T10:48:43.236214Z","shell.execute_reply.started":"2022-03-18T10:48:43.062981Z","shell.execute_reply":"2022-03-18T10:48:43.235326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-03-18T10:48:43.237745Z","iopub.execute_input":"2022-03-18T10:48:43.238048Z","iopub.status.idle":"2022-03-18T10:48:43.251822Z","shell.execute_reply.started":"2022-03-18T10:48:43.238007Z","shell.execute_reply":"2022-03-18T10:48:43.251141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-03-18T10:48:43.252831Z","iopub.execute_input":"2022-03-18T10:48:43.253642Z","iopub.status.idle":"2022-03-18T10:48:43.263515Z","shell.execute_reply.started":"2022-03-18T10:48:43.253604Z","shell.execute_reply":"2022-03-18T10:48:43.262584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sincere_questions=train_data[train_data['target']==0]\ninsincere_questions=train_data[train_data['target']==1]\nnum_of_sinc=sincere_questions.shape[0]\nnum_of_insinc=insincere_questions.shape[0]\npercentage_of_sincere=((num_of_sinc)/(num_of_sinc+num_of_insinc))*100\npercentage_of_insincere=((num_of_insinc)/(num_of_sinc+num_of_insinc))*100\nprint(\"No. of sincere questions\",num_of_sinc,\"Percentage:\",math.floor(percentage_of_sincere),\"%\")\nprint(\"No. of Insincere questions\",num_of_insinc,\"Percentage:\",math.ceil(percentage_of_insincere),\"%\")\nq=[num_of_sinc,num_of_insinc]\nlabels=['Sincere Questions','Insincere Questions']\nplt.bar(labels,q)\nplt.title(\"Target Distribution\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-18T10:48:43.265362Z","iopub.execute_input":"2022-03-18T10:48:43.266102Z","iopub.status.idle":"2022-03-18T10:48:43.788686Z","shell.execute_reply.started":"2022-03-18T10:48:43.266060Z","shell.execute_reply":"2022-03-18T10:48:43.787680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Word Cloud**","metadata":{}},{"cell_type":"code","source":"def black_color_func(word, font_size, position,orientation,random_state=None, **kwargs):\n    return(\"hsl(0,100%, 1%)\")\nwordcloud = WordCloud(background_color=\"white\", width=3000, height=2000, max_words=500).generate(\" \".join(sincere_questions.question_text))\nwordcloud.recolor(color_func = black_color_func)\nplt.figure(figsize=[15,10])\n# plot the wordcloud\nplt.imshow(wordcloud, interpolation=\"bilinear\")\n# remove plot axes\nplt.axis(\"off\")\n# save the image\nplt.savefig('wordcloud.png')","metadata":{"execution":{"iopub.status.busy":"2022-03-18T10:48:43.789844Z","iopub.execute_input":"2022-03-18T10:48:43.790078Z","iopub.status.idle":"2022-03-18T10:50:01.345643Z","shell.execute_reply.started":"2022-03-18T10:48:43.790050Z","shell.execute_reply":"2022-03-18T10:50:01.343948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 4/5 of the questions wil be used to train\n# the rest of them are used for validations\ntrain_ratio = 0.8 \nnum_of_train  = int(train_ratio * (num_of_sinc + num_of_insinc))\ntrain_sen = [] # array of training questions\nval_sen = [] # array of validating questions\ntest_sen = [] # array of testing questions\n\nfor i in range(0, len(train_data['question_text'])):\n    if i < num_of_train:\n        train_sen.append(train_data['question_text'].loc[i])\n    else:\n        val_sen.append(train_data['question_text'].loc[i])\n        \nfor i in range(0, len(test_data['question_text'])):\n    test_sen.append(test_data['question_text'].loc[i])\n\ntrain_label = [] # array of training questions' labels\nval_label = [] # array of validating questions' labels\nfor i in range(0, len(train_data['target'])):\n    if i < num_of_train:\n        train_label.append(float(train_data['target'].loc[i]))\n    else:\n        val_label.append(float(train_data['target'].loc[i]))","metadata":{"execution":{"iopub.status.busy":"2022-03-18T10:50:01.346966Z","iopub.execute_input":"2022-03-18T10:50:01.347478Z","iopub.status.idle":"2022-03-18T10:50:39.618667Z","shell.execute_reply.started":"2022-03-18T10:50:01.347430Z","shell.execute_reply":"2022-03-18T10:50:39.617868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3>Feature Engineering</h3>","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}