{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Basic Text Classification Approach\n1. Exploring Data\n2. Cleaning Data\n3. Developing a pipeline for data preprocessing\n4. Train/Val/test split into features and targets\n5. Model Selection and training \n6. Submission\n\n\n* **NOTE** :To introduce you to the process I have only used a small portion of the dataset.*","metadata":{}},{"cell_type":"markdown","source":"# 1. Exploring Data","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd \nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-14T15:58:42.990724Z","iopub.execute_input":"2022-02-14T15:58:42.991060Z","iopub.status.idle":"2022-02-14T15:58:43.018401Z","shell.execute_reply.started":"2022-02-14T15:58:42.990984Z","shell.execute_reply":"2022-02-14T15:58:43.017744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom tqdm import tqdm \nimport time\n#tqdm is used to create progress bars\ntqdm.pandas()","metadata":{"execution":{"iopub.status.busy":"2022-02-14T15:58:43.021067Z","iopub.execute_input":"2022-02-14T15:58:43.021428Z","iopub.status.idle":"2022-02-14T15:58:43.027822Z","shell.execute_reply.started":"2022-02-14T15:58:43.021401Z","shell.execute_reply":"2022-02-14T15:58:43.027041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain = pd.read_csv(\"../input/quora-insincere-questions-classification/train.csv\",index_col='qid')\ntest = pd.read_csv(\"../input/quora-insincere-questions-classification/test.csv\",index_col='qid')\nprint(\"Train shape : \",train.shape)\nprint(\"Test shape : \",test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-02-14T15:58:43.030520Z","iopub.execute_input":"2022-02-14T15:58:43.030813Z","iopub.status.idle":"2022-02-14T15:58:48.310741Z","shell.execute_reply.started":"2022-02-14T15:58:43.030780Z","shell.execute_reply":"2022-02-14T15:58:48.309059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-02-14T15:58:48.312888Z","iopub.execute_input":"2022-02-14T15:58:48.313135Z","iopub.status.idle":"2022-02-14T15:58:48.332148Z","shell.execute_reply.started":"2022-02-14T15:58:48.313099Z","shell.execute_reply":"2022-02-14T15:58:48.331311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train['target']==0].shape","metadata":{"execution":{"iopub.status.busy":"2022-02-14T15:58:48.333677Z","iopub.execute_input":"2022-02-14T15:58:48.333930Z","iopub.status.idle":"2022-02-14T15:58:48.433200Z","shell.execute_reply.started":"2022-02-14T15:58:48.333897Z","shell.execute_reply":"2022-02-14T15:58:48.432324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating a Shorter Training data","metadata":{}},{"cell_type":"code","source":"pos_df = train[train['target']==0].sample(frac=0.03)\nneg_df = train[train['target']==1].sample(frac=0.4)","metadata":{"execution":{"iopub.status.busy":"2022-02-14T15:58:48.435160Z","iopub.execute_input":"2022-02-14T15:58:48.435817Z","iopub.status.idle":"2022-02-14T15:58:48.579752Z","shell.execute_reply.started":"2022-02-14T15:58:48.435787Z","shell.execute_reply":"2022-02-14T15:58:48.578927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.concat([pos_df,neg_df])\ntrain_df['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-02-14T15:58:48.580963Z","iopub.execute_input":"2022-02-14T15:58:48.581209Z","iopub.status.idle":"2022-02-14T15:58:48.595207Z","shell.execute_reply.started":"2022-02-14T15:58:48.581176Z","shell.execute_reply":"2022-02-14T15:58:48.594400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"que_eg = train_df.iloc[:1,0]\nsentences = que_eg.item().split()\nsentences","metadata":{"execution":{"iopub.status.busy":"2022-02-14T15:58:48.596627Z","iopub.execute_input":"2022-02-14T15:58:48.597303Z","iopub.status.idle":"2022-02-14T15:58:48.604235Z","shell.execute_reply.started":"2022-02-14T15:58:48.597267Z","shell.execute_reply":"2022-02-14T15:58:48.603358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-02-14T15:58:48.605604Z","iopub.execute_input":"2022-02-14T15:58:48.605858Z","iopub.status.idle":"2022-02-14T15:58:48.622630Z","shell.execute_reply.started":"2022-02-14T15:58:48.605823Z","shell.execute_reply":"2022-02-14T15:58:48.621945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df, val_df = \\\n              np.split(train_df.sample(frac=1, random_state=42), \n                       [int(.8*len(train_df))])\n\ntrain_df.shape , val_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-02-14T15:58:48.624962Z","iopub.execute_input":"2022-02-14T15:58:48.625710Z","iopub.status.idle":"2022-02-14T15:58:48.652573Z","shell.execute_reply.started":"2022-02-14T15:58:48.625672Z","shell.execute_reply":"2022-02-14T15:58:48.651962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Cleaning Data","metadata":{}},{"cell_type":"code","source":"import re\ndef cleanText(resumeText):\n    resumeText = re.sub('http\\S+\\s*', ' ', resumeText)  # remove URLs\n    resumeText = re.sub('RT|cc', ' ', resumeText)  # remove RT and cc\n    resumeText = re.sub('#\\S+', '', resumeText)  # remove hashtags\n    resumeText = re.sub('@\\S+', '  ', resumeText)  # remove mentions\n    resumeText = re.sub('[%s]' % re.escape(\"\"\"!\"#$%&'()*+,-./:;<=>@[\\]^_`{|}~\"\"\"), ' ', resumeText)  # remove punctuations\n    resumeText = re.sub(r'[^\\x00-\\x7f]',r' ', resumeText) # remove non-ascii characters\n    resumeText = re.sub('\\s+', ' ', resumeText)  # remove extra whitespace\n    resumeText = resumeText.lower() \n    return resumeText","metadata":{"execution":{"iopub.status.busy":"2022-02-14T15:58:48.654129Z","iopub.execute_input":"2022-02-14T15:58:48.654519Z","iopub.status.idle":"2022-02-14T15:58:48.661403Z","shell.execute_reply.started":"2022-02-14T15:58:48.654486Z","shell.execute_reply":"2022-02-14T15:58:48.660640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# One Hot Encoding ","metadata":{}},{"cell_type":"code","source":"all_questns = [cleanText(sent) for sent in train_df.question_text.to_list()] # a list of all the quetions in the train data\nall_questns[:5]","metadata":{"execution":{"iopub.status.busy":"2022-02-14T15:58:48.664309Z","iopub.execute_input":"2022-02-14T15:58:48.664526Z","iopub.status.idle":"2022-02-14T15:58:49.921038Z","shell.execute_reply.started":"2022-02-14T15:58:48.664503Z","shell.execute_reply":"2022-02-14T15:58:49.920335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow\nfrom tensorflow.keras.preprocessing.text import one_hot\n#vocab \nmax_vocab = 25000\n#one hot representation\n\nonehot_que = [one_hot(questns,max_vocab) for questns in all_questns]","metadata":{"execution":{"iopub.status.busy":"2022-02-14T15:58:49.923218Z","iopub.execute_input":"2022-02-14T15:58:49.923652Z","iopub.status.idle":"2022-02-14T15:58:55.416552Z","shell.execute_reply.started":"2022-02-14T15:58:49.923600Z","shell.execute_reply":"2022-02-14T15:58:55.415734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Take a look at the emmbedings","metadata":{}},{"cell_type":"code","source":"pd.DataFrame({'Words' : [word for word in all_questns[0].split()], 'Encoding' : onehot_que[0]})","metadata":{"execution":{"iopub.status.busy":"2022-02-14T15:58:55.417978Z","iopub.execute_input":"2022-02-14T15:58:55.418222Z","iopub.status.idle":"2022-02-14T15:58:55.430493Z","shell.execute_reply.started":"2022-02-14T15:58:55.418188Z","shell.execute_reply":"2022-02-14T15:58:55.429266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Word Embedding","metadata":{}},{"cell_type":"markdown","source":"Before we embed, we need some parameters defined, \n1. Maximum length of the sentence\n2. Number of features to use in embedding","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.sequence import pad_sequences\n\nmax_len = max([len(sent.split()) for sent in all_questns])\nmax_len \n\n#padding the setences \npadded_que = pad_sequences(onehot_que,padding='post',maxlen=max_len)\npadded_que","metadata":{"execution":{"iopub.status.busy":"2022-02-14T15:58:55.431834Z","iopub.execute_input":"2022-02-14T15:58:55.432088Z","iopub.status.idle":"2022-02-14T15:58:55.812692Z","shell.execute_reply.started":"2022-02-14T15:58:55.432053Z","shell.execute_reply":"2022-02-14T15:58:55.811956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating a Pipeline for data processing\nSo basically all the preprocessing steps that we did in the above cells, we'll create a function just for that.","metadata":{}},{"cell_type":"code","source":"def preprocessing(df,y=True , max_vocab=25000 ,max_len=64, ): \n    sentences = df.question_text.to_list()\n    sentences = [cleanText(sent) for sent in sentences]\n    max_vocab = 25000\n#one hot representation\n    X = [one_hot(questns,max_vocab) for questns in sentences]\n    X = pad_sequences(X, padding='post' , maxlen=max_len)\n    if y :\n        target = np.array(df.target)\n        return X, target\n    else : \n        return X","metadata":{"execution":{"iopub.status.busy":"2022-02-14T15:58:55.814038Z","iopub.execute_input":"2022-02-14T15:58:55.814283Z","iopub.status.idle":"2022-02-14T15:58:55.822039Z","shell.execute_reply.started":"2022-02-14T15:58:55.814249Z","shell.execute_reply":"2022-02-14T15:58:55.821231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparing The Features and Targets","metadata":{}},{"cell_type":"code","source":"test_inputs = preprocessing(test,y=False)\ntest_inputs.shape","metadata":{"execution":{"iopub.status.busy":"2022-02-14T16:41:12.504977Z","iopub.execute_input":"2022-02-14T16:41:12.505521Z","iopub.status.idle":"2022-02-14T16:41:28.421042Z","shell.execute_reply.started":"2022-02-14T16:41:12.505483Z","shell.execute_reply":"2022-02-14T16:41:28.420365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train , y_train = preprocessing(train_df)\nX_val , y_val = preprocessing(val_df)\ndim=10\nX_train.shape , X_val.shape ,y_train.shape ,y_val.shape","metadata":{"execution":{"iopub.status.busy":"2022-02-14T15:58:55.823572Z","iopub.execute_input":"2022-02-14T15:58:55.824025Z","iopub.status.idle":"2022-02-14T15:58:59.098010Z","shell.execute_reply.started":"2022-02-14T15:58:55.823987Z","shell.execute_reply":"2022-02-14T15:58:59.097363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import Embedding,LSTM, Dense,Dropout\nfrom tensorflow.keras.models import Sequential\n\nmodel = Sequential()\nmodel.add(Embedding(max_vocab,dim, input_length=max_len))\nmodel.add(Dense(512 , activation='relu'))\n# model.add(Dropout(0.2))\n# model.add(Dense(512 , activation='relu'))\n# #model.add(Dropout(0.2))\n# model.add(Dense(512 , activation='relu'))\nmodel.add(Dense(512 , activation='relu'))\nmodel.add(Dense(256 , activation='relu'))\nmodel.add(Dense(512 , activation='relu'))\nmodel.add(Dense(1,activation='sigmoid'))\nmodel.compile('adam',['binary_crossentropy'])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-02-14T12:05:29.046294Z","iopub.execute_input":"2022-02-14T12:05:29.046556Z","iopub.status.idle":"2022-02-14T12:05:31.407941Z","shell.execute_reply.started":"2022-02-14T12:05:29.046523Z","shell.execute_reply":"2022-02-14T12:05:31.407189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nhistory = model.fit(X_train,y_train,validation_data=(X_val,y_val) , epochs=10,batch_size=150)","metadata":{"execution":{"iopub.status.busy":"2022-02-14T12:05:31.409455Z","iopub.execute_input":"2022-02-14T12:05:31.409719Z","iopub.status.idle":"2022-02-14T12:06:17.413971Z","shell.execute_reply.started":"2022-02-14T12:05:31.409685Z","shell.execute_reply":"2022-02-14T12:06:17.413231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history.history","metadata":{"execution":{"iopub.status.busy":"2022-02-14T12:06:17.415482Z","iopub.execute_input":"2022-02-14T12:06:17.416292Z","iopub.status.idle":"2022-02-14T12:06:17.423637Z","shell.execute_reply.started":"2022-02-14T12:06:17.416252Z","shell.execute_reply":"2022-02-14T12:06:17.422849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_df = pd.DataFrame(history.history)\nhistory_df.loc[: , ['loss' , 'val_loss']].plot()","metadata":{"execution":{"iopub.status.busy":"2022-02-14T12:06:17.424940Z","iopub.execute_input":"2022-02-14T12:06:17.425208Z","iopub.status.idle":"2022-02-14T12:06:17.655739Z","shell.execute_reply.started":"2022-02-14T12:06:17.425174Z","shell.execute_reply":"2022-02-14T12:06:17.655084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \nfrom sklearn.linear_model import LogisticRegression\nmodel1 = {\n    'LogisticRegression' : {\n        'model' : LogisticRegression(solver='liblinear' ),\n        'params': {\n            'C': [1,5,10,50]\n        }\n    }\n}\n\nfrom sklearn.model_selection import GridSearchCV\nimport pandas as pd\nscores = []\n\nfor model_name, mp in model1.items():\n    clf =  GridSearchCV(mp['model'], mp['params'], cv=4, return_train_score=False)\n    clf.fit(X_train, y_train)\n    scores.append({\n        'model': model_name,\n        'best_score': clf.best_score_,\n        'best_params': clf.best_params_\n    })\n    \ndf = pd.DataFrame(scores,columns=['model','best_score','best_params'])\ndf","metadata":{"execution":{"iopub.status.busy":"2022-02-14T12:07:31.941297Z","iopub.execute_input":"2022-02-14T12:07:31.941596Z","iopub.status.idle":"2022-02-14T12:07:53.358184Z","shell.execute_reply.started":"2022-02-14T12:07:31.941551Z","shell.execute_reply":"2022-02-14T12:07:53.357477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.ensemble.forest import RandomForestClassifier\nparams = [100,200,500,1000]\n\nscore = []\nfor estimator in params : \n    random_forest = RandomForestClassifier(n_estimators=estimator,max_depth=50 , random_state=1)\n    random_forest.fit(X_train, y_train)\n    train_acc = round(random_forest.score(X_train, y_train) * 100, 2)\n    val_acc = round(random_forest.score(X_val, y_val) * 100, 2)\n    score.append({\n        'n_estimator': estimator,\n        'train_accuracy' : train_acc ,\n        'val_accuracy' : val_acc\n    })\n","metadata":{"execution":{"iopub.status.busy":"2022-02-14T17:09:22.735776Z","iopub.execute_input":"2022-02-14T17:09:22.736237Z","iopub.status.idle":"2022-02-14T17:13:44.457387Z","shell.execute_reply.started":"2022-02-14T17:09:22.736190Z","shell.execute_reply":"2022-02-14T17:13:44.455671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"randomFroest_df = pd.DataFrame(score , columns=['n_estimator' , 'train_accuracy','val_accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-02-14T17:13:44.458962Z","iopub.execute_input":"2022-02-14T17:13:44.459279Z","iopub.status.idle":"2022-02-14T17:13:44.464827Z","shell.execute_reply.started":"2022-02-14T17:13:44.459239Z","shell.execute_reply":"2022-02-14T17:13:44.463941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred = random_forest.predict(test_inputs)\nY_pred","metadata":{"execution":{"iopub.status.busy":"2022-02-14T17:13:44.466208Z","iopub.execute_input":"2022-02-14T17:13:44.466479Z","iopub.status.idle":"2022-02-14T17:15:12.410035Z","shell.execute_reply.started":"2022-02-14T17:13:44.466426Z","shell.execute_reply":"2022-02-14T17:15:12.409305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'qid': test.index,'prediction': Y_pred} )\nsubmission.head()\nsubmission.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-14T17:15:12.411571Z","iopub.execute_input":"2022-02-14T17:15:12.411766Z","iopub.status.idle":"2022-02-14T17:15:13.143523Z","shell.execute_reply.started":"2022-02-14T17:15:12.411741Z","shell.execute_reply":"2022-02-14T17:15:13.142669Z"},"trusted":true},"execution_count":null,"outputs":[]}]}