{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-12T05:18:20.053316Z","iopub.execute_input":"2022-08-12T05:18:20.053819Z","iopub.status.idle":"2022-08-12T05:18:20.086139Z","shell.execute_reply.started":"2022-08-12T05:18:20.053716Z","shell.execute_reply":"2022-08-12T05:18:20.084959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train=pd.read_csv('/kaggle/input/nlp-getting-started/train.csv')\ndf_test=pd.read_csv('/kaggle/input/nlp-getting-started/test.csv')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T05:21:53.705653Z","iopub.execute_input":"2022-08-12T05:21:53.706819Z","iopub.status.idle":"2022-08-12T05:21:53.774011Z","shell.execute_reply.started":"2022-08-12T05:21:53.706780Z","shell.execute_reply":"2022-08-12T05:21:53.773063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T05:22:01.030569Z","iopub.execute_input":"2022-08-12T05:22:01.031054Z","iopub.status.idle":"2022-08-12T05:22:01.050339Z","shell.execute_reply.started":"2022-08-12T05:22:01.031021Z","shell.execute_reply":"2022-08-12T05:22:01.049408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-12T05:28:16.496725Z","iopub.execute_input":"2022-08-12T05:28:16.497720Z","iopub.status.idle":"2022-08-12T05:28:16.506465Z","shell.execute_reply.started":"2022-08-12T05:28:16.497648Z","shell.execute_reply":"2022-08-12T05:28:16.505169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-12T05:28:24.889850Z","iopub.execute_input":"2022-08-12T05:28:24.891004Z","iopub.status.idle":"2022-08-12T05:28:24.898589Z","shell.execute_reply.started":"2022-08-12T05:28:24.890951Z","shell.execute_reply":"2022-08-12T05:28:24.897646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train.columns)\nprint(df_test.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T05:28:50.704934Z","iopub.execute_input":"2022-08-12T05:28:50.706259Z","iopub.status.idle":"2022-08-12T05:28:50.713331Z","shell.execute_reply.started":"2022-08-12T05:28:50.706207Z","shell.execute_reply":"2022-08-12T05:28:50.712120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_colwidth', None) \nx=df_train[df_train['target']==1]\nx['text'][0:15]","metadata":{"execution":{"iopub.status.busy":"2022-08-12T05:54:35.439378Z","iopub.execute_input":"2022-08-12T05:54:35.439807Z","iopub.status.idle":"2022-08-12T05:54:35.451678Z","shell.execute_reply.started":"2022-08-12T05:54:35.439769Z","shell.execute_reply":"2022-08-12T05:54:35.450748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\ndef review_to_words(raw_review):\n    # 2. Remove all other letters apart from ASCII English letter ( basically remove all digits, puncuations etc.)\n    letters_only = re.sub('[^a-zA-Z]', ' ', raw_review)\n    # 3. lower letters\n    words = letters_only.lower()\n    return(words)\n\ndf_train['text_clean']=df_train['text'].apply(review_to_words)\n  \n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T05:45:49.197108Z","iopub.execute_input":"2022-08-12T05:45:49.197498Z","iopub.status.idle":"2022-08-12T05:45:49.271049Z","shell.execute_reply.started":"2022-08-12T05:45:49.197465Z","shell.execute_reply":"2022-08-12T05:45:49.270115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[['text','text_clean']][:15]","metadata":{"execution":{"iopub.status.busy":"2022-08-12T05:54:07.132815Z","iopub.execute_input":"2022-08-12T05:54:07.133234Z","iopub.status.idle":"2022-08-12T05:54:07.146830Z","shell.execute_reply.started":"2022-08-12T05:54:07.133202Z","shell.execute_reply":"2022-08-12T05:54:07.145616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import feature_extraction, linear_model, model_selection, preprocessing #import train_test_split\nfrom sklearn.feature_extraction.text import TfidfVectorizer #import TfidfVectorizer \nfrom sklearn.metrics import confusion_matrix #import confusion_matrix\nfrom sklearn.naive_bayes import MultinomialNB #import MultinomialNB\nfrom sklearn.ensemble import RandomForestClassifier  #import RandomForestClassifier\nfrom sklearn.linear_model import LogisticRegression, RidgeClassifier","metadata":{"execution":{"iopub.status.busy":"2022-08-12T06:23:37.550869Z","iopub.execute_input":"2022-08-12T06:23:37.551748Z","iopub.status.idle":"2022-08-12T06:23:37.557285Z","shell.execute_reply.started":"2022-08-12T06:23:37.551702Z","shell.execute_reply":"2022-08-12T06:23:37.556314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorizer = TfidfVectorizer()\nreviews_corpus = vectorizer.fit_transform(df_train.text_clean)\nreviews_corpus.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-12T06:13:05.422736Z","iopub.execute_input":"2022-08-12T06:13:05.423308Z","iopub.status.idle":"2022-08-12T06:13:05.717976Z","shell.execute_reply.started":"2022-08-12T06:13:05.423260Z","shell.execute_reply":"2022-08-12T06:13:05.716787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target=df_train['target']","metadata":{"execution":{"iopub.status.busy":"2022-08-12T06:13:07.574031Z","iopub.execute_input":"2022-08-12T06:13:07.574435Z","iopub.status.idle":"2022-08-12T06:13:07.579383Z","shell.execute_reply.started":"2022-08-12T06:13:07.574402Z","shell.execute_reply":"2022-08-12T06:13:07.578265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf=RandomForestClassifier()\nscores = model_selection.cross_val_score(clf, reviews_corpus,target, cv=3, scoring=\"f1\")\nscores\n\n# pred = clf.predict(X_test)\n\n# print(\"Accuracy: %s\" % str(clf.score(X_test, Y_test)))\n# print(\"Confusion Matrix\")\n# print(confusion_matrix(Y_test, pred))","metadata":{"execution":{"iopub.status.busy":"2022-08-12T06:18:13.668969Z","iopub.execute_input":"2022-08-12T06:18:13.669405Z","iopub.status.idle":"2022-08-12T06:18:50.120460Z","shell.execute_reply.started":"2022-08-12T06:18:13.669367Z","shell.execute_reply":"2022-08-12T06:18:50.119605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf=LogisticRegression()\nscores = model_selection.cross_val_score(clf, reviews_corpus,target, cv=3, scoring=\"f1\")\nscores","metadata":{"execution":{"iopub.status.busy":"2022-08-12T06:19:56.568972Z","iopub.execute_input":"2022-08-12T06:19:56.569357Z","iopub.status.idle":"2022-08-12T06:19:57.270236Z","shell.execute_reply.started":"2022-08-12T06:19:56.569326Z","shell.execute_reply":"2022-08-12T06:19:57.269009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf=RidgeClassifier()\nscores = model_selection.cross_val_score(clf, reviews_corpus,target, cv=3, scoring=\"f1\")\nscores","metadata":{"execution":{"iopub.status.busy":"2022-08-12T06:23:54.056285Z","iopub.execute_input":"2022-08-12T06:23:54.056702Z","iopub.status.idle":"2022-08-12T06:23:54.178780Z","shell.execute_reply.started":"2022-08-12T06:23:54.056656Z","shell.execute_reply":"2022-08-12T06:23:54.177537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us remove the stopwords and also see how this works. ","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}