{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-05-25T05:07:54.581484Z","iopub.execute_input":"2021-05-25T05:07:54.581912Z","iopub.status.idle":"2021-05-25T05:07:54.589896Z","shell.execute_reply.started":"2021-05-25T05:07:54.581873Z","shell.execute_reply":"2021-05-25T05:07:54.589004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom xgboost import XGBClassifier\n\ndf_test=pd.read_csv(\"../input/quora-insincere-questions-classification/test.csv\",sep=',')\ndf_train=pd.read_csv(\"../input/quora-insincere-questions-classification/train.csv\",sep=',')\n\nimport re\nimport nltk\nnltk.download('stopwords')\nfrom nltk.corpus import stopwords\nfrom nltk.stem import WordNetLemmatizer\nfrom nltk.stem.porter import PorterStemmer\nps= PorterStemmer()","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:07:54.734064Z","iopub.execute_input":"2021-05-25T05:07:54.734725Z","iopub.status.idle":"2021-05-25T05:08:02.608312Z","shell.execute_reply.started":"2021-05-25T05:07:54.734682Z","shell.execute_reply":"2021-05-25T05:08:02.607343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head(5)\n","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:02.609955Z","iopub.execute_input":"2021-05-25T05:08:02.610478Z","iopub.status.idle":"2021-05-25T05:08:02.638401Z","shell.execute_reply.started":"2021-05-25T05:08:02.610442Z","shell.execute_reply":"2021-05-25T05:08:02.636999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:02.640642Z","iopub.execute_input":"2021-05-25T05:08:02.640991Z","iopub.status.idle":"2021-05-25T05:08:02.652752Z","shell.execute_reply.started":"2021-05-25T05:08:02.640959Z","shell.execute_reply":"2021-05-25T05:08:02.651289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:02.654796Z","iopub.execute_input":"2021-05-25T05:08:02.655167Z","iopub.status.idle":"2021-05-25T05:08:02.920858Z","shell.execute_reply.started":"2021-05-25T05:08:02.655132Z","shell.execute_reply":"2021-05-25T05:08:02.919556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:02.922035Z","iopub.execute_input":"2021-05-25T05:08:02.922331Z","iopub.status.idle":"2021-05-25T05:08:03.004005Z","shell.execute_reply.started":"2021-05-25T05:08:02.922301Z","shell.execute_reply":"2021-05-25T05:08:03.002978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.target.value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:03.005196Z","iopub.execute_input":"2021-05-25T05:08:03.0057Z","iopub.status.idle":"2021-05-25T05:08:03.032937Z","shell.execute_reply.started":"2021-05-25T05:08:03.005637Z","shell.execute_reply":"2021-05-25T05:08:03.031629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"g=sns.countplot(x='target', data=df_train)\ng.set_xticklabels(['Sincere', 'Insincere'])","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:03.034734Z","iopub.execute_input":"2021-05-25T05:08:03.035222Z","iopub.status.idle":"2021-05-25T05:08:03.334409Z","shell.execute_reply.started":"2021-05-25T05:08:03.035176Z","shell.execute_reply":"2021-05-25T05:08:03.332991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.corr()","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:03.33618Z","iopub.execute_input":"2021-05-25T05:08:03.3365Z","iopub.status.idle":"2021-05-25T05:08:03.367991Z","shell.execute_reply.started":"2021-05-25T05:08:03.336471Z","shell.execute_reply":"2021-05-25T05:08:03.366789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corpous_train=[]\nfor i in range(0,50):\n    review1 = re.sub('[^a-zA-Z]',' ',df_train['question_text'][i])\n    review1 = review1.lower()\n    review1 = review1.split()\n    review1 = [ps.stem(word) for word in review1 if not word in stopwords.words('english')]\n    review1 =' '.join(review1)\n    corpous_train.append(review1)","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:03.370932Z","iopub.execute_input":"2021-05-25T05:08:03.371245Z","iopub.status.idle":"2021-05-25T05:08:03.504184Z","shell.execute_reply.started":"2021-05-25T05:08:03.371207Z","shell.execute_reply":"2021-05-25T05:08:03.502867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corpous_train","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:03.506195Z","iopub.execute_input":"2021-05-25T05:08:03.506566Z","iopub.status.idle":"2021-05-25T05:08:03.51503Z","shell.execute_reply.started":"2021-05-25T05:08:03.50653Z","shell.execute_reply":"2021-05-25T05:08:03.513779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corpous_test=[]\nfor i in range(0,50):\n    review = re.sub('[^a-zA-Z]',' ',df_test['question_text'][i])\n    review = review.lower()\n    review = review.split()\n    review = [ps.stem(word) for word in review if not word in stopwords.words('english')]\n    review =' '.join(review)\n    corpous_test.append(review)","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:03.516746Z","iopub.execute_input":"2021-05-25T05:08:03.517119Z","iopub.status.idle":"2021-05-25T05:08:03.622744Z","shell.execute_reply.started":"2021-05-25T05:08:03.517084Z","shell.execute_reply":"2021-05-25T05:08:03.621558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m=pd.DataFrame(corpous_train)\nm","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:25:09.457769Z","iopub.execute_input":"2021-05-25T05:25:09.458175Z","iopub.status.idle":"2021-05-25T05:25:09.476765Z","shell.execute_reply.started":"2021-05-25T05:25:09.458135Z","shell.execute_reply":"2021-05-25T05:25:09.475469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"k=pd.DataFrame(corpous_test)\n","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:26:12.640192Z","iopub.execute_input":"2021-05-25T05:26:12.64085Z","iopub.status.idle":"2021-05-25T05:26:12.647057Z","shell.execute_reply.started":"2021-05-25T05:26:12.640805Z","shell.execute_reply":"2021-05-25T05:26:12.645448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_text = pd.concat([m,k])\nall_text","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:26:50.544892Z","iopub.execute_input":"2021-05-25T05:26:50.545303Z","iopub.status.idle":"2021-05-25T05:26:50.560527Z","shell.execute_reply.started":"2021-05-25T05:26:50.545268Z","shell.execute_reply":"2021-05-25T05:26:50.559388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\ncv=CountVectorizer(max_features=5000)\nx_train=cv.fit_transform(corpous_train).toarray()\n\nprint(x_train)","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:03.632177Z","iopub.execute_input":"2021-05-25T05:08:03.632458Z","iopub.status.idle":"2021-05-25T05:08:03.6515Z","shell.execute_reply.started":"2021-05-25T05:08:03.63243Z","shell.execute_reply":"2021-05-25T05:08:03.65072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\ntf_idf= TfidfVectorizer(stop_words=\"english\",strip_accents='ascii',max_features=300)\ntf_idf_matrix = tf_idf.fit_transform(all_text)\nprint(tf_idf_matrix[:4])","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:30:45.109727Z","iopub.execute_input":"2021-05-25T05:30:45.110455Z","iopub.status.idle":"2021-05-25T05:30:45.146665Z","shell.execute_reply.started":"2021-05-25T05:30:45.110401Z","shell.execute_reply":"2021-05-25T05:30:45.144682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf_idf_matrix","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:03.673319Z","iopub.execute_input":"2021-05-25T05:08:03.673827Z","iopub.status.idle":"2021-05-25T05:08:03.682431Z","shell.execute_reply.started":"2021-05-25T05:08:03.673777Z","shell.execute_reply":"2021-05-25T05:08:03.680919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#data_extra_features = pd.concat([df_train,pd.DataFrame(tf_idf_matrix,columns=tf_idf.get_feature_names())],axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:03.684834Z","iopub.execute_input":"2021-05-25T05:08:03.685353Z","iopub.status.idle":"2021-05-25T05:08:03.693473Z","shell.execute_reply.started":"2021-05-25T05:08:03.685297Z","shell.execute_reply":"2021-05-25T05:08:03.692058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#data_extra_features","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:03.695429Z","iopub.execute_input":"2021-05-25T05:08:03.695954Z","iopub.status.idle":"2021-05-25T05:08:03.707023Z","shell.execute_reply.started":"2021-05-25T05:08:03.695907Z","shell.execute_reply":"2021-05-25T05:08:03.705527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X=data_extra_features\n#features = X.columns.drop([\"Value\",\"Text\",\"Value_num\"])\n#target = [\"Value\"]\n#X_train,X_test,y_train,y_test = train_test_split(X[features],X[target])","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:03.709273Z","iopub.execute_input":"2021-05-25T05:08:03.709829Z","iopub.status.idle":"2021-05-25T05:08:03.719184Z","shell.execute_reply.started":"2021-05-25T05:08:03.709759Z","shell.execute_reply":"2021-05-25T05:08:03.717862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#feature_names = cv.get_feature_names() \n#train_set = [] \n#for i, single_sample in enumerate(x_train): \n  #  single_feature_dict = {} \n  #  for j, single_feature in enumerate(single_sample): \n  #      single_feature_dict[feature_names[j]]=single_feature \n #   train_set.append((single_feature_dict, y[i]))","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:03.721246Z","iopub.execute_input":"2021-05-25T05:08:03.721865Z","iopub.status.idle":"2021-05-25T05:08:03.731469Z","shell.execute_reply.started":"2021-05-25T05:08:03.721818Z","shell.execute_reply":"2021-05-25T05:08:03.730587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y=pd.get_dummies(df_train['target'])\ny_train=y.iloc[:50,1].values\ny_train","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:03.732551Z","iopub.execute_input":"2021-05-25T05:08:03.733067Z","iopub.status.idle":"2021-05-25T05:08:03.772861Z","shell.execute_reply.started":"2021-05-25T05:08:03.733034Z","shell.execute_reply":"2021-05-25T05:08:03.771491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv2=CountVectorizer(max_features=5000)\nx_test=cv2.fit_transform(corpous_test).toarray()\nx_test","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:03.774748Z","iopub.execute_input":"2021-05-25T05:08:03.775212Z","iopub.status.idle":"2021-05-25T05:08:03.786173Z","shell.execute_reply.started":"2021-05-25T05:08:03.775167Z","shell.execute_reply":"2021-05-25T05:08:03.785068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmodel = XGBClassifier(objective=\"binary:logistic\")\nmodel.fit(x_train,y_train)\ny_test=model.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:03.787976Z","iopub.execute_input":"2021-05-25T05:08:03.788354Z","iopub.status.idle":"2021-05-25T05:08:03.944065Z","shell.execute_reply.started":"2021-05-25T05:08:03.788318Z","shell.execute_reply":"2021-05-25T05:08:03.94199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_model = XGBClassifier(objective=\"binary:logistic\")\nmy_model.fit(x_train,y_train,early_stopping_rounds=5,verbose=False,eval_set=x_test)\n\ny_test=my_model.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2021-05-25T05:08:03.945407Z","iopub.status.idle":"2021-05-25T05:08:03.946295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}