{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.041004,"end_time":"2021-04-19T02:16:46.093875","exception":false,"start_time":"2021-04-19T02:16:46.052871","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-22T08:50:59.377175Z","iopub.execute_input":"2021-05-22T08:50:59.377856Z","iopub.status.idle":"2021-05-22T08:50:59.390539Z","shell.execute_reply.started":"2021-05-22T08:50:59.377822Z","shell.execute_reply":"2021-05-22T08:50:59.389846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"papermill":{"duration":1.13049,"end_time":"2021-04-19T02:16:47.248348","exception":false,"start_time":"2021-04-19T02:16:46.117858","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-22T08:50:59.391964Z","iopub.execute_input":"2021-05-22T08:50:59.392421Z","iopub.status.idle":"2021-05-22T08:50:59.398113Z","shell.execute_reply.started":"2021-05-22T08:50:59.392393Z","shell.execute_reply":"2021-05-22T08:50:59.397548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')","metadata":{"papermill":{"duration":5.731754,"end_time":"2021-04-19T02:16:53.002993","exception":false,"start_time":"2021-04-19T02:16:47.271239","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-22T08:50:59.399504Z","iopub.execute_input":"2021-05-22T08:50:59.399972Z","iopub.status.idle":"2021-05-22T08:51:01.708605Z","shell.execute_reply.started":"2021-05-22T08:50:59.399944Z","shell.execute_reply":"2021-05-22T08:51:01.707557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nfrom nltk.stem import WordNetLemmatizer\nfrom nltk.stem import SnowballStemmer\n\nstop_words = stopwords.words('english')\nstop_words.remove('not')\nlemmatizer = WordNetLemmatizer()","metadata":{"execution":{"iopub.status.busy":"2021-05-22T08:51:01.710345Z","iopub.execute_input":"2021-05-22T08:51:01.710911Z","iopub.status.idle":"2021-05-22T08:51:01.716295Z","shell.execute_reply.started":"2021-05-22T08:51:01.710871Z","shell.execute_reply":"2021-05-22T08:51:01.715389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"contraction_dict = {\"dont\": \"do not\", \"aint\": \"is not\", \"isnt\": \"is not\", \"doesnt\": \"does not\", \"cant\": \"cannot\", \"mustnt\": \"must not\", \"hasnt\": \"has not\", \"havent\": \"have not\", \"arent\": \"are not\", \"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"‘cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\", \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\", \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"Iam\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\", \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\", \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\"}\n\n# if contraction_dict.has_key('aya'):\n#     print('yaya')\n\ndef replace_contractions(question):\n    return [contraction_dict.get(token) if (contraction_dict.get(token) != None) else token for token in question]","metadata":{"execution":{"iopub.status.busy":"2021-05-22T08:51:01.717454Z","iopub.execute_input":"2021-05-22T08:51:01.717804Z","iopub.status.idle":"2021-05-22T08:51:01.732140Z","shell.execute_reply.started":"2021-05-22T08:51:01.717768Z","shell.execute_reply":"2021-05-22T08:51:01.731557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def qes_preprocessing(qes):\n    # Data cleaning:\n    qes = re.sub(re.compile('<.*?>'), '', qes)\n    qes = re.sub('[^A-Za-z0-9]+', ' ', qes)\n\n    # Lowercase:\n    qes = qes.lower()\n\n    # Tokenization:\n    tokens = word_tokenize(qes)\n\n    # Contractions replacement:\n    tokens = [contraction_dict.get(token) if (contraction_dict.get(token) != None) else token for token in tokens]\n\n    # Stop words removal:\n    tokens = [w for w in tokens if w not in stop_words]\n    \n    # Lemmatization:\n    tokens = [lemmatizer.lemmatize(w) for w in tokens]\n\n    # Join words after preprocessed:\n    qes = ' '.join(tokens) \n\n    return qes","metadata":{"execution":{"iopub.status.busy":"2021-05-22T08:51:01.733031Z","iopub.execute_input":"2021-05-22T08:51:01.733399Z","iopub.status.idle":"2021-05-22T08:51:01.747745Z","shell.execute_reply.started":"2021-05-22T08:51:01.733341Z","shell.execute_reply":"2021-05-22T08:51:01.746929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\n\ntqdm.pandas()\ntrain_df['preprocessed_questions'] = train_df['question_text'].progress_apply(qes_preprocessing)","metadata":{"papermill":{"duration":0.05742,"end_time":"2021-04-19T02:16:53.085408","exception":false,"start_time":"2021-04-19T02:16:53.027988","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-22T08:51:01.748902Z","iopub.execute_input":"2021-05-22T08:51:01.749431Z","iopub.status.idle":"2021-05-22T08:56:12.141680Z","shell.execute_reply.started":"2021-05-22T08:51:01.749393Z","shell.execute_reply":"2021-05-22T08:56:12.140817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"papermill":{"duration":0.313762,"end_time":"2021-04-19T02:16:53.422671","exception":false,"start_time":"2021-04-19T02:16:53.108909","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-22T08:56:12.143797Z","iopub.execute_input":"2021-05-22T08:56:12.144041Z","iopub.status.idle":"2021-05-22T08:56:12.492910Z","shell.execute_reply.started":"2021-05-22T08:56:12.144017Z","shell.execute_reply":"2021-05-22T08:56:12.491795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"insincere = train_df[train_df['target'] == 0]\nsincere = train_df[train_df['target'] == 1]","metadata":{"execution":{"iopub.status.busy":"2021-05-22T08:56:12.494990Z","iopub.execute_input":"2021-05-22T08:56:12.495390Z","iopub.status.idle":"2021-05-22T08:56:12.792757Z","shell.execute_reply.started":"2021-05-22T08:56:12.495328Z","shell.execute_reply":"2021-05-22T08:56:12.791964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(sincere)","metadata":{"execution":{"iopub.status.busy":"2021-05-22T08:56:12.793843Z","iopub.execute_input":"2021-05-22T08:56:12.794085Z","iopub.status.idle":"2021-05-22T08:56:12.799060Z","shell.execute_reply.started":"2021-05-22T08:56:12.794061Z","shell.execute_reply":"2021-05-22T08:56:12.798275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"insincere_batch = insincere[:len(sincere)]","metadata":{"execution":{"iopub.status.busy":"2021-05-22T08:56:12.800182Z","iopub.execute_input":"2021-05-22T08:56:12.800473Z","iopub.status.idle":"2021-05-22T08:56:12.810220Z","shell.execute_reply.started":"2021-05-22T08:56:12.800448Z","shell.execute_reply":"2021-05-22T08:56:12.809186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"insincere_batch.shape","metadata":{"execution":{"iopub.status.busy":"2021-05-22T08:59:16.237137Z","iopub.execute_input":"2021-05-22T08:59:16.237652Z","iopub.status.idle":"2021-05-22T08:59:16.243429Z","shell.execute_reply.started":"2021-05-22T08:59:16.237611Z","shell.execute_reply":"2021-05-22T08:59:16.242510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"insincere_batch.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-22T08:56:12.811199Z","iopub.execute_input":"2021-05-22T08:56:12.811440Z","iopub.status.idle":"2021-05-22T08:56:12.827011Z","shell.execute_reply.started":"2021-05-22T08:56:12.811417Z","shell.execute_reply":"2021-05-22T08:56:12.826209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_batch = pd.concat([insincere_batch, sincere])","metadata":{"execution":{"iopub.status.busy":"2021-05-22T08:56:12.828146Z","iopub.execute_input":"2021-05-22T08:56:12.828602Z","iopub.status.idle":"2021-05-22T08:56:12.877157Z","shell.execute_reply.started":"2021-05-22T08:56:12.828575Z","shell.execute_reply":"2021-05-22T08:56:12.876523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_batch","metadata":{"execution":{"iopub.status.busy":"2021-05-22T08:56:12.878076Z","iopub.execute_input":"2021-05-22T08:56:12.878473Z","iopub.status.idle":"2021-05-22T08:56:12.892303Z","shell.execute_reply.started":"2021-05-22T08:56:12.878435Z","shell.execute_reply":"2021-05-22T08:56:12.891710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.linear_model import LogisticRegression\n\npipeline = Pipeline([(\"cv\", CountVectorizer(analyzer=\"word\", ngram_range=(1,4), max_df=0.9)),\n                     (\"clf\", LogisticRegression(solver=\"saga\", class_weight=\"balanced\", C=0.45, max_iter=250, verbose=1, n_jobs=-1))])\n\nX_train, X_test, y_train, y_test = train_test_split(test_batch['preprocessed_questions'], test_batch.target, test_size=0.2, stratify = test_batch.target.values)","metadata":{"papermill":{"duration":1.566687,"end_time":"2021-04-19T02:25:50.200671","exception":false,"start_time":"2021-04-19T02:25:48.633984","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-22T08:56:12.893385Z","iopub.execute_input":"2021-05-22T08:56:12.893804Z","iopub.status.idle":"2021-05-22T08:56:13.015618Z","shell.execute_reply.started":"2021-05-22T08:56:12.893778Z","shell.execute_reply":"2021-05-22T08:56:13.014839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(15):\n    print('Batch: ', i)\n    insincere_batch = insincere[:((i + 1) * len(sincere))]\n    test_batch = pd.concat([insincere_batch, sincere])\n    \n    X_train, X_test, y_train, y_test = train_test_split(test_batch['preprocessed_questions'], test_batch.target, test_size=0.2, stratify = test_batch.target.values)\n    \n    lr_model = pipeline.fit(X_train, y_train)\n    \n    get_fscore_matrix(lr_model, 'Linear Regression')","metadata":{"execution":{"iopub.status.busy":"2021-05-22T09:05:56.586294Z","iopub.execute_input":"2021-05-22T09:05:56.586682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lr_model = pipeline.fit(X_train, y_train)","metadata":{"papermill":{"duration":663.849689,"end_time":"2021-04-19T02:36:54.081296","exception":false,"start_time":"2021-04-19T02:25:50.231607","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-22T08:56:13.016707Z","iopub.execute_input":"2021-05-22T08:56:13.017148Z","iopub.status.idle":"2021-05-22T08:57:23.224646Z","shell.execute_reply.started":"2021-05-22T08:56:13.017119Z","shell.execute_reply":"2021-05-22T08:57:23.223678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_fscore_matrix(fitted_clf, model_name):\n    print(model_name, ' :')\n    \n    # get classes predictions for the classification report \n    y_train_pred, y_pred = fitted_clf.predict(X_train), fitted_clf.predict(X_test)\n    print(classification_report(y_test, y_pred), '\\n') # target_names=y\n    \n    # computes probabilities keep the ones for the positive outcome only      \n    print(f'F1-score = {f1_score(y_test, y_pred):.2f}')","metadata":{"papermill":{"duration":0.043077,"end_time":"2021-04-19T02:36:54.156667","exception":false,"start_time":"2021-04-19T02:36:54.11359","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-22T08:57:23.225645Z","iopub.execute_input":"2021-05-22T08:57:23.225903Z","iopub.status.idle":"2021-05-22T08:57:23.230721Z","shell.execute_reply.started":"2021-05-22T08:57:23.225879Z","shell.execute_reply":"2021-05-22T08:57:23.229919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.metrics import f1_score, confusion_matrix, classification_report\n\n# get_fscore_matrix(lr_model, 'Linear Regression')","metadata":{"papermill":{"duration":54.892736,"end_time":"2021-04-19T02:37:49.081554","exception":false,"start_time":"2021-04-19T02:36:54.188818","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-22T08:57:23.232173Z","iopub.execute_input":"2021-05-22T08:57:23.232577Z","iopub.status.idle":"2021-05-22T08:57:29.607300Z","shell.execute_reply.started":"2021-05-22T08:57:23.232541Z","shell.execute_reply":"2021-05-22T08:57:29.606260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/test.csv')","metadata":{"papermill":{"duration":1.685601,"end_time":"2021-04-19T02:37:50.800027","exception":false,"start_time":"2021-04-19T02:37:49.114426","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"papermill":{"duration":0.050049,"end_time":"2021-04-19T02:37:50.884359","exception":false,"start_time":"2021-04-19T02:37:50.83431","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-22T08:57:30.266433Z","iopub.execute_input":"2021-05-22T08:57:30.266779Z","iopub.status.idle":"2021-05-22T08:57:30.277036Z","shell.execute_reply.started":"2021-05-22T08:57:30.266744Z","shell.execute_reply":"2021-05-22T08:57:30.276161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['preprocessed'] = test_df['question_text'].apply(qes_preprocessing)","metadata":{"papermill":{"duration":145.458842,"end_time":"2021-04-19T02:40:16.377521","exception":false,"start_time":"2021-04-19T02:37:50.918679","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-22T08:57:30.278491Z","iopub.execute_input":"2021-05-22T08:57:30.278834Z","iopub.status.idle":"2021-05-22T08:58:56.908922Z","shell.execute_reply.started":"2021-05-22T08:57:30.278800Z","shell.execute_reply":"2021-05-22T08:58:56.907955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = lr_model.predict(test_df['preprocessed'])","metadata":{"papermill":{"duration":14.238933,"end_time":"2021-04-19T02:40:30.651523","exception":false,"start_time":"2021-04-19T02:40:16.41259","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(predictions))","metadata":{"papermill":{"duration":0.043305,"end_time":"2021-04-19T02:40:30.729722","exception":false,"start_time":"2021-04-19T02:40:30.686417","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['prediction'] = predictions\nresults = test_df[['qid', 'prediction']]\nresults.to_csv('submission.csv', index=False)\nresults.shape","metadata":{"papermill":{"duration":1.035671,"end_time":"2021-04-19T02:40:31.800683","exception":false,"start_time":"2021-04-19T02:40:30.765012","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{"papermill":{"duration":0.035117,"end_time":"2021-04-19T02:40:31.871296","exception":false,"start_time":"2021-04-19T02:40:31.836179","status":"completed"},"tags":[]}}]}