{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### Experiment Log:\n\n\n| Version | Models Used | CV Score | LB Score | Comments |\n| --- | --- | --- | --- | --- |\n| v1 | LogisticRegression <br> RandomForestClassifier | - | - | Baseline (Errored Out)\n| v2 | LogisticRegression <br> RandomForestClassifier | 0.8898984 <br> 0.8877021 | - | Baseline\n| v3 | LogisticRegression <br> RandomForestClassifier | 0.8898984 <br> 0.8877021 | 0.868 | Baseline\n| v4 | LogisticRegression <br> RandomForestClassifier | 0.7463457 <br> 0.8374394 | 0.784 | Text Preprocessing <br> TF-IDF\n| v5 | LogisticRegression <br> RandomForestClassifier | 0.7534682 <br> 0.7823450 | 0.761 | Text Preprocessing <br> Lemmatization <br> TF-IDF <br> Glove Embeddings\n| v6 | LogisticRegression <br> XGBoost <br> LightGBM | 0.7534682 <br> 0.6215837 <br> 0.5861382 | 0.695 | Text Preprocessing <br> Lemmatization <br> TF-IDF <br> Glove Embeddings\n| v7 | LogisticRegression <br> XGBoost <br> LightGBM | 0.7932787 <br> 0.6045388 <br> 0.5755040 | 0.685 | Text Preprocessing <br> Lemmatization <br> Glove Embeddings\n| v8 | LogisticRegression <br> XGBoost <br> LightGBM | 0.7932841 <br> 0.6041433 <br> 0.5748984 | 0.677 | Text Preprocessing <br> Lemmatization <br> Glove Embeddings <br> New features added\n| v9 | LogisticRegression <br> XGBoost <br> LightGBM | 0.7013439 <br> 0.6039653 <br> 0.5749525 | 0.673 | Text Preprocessing <br> Lemmatization <br> Glove Embeddings <br> Quantile Transformer\n| v10 | LogisticRegression <br> XGBoost <br> LightGBM | 0.7012418 <br> 0.5843301 <br> 0.5770625 | 0.679 | Text Preprocessing <br> Lemmatization <br> Glove Embeddings <br> Quantile Transformer\n| v11 | LogisticRegression <br> XGBoost <br> LightGBM <br> Voting Classifier | Error | - | Text Preprocessing <br> Lemmatization <br> Glove Embeddings <br> Quantile Transformer\n| v13 | LogisticRegression <br> XGBoost <br> LightGBM <br> Voting Classifier | 0.2121148 | 0.681 | Text Preprocessing <br> Lemmatization <br> Glove Embeddings <br> Quantile Transformer\n| v14 | LogisticRegression <br> XGBoost <br> LightGBM | 0.7034141 <br> 0.6002171 <br> 0.5768558 | 0.669 | Text Preprocessing <br> Lemmatization <br> Glove + FastText Embeddings <br> Quantile Transformer\n| v15 | LogisticRegression <br> XGBoost <br> LightGBM | 0.6929070 <br> 0.6058480 <br> 0.5769126 | 0.685 | Sentence-Transformers\n| v16 | XGBoost <br> LightGBM | Error | - | Text Preprocessing (handle OOV words) <br> Lemmatization <br> Glove + FastText Embeddings <br> Quantile Transformer\n| v17 | XGBoost <br> LightGBM | 0.5981064 <br> 0.5738860 | 0.673 | Text Preprocessing (handle OOV words) <br> Lemmatization <br> Glove + FastText Embeddings <br> Quantile Transformer\n| v18 | LogisticRegression <br> XGBoost <br> LightGBM | 0.8130772 <br> 0.6800559 <br> 0.6760178 | 0.672 | Text Preprocessing <br> Lemmatization <br> Glove + FastText Embeddings <br> Quantile Transformer <br> GroupKFold\n| v19 | LogisticRegression <br> XGBoost <br> LightGBM | 0.8135349 <br> 0.6804664 <br> 0.6747677 | - | Text Preprocessing <br> Lemmatization <br> Glove + FastText Embeddings <br> TextBlob to handle OOV tokens <br> Quantile Transformer <br> GroupKFold\n| v21 | LogisticRegression <br> XGBoost <br> LightGBM | 0.7428862 <br> 0.6817771 <br> 0.6779332 | 0.672 | Preprocessing <br> Lemmatization <br> Glove + FastText Embeddings <br> Min Max Scaler <br> Stratified GroupKFold\n| v22 | LogisticRegression <br> XGBoost <br> LightGBM | 0.7022837 <br> 0.5999362 <br> 0.5741513 | 0.669 | Text Preprocessing <br> Lemmatization <br> Glove + FastText Embeddings <br> Quantile Transformer\n| v25 | LogisticRegression <br> XGBoost <br> LightGBM | 0.6991969 <br> 0.5993221 <br> 0.5729388 | 0.666 | Text Preprocessing <br> Glove + FastText Embeddings <br> Quantile Transformer\n| v26 | LogisticRegression <br> XGBoost <br> LightGBM | 0.7008197 <br> 0.6003453 <br> 0.5755974 | 0.667 | Text Preprocessing <br> Glove + Paragram Embeddings <br> Quantile Transformer\n| v27 | LogisticRegression <br> XGBoost <br> LightGBM | - | - | Text Preprocessing <br> Glove + Paragram Embeddings <br> Quantile Transformer <br> New features added","metadata":{"papermill":{"duration":0.0175,"end_time":"2022-06-18T13:54:33.602611","exception":false,"start_time":"2022-06-18T13:54:33.585111","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## Import libraries","metadata":{"papermill":{"duration":0.015336,"end_time":"2022-06-18T13:54:33.633804","exception":false,"start_time":"2022-06-18T13:54:33.618468","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import gc\nimport pickle\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\nimport re\nimport nltk\nimport string\nfrom textblob import TextBlob\nfrom collections import Counter\nfrom nltk.corpus import wordnet\nfrom nltk.corpus import stopwords\nfrom nltk.stem import WordNetLemmatizer\n\nfrom sklearn.metrics import log_loss\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import QuantileTransformer\n\nimport lightgbm as lgb\nfrom xgboost import XGBClassifier\nfrom sklearn.linear_model import LogisticRegression\n\ntqdm.pandas()\nnp.random.seed(42)","metadata":{"papermill":{"duration":2.709802,"end_time":"2022-06-18T13:54:36.359357","exception":false,"start_time":"2022-06-18T13:54:33.649555","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:35:10.919721Z","iopub.execute_input":"2022-07-07T04:35:10.920147Z","iopub.status.idle":"2022-07-07T04:35:14.104648Z","shell.execute_reply.started":"2022-07-07T04:35:10.920065Z","shell.execute_reply":"2022-07-07T04:35:14.103731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load source datasets","metadata":{"papermill":{"duration":0.015443,"end_time":"2022-06-18T13:54:36.390832","exception":false,"start_time":"2022-06-18T13:54:36.375389","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train = pd.read_csv('../input/feedback-prize-effectiveness/train.csv')\ntrain[\"essay_text\"] = train[\"essay_id\"].apply(lambda x: open(f'../input/feedback-prize-effectiveness/train/{x}.txt').read())\nprint(f\"train: {train.shape}\")\ntrain.head()","metadata":{"papermill":{"duration":39.588764,"end_time":"2022-06-18T13:55:15.996151","exception":false,"start_time":"2022-06-18T13:54:36.407387","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:35:14.106400Z","iopub.execute_input":"2022-07-07T04:35:14.106747Z","iopub.status.idle":"2022-07-07T04:35:44.893069Z","shell.execute_reply.started":"2022-07-07T04:35:14.106717Z","shell.execute_reply":"2022-07-07T04:35:44.892203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/feedback-prize-effectiveness/test.csv')\ntest[\"essay_text\"] = test[\"essay_id\"].apply(lambda x: open(f'../input/feedback-prize-effectiveness/test/{x}.txt').read())\nprint(f\"test: {test.shape}\")\ntest.head()","metadata":{"papermill":{"duration":0.052206,"end_time":"2022-06-18T13:55:16.065142","exception":false,"start_time":"2022-06-18T13:55:16.012936","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:35:44.894364Z","iopub.execute_input":"2022-07-07T04:35:44.894679Z","iopub.status.idle":"2022-07-07T04:35:44.925457Z","shell.execute_reply.started":"2022-07-07T04:35:44.894651Z","shell.execute_reply":"2022-07-07T04:35:44.924312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"discourse_effectiveness\"] = train[\"discourse_effectiveness\"].map({\n    \"Adequate\":1,\n    \"Effective\":2,\n    \"Ineffective\":0\n})\ntrain['discourse_effectiveness'].value_counts()","metadata":{"papermill":{"duration":0.041768,"end_time":"2022-06-18T13:55:16.123815","exception":false,"start_time":"2022-06-18T13:55:16.082047","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:35:44.927297Z","iopub.execute_input":"2022-07-07T04:35:44.927641Z","iopub.status.idle":"2022-07-07T04:35:44.950144Z","shell.execute_reply.started":"2022-07-07T04:35:44.927611Z","shell.execute_reply":"2022-07-07T04:35:44.949268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Ytrain = train['discourse_effectiveness'].values\ntrain.drop(['discourse_effectiveness'], inplace=True, axis=1)\n\nprint(f\"train: {train.shape} \\ntest: {test.shape} \\nYtrain: {Ytrain.shape}\")","metadata":{"papermill":{"duration":0.041446,"end_time":"2022-06-18T13:55:16.181993","exception":false,"start_time":"2022-06-18T13:55:16.140547","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:35:44.951203Z","iopub.execute_input":"2022-07-07T04:35:44.951570Z","iopub.status.idle":"2022-07-07T04:35:44.974767Z","shell.execute_reply.started":"2022-07-07T04:35:44.951543Z","shell.execute_reply":"2022-07-07T04:35:44.973255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{"papermill":{"duration":0.016516,"end_time":"2022-06-18T13:55:16.21515","exception":false,"start_time":"2022-06-18T13:55:16.198634","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### Helper Functions","metadata":{"papermill":{"duration":0.016777,"end_time":"2022-06-18T13:55:16.248791","exception":false,"start_time":"2022-06-18T13:55:16.232014","status":"completed"},"tags":[]}},{"cell_type":"code","source":"misspell_mapping = {\n    'studentdesigned': 'student designed',\n    'teacherdesigned': 'teacher designed',\n    'genericname': 'generic name',\n    'winnertakeall': 'winner take all',\n    'studentname': 'student name',\n    'driveless': 'driverless',\n    'teachername': 'teacher name',\n    'propername': 'proper name',\n    'bestlaid': 'best laid',\n    'genericschool': 'generic school',\n    'schoolname': 'school name',\n    'winnertakesall': 'winner take all',\n    'elctoral': 'electoral',\n    'eletoral': 'electoral',\n    'genericcity': 'generic city',\n    'elctors': 'electoral',\n    'venuse': 'venue',\n    'blimplike': 'blimp like',\n    'selfdriving': 'self driving',\n    'electorals': 'electoral',\n    'nearrecord': 'near record',\n    'egyptianstyle': 'egyptian style',\n    'oddnumbered': 'odd numbered',\n    'carintensive': 'car intensive',\n    'elecoral': 'electoral',\n    'oction': 'auction',\n    'electroal': 'electoral',\n    'evennumbered': 'even numbered',\n    'mesalandforms': 'mesa landforms',\n    'electoralvote': 'electoral vote',\n    'relativename': 'relative name',\n    '22euro': 'twenty two euro',\n    'ellectoral': 'electoral',\n    'thirtyplus': 'thirty plus',\n    'collegewon': 'college won',\n    'hisher': 'higher',\n    'teacherbased': 'teacher based',\n    'computeranimated': 'computer animated',\n    'canadidate': 'candidate',\n    'studentbased': 'student based',\n    'gorethanks': 'gore thanks',\n    'clouddraped': 'cloud draped',\n    'edgarsnyder': 'edgar snyder',\n    'emotionrecognition': 'emotion recognition',\n    'landfrom': 'land form',\n    'fivedays': 'five days',\n    'electoal': 'electoral',\n    'lanform': 'land form',\n    'electral': 'electoral',\n    'presidentbut': 'president but',\n    'teacherassigned': 'teacher assigned',\n    'beacuas': 'because',\n    'positionestimating': 'position estimating',\n    'selfeducation': 'self education',\n    'diverless': 'driverless',\n    'computerdriven': 'computer driven',\n    'outofcontrol': 'out of control',\n    'faultthe': 'fault the',\n    'unfairoutdated': 'unfair outdated',\n    'aviods': 'avoid',\n    'momdad': 'mom dad',\n    'statesbig': 'states big',\n    'presidentswing': 'president swing',\n    'inconclusion': 'in conclusion',\n    'handsonlearning': 'hands on learning',\n    'electroral': 'electoral',\n    'carowner': 'car owner',\n    'elecotral': 'electoral',\n    'studentassigned': 'student assigned',\n    'collegefive': 'college five',\n    'presidant': 'president',\n    'unfairoutdatedand': 'unfair outdated and',\n    'nixonjimmy': 'nixon jimmy',\n    'canadates': 'candidate',\n    'tabletennis': 'table tennis',\n    'himher': 'him her',\n    'studentsummerpacketdesigners': 'student summer packet designers',\n    'studentdesign': 'student designed',\n    'limting': 'limiting',\n    'electrol': 'electoral',\n    'campaignto': 'campaign to',\n    'presendent': 'president',\n    'thezebra': 'the zebra',\n    'landformation': 'land formation',\n    'eyetoeye': 'eye to eye',\n    'selfreliance': 'self reliance',\n    'studentdriven': 'student driven',\n    'winnertake': 'winner take',\n    'alliens': 'aliens',\n    '2000but': '2000 but',\n    'electionto': 'election to',\n    'candidatesas': 'candidates as',\n    'electers': 'electoral',\n    'winnertakes': 'winner takes',\n    'isfeet': 'is feet',\n    'incar': 'incur',\n    'covid19': 'something',\n    'aflcio': '',\n    'outdatedand': 'outdated and',\n    'httpswww': '',\n    '51998': '',\n    'iswing': '',\n    'ascertainments': '',\n    'athome': '',\n    'risorius': '',\n    'votes538': '',\n    '41971': '',\n    'palpabraeus': '',\n    'figurelandform': 'figure landform',\n    'possibleit': 'possible it',\n    'takeall': 'take all',\n    'inschool': 'in school',\n    'fouces': 'focus',\n    'presidentand': 'president and',\n    'elecotrs': 'electoral',\n    'formationwhich': 'formation which',\n    'electorswho': 'electoral who',\n    'presidnt': 'president',\n    'eletors': 'electoral',\n    'sinceraly': 'sincerely',\n    'emotionshappiness': 'emotions happiness',\n    'carterbob': 'carter bob',\n    'donÃ£Ã¢t': 'do not',\n    'eyesnose': 'eyes nose',\n    'smartroad': 'smart road',\n    'systemvoters': 'system voters',\n    'emtions': 'emotions',\n    'statedemocrats': 'state democrats',\n    'lowcar': 'low car',\n    'elcetoral': 'electoral',\n    'expressivefor': 'expressive for',\n    'animails': 'animals',\n    'oppertonuty': 'opportunity',\n    'tempetures': 'temperature',\n    'recevies': 'receives',\n    'twoseat': 'two seat',\n    'consistution': 'constitution',\n    'horsesyoung': 'horses young',\n    'semidriverless': 'semi driverless',\n    'presisdent': 'president',\n    'exspression': 'expression',\n    'valcanoes': 'volcano',\n    'actiry': '',\n    'lifejust': 'life just',\n    'selfreliant': 'self reliant',\n    'comcaraccidentcauseofaccidentcellphonecellphonestatistics': 'car accident cause of accident cellphone statistics',\n    'vaubangermany': 'germany',\n    'fourtyfour': 'fourty four',\n    'atomspheric': 'atmospheric',\n    'mid1990': '',\n    'activitis': 'activities',\n    'paragrpah': 'paragraph',\n    'electora': 'electoral',\n    'elcetion': 'election',\n    'stressfree': 'stress free',\n    'seegoing': 'see going',\n    'coferencing': 'conferencing',\n    'ctrdot': '',\n    'segoing': '',\n    'teacherdesign': 'teacher design',\n    'kidsteens': 'kids teens',\n    'elcetors': 'electoral',\n    'poulltion': 'pollution',\n    'surportive': 'supportive',\n    'presisent': 'president',\n    'technollogy': 'technology',\n    'precidency': 'president',\n    'voteswhile': 'votes while',\n    'headformed': 'head formed',\n    'swingstates': 'swing states',\n    'candates': 'candidate',\n    'locationname': 'location name',\n    'venuss': 'venues',\n    'astronmers': 'astronomers',\n    'democtratic': 'democratic',\n    'canadent': 'candidate',\n    'cyndonia': '',\n    'computure': 'computer',\n    'nasas': 'nasa',\n    'onehalf': 'one half',\n    'preident': 'president',\n    'ressons': 'reasons',\n    'presidentvice': 'president vice',\n    'nonswing': 'non swing',\n    'thirtyeight': 'thirty eight',\n    'processnot': 'process not',\n    'facetoface': 'face to face',\n    'teendriversource': 'teen driver source',\n    'sadnessand': 'sadness and',\n    'abloish': 'abolish',\n    'driveing': 'driving',\n    'navagating': 'navigating',\n    'electorsthe': 'electoral',\n    'vothing': 'voting',\n    'callage': 'college',\n    'senseit': 'sense it',\n    'mercedesbenz': 'mercedes benz',\n    'electorall': 'electoral'\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-07T04:35:44.976425Z","iopub.execute_input":"2022-07-07T04:35:44.976850Z","iopub.status.idle":"2022-07-07T04:35:45.005372Z","shell.execute_reply.started":"2022-07-07T04:35:44.976814Z","shell.execute_reply":"2022-07-07T04:35:45.004238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def contraction_count(sent):\n    count = 0\n    count += re.subn(r\"won\\'t\", '', sent)[1]\n    count += re.subn(r\"can\\'t\", '', sent)[1]\n    count += re.subn(r\"n\\'t\", '', sent)[1]\n    count += re.subn(r\"\\'re\", '', sent)[1]\n    count += re.subn(r\"\\'s\", '', sent)[1]\n    count += re.subn(r\"\\'d\", '', sent)[1]\n    count += re.subn(r\"\\'ll\", '', sent)[1]\n    count += re.subn(r\"\\'t\", '', sent)[1]\n    count += re.subn(r\"\\'ve\", '', sent)[1]\n    count += re.subn(r\"\\'m\", '', sent)[1]\n    return count","metadata":{"papermill":{"duration":0.029184,"end_time":"2022-06-18T13:55:16.29518","exception":false,"start_time":"2022-06-18T13:55:16.265996","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:35:45.006855Z","iopub.execute_input":"2022-07-07T04:35:45.007320Z","iopub.status.idle":"2022-07-07T04:35:45.020361Z","shell.execute_reply.started":"2022-07-07T04:35:45.007277Z","shell.execute_reply":"2022-07-07T04:35:45.019294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pos_count(sent):\n    nn_count = 0   #Noun\n    pr_count = 0   #Pronoun\n    vb_count = 0   #Verb\n    jj_count = 0   #Adjective\n    uh_count = 0   #Interjection\n    cd_count = 0   #Numerics\n    \n    sent = nltk.word_tokenize(sent)\n    sent = nltk.pos_tag(sent)\n\n    for token in sent:\n        if token[1] in ['NN','NNP','NNS']:\n            nn_count += 1\n\n        if token[1] in ['PRP','PRP$']:\n            pr_count += 1\n\n        if token[1] in ['VB','VBD','VBG','VBN','VBP','VBZ']:\n            vb_count += 1\n\n        if token[1] in ['JJ','JJR','JJS']:\n            jj_count += 1\n\n        if token[1] in ['UH']:\n            uh_count += 1\n\n        if token[1] in ['CD']:\n            cd_count += 1\n    \n    return pd.Series([nn_count, pr_count, vb_count, jj_count, uh_count, cd_count])","metadata":{"papermill":{"duration":0.030738,"end_time":"2022-06-18T13:55:16.342772","exception":false,"start_time":"2022-06-18T13:55:16.312034","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:35:45.021739Z","iopub.execute_input":"2022-07-07T04:35:45.022275Z","iopub.status.idle":"2022-07-07T04:35:45.037995Z","shell.execute_reply.started":"2022-07-07T04:35:45.022241Z","shell.execute_reply":"2022-07-07T04:35:45.036809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decontraction(phrase):\n    phrase = re.sub(r\"won\\'t\", \"will not\", phrase)\n    phrase = re.sub(r\"can\\'t\", \"can not\", phrase)\n    phrase = re.sub(r\"n\\'t\", \" not\", phrase)\n    phrase = re.sub(r\"\\'re\", \" are\", phrase)\n    phrase = re.sub(r\"\\'s\", \" is\", phrase)\n    phrase = re.sub(r\"\\'d\", \" would\", phrase)\n    phrase = re.sub(r\"\\'ll\", \" will\", phrase)\n    phrase = re.sub(r\"\\'t\", \" not\", phrase)\n    phrase = re.sub(r\"\\'ve\", \" have\", phrase)\n    phrase = re.sub(r\"\\'m\", \" am\", phrase)\n    phrase = re.sub(r\"he's\", \"he is\", phrase)\n    phrase = re.sub(r\"there's\", \"there is\", phrase)\n    phrase = re.sub(r\"We're\", \"We are\", phrase)\n    phrase = re.sub(r\"That's\", \"That is\", phrase)\n    phrase = re.sub(r\"won't\", \"will not\", phrase)\n    phrase = re.sub(r\"they're\", \"they are\", phrase)\n    phrase = re.sub(r\"Can't\", \"Cannot\", phrase)\n    phrase = re.sub(r\"wasn't\", \"was not\", phrase)\n    phrase = re.sub(r\"don\\x89Ûªt\", \"do not\", phrase)\n    phrase = re.sub(r\"donãât\", \"do not\", phrase)\n    phrase = re.sub(r\"aren't\", \"are not\", phrase)\n    phrase = re.sub(r\"isn't\", \"is not\", phrase)\n    phrase = re.sub(r\"What's\", \"What is\", phrase)\n    phrase = re.sub(r\"haven't\", \"have not\", phrase)\n    phrase = re.sub(r\"hasn't\", \"has not\", phrase)\n    phrase = re.sub(r\"There's\", \"There is\", phrase)\n    phrase = re.sub(r\"He's\", \"He is\", phrase)\n    phrase = re.sub(r\"It's\", \"It is\", phrase)\n    phrase = re.sub(r\"You're\", \"You are\", phrase)\n    phrase = re.sub(r\"I'M\", \"I am\", phrase)\n    phrase = re.sub(r\"shouldn't\", \"should not\", phrase)\n    phrase = re.sub(r\"wouldn't\", \"would not\", phrase)\n    phrase = re.sub(r\"i'm\", \"I am\", phrase)\n    phrase = re.sub(r\"I\\x89Ûªm\", \"I am\", phrase)\n    phrase = re.sub(r\"I'm\", \"I am\", phrase)\n    phrase = re.sub(r\"Isn't\", \"is not\", phrase)\n    phrase = re.sub(r\"Here's\", \"Here is\", phrase)\n    phrase = re.sub(r\"you've\", \"you have\", phrase)\n    phrase = re.sub(r\"you\\x89Ûªve\", \"you have\", phrase)\n    phrase = re.sub(r\"we're\", \"we are\", phrase)\n    phrase = re.sub(r\"what's\", \"what is\", phrase)\n    phrase = re.sub(r\"couldn't\", \"could not\", phrase)\n    phrase = re.sub(r\"we've\", \"we have\", phrase)\n    phrase = re.sub(r\"it\\x89Ûªs\", \"it is\", phrase)\n    phrase = re.sub(r\"doesn\\x89Ûªt\", \"does not\", phrase)\n    phrase = re.sub(r\"It\\x89Ûªs\", \"It is\", phrase)\n    phrase = re.sub(r\"Here\\x89Ûªs\", \"Here is\", phrase)\n    phrase = re.sub(r\"who's\", \"who is\", phrase)\n    phrase = re.sub(r\"I\\x89Ûªve\", \"I have\", phrase)\n    phrase = re.sub(r\"y'all\", \"you all\", phrase)\n    phrase = re.sub(r\"can\\x89Ûªt\", \"cannot\", phrase)\n    phrase = re.sub(r\"would've\", \"would have\", phrase)\n    phrase = re.sub(r\"it'll\", \"it will\", phrase)\n    phrase = re.sub(r\"we'll\", \"we will\", phrase)\n    phrase = re.sub(r\"wouldn\\x89Ûªt\", \"would not\", phrase)\n    phrase = re.sub(r\"We've\", \"We have\", phrase)\n    phrase = re.sub(r\"he'll\", \"he will\", phrase)\n    phrase = re.sub(r\"Y'all\", \"You all\", phrase)\n    phrase = re.sub(r\"Weren't\", \"Were not\", phrase)\n    phrase = re.sub(r\"Didn't\", \"Did not\", phrase)\n    phrase = re.sub(r\"they'll\", \"they will\", phrase)\n    phrase = re.sub(r\"they'd\", \"they would\", phrase)\n    phrase = re.sub(r\"DON'T\", \"DO NOT\", phrase)\n    phrase = re.sub(r\"That\\x89Ûªs\", \"That is\", phrase)\n    phrase = re.sub(r\"they've\", \"they have\", phrase)\n    phrase = re.sub(r\"i'd\", \"I would\", phrase)\n    phrase = re.sub(r\"should've\", \"should have\", phrase)\n    phrase = re.sub(r\"You\\x89Ûªre\", \"You are\", phrase)\n    phrase = re.sub(r\"where's\", \"where is\", phrase)\n    phrase = re.sub(r\"Don\\x89Ûªt\", \"Do not\", phrase)\n    phrase = re.sub(r\"we'd\", \"we would\", phrase)\n    phrase = re.sub(r\"i'll\", \"I will\", phrase)\n    phrase = re.sub(r\"weren't\", \"were not\", phrase)\n    phrase = re.sub(r\"They're\", \"They are\", phrase)\n    phrase = re.sub(r\"Can\\x89Ûªt\", \"Cannot\", phrase)\n    phrase = re.sub(r\"you\\x89Ûªll\", \"you will\", phrase)\n    phrase = re.sub(r\"I\\x89Ûªd\", \"I would\", phrase)\n    phrase = re.sub(r\"let's\", \"let us\", phrase)\n    phrase = re.sub(r\"it's\", \"it is\", phrase)\n    phrase = re.sub(r\"can't\", \"cannot\", phrase)\n    phrase = re.sub(r\"don't\", \"do not\", phrase)\n    phrase = re.sub(r\"you're\", \"you are\", phrase)\n    phrase = re.sub(r\"i've\", \"I have\", phrase)\n    phrase = re.sub(r\"that's\", \"that is\", phrase)\n    phrase = re.sub(r\"i'll\", \"I will\", phrase)\n    phrase = re.sub(r\"doesn't\", \"does not\",phrase)\n    phrase = re.sub(r\"i'd\", \"I would\", phrase)\n    phrase = re.sub(r\"didn't\", \"did not\", phrase)\n    phrase = re.sub(r\"ain't\", \"am not\", phrase)\n    phrase = re.sub(r\"you'll\", \"you will\", phrase)\n    phrase = re.sub(r\"I've\", \"I have\", phrase)\n    phrase = re.sub(r\"Don't\", \"do not\", phrase)\n    phrase = re.sub(r\"I'll\", \"I will\", phrase)\n    phrase = re.sub(r\"I'd\", \"I would\", phrase)\n    phrase = re.sub(r\"Let's\", \"Let us\", phrase)\n    phrase = re.sub(r\"you'd\", \"You would\", phrase)\n    phrase = re.sub(r\"It's\", \"It is\", phrase)\n    phrase = re.sub(r\"Ain't\", \"am not\", phrase)\n    phrase = re.sub(r\"Haven't\", \"Have not\", phrase)\n    phrase = re.sub(r\"Could've\", \"Could have\", phrase)\n    phrase = re.sub(r\"youve\", \"you have\", phrase)  \n    phrase = re.sub(r\"donå«t\", \"do not\", phrase)\n    return phrase","metadata":{"papermill":{"duration":0.027102,"end_time":"2022-06-18T13:55:16.386606","exception":false,"start_time":"2022-06-18T13:55:16.359504","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:35:45.039884Z","iopub.execute_input":"2022-07-07T04:35:45.042538Z","iopub.status.idle":"2022-07-07T04:35:45.078262Z","shell.execute_reply.started":"2022-07-07T04:35:45.042495Z","shell.execute_reply":"2022-07-07T04:35:45.076800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_punctuations(text):\n    for punctuation in list(string.punctuation):\n        text = text.replace(punctuation, '')\n    return text","metadata":{"papermill":{"duration":0.025399,"end_time":"2022-06-18T13:55:16.428615","exception":false,"start_time":"2022-06-18T13:55:16.403216","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:35:45.081051Z","iopub.execute_input":"2022-07-07T04:35:45.081416Z","iopub.status.idle":"2022-07-07T04:35:45.098371Z","shell.execute_reply.started":"2022-07-07T04:35:45.081386Z","shell.execute_reply":"2022-07-07T04:35:45.097024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lemmatize_words(text):\n    lemmatizer = WordNetLemmatizer()\n    wordnet_map = {\n        \"N\": wordnet.NOUN, \n        \"V\": wordnet.VERB, \n        \"J\": wordnet.ADJ, \n        \"R\": wordnet.ADV\n    }\n    pos_tagged_text = nltk.pos_tag(text.split())\n    return \" \".join([lemmatizer.lemmatize(word, wordnet_map.get(pos[0], wordnet.NOUN)) for word, pos in pos_tagged_text])","metadata":{"papermill":{"duration":0.02645,"end_time":"2022-06-18T13:55:16.47227","exception":false,"start_time":"2022-06-18T13:55:16.44582","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:35:45.100557Z","iopub.execute_input":"2022-07-07T04:35:45.101176Z","iopub.status.idle":"2022-07-07T04:35:45.115537Z","shell.execute_reply.started":"2022-07-07T04:35:45.101043Z","shell.execute_reply":"2022-07-07T04:35:45.113924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_number(text):\n    text = re.sub(r'(\\d+)([a-zA-Z])', '\\g<1> \\g<2>', text)\n    text = re.sub(r'(\\d+) (th|st|nd|rd) ', '\\g<1>\\g<2> ', text)\n    text = re.sub(r'(\\d+),(\\d+)', '\\g<1>\\g<2>', text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-07-07T04:35:45.117096Z","iopub.execute_input":"2022-07-07T04:35:45.117634Z","iopub.status.idle":"2022-07-07T04:35:45.129994Z","shell.execute_reply.started":"2022-07-07T04:35:45.117600Z","shell.execute_reply":"2022-07-07T04:35:45.128898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_misspell(text):\n    for bad_word in misspell_mapping:\n        if bad_word in text:\n            text = text.replace(bad_word, misspell_mapping[bad_word])\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-07-07T04:35:45.131385Z","iopub.execute_input":"2022-07-07T04:35:45.131980Z","iopub.status.idle":"2022-07-07T04:35:45.142385Z","shell.execute_reply.started":"2022-07-07T04:35:45.131941Z","shell.execute_reply":"2022-07-07T04:35:45.141442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sent2vec(text):\n    words = nltk.word_tokenize(text)\n    words = [w for w in words if w.isalpha()]\n    \n    M = []\n    for w in words:\n        try:\n            M.append(embeddings_index[w])\n        except:\n            M.append(embeddings_index['unk'])\n            continue\n    \n    M = np.array(M)\n    v = M.sum(axis=0)\n    if type(v) != np.ndarray:\n        return np.zeros(300)\n    \n    return v / np.sqrt((v ** 2).sum())","metadata":{"papermill":{"duration":0.027843,"end_time":"2022-06-18T13:55:16.516849","exception":false,"start_time":"2022-06-18T13:55:16.489006","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:35:45.143971Z","iopub.execute_input":"2022-07-07T04:35:45.144482Z","iopub.status.idle":"2022-07-07T04:35:45.155844Z","shell.execute_reply.started":"2022-07-07T04:35:45.144434Z","shell.execute_reply":"2022-07-07T04:35:45.155075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create basic text features","metadata":{"papermill":{"duration":0.016269,"end_time":"2022-06-18T13:55:16.549979","exception":false,"start_time":"2022-06-18T13:55:16.53371","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def text_features(df, col):\n    df[f\"{col}_num_words\"] = df[col].progress_apply(lambda x: len(str(x).split()))\n    df[f\"{col}_num_unique_words\"] = df[col].progress_apply(lambda x: len(set(str(x).split())))\n    df[f\"{col}_num_chars\"] = df[col].progress_apply(lambda x: len(str(x)))\n    df[f\"{col}_num_stopwords\"] = df[col].progress_apply(lambda x: len([w for w in str(x).lower().split() if w in stopwords.words('english')]))\n    df[f\"{col}_num_punctuations\"] = df[col].progress_apply(lambda x: len([c for c in str(x) if c in list(string.punctuation)]))\n    df[f\"{col}_num_words_upper\"] = df[col].progress_apply(lambda x: len([w for w in str(x).split() if w.isupper()]))\n    df[f\"{col}_num_words_title\"] = df[col].progress_apply(lambda x: len([w for w in str(x).split() if w.istitle()]))\n    df[f\"{col}_mean_word_len\"] = df[col].progress_apply(lambda x: np.mean([len(w) for w in str(x).split()]))\n    df[f\"{col}_num_paragraphs\"] = df[col].progress_apply(lambda x: len(x.split('\\n')))\n    df[f\"{col}_num_contractions\"] = df[col].progress_apply(contraction_count)\n    df[f\"{col}_polarity\"] = df[col].progress_apply(lambda x: TextBlob(x).sentiment[0])\n    df[f\"{col}_subjectivity\"] = df[col].progress_apply(lambda x: TextBlob(x).sentiment[1])\n    df[[f'{col}_nn_count',f'{col}_pr_count',f'{col}_vb_count',f'{col}_jj_count',f'{col}_uh_count',f'{col}_cd_count']] = df[col].progress_apply(pos_count)\n    return df","metadata":{"papermill":{"duration":0.034727,"end_time":"2022-06-18T13:55:16.601324","exception":false,"start_time":"2022-06-18T13:55:16.566597","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:35:45.156813Z","iopub.execute_input":"2022-07-07T04:35:45.157534Z","iopub.status.idle":"2022-07-07T04:35:45.177083Z","shell.execute_reply.started":"2022-07-07T04:35:45.157497Z","shell.execute_reply":"2022-07-07T04:35:45.175607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"discourse_train = train[['discourse_id','discourse_text']].copy()\ndiscourse_train.drop_duplicates(inplace=True)\nprint(f\"discourse_train: {discourse_train.shape}\")\n\ndiscourse_train = text_features(discourse_train, \"discourse_text\")\ndiscourse_train.head()","metadata":{"papermill":{"duration":416.20836,"end_time":"2022-06-18T14:02:12.826421","exception":false,"start_time":"2022-06-18T13:55:16.618061","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:35:45.178398Z","iopub.execute_input":"2022-07-07T04:35:45.179125Z","iopub.status.idle":"2022-07-07T04:42:21.988435Z","shell.execute_reply.started":"2022-07-07T04:35:45.179086Z","shell.execute_reply":"2022-07-07T04:42:21.987473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"essay_train = train[['essay_id','essay_text']].copy()\nessay_train.drop_duplicates(inplace=True)\nprint(f\"essay_train: {essay_train.shape}\")\n\nessay_train = text_features(essay_train, \"essay_text\")\nessay_train.head()","metadata":{"papermill":{"duration":376.86544,"end_time":"2022-06-18T14:08:29.955897","exception":false,"start_time":"2022-06-18T14:02:13.090457","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:42:21.989787Z","iopub.execute_input":"2022-07-07T04:42:21.990986Z","iopub.status.idle":"2022-07-07T04:48:41.833028Z","shell.execute_reply.started":"2022-07-07T04:42:21.990947Z","shell.execute_reply":"2022-07-07T04:48:41.832079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"discourse_test = test[['discourse_id','discourse_text']].copy()\ndiscourse_test.drop_duplicates(inplace=True)\nprint(f\"discourse_test: {discourse_test.shape}\")\n\ndiscourse_test = text_features(discourse_test, \"discourse_text\")\ndiscourse_test.head()","metadata":{"papermill":{"duration":0.646569,"end_time":"2022-06-18T14:08:31.083616","exception":false,"start_time":"2022-06-18T14:08:30.437047","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:48:41.834474Z","iopub.execute_input":"2022-07-07T04:48:41.834883Z","iopub.status.idle":"2022-07-07T04:48:42.023924Z","shell.execute_reply.started":"2022-07-07T04:48:41.834851Z","shell.execute_reply":"2022-07-07T04:48:42.023263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"essay_test = test[['essay_id','essay_text']].copy()\nessay_test.drop_duplicates(inplace=True)\nprint(f\"essay_test: {essay_test.shape}\")\n\nessay_test = text_features(essay_test, \"essay_text\")\nessay_test.head()","metadata":{"papermill":{"duration":0.645828,"end_time":"2022-06-18T14:08:32.187398","exception":false,"start_time":"2022-06-18T14:08:31.54157","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:48:42.025029Z","iopub.execute_input":"2022-07-07T04:48:42.025361Z","iopub.status.idle":"2022-07-07T04:48:42.202219Z","shell.execute_reply.started":"2022-07-07T04:48:42.025331Z","shell.execute_reply":"2022-07-07T04:48:42.201281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Text Preprocessing","metadata":{"papermill":{"duration":0.460228,"end_time":"2022-06-18T14:08:33.177053","exception":false,"start_time":"2022-06-18T14:08:32.716825","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def clean_text(text):\n    text = decontraction(text)\n    text = text.lower()\n    text = re.sub(r'[^\\w\\s]','',text, re.UNICODE)\n    text = remove_punctuations(text)\n    text = clean_number(text)\n    text = clean_misspell(text)\n    return text","metadata":{"papermill":{"duration":0.476152,"end_time":"2022-06-18T14:08:34.11375","exception":false,"start_time":"2022-06-18T14:08:33.637598","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:48:42.203452Z","iopub.execute_input":"2022-07-07T04:48:42.204291Z","iopub.status.idle":"2022-07-07T04:48:42.210491Z","shell.execute_reply.started":"2022-07-07T04:48:42.204249Z","shell.execute_reply":"2022-07-07T04:48:42.209488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"discourse_train['discourse_text'] = discourse_train['discourse_text'].progress_apply(clean_text)\ndiscourse_test['discourse_text'] = discourse_test['discourse_text'].progress_apply(clean_text)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T04:48:42.211724Z","iopub.execute_input":"2022-07-07T04:48:42.212033Z","iopub.status.idle":"2022-07-07T04:48:52.822719Z","shell.execute_reply.started":"2022-07-07T04:48:42.212005Z","shell.execute_reply":"2022-07-07T04:48:52.821646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"essay_train['essay_text'] = essay_train['essay_text'].progress_apply(clean_text)\nessay_test['essay_text'] = essay_test['essay_text'].progress_apply(clean_text)","metadata":{"papermill":{"duration":99.96913,"end_time":"2022-06-18T14:10:14.604652","exception":false,"start_time":"2022-06-18T14:08:34.635522","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:48:52.823928Z","iopub.execute_input":"2022-07-07T04:48:52.824862Z","iopub.status.idle":"2022-07-07T04:48:58.216092Z","shell.execute_reply.started":"2022-07-07T04:48:52.824801Z","shell.execute_reply":"2022-07-07T04:48:58.215233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Glove Embeddings","metadata":{"papermill":{"duration":0.583651,"end_time":"2022-06-18T14:11:51.272897","exception":false,"start_time":"2022-06-18T14:11:50.689246","status":"completed"},"tags":[]}},{"cell_type":"code","source":"with open(\"../input/nlp-word-embeddings/Glove_Embeddings.txt\", 'rb') as handle: \n    data = handle.read()\n\nprocessed_data = pickle.loads(data)\nembeddings_index = processed_data['glove_embeddings_index']\nprint('Word vectors found: {}'.format(len(embeddings_index)))\n\ndel processed_data\ngc.collect()","metadata":{"papermill":{"duration":41.133595,"end_time":"2022-06-18T14:12:33.046527","exception":false,"start_time":"2022-06-18T14:11:51.912932","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:48:58.217652Z","iopub.execute_input":"2022-07-07T04:48:58.217974Z","iopub.status.idle":"2022-07-07T04:49:42.449397Z","shell.execute_reply.started":"2022-07-07T04:48:58.217946Z","shell.execute_reply":"2022-07-07T04:49:42.448262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"discourse_train.set_index('discourse_id', inplace=True)\n\nglove_vec = [sent2vec(x) for x in tqdm(discourse_train[\"discourse_text\"].values)]\ncol_list = ['discourse_glove_'+str(i) for i in range(300)]\nglove_vec_df = pd.DataFrame(np.array(glove_vec), columns=col_list, index=discourse_train.index)\nprint(f\"glove_vec_df: {glove_vec_df.shape}\")\n\ndiscourse_train = pd.merge(\n    discourse_train, \n    glove_vec_df, \n    how=\"inner\", \n    on=\"discourse_id\", \n    sort=False\n)\n\ndel glove_vec, glove_vec_df\ngc.collect()\n\nprint(f\"discourse_train: {discourse_train.shape}\")\ndiscourse_train.head()","metadata":{"papermill":{"duration":23.821942,"end_time":"2022-06-18T14:12:57.44866","exception":false,"start_time":"2022-06-18T14:12:33.626718","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:49:42.450874Z","iopub.execute_input":"2022-07-07T04:49:42.451249Z","iopub.status.idle":"2022-07-07T04:50:05.962036Z","shell.execute_reply.started":"2022-07-07T04:49:42.451193Z","shell.execute_reply":"2022-07-07T04:50:05.961027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"essay_train.set_index('essay_id', inplace=True)\n\nglove_vec = [sent2vec(x) for x in tqdm(essay_train[\"essay_text\"].values)]\ncol_list = ['essay_glove_'+str(i) for i in range(300)]\nglove_vec_df = pd.DataFrame(np.array(glove_vec), columns=col_list, index=essay_train.index)\nprint(f\"glove_vec_df: {glove_vec_df.shape}\")\n\nessay_train = pd.merge(\n    essay_train, \n    glove_vec_df, \n    how=\"inner\", \n    on=\"essay_id\", \n    sort=False\n)\n\ndel glove_vec, glove_vec_df\ngc.collect()\n\nprint(f\"essay_train: {essay_train.shape}\")\nessay_train.head()","metadata":{"papermill":{"duration":18.326306,"end_time":"2022-06-18T14:13:16.374982","exception":false,"start_time":"2022-06-18T14:12:58.048676","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:50:05.963694Z","iopub.execute_input":"2022-07-07T04:50:05.964116Z","iopub.status.idle":"2022-07-07T04:50:24.462163Z","shell.execute_reply.started":"2022-07-07T04:50:05.964082Z","shell.execute_reply":"2022-07-07T04:50:24.461184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"discourse_test.set_index('discourse_id', inplace=True)\n\nglove_vec = [sent2vec(x) for x in tqdm(discourse_test[\"discourse_text\"].values)]\ncol_list = ['discourse_glove_'+str(i) for i in range(300)]\nglove_vec_df = pd.DataFrame(np.array(glove_vec), columns=col_list, index=discourse_test.index)\nprint(f\"glove_vec_df: {glove_vec_df.shape}\")\n\ndiscourse_test = pd.merge(\n    discourse_test, \n    glove_vec_df, \n    how=\"inner\", \n    on=\"discourse_id\", \n    sort=False\n)\n\ndel glove_vec, glove_vec_df\ngc.collect()\n\nprint(f\"discourse_test: {discourse_test.shape}\")\ndiscourse_test.head()","metadata":{"papermill":{"duration":0.882551,"end_time":"2022-06-18T14:13:17.969988","exception":false,"start_time":"2022-06-18T14:13:17.087437","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:50:24.463462Z","iopub.execute_input":"2022-07-07T04:50:24.463809Z","iopub.status.idle":"2022-07-07T04:50:24.651432Z","shell.execute_reply.started":"2022-07-07T04:50:24.463780Z","shell.execute_reply":"2022-07-07T04:50:24.650267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"essay_test.set_index('essay_id', inplace=True)\n\nglove_vec = [sent2vec(x) for x in tqdm(essay_test[\"essay_text\"].values)]\ncol_list = ['essay_glove_'+str(i) for i in range(300)]\nglove_vec_df = pd.DataFrame(np.array(glove_vec), columns=col_list, index=essay_test.index)\nprint(f\"glove_vec_df: {glove_vec_df.shape}\")\n\nessay_test = pd.merge(\n    essay_test, \n    glove_vec_df, \n    how=\"inner\", \n    on=\"essay_id\", \n    sort=False\n)\n\ndel glove_vec, glove_vec_df\ngc.collect()\n\nprint(f\"essay_test: {essay_test.shape}\")\nessay_test.head()","metadata":{"papermill":{"duration":0.91535,"end_time":"2022-06-18T14:13:19.540176","exception":false,"start_time":"2022-06-18T14:13:18.624826","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:50:24.652730Z","iopub.execute_input":"2022-07-07T04:50:24.653041Z","iopub.status.idle":"2022-07-07T04:50:24.839616Z","shell.execute_reply.started":"2022-07-07T04:50:24.653013Z","shell.execute_reply":"2022-07-07T04:50:24.838654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del embeddings_index\ngc.collect()","metadata":{"papermill":{"duration":1.884248,"end_time":"2022-06-18T14:13:22.038052","exception":false,"start_time":"2022-06-18T14:13:20.153804","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:50:24.844640Z","iopub.execute_input":"2022-07-07T04:50:24.844993Z","iopub.status.idle":"2022-07-07T04:50:25.933385Z","shell.execute_reply.started":"2022-07-07T04:50:24.844965Z","shell.execute_reply":"2022-07-07T04:50:25.932343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Paragram Embeddings","metadata":{"papermill":{"duration":0.611144,"end_time":"2022-06-18T14:13:23.260557","exception":false,"start_time":"2022-06-18T14:13:22.649413","status":"completed"},"tags":[]}},{"cell_type":"code","source":"with open(\"../input/nlp-word-embeddings/Para_Embeddings.txt\", 'rb') as handle: \n    data = handle.read()\n\nprocessed_data = pickle.loads(data)\nembeddings_index = processed_data['para_embeddings_index']\nprint('Word vectors found: {}'.format(len(embeddings_index)))\n\ndel processed_data\ngc.collect()","metadata":{"papermill":{"duration":14.170462,"end_time":"2022-06-18T14:13:38.108634","exception":false,"start_time":"2022-06-18T14:13:23.938172","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:50:25.935106Z","iopub.execute_input":"2022-07-07T04:50:25.935459Z","iopub.status.idle":"2022-07-07T04:50:53.055619Z","shell.execute_reply.started":"2022-07-07T04:50:25.935431Z","shell.execute_reply":"2022-07-07T04:50:53.054931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"para_vec = [sent2vec(x) for x in tqdm(discourse_train[\"discourse_text\"].values)]\ncol_list = ['discourse_para_'+str(i) for i in range(300)]\npara_vec_df = pd.DataFrame(np.array(para_vec), columns=col_list, index=discourse_train.index)\nprint(f\"para_vec_df: {para_vec_df.shape}\")\n\ndiscourse_train = pd.merge(\n    discourse_train, \n    para_vec_df, \n    how=\"inner\", \n    on=\"discourse_id\", \n    sort=False\n)\n\ndel para_vec, para_vec_df\ngc.collect()\n\ndiscourse_train.drop('discourse_text', axis=1, inplace=True)\nprint(f\"discourse_train: {discourse_train.shape}\")\ndiscourse_train.head()","metadata":{"papermill":{"duration":23.49866,"end_time":"2022-06-18T14:14:02.221393","exception":false,"start_time":"2022-06-18T14:13:38.722733","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:50:53.056693Z","iopub.execute_input":"2022-07-07T04:50:53.057550Z","iopub.status.idle":"2022-07-07T04:51:16.645237Z","shell.execute_reply.started":"2022-07-07T04:50:53.057518Z","shell.execute_reply":"2022-07-07T04:51:16.644298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"para_vec = [sent2vec(x) for x in tqdm(essay_train[\"essay_text\"].values)]\ncol_list = ['essay_para_'+str(i) for i in range(300)]\npara_vec_df = pd.DataFrame(np.array(para_vec), columns=col_list, index=essay_train.index)\nprint(f\"para_vec_df: {para_vec_df.shape}\")\n\nessay_train = pd.merge(\n    essay_train, \n    para_vec_df, \n    how=\"inner\", \n    on=\"essay_id\", \n    sort=False\n)\n\ndel para_vec, para_vec_df\ngc.collect()\n\nessay_train.drop('essay_text', axis=1, inplace=True)\nprint(f\"essay_train: {essay_train.shape}\")\nessay_train.head()","metadata":{"papermill":{"duration":18.174534,"end_time":"2022-06-18T14:14:21.09511","exception":false,"start_time":"2022-06-18T14:14:02.920576","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:51:16.646432Z","iopub.execute_input":"2022-07-07T04:51:16.646780Z","iopub.status.idle":"2022-07-07T04:51:34.869553Z","shell.execute_reply.started":"2022-07-07T04:51:16.646749Z","shell.execute_reply":"2022-07-07T04:51:34.868584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"para_vec = [sent2vec(x) for x in tqdm(discourse_test[\"discourse_text\"].values)]\ncol_list = ['discourse_para_'+str(i) for i in range(300)]\npara_vec_df = pd.DataFrame(np.array(para_vec), columns=col_list, index=discourse_test.index)\nprint(f\"para_vec_df: {para_vec_df.shape}\")\n\ndiscourse_test = pd.merge(\n    discourse_test, \n    para_vec_df, \n    how=\"inner\", \n    on=\"discourse_id\", \n    sort=False\n)\n\ndel para_vec, para_vec_df\ngc.collect()\n\ndiscourse_test.drop('discourse_text', axis=1, inplace=True)\nprint(f\"discourse_test: {discourse_test.shape}\")\ndiscourse_test.head()","metadata":{"papermill":{"duration":0.930581,"end_time":"2022-06-18T14:14:22.77541","exception":false,"start_time":"2022-06-18T14:14:21.844829","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:51:34.871269Z","iopub.execute_input":"2022-07-07T04:51:34.871711Z","iopub.status.idle":"2022-07-07T04:51:35.055051Z","shell.execute_reply.started":"2022-07-07T04:51:34.871668Z","shell.execute_reply":"2022-07-07T04:51:35.053953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"para_vec = [sent2vec(x) for x in tqdm(essay_test[\"essay_text\"].values)]\ncol_list = ['essay_para_'+str(i) for i in range(300)]\npara_vec_df = pd.DataFrame(np.array(para_vec), columns=col_list, index=essay_test.index)\nprint(f\"para_vec_df: {para_vec_df.shape}\")\n\nessay_test = pd.merge(\n    essay_test, \n    para_vec_df, \n    how=\"inner\", \n    on=\"essay_id\", \n    sort=False\n)\n\ndel para_vec, para_vec_df\ngc.collect()\n\nessay_test.drop('essay_text', axis=1, inplace=True)\nprint(f\"essay_test: {essay_test.shape}\")\nessay_test.head()","metadata":{"papermill":{"duration":0.936823,"end_time":"2022-06-18T14:14:24.361313","exception":false,"start_time":"2022-06-18T14:14:23.42449","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:51:35.056844Z","iopub.execute_input":"2022-07-07T04:51:35.057306Z","iopub.status.idle":"2022-07-07T04:51:35.240746Z","shell.execute_reply.started":"2022-07-07T04:51:35.057262Z","shell.execute_reply":"2022-07-07T04:51:35.239846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del embeddings_index\ngc.collect()","metadata":{"papermill":{"duration":1.303531,"end_time":"2022-06-18T14:14:26.308266","exception":false,"start_time":"2022-06-18T14:14:25.004735","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:51:35.242097Z","iopub.execute_input":"2022-07-07T04:51:35.242438Z","iopub.status.idle":"2022-07-07T04:51:36.108270Z","shell.execute_reply.started":"2022-07-07T04:51:35.242409Z","shell.execute_reply":"2022-07-07T04:51:36.107256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Merge all datasets","metadata":{"papermill":{"duration":0.641873,"end_time":"2022-06-18T14:14:27.653902","exception":false,"start_time":"2022-06-18T14:14:27.012029","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train = pd.merge(\n    train,\n    discourse_train,\n    how='inner',\n    on='discourse_id',\n    sort=False\n)\n\ntrain = pd.merge(\n    train,\n    essay_train,\n    how='inner',\n    on='essay_id',\n    sort=False\n)\n\nprint(f\"train: {train.shape}\")\ntrain.head()","metadata":{"papermill":{"duration":1.099257,"end_time":"2022-06-18T14:14:29.392128","exception":false,"start_time":"2022-06-18T14:14:28.292871","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:51:36.109360Z","iopub.execute_input":"2022-07-07T04:51:36.109672Z","iopub.status.idle":"2022-07-07T04:51:36.343975Z","shell.execute_reply.started":"2022-07-07T04:51:36.109646Z","shell.execute_reply":"2022-07-07T04:51:36.342924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.merge(\n    test,\n    discourse_test,\n    how='inner',\n    on='discourse_id',\n    sort=False\n)\n\ntest = pd.merge(\n    test,\n    essay_test,\n    how='inner',\n    on='essay_id',\n    sort=False\n)\n\nprint(f\"test: {test.shape}\")\ntest.head()","metadata":{"papermill":{"duration":0.7421,"end_time":"2022-06-18T14:14:30.774809","exception":false,"start_time":"2022-06-18T14:14:30.032709","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:51:36.345309Z","iopub.execute_input":"2022-07-07T04:51:36.345749Z","iopub.status.idle":"2022-07-07T04:51:36.379406Z","shell.execute_reply.started":"2022-07-07T04:51:36.345714Z","shell.execute_reply":"2022-07-07T04:51:36.378437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del discourse_train, essay_train\ndel discourse_test, essay_test\ngc.collect()","metadata":{"papermill":{"duration":0.845233,"end_time":"2022-06-18T14:14:32.261854","exception":false,"start_time":"2022-06-18T14:14:31.416621","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:51:36.380689Z","iopub.execute_input":"2022-07-07T04:51:36.381030Z","iopub.status.idle":"2022-07-07T04:51:36.526478Z","shell.execute_reply.started":"2022-07-07T04:51:36.380999Z","shell.execute_reply":"2022-07-07T04:51:36.525344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Additional features","metadata":{"papermill":{"duration":0.660315,"end_time":"2022-06-18T14:14:33.634819","exception":false,"start_time":"2022-06-18T14:14:32.974504","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train['discourse_index'] = train.apply(lambda x: x['essay_text'].find(x['discourse_text'].strip()), axis=1)\ntrain['num_words_ratio'] = train['discourse_text_num_words']/train['essay_text_num_words']\ntrain['num_unique_words_ratio'] = train['discourse_text_num_unique_words']/train['essay_text_num_unique_words']\ntrain['num_chars_ratio'] = train['discourse_text_num_chars']/train['essay_text_num_chars']\ntrain['num_stopwords_ratio'] = train['discourse_text_num_stopwords']/train['essay_text_num_stopwords']\ntrain['num_punctuations_ratio'] = train['discourse_text_num_punctuations']/train['essay_text_num_punctuations']\ntrain['mean_word_len_ratio'] = train['discourse_text_mean_word_len']/train['essay_text_mean_word_len']\ntrain.head()","metadata":{"papermill":{"duration":5.488903,"end_time":"2022-06-18T14:14:39.765626","exception":false,"start_time":"2022-06-18T14:14:34.276723","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:52:42.939137Z","iopub.execute_input":"2022-07-07T04:52:42.939589Z","iopub.status.idle":"2022-07-07T04:52:47.765424Z","shell.execute_reply.started":"2022-07-07T04:52:42.939556Z","shell.execute_reply":"2022-07-07T04:52:47.764284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['discourse_index'] = test.apply(lambda x: x['essay_text'].find(x['discourse_text'].strip()), axis=1)\ntest['num_words_ratio'] = test['discourse_text_num_words']/test['essay_text_num_words']\ntest['num_unique_words_ratio'] = test['discourse_text_num_unique_words']/test['essay_text_num_unique_words']\ntest['num_chars_ratio'] = test['discourse_text_num_chars']/test['essay_text_num_chars']\ntest['num_stopwords_ratio'] = test['discourse_text_num_stopwords']/test['essay_text_num_stopwords']\ntest['num_punctuations_ratio'] = test['discourse_text_num_punctuations']/test['essay_text_num_punctuations']\ntest['mean_word_len_ratio'] = test['discourse_text_mean_word_len']/test['essay_text_mean_word_len']\ntest.head()","metadata":{"papermill":{"duration":0.748535,"end_time":"2022-06-18T14:14:41.155118","exception":false,"start_time":"2022-06-18T14:14:40.406583","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:52:47.767396Z","iopub.execute_input":"2022-07-07T04:52:47.767831Z","iopub.status.idle":"2022-07-07T04:52:47.809322Z","shell.execute_reply.started":"2022-07-07T04:52:47.767791Z","shell.execute_reply":"2022-07-07T04:52:47.808336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = train.groupby('essay_id')\\\n        .agg({'discourse_id':'count'})\\\n        .reset_index()\\\n        .rename(columns={'discourse_id':'discourse_count'})\n\ntrain = pd.merge(\n    train,\n    df,\n    how='inner',\n    on='essay_id',\n    sort=False\n)\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T04:56:11.379918Z","iopub.execute_input":"2022-07-07T04:56:11.380378Z","iopub.status.idle":"2022-07-07T04:56:11.627759Z","shell.execute_reply.started":"2022-07-07T04:56:11.380344Z","shell.execute_reply":"2022-07-07T04:56:11.626641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = train.groupby(['essay_id','discourse_type']).agg({\n    'discourse_id': 'count',\n    'discourse_text_num_words': 'mean',\n    'discourse_text_num_chars': 'mean'\n}).reset_index().rename(columns={\n    'discourse_id':'discourse_type_count',\n    'discourse_text_num_words':'discourse_text_num_words_mean',\n    'discourse_text_num_chars':'discourse_text_num_chars_mean'\n})\n\ntrain = pd.merge(\n    train,\n    df,\n    how='inner',\n    on=['essay_id','discourse_type'],\n    sort=False\n)\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T04:56:21.637703Z","iopub.execute_input":"2022-07-07T04:56:21.638137Z","iopub.status.idle":"2022-07-07T04:56:21.809387Z","shell.execute_reply.started":"2022-07-07T04:56:21.638103Z","shell.execute_reply":"2022-07-07T04:56:21.808387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = test.groupby('essay_id')\\\n        .agg({'discourse_id':'count'})\\\n        .reset_index()\\\n        .rename(columns={'discourse_id':'discourse_count'})\n\ntest = pd.merge(\n    test,\n    df,\n    how='inner',\n    on='essay_id',\n    sort=False\n)\n\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T04:57:24.238485Z","iopub.execute_input":"2022-07-07T04:57:24.238914Z","iopub.status.idle":"2022-07-07T04:57:24.424083Z","shell.execute_reply.started":"2022-07-07T04:57:24.238881Z","shell.execute_reply":"2022-07-07T04:57:24.422294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = test.groupby(['essay_id','discourse_type']).agg({\n    'discourse_id': 'count',\n    'discourse_text_num_words': 'mean',\n    'discourse_text_num_chars': 'mean'\n}).reset_index().rename(columns={\n    'discourse_id':'discourse_type_count',\n    'discourse_text_num_words':'discourse_text_num_words_mean',\n    'discourse_text_num_chars':'discourse_text_num_chars_mean'\n})\n\ntest = pd.merge(\n    test,\n    df,\n    how='inner',\n    on=['essay_id','discourse_type'],\n    sort=False\n)\n\ntest.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Label Encoding and Feature Scaling","metadata":{"papermill":{"duration":0.642306,"end_time":"2022-06-18T14:14:42.439969","exception":false,"start_time":"2022-06-18T14:14:41.797663","status":"completed"},"tags":[]}},{"cell_type":"code","source":"le = LabelEncoder().fit(train['discourse_type'].append(test['discourse_type']))\ntrain['discourse_type'] = le.transform(train['discourse_type'])\ntest['discourse_type'] = le.transform(test['discourse_type'])\ntrain.head()","metadata":{"papermill":{"duration":0.689621,"end_time":"2022-06-18T14:14:43.831307","exception":false,"start_time":"2022-06-18T14:14:43.141686","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:56:28.337864Z","iopub.execute_input":"2022-07-07T04:56:28.338278Z","iopub.status.idle":"2022-07-07T04:56:28.383233Z","shell.execute_reply.started":"2022-07-07T04:56:28.338240Z","shell.execute_reply":"2022-07-07T04:56:28.382304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop([\n    'discourse_id',\n    'essay_id',\n    'discourse_text',\n    'essay_text'\n], axis=1, inplace=True)\n\n\ntest.drop([\n    'discourse_id',\n    'essay_id',\n    'discourse_text',\n    'essay_text'\n], axis=1, inplace=True)\n\ntrain.fillna(0, inplace=True)\ntest.fillna(0, inplace=True)","metadata":{"papermill":{"duration":0.874225,"end_time":"2022-06-18T14:14:45.44275","exception":false,"start_time":"2022-06-18T14:14:44.568525","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-07T04:56:28.736610Z","iopub.execute_input":"2022-07-07T04:56:28.737117Z","iopub.status.idle":"2022-07-07T04:56:28.961260Z","shell.execute_reply.started":"2022-07-07T04:56:28.737086Z","shell.execute_reply":"2022-07-07T04:56:28.960308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = test.columns.tolist()\n\nqt = QuantileTransformer(n_quantiles=1000, \n                         output_distribution='normal', \n                         random_state=42).fit(train[features])\n\ntrain[features] = qt.transform(train[features])\ntest[features] = qt.transform(test[features])","metadata":{"execution":{"iopub.execute_input":"2022-06-18T14:14:46.805826Z","iopub.status.busy":"2022-06-18T14:14:46.805392Z","iopub.status.idle":"2022-06-18T14:15:07.05182Z","shell.execute_reply":"2022-06-18T14:15:07.050867Z"},"papermill":{"duration":20.962986,"end_time":"2022-06-18T14:15:07.054368","exception":false,"start_time":"2022-06-18T14:14:46.091382","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Xtrain = train.copy()\nXtest = test.copy()\nprint(f\"Xtrain: {Xtrain.shape} \\nXtest: {Xtest.shape}\")","metadata":{"execution":{"iopub.execute_input":"2022-06-18T14:15:08.352006Z","iopub.status.busy":"2022-06-18T14:15:08.351582Z","iopub.status.idle":"2022-06-18T14:15:08.725971Z","shell.execute_reply":"2022-06-18T14:15:08.725253Z"},"papermill":{"duration":1.028595,"end_time":"2022-06-18T14:15:08.728217","exception":false,"start_time":"2022-06-18T14:15:07.699622","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train, test, qt\ngc.collect()","metadata":{"execution":{"iopub.execute_input":"2022-06-18T14:15:10.083155Z","iopub.status.busy":"2022-06-18T14:15:10.082541Z","iopub.status.idle":"2022-06-18T14:15:10.293588Z","shell.execute_reply":"2022-06-18T14:15:10.292664Z"},"papermill":{"duration":0.858104,"end_time":"2022-06-18T14:15:10.295599","exception":false,"start_time":"2022-06-18T14:15:09.437495","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Models Training","metadata":{"papermill":{"duration":0.649306,"end_time":"2022-06-18T14:15:11.65603","exception":false,"start_time":"2022-06-18T14:15:11.006724","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### Logistic Regression","metadata":{"papermill":{"duration":0.738428,"end_time":"2022-06-18T14:15:13.055564","exception":false,"start_time":"2022-06-18T14:15:12.317136","status":"completed"},"tags":[]}},{"cell_type":"code","source":"FOLD = 5\nSEEDS = [42]\n\ncounter = 0\noof_score = 0\ny_pred_final_lr = np.zeros((Xtest.shape[0], 3))\n\n\nfor sidx, seed in enumerate(SEEDS):\n    seed_score = 0\n    \n    kfold = StratifiedKFold(n_splits=FOLD, shuffle=True, random_state=seed)\n\n    for idx, (train, val) in enumerate(kfold.split(Xtrain, Ytrain)):\n        counter += 1\n\n        train_x, train_y = Xtrain.iloc[train], Ytrain[train]\n        val_x, val_y = Xtrain.iloc[val], Ytrain[val]\n\n        model = LogisticRegression(max_iter=2000, random_state=42)\n        model.fit(train_x, train_y)\n        \n        y_pred = model.predict_proba(val_x)\n        y_pred_final_lr += model.predict_proba(Xtest)\n        \n        score = log_loss(val_y, y_pred)\n        oof_score += score\n        seed_score += score\n        print(\"Seed-{} | Fold-{} | OOF Score: {}\".format(seed, idx, score))\n        \n        with open(f'FPE_LR_Model_{counter}.pkl', 'wb') as file:\n            pickle.dump(model, file)\n        \n        del model, y_pred\n        del train_x, train_y\n        del val_x, val_y\n        gc.collect()\n    \n    print(\"\\nSeed: {} | Aggregate OOF Score: {}\\n\\n\".format(seed, (seed_score / FOLD)))\n\n\ny_pred_final_lr = y_pred_final_lr / float(counter)\noof_score /= float(counter)\nprint(\"Aggregate OOF Score: {}\".format(oof_score))","metadata":{"execution":{"iopub.execute_input":"2022-06-18T14:15:14.476136Z","iopub.status.busy":"2022-06-18T14:15:14.475771Z","iopub.status.idle":"2022-06-18T14:27:30.783966Z","shell.execute_reply":"2022-06-18T14:27:30.782951Z"},"papermill":{"duration":737.773703,"end_time":"2022-06-18T14:27:31.506868","exception":false,"start_time":"2022-06-18T14:15:13.733165","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### XGBoost","metadata":{"papermill":{"duration":0.70974,"end_time":"2022-06-18T14:27:32.869614","exception":false,"start_time":"2022-06-18T14:27:32.159874","status":"completed"},"tags":[]}},{"cell_type":"code","source":"FOLD = 5\nSEEDS = [42]\n\ncounter = 0\noof_score = 0\ny_pred_final_xgb = np.zeros((Xtest.shape[0], 3))\n\n\nfor sidx, seed in enumerate(SEEDS):\n    seed_score = 0\n    \n    kfold = StratifiedKFold(n_splits=FOLD, shuffle=True, random_state=seed)\n\n    for idx, (train, val) in enumerate(kfold.split(Xtrain, Ytrain)):\n        counter += 1\n\n        train_x, train_y = Xtrain.iloc[train], Ytrain[train]\n        val_x, val_y = Xtrain.iloc[val], Ytrain[val]\n\n        model = XGBClassifier(\n            objective='multi:softproba',\n            eval_metric='mlogloss',\n            booster='gbtree',\n            sample_type='weighted',\n            tree_method='hist',\n            grow_policy='lossguide',\n            use_label_encoder=False,\n            num_round=5000,\n            num_class=3,\n            max_depth=9, \n            max_leaves=36,\n            learning_rate=0.095,\n            subsample=0.7024,\n            colsample_bytree=0.5289,\n            min_child_weight=15,\n            reg_lambda=0.05465,\n            verbosity=0,\n            random_state=42\n        )\n        \n        model.fit(train_x, train_y, eval_set=[(train_x, train_y), (val_x, val_y)], \n                  early_stopping_rounds=100, verbose=50)\n        \n        y_pred = model.predict_proba(val_x, iteration_range=(0, model.best_iteration))\n        y_pred_final_xgb += model.predict_proba(Xtest, iteration_range=(0, model.best_iteration))\n        \n        score = log_loss(val_y, y_pred)\n        oof_score += score\n        seed_score += score\n        print(\"Seed-{} | Fold-{} | OOF Score: {}\".format(seed, idx, score))\n        \n        with open(f'FPE_XGB_Model_{counter}.pkl', 'wb') as file:\n            pickle.dump(model, file)\n        \n        del model, y_pred\n        del train_x, train_y\n        del val_x, val_y\n        gc.collect()\n    \n    print(\"\\nSeed: {} | Aggregate OOF Score: {}\\n\\n\".format(seed, (seed_score / FOLD)))\n\n\ny_pred_final_xgb = y_pred_final_xgb / float(counter)\noof_score /= float(counter)\nprint(\"Aggregate OOF Score: {}\".format(oof_score))","metadata":{"execution":{"iopub.execute_input":"2022-06-18T14:27:34.189131Z","iopub.status.busy":"2022-06-18T14:27:34.188715Z","iopub.status.idle":"2022-06-18T14:44:32.855324Z","shell.execute_reply":"2022-06-18T14:44:32.854551Z"},"papermill":{"duration":1019.609734,"end_time":"2022-06-18T14:44:33.128701","exception":false,"start_time":"2022-06-18T14:27:33.518967","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### LightGBM","metadata":{"papermill":{"duration":0.72434,"end_time":"2022-06-18T14:44:34.500075","exception":false,"start_time":"2022-06-18T14:44:33.775735","status":"completed"},"tags":[]}},{"cell_type":"code","source":"params = {}\nparams[\"objective\"] = 'multiclass'\nparams['metric'] = 'multi_logloss'\nparams['boosting'] = 'gbdt'\nparams['num_class'] = 3\nparams['is_unbalance'] = True\nparams[\"learning_rate\"] = 0.05\nparams[\"lambda_l2\"] = 0.0256\nparams[\"num_leaves\"] = 52\nparams[\"max_depth\"] = 10\nparams[\"feature_fraction\"] = 0.503\nparams[\"bagging_fraction\"] = 0.741\nparams[\"bagging_freq\"] = 8\nparams[\"bagging_seed\"] = 10\nparams[\"min_data_in_leaf\"] = 10\nparams[\"verbosity\"] = -1\nparams[\"random_state\"] = 42\nnum_rounds = 5000","metadata":{"execution":{"iopub.execute_input":"2022-06-18T14:44:35.802071Z","iopub.status.busy":"2022-06-18T14:44:35.801107Z","iopub.status.idle":"2022-06-18T14:44:35.808843Z","shell.execute_reply":"2022-06-18T14:44:35.808152Z"},"papermill":{"duration":0.65922,"end_time":"2022-06-18T14:44:35.81091","exception":false,"start_time":"2022-06-18T14:44:35.15169","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FOLD = 5\nSEEDS = [42]\n\ncounter = 0\noof_score = 0\ny_pred_final_lgb = np.zeros((Xtest.shape[0], 3))\n\n\nfor sidx, seed in enumerate(SEEDS):\n    seed_score = 0\n    \n    kfold = StratifiedKFold(n_splits=FOLD, shuffle=True, random_state=seed)\n\n    for idx, (train, val) in enumerate(kfold.split(Xtrain, Ytrain)):\n        counter += 1\n\n        train_x, train_y = Xtrain.iloc[train], Ytrain[train]\n        val_x, val_y = Xtrain.iloc[val], Ytrain[val]\n        \n        lgtrain = lgb.Dataset(train_x, label=train_y.ravel())\n        lgvalidation = lgb.Dataset(val_x, label=val_y.ravel())\n\n        model = lgb.train(params, lgtrain, num_rounds, \n                          valid_sets=[lgtrain, lgvalidation], \n                          early_stopping_rounds=100, verbose_eval=100)\n        \n        y_pred = model.predict(val_x, num_iteration=model.best_iteration)\n        y_pred_final_lgb += model.predict(Xtest, num_iteration=model.best_iteration)\n        \n        score = log_loss(val_y, y_pred)\n        oof_score += score\n        seed_score += score\n        print(\"Seed-{} | Fold-{} | OOF Score: {}\".format(seed, idx, score))\n        \n        with open(f'FPE_LGB_Model_{counter}.pkl', 'wb') as file:\n            pickle.dump(model, file)\n        \n        del model, y_pred\n        del train_x, train_y\n        del val_x, val_y\n        gc.collect()\n    \n    print(\"\\nSeed: {} | Aggregate OOF Score: {}\\n\\n\".format(seed, (seed_score / FOLD)))\n\n\ny_pred_final_lgb = y_pred_final_lgb / float(counter)\noof_score /= float(counter)\nprint(\"Aggregate OOF Score: {}\".format(oof_score))","metadata":{"execution":{"iopub.execute_input":"2022-06-18T14:44:37.203255Z","iopub.status.busy":"2022-06-18T14:44:37.202492Z","iopub.status.idle":"2022-06-18T14:59:16.851683Z","shell.execute_reply":"2022-06-18T14:59:16.849849Z"},"papermill":{"duration":880.324885,"end_time":"2022-06-18T14:59:16.8547","exception":false,"start_time":"2022-06-18T14:44:36.529815","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create submission file","metadata":{"papermill":{"duration":0.652442,"end_time":"2022-06-18T14:59:18.183586","exception":false,"start_time":"2022-06-18T14:59:17.531144","status":"completed"},"tags":[]}},{"cell_type":"code","source":"y_pred_final = (y_pred_final_lr * 0.1) + (y_pred_final_xgb * 0.5) + (y_pred_final_lgb * 0.4)\n\nsubmission = pd.read_csv(\"../input/feedback-prize-effectiveness/sample_submission.csv\")\nsubmission['Ineffective'] = y_pred_final[:,0]\nsubmission['Adequate'] = y_pred_final[:,1]\nsubmission['Effective'] = y_pred_final[:,2]\nsubmission.to_csv(\"./submission.csv\", index=False)\nsubmission.head()","metadata":{"execution":{"iopub.execute_input":"2022-06-18T14:59:19.564036Z","iopub.status.busy":"2022-06-18T14:59:19.563619Z","iopub.status.idle":"2022-06-18T14:59:19.593549Z","shell.execute_reply":"2022-06-18T14:59:19.592779Z"},"papermill":{"duration":0.686819,"end_time":"2022-06-18T14:59:19.595567","exception":false,"start_time":"2022-06-18T14:59:18.908748","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Good Day!!","metadata":{"execution":{"iopub.execute_input":"2022-06-18T14:59:20.911293Z","iopub.status.busy":"2022-06-18T14:59:20.910253Z","iopub.status.idle":"2022-06-18T14:59:20.914682Z","shell.execute_reply":"2022-06-18T14:59:20.913899Z"},"papermill":{"duration":0.663324,"end_time":"2022-06-18T14:59:20.916754","exception":false,"start_time":"2022-06-18T14:59:20.25343","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}