{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-04T14:20:52.826590Z","iopub.execute_input":"2022-08-04T14:20:52.827166Z","iopub.status.idle":"2022-08-04T14:20:53.806805Z","shell.execute_reply.started":"2022-08-04T14:20:52.827046Z","shell.execute_reply":"2022-08-04T14:20:53.805398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/feedback-prize-effectiveness/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:20:53.808727Z","iopub.execute_input":"2022-08-04T14:20:53.810066Z","iopub.status.idle":"2022-08-04T14:20:54.001913Z","shell.execute_reply.started":"2022-08-04T14:20:53.810002Z","shell.execute_reply":"2022-08-04T14:20:54.000792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:20:54.003946Z","iopub.execute_input":"2022-08-04T14:20:54.004863Z","iopub.status.idle":"2022-08-04T14:20:54.023510Z","shell.execute_reply.started":"2022-08-04T14:20:54.004812Z","shell.execute_reply":"2022-08-04T14:20:54.022107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/feedback-prize-effectiveness/test.csv')\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:20:54.026844Z","iopub.execute_input":"2022-08-04T14:20:54.027290Z","iopub.status.idle":"2022-08-04T14:20:54.043374Z","shell.execute_reply.started":"2022-08-04T14:20:54.027253Z","shell.execute_reply":"2022-08-04T14:20:54.042026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ***EDA***","metadata":{}},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:20:54.044802Z","iopub.execute_input":"2022-08-04T14:20:54.045976Z","iopub.status.idle":"2022-08-04T14:20:54.056424Z","shell.execute_reply.started":"2022-08-04T14:20:54.045922Z","shell.execute_reply":"2022-08-04T14:20:54.054720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:20:54.059023Z","iopub.execute_input":"2022-08-04T14:20:54.060304Z","iopub.status.idle":"2022-08-04T14:20:54.154631Z","shell.execute_reply.started":"2022-08-04T14:20:54.060226Z","shell.execute_reply":"2022-08-04T14:20:54.152932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nax = sns.countplot(x=\"discourse_effectiveness\", data=df_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:20:54.157017Z","iopub.execute_input":"2022-08-04T14:20:54.157458Z","iopub.status.idle":"2022-08-04T14:20:54.984071Z","shell.execute_reply.started":"2022-08-04T14:20:54.157421Z","shell.execute_reply":"2022-08-04T14:20:54.982472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col = 'discourse_effectiveness'\nfig, ax = plt.subplots(nrows = 1, ncols = 2, figsize = (18, 6))\nfig.suptitle(col, fontsize = 16)\nsns.countplot(data = df_train,\n                  x = col,\n                  ax = ax[0],\n                  palette= 'tab10',\n                  order =  df_train[col].value_counts().index)\nax[0].set_xlabel('')\n\npie_cmap = plt.get_cmap('tab10')\nnormalize = lambda x: (x - np.min(x)) / (np.max(x) - np.min(x)) \ndf_train[col].value_counts().plot.pie(autopct='%1.1f%%',\n                                      textprops={'fontsize': 12},\n                                      ax=ax[1],\n                                      colors = pie_cmap(normalize(df_train[col].value_counts())))\nax[1].set_ylabel('')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:20:54.985529Z","iopub.execute_input":"2022-08-04T14:20:54.985932Z","iopub.status.idle":"2022-08-04T14:20:55.368883Z","shell.execute_reply.started":"2022-08-04T14:20:54.985881Z","shell.execute_reply":"2022-08-04T14:20:55.367017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col = 'discourse_type'\nfig, ax = plt.subplots(nrows = 1, ncols = 2, figsize = (18, 6))\nfig.suptitle(col, fontsize = 16)\nsns.countplot(data = df_train,\n                  x = col,\n                  ax = ax[0],\n                  palette= 'tab10',\n                  order =  df_train[col].value_counts().index)\nax[0].set_xlabel('')\n\npie_cmap = plt.get_cmap('tab10')\nnormalize = lambda x: (x - np.min(x)) / (np.max(x) - np.min(x)) \ndf_train[col].value_counts().plot.pie(autopct='%1.1f%%',\n                                      textprops={'fontsize': 12},\n                                      ax=ax[1],\n                                      colors = pie_cmap(normalize(df_train[col].value_counts())))\nax[1].set_ylabel('')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:20:55.371081Z","iopub.execute_input":"2022-08-04T14:20:55.371879Z","iopub.status.idle":"2022-08-04T14:20:55.767920Z","shell.execute_reply.started":"2022-08-04T14:20:55.371814Z","shell.execute_reply":"2022-08-04T14:20:55.766987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(rc={'figure.figsize':(18,8)})\nsns.countplot(data = df_train,\n              x = 'discourse_type',\n              hue ='discourse_effectiveness',\n              palette = 'tab10')\nplt.legend(loc = 'best', prop={'size': 14})\nplt.title('Discourse Type & Discourse Effectiveness', size = 14)\nplt.yticks(size = 14)\nplt.xticks(size = 14)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:20:55.769464Z","iopub.execute_input":"2022-08-04T14:20:55.770183Z","iopub.status.idle":"2022-08-04T14:20:56.183274Z","shell.execute_reply.started":"2022-08-04T14:20:55.770132Z","shell.execute_reply":"2022-08-04T14:20:56.181995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://stackoverflow.com/questions/41409570/cant-install-wordcloud-in-python-anaconda#:~:text=1)%20Download%20the%20latest%20version,or%20sudo%20pip%20install%20wordcloud%20.)\n# https://www.geeksforgeeks.org/generating-word-cloud-python/\nfrom wordcloud import WordCloud, STOPWORDS\ncomment_words = ''\nstopwords = set(STOPWORDS)\nfor val in df_train['discourse_text']:\n     \n    # typecaste each val to string\n    val = str(val)\n \n    # split the value\n    tokens = val.split()\n     \n    # Converts each token into lowercase\n    for i in range(len(tokens)):\n        tokens[i] = tokens[i].lower()\n     \n    comment_words += \" \".join(tokens)+\" \"\n \nwordcloud = WordCloud(width = 800, height = 800,\n                background_color ='white',\n                stopwords = stopwords,\n                min_font_size = 10).generate(comment_words)\n \n# plot the WordCloud image                      \nplt.figure(figsize = (8, 8), facecolor = None)\nplt.imshow(wordcloud)\nplt.axis(\"off\")\nplt.tight_layout(pad = 0)\n \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:20:56.185106Z","iopub.execute_input":"2022-08-04T14:20:56.186289Z","iopub.status.idle":"2022-08-04T14:21:03.616186Z","shell.execute_reply.started":"2022-08-04T14:20:56.186234Z","shell.execute_reply":"2022-08-04T14:21:03.614999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import Counter\nfrom nltk.corpus import stopwords\ndef wordBarGraphFunction(df,column,title):\n    topic_words = [ z.lower() for y in\n                       [ x.split() for x in df[column] if isinstance(x, str)]\n                       for z in y]\n    word_count_dict = dict(Counter(topic_words))\n    popular_words = sorted(word_count_dict, key = word_count_dict.get, reverse = True)\n    popular_words_nonstop = [w for w in popular_words if w not in stopwords.words(\"english\")]\n    plt.barh(range(50), [word_count_dict[w] for w in reversed(popular_words_nonstop[0:50])])\n    plt.yticks([x + 0.5 for x in range(50)], reversed(popular_words_nonstop[0:50]))\n    plt.title(title)\n    plt.show()\nplt.figure(figsize=(10,10))\nwordBarGraphFunction(df_train,'discourse_text',\"Popular Words\")","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:21:03.617730Z","iopub.execute_input":"2022-08-04T14:21:03.619007Z","iopub.status.idle":"2022-08-04T14:21:12.397783Z","shell.execute_reply.started":"2022-08-04T14:21:03.618957Z","shell.execute_reply":"2022-08-04T14:21:12.396281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def CommonWords(data, col, numWords, title, bar_limit = 20):\n    \n    topic_words = [z.lower() for y in [x.split() for x in data[col] \n                                       if isinstance(x, str)] for z in y]\n    word_count_dict = dict(Counter(topic_words))\n    popular_words = sorted(word_count_dict, key = word_count_dict.get, reverse = True)\n    popular_words_nonstop = [word for word in popular_words if word not in stopwords.words(\"english\")]\n    word_string=str(popular_words_nonstop)\n    wordcloud = WordCloud(stopwords=STOPWORDS,\n                          background_color='white',\n                          max_words=numWords,\n                          width= 500,\n                          height= 500,\n                          colormap = 'tab10'\n                         ).generate(word_string)\n    \n    fig, (ax1, ax2) = plt.subplots(nrows = 1, ncols= 2, figsize = (18, 8))\n    fig.suptitle(title, fontsize=15, y = 0.92)\n    ax1.imshow(wordcloud)\n    ax1.axis('off')\n    \n    bar_cmap = plt.get_cmap(\"tab10\")\n    reversed_popular_words_nonstop = [word_count_dict[w] for w in reversed(popular_words_nonstop[0:bar_limit])]\n    normalize = lambda x: (x - np.min(x)) / (np.max(x) - np.min(x))\n    ax2.barh(range(bar_limit), [word_count_dict[w] for w in reversed(popular_words_nonstop[0:bar_limit])],\n             color=bar_cmap(normalize(reversed_popular_words_nonstop)))\n    ax2.set_yticks([x + 0.5 for x in range(bar_limit)], reversed(popular_words_nonstop[0:bar_limit]))\n    plt.show()\n\nfor dis_eff in df_train['discourse_effectiveness'].unique():\n    sub_df = df_train.loc[df_train['discourse_effectiveness'] == dis_eff]\n    title = f'Most Common Words for Discourse Effectiveness: {dis_eff}'\n    CommonWords(sub_df,'discourse_text',1000, title)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:21:12.402394Z","iopub.execute_input":"2022-08-04T14:21:12.403763Z","iopub.status.idle":"2022-08-04T14:21:31.089596Z","shell.execute_reply.started":"2022-08-04T14:21:12.403714Z","shell.execute_reply":"2022-08-04T14:21:31.088619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ***Text Preprocessing***","metadata":{}},{"cell_type":"code","source":"# https://stackoverflow.com/a/47091490/4084039\nimport re\n# changing words which contain short forms like change won't to will not and etc\ndef decontracted(phrase):\n    \n    phrase = re.sub(r\"won't\", \"will not\", phrase)\n    phrase = re.sub(r\"can\\'t\", \"can not\", phrase)\n    \n    phrase = re.sub(r\"n\\'t\", \" not\", phrase)\n    phrase = re.sub(r\"\\'re\", \" are\", phrase)\n    phrase = re.sub(r\"\\'s\", \" is\", phrase)\n    phrase = re.sub(r\"\\'d\", \" would\", phrase)\n    phrase = re.sub(r\"\\'ll\", \" will\", phrase)\n    phrase = re.sub(r\"\\'t\", \" not\", phrase)\n    phrase = re.sub(r\"\\'ve\", \" have\", phrase)\n    phrase = re.sub(r\"\\'m\", \" am\", phrase)\n    return phrase","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:21:31.091024Z","iopub.execute_input":"2022-08-04T14:21:31.092289Z","iopub.status.idle":"2022-08-04T14:21:31.100271Z","shell.execute_reply.started":"2022-08-04T14:21:31.092244Z","shell.execute_reply":"2022-08-04T14:21:31.098861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# removing the words from the stop words list: 'no', 'nor', 'not'\nstopwords= ['i', 'me', 'my', 'myself', 'we', 'our', 'ours', 'ourselves', 'you', \"you're\", \"you've\",\\\n            \"you'll\", \"you'd\", 'your', 'yours', 'yourself', 'yourselves', 'he', 'him', 'his', 'himself', \\\n            'she', \"she's\", 'her', 'hers', 'herself', 'it', \"it's\", 'its', 'itself', 'they', 'them', 'their',\\\n            'theirs', 'themselves', 'what', 'which', 'who', 'whom', 'this', 'that', \"that'll\", 'these', 'those', \\\n            'am', 'is', 'are', 'was', 'were', 'be', 'been', 'being', 'have', 'has', 'had', 'having', 'do', 'does', \\\n            'did', 'doing', 'a', 'an', 'the', 'and', 'but', 'if', 'or', 'because', 'as', 'until', 'while', 'of', \\\n            'at', 'by', 'for', 'with', 'about', 'against', 'between', 'into', 'through', 'during', 'before', 'after',\\\n            'above', 'below', 'to', 'from', 'up', 'down', 'in', 'out', 'on', 'off', 'over', 'under', 'again', 'further',\\\n            'then', 'once', 'here', 'there', 'when', 'where', 'why', 'how', 'all', 'any', 'both', 'each', 'few', 'more',\\\n            'most', 'other', 'some', 'such', 'only', 'own', 'same', 'so', 'than', 'too', 'very', \\\n            's', 't', 'can', 'will', 'just', 'don', \"don't\", 'should', \"should've\", 'now', 'd', 'll', 'm', 'o', 're', \\\n            've', 'y', 'ain', 'aren', \"aren't\", 'couldn', \"couldn't\", 'didn', \"didn't\", 'doesn', \"doesn't\", 'hadn',\\\n            \"hadn't\", 'hasn', \"hasn't\", 'haven', \"haven't\", 'isn', \"isn't\", 'ma', 'mightn', \"mightn't\", 'mustn',\\\n            \"mustn't\", 'needn', \"needn't\", 'shan', \"shan't\", 'shouldn', \"shouldn't\", 'wasn', \"wasn't\", 'weren', \"weren't\", \\\n            'won', \"won't\", 'wouldn', \"wouldn't\"]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:21:31.101867Z","iopub.execute_input":"2022-08-04T14:21:31.102264Z","iopub.status.idle":"2022-08-04T14:21:31.118967Z","shell.execute_reply.started":"2022-08-04T14:21:31.102231Z","shell.execute_reply":"2022-08-04T14:21:31.117589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# text before preprocessing\nprint(0, df_train['discourse_text'].values[0])\nprint(1, df_train['discourse_text'].values[1])\nprint(2, df_train['discourse_text'].values[2])\nprint(3, df_train['discourse_text'].values[3])\nprint(4, df_train['discourse_text'].values[4])","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:21:31.120607Z","iopub.execute_input":"2022-08-04T14:21:31.122245Z","iopub.status.idle":"2022-08-04T14:21:31.137853Z","shell.execute_reply.started":"2022-08-04T14:21:31.122183Z","shell.execute_reply":"2022-08-04T14:21:31.136426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(0, df_test['discourse_text'].values[0])\nprint(1, df_test['discourse_text'].values[1])\nprint(2, df_test['discourse_text'].values[2])\nprint(3, df_test['discourse_text'].values[3])\nprint(4, df_test['discourse_text'].values[4])","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:21:31.139464Z","iopub.execute_input":"2022-08-04T14:21:31.139981Z","iopub.status.idle":"2022-08-04T14:21:31.153430Z","shell.execute_reply.started":"2022-08-04T14:21:31.139932Z","shell.execute_reply":"2022-08-04T14:21:31.151872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replacing // with empty space and etc if present\n# https://stackoverflow.com/questions/33404752/removing-emojis-from-a-string-in-python\nfrom tqdm import tqdm\ndef preprocess_text(text_data):\n    preprocessed_text = []\n    # tqdm is for printing the status bar\n    for sentance in tqdm(text_data):\n        sent = decontracted(sentance)\n        sent = sent.replace('\\\\r', ' ')\n        sent = sent.replace('\\\\n', ' ')\n        sent = sent.replace('\\\\\"', ' ')\n        sent = re.sub('[^A-Za-z0-9]+', ' ', sent)\n        # https://gist.github.com/sebleier/554280\n        emoji_pattern = re.compile(\"[\"\n        u\"\\U0001F600-\\U0001F64F\"  # emoticons\n        u\"\\U0001F300-\\U0001F5FF\"  # symbols & pictographs\n        u\"\\U0001F680-\\U0001F6FF\"  # transport & map symbols\n        u\"\\U0001F1E0-\\U0001F1FF\"  # flags (iOS)\n                           \"]+\", flags=re.UNICODE)\n        sent = emoji_pattern.sub(r'', sent)\n        sent = ' '.join(e for e in sent.split() if e.lower() not in stopwords)\n        preprocessed_text.append(sent.lower().strip())\n    return preprocessed_text","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:21:31.156052Z","iopub.execute_input":"2022-08-04T14:21:31.156602Z","iopub.status.idle":"2022-08-04T14:21:31.171086Z","shell.execute_reply.started":"2022-08-04T14:21:31.156552Z","shell.execute_reply":"2022-08-04T14:21:31.169634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessed_discourse_text_train = preprocess_text(df_train['discourse_text'].values)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:21:31.172213Z","iopub.execute_input":"2022-08-04T14:21:31.173287Z","iopub.status.idle":"2022-08-04T14:21:36.652749Z","shell.execute_reply.started":"2022-08-04T14:21:31.173242Z","shell.execute_reply":"2022-08-04T14:21:36.651472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessed_discourse_text_test = preprocess_text(df_test['discourse_text'].values)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:21:36.654727Z","iopub.execute_input":"2022-08-04T14:21:36.655095Z","iopub.status.idle":"2022-08-04T14:21:36.667598Z","shell.execute_reply.started":"2022-08-04T14:21:36.655061Z","shell.execute_reply":"2022-08-04T14:21:36.666684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# text after preprocessing\nprint(0, preprocessed_discourse_text_train[0])\nprint(1, preprocessed_discourse_text_train[1])\nprint(2, preprocessed_discourse_text_train[2])\nprint(3, preprocessed_discourse_text_train[3])\nprint(4, preprocessed_discourse_text_train[4])","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:21:36.668917Z","iopub.execute_input":"2022-08-04T14:21:36.669239Z","iopub.status.idle":"2022-08-04T14:21:36.678557Z","shell.execute_reply.started":"2022-08-04T14:21:36.669209Z","shell.execute_reply":"2022-08-04T14:21:36.677174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(0, preprocessed_discourse_text_test[0])\nprint(1, preprocessed_discourse_text_test[1])\nprint(2, preprocessed_discourse_text_test[2])\nprint(3, preprocessed_discourse_text_test[3])\nprint(4, preprocessed_discourse_text_test[4])","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:21:36.681050Z","iopub.execute_input":"2022-08-04T14:21:36.682594Z","iopub.status.idle":"2022-08-04T14:21:36.689794Z","shell.execute_reply.started":"2022-08-04T14:21:36.682546Z","shell.execute_reply":"2022-08-04T14:21:36.688461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = df_train['discourse_effectiveness']\ndic = ['Effective', 'Adequate', 'Ineffective']\nY = []\nfor l in labels:\n    Y.append(dic.index(l))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:21:36.691051Z","iopub.execute_input":"2022-08-04T14:21:36.692509Z","iopub.status.idle":"2022-08-04T14:21:36.715930Z","shell.execute_reply.started":"2022-08-04T14:21:36.692456Z","shell.execute_reply":"2022-08-04T14:21:36.714564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(preprocessed_discourse_text_train, Y, test_size=0.2, stratify=Y, random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:21:36.717697Z","iopub.execute_input":"2022-08-04T14:21:36.718110Z","iopub.status.idle":"2022-08-04T14:21:36.757798Z","shell.execute_reply.started":"2022-08-04T14:21:36.718073Z","shell.execute_reply":"2022-08-04T14:21:36.756238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ***Preparing data for modeling***","metadata":{}},{"cell_type":"code","source":"#tfidf vectorizer\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nvectorizer_tfidf = TfidfVectorizer(min_df=150,ngram_range=(1,4))\nX_tr = vectorizer_tfidf.fit_transform(X_train)\nX_va = vectorizer_tfidf.transform(X_val)\nX_te = vectorizer_tfidf.transform(preprocessed_discourse_text_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:21:36.759682Z","iopub.execute_input":"2022-08-04T14:21:36.760099Z","iopub.status.idle":"2022-08-04T14:21:45.077698Z","shell.execute_reply.started":"2022-08-04T14:21:36.760061Z","shell.execute_reply":"2022-08-04T14:21:45.076683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.model_selection import GridSearchCV\n\nparams = {'max_depth': [1,2,3,4], 'n_estimators': [5, 10, 15, 20]}\n\ngbdt_tfidf = GradientBoostingClassifier(learning_rate=0.1)\n\ngbdtclf_set_1 = GridSearchCV(gbdt_tfidf, params, cv=5, scoring='accuracy', return_train_score=True, n_jobs=-1)\n\ngbdtclf_set_1.fit(X_tr,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:21:45.079489Z","iopub.execute_input":"2022-08-04T14:21:45.080132Z","iopub.status.idle":"2022-08-04T14:24:18.858931Z","shell.execute_reply.started":"2022-08-04T14:21:45.080093Z","shell.execute_reply":"2022-08-04T14:24:18.857461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Best score: ',gbdtclf_set_1.best_score_)\nprint('alpha value with best score: ',gbdtclf_set_1.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:24:18.861573Z","iopub.execute_input":"2022-08-04T14:24:18.862148Z","iopub.status.idle":"2022-08-04T14:24:18.869501Z","shell.execute_reply.started":"2022-08-04T14:24:18.862092Z","shell.execute_reply":"2022-08-04T14:24:18.868187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gbdt_tfidf_best = GradientBoostingClassifier(learning_rate=0.1,max_depth=4,n_estimators=20) \ngbdt_tfidf_best.fit(X_tr,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:24:18.871364Z","iopub.execute_input":"2022-08-04T14:24:18.871852Z","iopub.status.idle":"2022-08-04T14:24:33.095352Z","shell.execute_reply.started":"2022-08-04T14:24:18.871802Z","shell.execute_reply":"2022-08-04T14:24:33.093544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr_pred = gbdt_tfidf_best.predict(X_tr)\nva_pred = gbdt_tfidf_best.predict(X_va)\nte_pred = gbdt_tfidf_best.predict(X_te)\n\nkey = np.array([0,1,2])","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:28:59.179302Z","iopub.execute_input":"2022-08-04T14:28:59.179766Z","iopub.status.idle":"2022-08-04T14:28:59.303705Z","shell.execute_reply.started":"2022-08-04T14:28:59.179731Z","shell.execute_reply":"2022-08-04T14:28:59.302542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#reference -> https://www.kaggle.com/code/petergeorgebeidler/very-simple-tfidf-linreg\nimport scipy\ndef to_logits(preds):\n    key = np.array([0,1,2])\n    out_preds=[]\n    \n    for p in preds:\n        out_preds.append(scipy.special.softmax(-np.abs(p-key)))\n    return out_preds","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:29:02.728384Z","iopub.execute_input":"2022-08-04T14:29:02.728890Z","iopub.status.idle":"2022-08-04T14:29:02.736407Z","shell.execute_reply.started":"2022-08-04T14:29:02.728849Z","shell.execute_reply":"2022-08-04T14:29:02.735027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s = pd.DataFrame(columns=['discourse_id', 'Ineffective', 'Adequate', 'Effective'])\nk = np.array([0,1,2])\ntest_preds_logits = to_logits(te_pred)\n\nfor i, row in df_test.iterrows():\n    p = test_preds_logits[i]\n    s.loc[i] = [row['discourse_id'], p[0], p[1], p[2]]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:38:03.169656Z","iopub.execute_input":"2022-08-04T14:38:03.171109Z","iopub.status.idle":"2022-08-04T14:38:04.670122Z","shell.execute_reply.started":"2022-08-04T14:38:03.171057Z","shell.execute_reply":"2022-08-04T14:38:04.669149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:38:11.279097Z","iopub.execute_input":"2022-08-04T14:38:11.279518Z","iopub.status.idle":"2022-08-04T14:38:11.289276Z","shell.execute_reply.started":"2022-08-04T14:38:11.279484Z","shell.execute_reply":"2022-08-04T14:38:11.288139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}