{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"feedback prize \n--\nPredict probs of effectiveness based on discourse_text & type ","metadata":{}},{"cell_type":"code","source":"import os\n\nimport pandas as pd\nimport numpy as np\n\nfrom gensim.models import Word2Vec\nfrom gensim.parsing.preprocessing import preprocess_string","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:33:23.587144Z","iopub.execute_input":"2022-07-09T17:33:23.587547Z","iopub.status.idle":"2022-07-09T17:33:23.593068Z","shell.execute_reply.started":"2022-07-09T17:33:23.587514Z","shell.execute_reply":"2022-07-09T17:33:23.592063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Define some functions Kaggle notebook does not support","metadata":{}},{"cell_type":"code","source":"from gensim import utils\n\ndef lower_to_unicode(text, encoding='utf8', errors='strict'):\n    return utils.to_unicode(text.lower(), encoding, errors)\n\ndef get_mean_vector(self, keys, weights=None, pre_normalize=True, post_normalize=False, ignore_missing=True):\n        if len(keys) == 0:\n            raise ValueError(\"cannot compute mean with no input\")\n        if isinstance(weights, list):\n            weights = np.array(weights)\n        if weights is None:\n            weights = np.ones(len(keys))\n        if len(keys) != weights.shape[0]:  # weights is a 1-D numpy array\n            raise ValueError(\n                \"keys and weights array must have same number of elements\"\n            )\n\n        mean = np.zeros(self.vector_size, self.vectors.dtype)\n\n        total_weight = 0\n        for idx, key in enumerate(keys):\n            if isinstance(key, np.ndarray):\n                mean += weights[idx] * key\n                total_weight += abs(weights[idx])\n            elif self.__contains__(key):\n                vec = self.get_vector(key, norm=pre_normalize)\n                mean += weights[idx] * vec\n                total_weight += abs(weights[idx])\n            elif not ignore_missing:\n                raise KeyError(f\"Key '{key}' not present in vocabulary\")\n\n        if(total_weight > 0):\n            mean = mean / total_weight\n        if post_normalize:\n            mean = matutils.unitvec(mean).astype(REAL)\n        return mean","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:28:45.813362Z","iopub.execute_input":"2022-07-09T17:28:45.813989Z","iopub.status.idle":"2022-07-09T17:28:45.826504Z","shell.execute_reply.started":"2022-07-09T17:28:45.813955Z","shell.execute_reply":"2022-07-09T17:28:45.825413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root_path = '/kaggle/input/feedback-prize-effectiveness/'","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:28:45.827726Z","iopub.execute_input":"2022-07-09T17:28:45.828322Z","iopub.status.idle":"2022-07-09T17:28:45.840916Z","shell.execute_reply.started":"2022-07-09T17:28:45.828289Z","shell.execute_reply":"2022-07-09T17:28:45.840164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(os.path.join(root_path, 'train.csv'))\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:28:45.844289Z","iopub.execute_input":"2022-07-09T17:28:45.844701Z","iopub.status.idle":"2022-07-09T17:28:46.194670Z","shell.execute_reply.started":"2022-07-09T17:28:45.844663Z","shell.execute_reply":"2022-07-09T17:28:46.193505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['discourse_type']","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:28:46.196181Z","iopub.execute_input":"2022-07-09T17:28:46.196625Z","iopub.status.idle":"2022-07-09T17:28:46.208830Z","shell.execute_reply.started":"2022-07-09T17:28:46.196579Z","shell.execute_reply":"2022-07-09T17:28:46.207635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['discourse_effectiveness']","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:28:46.210621Z","iopub.execute_input":"2022-07-09T17:28:46.211397Z","iopub.status.idle":"2022-07-09T17:28:46.220521Z","shell.execute_reply.started":"2022-07-09T17:28:46.211363Z","shell.execute_reply":"2022-07-09T17:28:46.219772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cate_cols = ['discourse_id', 'essay_id', 'discourse_type', 'discourse_effectiveness']\n\nfor col in cate_cols:\n    print(col, len(df[col].unique()))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:28:46.222079Z","iopub.execute_input":"2022-07-09T17:28:46.223201Z","iopub.status.idle":"2022-07-09T17:28:46.249195Z","shell.execute_reply.started":"2022-07-09T17:28:46.223154Z","shell.execute_reply":"2022-07-09T17:28:46.248069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**preprocess_string** function\n\n* remove_stopwords \n* strip numeric\n* strip short\n* stem text","metadata":{}},{"cell_type":"code","source":"example_string = df['discourse_text'][0]\nexample_string","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:28:46.250654Z","iopub.execute_input":"2022-07-09T17:28:46.251163Z","iopub.status.idle":"2022-07-09T17:28:46.258314Z","shell.execute_reply.started":"2022-07-09T17:28:46.251133Z","shell.execute_reply":"2022-07-09T17:28:46.257093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocess_string(example_string)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:28:46.259919Z","iopub.execute_input":"2022-07-09T17:28:46.260697Z","iopub.status.idle":"2022-07-09T17:28:46.268867Z","shell.execute_reply.started":"2022-07-09T17:28:46.260650Z","shell.execute_reply":"2022-07-09T17:28:46.267779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**w2v train**","metadata":{}},{"cell_type":"code","source":"from gensim.parsing.preprocessing import remove_stopwords, strip_multiple_whitespaces, strip_short, strip_numeric, stem_text, strip_tags, strip_punctuation\n\nmodel = Word2Vec(sentences=df['discourse_text'].apply(func=preprocess_string, filters=[lower_to_unicode, strip_multiple_whitespaces, strip_numeric, strip_tags, strip_punctuation, strip_short]),\n                 vector_size=100, \n                 window=5, \n                 min_count=1, \n                 workers=16)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:28:46.274420Z","iopub.execute_input":"2022-07-09T17:28:46.275103Z","iopub.status.idle":"2022-07-09T17:28:58.310550Z","shell.execute_reply.started":"2022-07-09T17:28:46.275059Z","shell.execute_reply":"2022-07-09T17:28:58.308925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, word in enumerate(model.wv.index_to_key):\n    if index == 10:\n        break\n    print(f\"word #{index}/{len(model.wv.index_to_key)} is {word}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:28:58.313311Z","iopub.execute_input":"2022-07-09T17:28:58.313984Z","iopub.status.idle":"2022-07-09T17:28:58.322376Z","shell.execute_reply.started":"2022-07-09T17:28:58.313939Z","shell.execute_reply":"2022-07-09T17:28:58.321264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**get one-hot vector** & **get mean_vector of documents**","metadata":{}},{"cell_type":"code","source":"use_pretrained = False\nX = []\nfor token_list in df['discourse_text'].apply(func=preprocess_string, filters=[lower_to_unicode, strip_multiple_whitespaces, strip_numeric, strip_tags, strip_punctuation, strip_short]):\n    if token_list:\n        if not use_pretrained:\n            mean_vec = get_mean_vector(model.wv, token_list)\n        else:\n            mean_vec = get_mean_vector(glove_model, token_list)\n    X.append(mean_vec)\nX = np.array(X)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:28:58.323811Z","iopub.execute_input":"2022-07-09T17:28:58.324769Z","iopub.status.idle":"2022-07-09T17:29:18.645347Z","shell.execute_reply.started":"2022-07-09T17:28:58.324737Z","shell.execute_reply":"2022-07-09T17:29:18.643821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:29:18.647343Z","iopub.execute_input":"2022-07-09T17:29:18.647797Z","iopub.status.idle":"2022-07-09T17:29:18.655218Z","shell.execute_reply.started":"2022-07-09T17:29:18.647665Z","shell.execute_reply":"2022-07-09T17:29:18.654110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['discourse_type'] = df['discourse_type'].astype('category') \ndf['discourse_effectiveness'] = df['discourse_effectiveness'].astype('category') \n\nX_one_hot = pd.concat(\n    [df, pd.get_dummies(df['discourse_type'], prefix='type')], axis=1\n    ).drop(\n    ['discourse_type', 'discourse_id', 'essay_id', 'discourse_text', 'discourse_effectiveness'], axis=1\n    )\n\ntrain_y = pd.get_dummies(df['discourse_effectiveness']).to_numpy()\ntrain_x = np.concatenate([X, X_one_hot.to_numpy()], axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:29:18.656956Z","iopub.execute_input":"2022-07-09T17:29:18.657360Z","iopub.status.idle":"2022-07-09T17:29:18.692829Z","shell.execute_reply.started":"2022-07-09T17:29:18.657326Z","shell.execute_reply":"2022-07-09T17:29:18.691852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['discourse_effectiveness'][:10]","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:29:18.694963Z","iopub.execute_input":"2022-07-09T17:29:18.695295Z","iopub.status.idle":"2022-07-09T17:29:18.703496Z","shell.execute_reply.started":"2022-07-09T17:29:18.695267Z","shell.execute_reply":"2022-07-09T17:29:18.702606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_y[:10]","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:29:18.704935Z","iopub.execute_input":"2022-07-09T17:29:18.705301Z","iopub.status.idle":"2022-07-09T17:29:18.715250Z","shell.execute_reply.started":"2022-07-09T17:29:18.705272Z","shell.execute_reply":"2022-07-09T17:29:18.714060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_valid, y_train, y_valid = train_test_split(train_x, train_y, test_size=0.2, random_state=42, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:29:18.716250Z","iopub.execute_input":"2022-07-09T17:29:18.716571Z","iopub.status.idle":"2022-07-09T17:29:18.840397Z","shell.execute_reply.started":"2022-07-09T17:29:18.716543Z","shell.execute_reply":"2022-07-09T17:29:18.839375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:29:18.841654Z","iopub.execute_input":"2022-07-09T17:29:18.842060Z","iopub.status.idle":"2022-07-09T17:29:18.849494Z","shell.execute_reply.started":"2022-07-09T17:29:18.842019Z","shell.execute_reply":"2022-07-09T17:29:18.848602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**build model**","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\n\nclass SimpleNN(nn.Module):\n    def __init__(self, input_dim):\n        super(SimpleNN, self).__init__()\n        self.fc1 = nn.Linear(input_dim, input_dim * 2)  \n        self.fc2 = nn.Linear(input_dim * 2, input_dim )\n        self.fc3 = nn.Linear(input_dim , input_dim // 2)\n        self.fc4 = nn.Linear(input_dim // 2, 3)\n\n    def forward(self, x):\n        x = F.relu(self.fc1(x))\n        x = F.relu(self.fc2(x))\n        x = F.relu(self.fc3(x))\n        x = F.softmax(self.fc4(x), dim=-1)\n        return x\n\n\nsimple_nn = SimpleNN(input_dim = train_x.shape[1])\nprint(simple_nn)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:29:18.850557Z","iopub.execute_input":"2022-07-09T17:29:18.850828Z","iopub.status.idle":"2022-07-09T17:29:20.410040Z","shell.execute_reply.started":"2022-07-09T17:29:18.850803Z","shell.execute_reply":"2022-07-09T17:29:20.408874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if torch.cuda.is_available():\n    device = torch.device('cuda')\nelse:\n    device = torch.device('cpu')\ndevice","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:29:20.411829Z","iopub.execute_input":"2022-07-09T17:29:20.412615Z","iopub.status.idle":"2022-07-09T17:29:20.421115Z","shell.execute_reply.started":"2022-07-09T17:29:20.412574Z","shell.execute_reply":"2022-07-09T17:29:20.419806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"simple_nn(torch.Tensor(train_x[0:3]))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:29:20.423263Z","iopub.execute_input":"2022-07-09T17:29:20.423729Z","iopub.status.idle":"2022-07-09T17:29:20.446962Z","shell.execute_reply.started":"2022-07-09T17:29:20.423686Z","shell.execute_reply":"2022-07-09T17:29:20.446076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_tensor = torch.Tensor(X_train).to(device)\ny_train_tensor = torch.Tensor(y_train).to(device)\n\nX_valid_tensor = torch.Tensor(X_valid).to(device)\ny_valid_tensor = torch.Tensor(y_valid).to(device)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:29:20.448365Z","iopub.execute_input":"2022-07-09T17:29:20.448645Z","iopub.status.idle":"2022-07-09T17:29:20.454263Z","shell.execute_reply.started":"2022-07-09T17:29:20.448611Z","shell.execute_reply":"2022-07-09T17:29:20.453360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def test_loop(X, y, model, loss_fn):\n    test_loss, correct = 0, 0\n\n    pred = model(X)\n    test_loss += loss_fn(pred, y).item()\n    correct += (pred.argmax(1) == y.argmax(1)).type(torch.float).sum().item()\n        \n    correct /= len(y)\n    print(f\"Validation Error: \\n Accuracy: {(100*correct):>0.1f}%, Loss: {test_loss:>8f} \\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:29:20.455747Z","iopub.execute_input":"2022-07-09T17:29:20.456100Z","iopub.status.idle":"2022-07-09T17:29:20.463615Z","shell.execute_reply.started":"2022-07-09T17:29:20.456069Z","shell.execute_reply":"2022-07-09T17:29:20.462736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epoch_n = 1000\ncriterion = nn.BCELoss()\nsimple_nn = SimpleNN(input_dim = train_x.shape[1]).to(device)\n\nfor ep in range(epoch_n + 1):\n    # create your optimizer\n    optimizer = optim.Adam(simple_nn.parameters(), lr=0.001)\n\n    # in your training loop:\n    optimizer.zero_grad()   # zero the gradient buffers\n    output = simple_nn(X_train_tensor)\n    loss = criterion(output, y_train_tensor)\n    if ep % 200 == 0:\n        with torch.no_grad():\n            accuracy = ((output.argmax(1) == y_train_tensor.argmax(1)).type(torch.float).mean().item()) * 100.\n            print('Train \\n Epochs: {}, Loss: {:>8f}, Accuracy: {:>.3f}%'.format(ep, loss.item(), accuracy))\n            test_loop(X_valid_tensor, y_valid_tensor, simple_nn, criterion)\n    loss.backward()\n    optimizer.step()    # Does the update","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:29:20.465200Z","iopub.execute_input":"2022-07-09T17:29:20.465939Z","iopub.status.idle":"2022-07-09T17:31:22.527484Z","shell.execute_reply.started":"2022-07-09T17:29:20.465893Z","shell.execute_reply":"2022-07-09T17:31:22.526081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**save test_submission**","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv(os.path.join(root_path, 'test.csv'))\ntest_df","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:31:22.528894Z","iopub.execute_input":"2022-07-09T17:31:22.529236Z","iopub.status.idle":"2022-07-09T17:31:22.549321Z","shell.execute_reply.started":"2022-07-09T17:31:22.529205Z","shell.execute_reply":"2022-07-09T17:31:22.548241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"use_pretrained = False\ntest_X = []\nfor token_list in test_df['discourse_text'].apply(func=preprocess_string, \n                                                  filters=[lower_to_unicode, \n                                                           strip_multiple_whitespaces,\n                                                           strip_numeric, \n                                                           strip_tags, \n                                                           strip_punctuation, \n                                                           strip_short]):\n    if token_list:\n        if not use_pretrained:\n            mean_vec = get_mean_vector(model.wv, token_list)\n        else:\n            mean_vec = glove_model.get_mean_vector(token_list)\n    test_X.append(mean_vec)\ntest_X = np.array(test_X)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:31:22.550913Z","iopub.execute_input":"2022-07-09T17:31:22.551361Z","iopub.status.idle":"2022-07-09T17:31:22.565618Z","shell.execute_reply.started":"2022-07-09T17:31:22.551320Z","shell.execute_reply":"2022-07-09T17:31:22.564449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['discourse_type'] = test_df['discourse_type'].astype('category') \ntest_df['discourse_type'] = test_df['discourse_type'].cat.add_categories(['Counterclaim', 'Rebuttal'])","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:31:22.567124Z","iopub.execute_input":"2022-07-09T17:31:22.567551Z","iopub.status.idle":"2022-07-09T17:31:22.577580Z","shell.execute_reply.started":"2022-07-09T17:31:22.567508Z","shell.execute_reply":"2022-07-09T17:31:22.576607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_one_hot = pd.concat(\n    [test_df, pd.get_dummies(test_df['discourse_type'], prefix='type')], axis=1\n    ).drop(\n    ['discourse_type', 'discourse_id', 'essay_id', 'discourse_text'], axis=1\n    )\ntest_x = np.concatenate([test_X, X_test_one_hot.to_numpy()], axis=1)\ntest_x.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:31:22.582740Z","iopub.execute_input":"2022-07-09T17:31:22.583173Z","iopub.status.idle":"2022-07-09T17:31:22.593476Z","shell.execute_reply.started":"2022-07-09T17:31:22.583109Z","shell.execute_reply":"2022-07-09T17:31:22.592382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_X = torch.Tensor(test_x).to(device)\n\noutput = simple_nn(test_X).to(torch.device('cpu')).detach().numpy()\n\nsubmission_df = test_df.drop(['essay_id', 'discourse_text', 'discourse_type'], axis=1)\nsubmission_df = pd.concat([submission_df, pd.DataFrame(output)], axis=1)\nsubmission_df = submission_df.rename(columns= {0: 'Adequate', 1: 'Effective', 2: 'Ineffective'})\nsubmission_df = submission_df[['discourse_id', 'Ineffective', 'Adequate', 'Effective']]","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:31:22.595315Z","iopub.execute_input":"2022-07-09T17:31:22.596029Z","iopub.status.idle":"2022-07-09T17:31:22.610477Z","shell.execute_reply.started":"2022-07-09T17:31:22.595969Z","shell.execute_reply":"2022-07-09T17:31:22.609444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:31:22.612093Z","iopub.execute_input":"2022-07-09T17:31:22.612526Z","iopub.status.idle":"2022-07-09T17:31:22.626271Z","shell.execute_reply.started":"2022-07-09T17:31:22.612485Z","shell.execute_reply":"2022-07-09T17:31:22.625412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T17:31:22.627571Z","iopub.execute_input":"2022-07-09T17:31:22.628100Z","iopub.status.idle":"2022-07-09T17:31:22.640142Z","shell.execute_reply.started":"2022-07-09T17:31:22.628058Z","shell.execute_reply":"2022-07-09T17:31:22.639249Z"},"trusted":true},"execution_count":null,"outputs":[]}]}