{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom transformers import BertTokenizer, BertForSequenceClassification, get_linear_schedule_with_warmup\nimport torch\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom torch.utils.data import TensorDataset, DataLoader\nfrom torch.optim import AdamW","metadata":{"execution":{"iopub.status.busy":"2022-08-08T09:37:56.732963Z","iopub.execute_input":"2022-08-08T09:37:56.733339Z","iopub.status.idle":"2022-08-08T09:38:05.365642Z","shell.execute_reply.started":"2022-08-08T09:37:56.733261Z","shell.execute_reply":"2022-08-08T09:38:05.364495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = BertTokenizer.from_pretrained('../input/huggingface-bert-variants/bert-base-uncased/bert-base-uncased', do_lower_case=True)\ndef sentences_to_ids_pad(sentences):\n    sentences = [f\"[CLS] {sent} [SEP]\" for sent in sentences]\n    sentences_tokenized = [tokenizer.tokenize(sent) for sent in sentences]\n    sentences_ids = [tokenizer.encode(sent, add_special_token=True) for sent in sentences_tokenized]\n    sentences_padded = pad_sequences(sentences_ids, maxlen=256, dtype='long', padding='post')\n    return sentences_padded","metadata":{"execution":{"iopub.status.busy":"2022-08-08T09:38:05.367671Z","iopub.execute_input":"2022-08-08T09:38:05.368713Z","iopub.status.idle":"2022-08-08T09:38:05.444402Z","shell.execute_reply.started":"2022-08-08T09:38:05.368675Z","shell.execute_reply":"2022-08-08T09:38:05.443457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = BertForSequenceClassification.from_pretrained('../input/huggingface-bert-variants/bert-base-uncased/bert-base-uncased', num_labels=3)\nmodel.cuda()\nmodel.load_state_dict(torch.load('../input/model-bert/model.pt'))","metadata":{"execution":{"iopub.status.busy":"2022-08-08T09:38:05.445708Z","iopub.execute_input":"2022-08-08T09:38:05.446085Z","iopub.status.idle":"2022-08-08T09:38:20.820752Z","shell.execute_reply.started":"2022-08-08T09:38:05.446049Z","shell.execute_reply":"2022-08-08T09:38:20.819809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = 'cuda' if torch.cuda.is_available() else 'cpu'\n\ntest_df = pd.read_csv('../input/feedback-prize-effectiveness/test.csv')\nids = test_df.discourse_id.values\ntest_sentences = test_df.discourse_text.values[:9]\n\ndef predict(ids, test_sentences):\n    test_sentences = sentences_to_ids_pad(test_sentences)\n    test_sentences = torch.tensor(test_sentences)\n\n    model.eval()\n\n    test_sentences = test_sentences.to(device)\n    with torch.no_grad():\n            preds = model(test_sentences)['logits'].detach().cpu().numpy()\n            \n    return preds\n\npreds = predict(ids, test_sentences)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T09:38:20.823267Z","iopub.execute_input":"2022-08-08T09:38:20.823629Z","iopub.status.idle":"2022-08-08T09:38:21.696186Z","shell.execute_reply.started":"2022-08-08T09:38:20.823593Z","shell.execute_reply":"2022-08-08T09:38:21.695234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.read_csv('/kaggle/input/feedback-prize-effectiveness/sample_submission.csv')\nout_df = pd.DataFrame({'discourse_id': ids})\n#preds = np.append(preds, np.zeros(((len(ids) - preds.shape[0]) * 3))).reshape(len(ids), 3)\n\nall_preds = np.zeros((len(ids), 3))\nif type(preds[0][0]) == np.ndarray:\n    all_preds[0][0] = preds[0][0][0]\nelse:\n    all_preds[0][0] = preds[0][0]\n#all_preds[0][1] = preds[0][1]\n#all_preds[0][2] = preds[0][2]\n#for i, chunk in enumerate(preds):\n#    all_preds[i] = chunk\n\nout_df[['Ineffective', 'Adequate', 'Effective']] = all_preds\nout_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T09:39:55.213540Z","iopub.execute_input":"2022-08-08T09:39:55.213998Z","iopub.status.idle":"2022-08-08T09:39:55.230012Z","shell.execute_reply.started":"2022-08-08T09:39:55.213964Z","shell.execute_reply":"2022-08-08T09:39:55.228850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}