{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-25T21:37:55.447409Z","iopub.execute_input":"2022-07-25T21:37:55.448582Z","iopub.status.idle":"2022-07-25T21:37:55.454243Z","shell.execute_reply.started":"2022-07-25T21:37:55.448538Z","shell.execute_reply":"2022-07-25T21:37:55.453069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/feedback-prize-effectiveness/train.csv')\ntest = pd.read_csv('../input/feedback-prize-effectiveness/test.csv')\nsubmission = pd.read_csv('../input/feedback-prize-effectiveness/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:37:55.488889Z","iopub.execute_input":"2022-07-25T21:37:55.491175Z","iopub.status.idle":"2022-07-25T21:37:55.684420Z","shell.execute_reply.started":"2022-07-25T21:37:55.491115Z","shell.execute_reply":"2022-07-25T21:37:55.683267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:37:55.687045Z","iopub.execute_input":"2022-07-25T21:37:55.687491Z","iopub.status.idle":"2022-07-25T21:37:55.706209Z","shell.execute_reply.started":"2022-07-25T21:37:55.687446Z","shell.execute_reply":"2022-07-25T21:37:55.704784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:37:55.709288Z","iopub.execute_input":"2022-07-25T21:37:55.710419Z","iopub.status.idle":"2022-07-25T21:37:55.726307Z","shell.execute_reply.started":"2022-07-25T21:37:55.710371Z","shell.execute_reply":"2022-07-25T21:37:55.724982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"discourse_effectivenessを推測するのが、このコンペの目的になる。\n\nちなみにdiscourseは論文の意味。","metadata":{}},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:37:55.729351Z","iopub.execute_input":"2022-07-25T21:37:55.730089Z","iopub.status.idle":"2022-07-25T21:37:55.746520Z","shell.execute_reply.started":"2022-07-25T21:37:55.730043Z","shell.execute_reply":"2022-07-25T21:37:55.745174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['discourse_effectiveness'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:37:55.748763Z","iopub.execute_input":"2022-07-25T21:37:55.749890Z","iopub.status.idle":"2022-07-25T21:37:55.764382Z","shell.execute_reply.started":"2022-07-25T21:37:55.749831Z","shell.execute_reply":"2022-07-25T21:37:55.762881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"論文への評価は、下記の3つ\n\n* Effective\n* Adequate\n* Ineffective\n\nEffective>Adequate>Ineffectiveの順で評価されている","metadata":{}},{"cell_type":"markdown","source":"discourse_idはすべてuniqueかを確認する","metadata":{}},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:37:55.768329Z","iopub.execute_input":"2022-07-25T21:37:55.769273Z","iopub.status.idle":"2022-07-25T21:37:55.906074Z","shell.execute_reply.started":"2022-07-25T21:37:55.769224Z","shell.execute_reply":"2022-07-25T21:37:55.904840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"気づき\n* discourse_idはすべてunique。\n* なので、それぞれのdiscourse_idがどのような評価をされているかを予測(分類)する必要がある。\n* discourse_textのいくつかは重複が存在している\n* essay_idでuniqueなのはかなり少ない。essay_idが何なのかがよくわかっていない。","metadata":{}},{"cell_type":"markdown","source":"essay_idが何なのかを把握する。","metadata":{}},{"cell_type":"code","source":"train[['essay_id', 'discourse_id']].head()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:37:55.908122Z","iopub.execute_input":"2022-07-25T21:37:55.908456Z","iopub.status.idle":"2022-07-25T21:37:55.925681Z","shell.execute_reply.started":"2022-07-25T21:37:55.908425Z","shell.execute_reply":"2022-07-25T21:37:55.924479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"007ACE74B050のessay_idのdiscourse_textがどのようなものになっているかを調べてみる。","metadata":{}},{"cell_type":"code","source":"train[train['essay_id'] == \"007ACE74B050\"].shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:37:55.927686Z","iopub.execute_input":"2022-07-25T21:37:55.928591Z","iopub.status.idle":"2022-07-25T21:37:55.949465Z","shell.execute_reply.started":"2022-07-25T21:37:55.928546Z","shell.execute_reply":"2022-07-25T21:37:55.948102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train['essay_id'] == \"007ACE74B050\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:37:55.953584Z","iopub.execute_input":"2022-07-25T21:37:55.954331Z","iopub.status.idle":"2022-07-25T21:37:55.980514Z","shell.execute_reply.started":"2022-07-25T21:37:55.954262Z","shell.execute_reply":"2022-07-25T21:37:55.978944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train['essay_id'] == \"007ACE74B050\"]['discourse_text']","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:37:55.982320Z","iopub.execute_input":"2022-07-25T21:37:55.982666Z","iopub.status.idle":"2022-07-25T21:37:56.001923Z","shell.execute_reply.started":"2022-07-25T21:37:55.982635Z","shell.execute_reply":"2022-07-25T21:37:56.000699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"全文が読めないので、pandasで表示する幅を最大にする。","metadata":{}},{"cell_type":"code","source":"pd.set_option(\"max_colwidth\", None)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:37:56.005158Z","iopub.execute_input":"2022-07-25T21:37:56.009109Z","iopub.status.idle":"2022-07-25T21:37:56.016037Z","shell.execute_reply.started":"2022-07-25T21:37:56.009018Z","shell.execute_reply":"2022-07-25T21:37:56.014887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train['essay_id'] == \"007ACE74B050\"]['discourse_text']","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:37:56.018885Z","iopub.execute_input":"2022-07-25T21:37:56.020228Z","iopub.status.idle":"2022-07-25T21:37:56.040320Z","shell.execute_reply.started":"2022-07-25T21:37:56.020182Z","shell.execute_reply":"2022-07-25T21:37:56.038402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"火星にある地形が自然が作成したものなのか、それとも宇宙人が作成したものかについて論じている。\n\n筆者の主張は、宇宙人は存在せず、自然に作成されているものであるとの主張になる。","metadata":{}},{"cell_type":"markdown","source":"この論文のそれぞれの論文タイプと評価を再確認する。","metadata":{}},{"cell_type":"code","source":"train[train['essay_id'] == \"007ACE74B050\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:37:56.043011Z","iopub.execute_input":"2022-07-25T21:37:56.043724Z","iopub.status.idle":"2022-07-25T21:37:56.070626Z","shell.execute_reply.started":"2022-07-25T21:37:56.043677Z","shell.execute_reply":"2022-07-25T21:37:56.069484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"discourse_typeの理解をする。\n\nkaggleの説明文をdeepLに突っ込んだ文章。\n\n=======================\n\nあなたのタスクは、人間のアノテーションを予測することです。あなたはまず、各エッセイを個別の修辞的および論証的要素（すなわち談話要素）に区分し、各要素を以下のいずれかに分類する必要があります。\n\n* リード - 統計、引用、説明、または読者の注意を引き、論文への方向を示すその他の装置で始まる導入部。\n* Lead - an introduction that begins with a statistic, a quotation, a description, or some other device to grab the reader’s attention and point toward the thesis\n* ポジション - 主要な質問に対する意見または結論\n* Position - an opinion or conclusion on the main question\n* 主張（Claim）-立場を支持する主張\n* Claim - a claim that supports the position\n* 反対意見 - 他の主張に反論する、またはその立場に反対する理由を示す主張。\n* Counterclaim - a claim that refutes another claim or gives an opposing reason to the position\n* Rebuttal（反論） - 反論に反論する主張。\n* Rebuttal - a claim that refutes a counterclaim\n* 証拠 - 主張、反論、または反駁をサポートするアイデアや例。\n* Evidence - ideas or examples that support claims, counterclaims, or rebuttals.\n* Conclusion Statement（結論の記述） - 主張を再表現する結論の記述。\n* Concluding Statement - a concluding statement that restates the claims","metadata":{}},{"cell_type":"markdown","source":"自分なりの解釈\n\n→\n* Lead: 導入部分。\n* position: 自分が今回の論文でどういう立場で意見を発するかを主張している。\n* claim: 主な議題に対しての自分なりの結論\n* counterclaim: 自分のpositionに対しての予想される反論\n* rebuttal: counterclaimに対しての反論\n* evidence: claimやcounterclaim、rebuttalなどの主張を裏付ける証拠\n* conclusion statement: claimを結論に再度、記述しているところ","metadata":{}},{"cell_type":"markdown","source":"今回のタスクの目的は、それぞれの分類において、effectivenessで三段階で評価を予測すること。\n\nそうすれば、機械が自動で評価してくれるので圧倒的に楽になる。","metadata":{}},{"cell_type":"markdown","source":"データ数を確認する。","metadata":{}},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:37:56.075960Z","iopub.execute_input":"2022-07-25T21:37:56.077079Z","iopub.status.idle":"2022-07-25T21:37:56.085988Z","shell.execute_reply.started":"2022-07-25T21:37:56.077028Z","shell.execute_reply":"2022-07-25T21:37:56.084825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:37:56.113972Z","iopub.execute_input":"2022-07-25T21:37:56.114705Z","iopub.status.idle":"2022-07-25T21:37:56.123127Z","shell.execute_reply.started":"2022-07-25T21:37:56.114668Z","shell.execute_reply":"2022-07-25T21:37:56.121799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"trainのデータがどのようになっているかを確認していく。\n\npandas_profilingを使って、データ全体を確認していく。","metadata":{}},{"cell_type":"code","source":"pip install pandas-profiling","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:37:56.155960Z","iopub.execute_input":"2022-07-25T21:37:56.157280Z","iopub.status.idle":"2022-07-25T21:38:09.262369Z","shell.execute_reply.started":"2022-07-25T21:37:56.157237Z","shell.execute_reply":"2022-07-25T21:38:09.260904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas_profiling as pdp","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:09.266256Z","iopub.execute_input":"2022-07-25T21:38:09.267719Z","iopub.status.idle":"2022-07-25T21:38:09.274347Z","shell.execute_reply.started":"2022-07-25T21:38:09.267659Z","shell.execute_reply":"2022-07-25T21:38:09.273131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pdp.ProfileReport(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:09.276112Z","iopub.execute_input":"2022-07-25T21:38:09.277192Z","iopub.status.idle":"2022-07-25T21:38:25.213193Z","shell.execute_reply.started":"2022-07-25T21:38:09.277144Z","shell.execute_reply":"2022-07-25T21:38:25.212058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"pandas profilingでの気付き\n* discourse_textで、なぜか何件かは重複している\n\n→一旦、今回は上記の重複は無視した上で、一度 他のkaggle notebookを参照して、予測結果を作成してみる。","metadata":{}},{"cell_type":"markdown","source":"BERTで自然言語処理を行ってみる。","metadata":{}},{"cell_type":"markdown","source":"BERTだと、基本的に前処理はほとんど行わなくてよさそう。\n\ndiscourse_effectivenessのlabel encodingだけを行っておく。","metadata":{}},{"cell_type":"code","source":"train['label'] = train['discourse_effectiveness'].replace(\n    {\n    \"Ineffective\": 0, \n    \"Adequate\": 1, \n    \"Effective\": 2\n    }\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:25.217550Z","iopub.execute_input":"2022-07-25T21:38:25.218346Z","iopub.status.idle":"2022-07-25T21:38:25.264631Z","shell.execute_reply.started":"2022-07-25T21:38:25.218301Z","shell.execute_reply":"2022-07-25T21:38:25.263458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:25.266908Z","iopub.execute_input":"2022-07-25T21:38:25.268434Z","iopub.status.idle":"2022-07-25T21:38:25.274183Z","shell.execute_reply.started":"2022-07-25T21:38:25.268349Z","shell.execute_reply":"2022-07-25T21:38:25.272937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Configuration\nBATCH_SIZE = 16\nMAX_LEN = 256 \nDROPOUT = 0.1 # 0.2\nLEARNING_RATE = 1e-5\nEPOCHS = 1#8\nAUTO = tf.data.experimental.AUTOTUNE\nMODEL = \"distilbert\" #\"bert\"\nMODEL_PATH = f\"../input/huggingface-bert-variants/{MODEL}-base-uncased/{MODEL}-base-uncased\"","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:25.276473Z","iopub.execute_input":"2022-07-25T21:38:25.277225Z","iopub.status.idle":"2022-07-25T21:38:25.286844Z","shell.execute_reply.started":"2022-07-25T21:38:25.277178Z","shell.execute_reply":"2022-07-25T21:38:25.285617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import transformers","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:25.340244Z","iopub.execute_input":"2022-07-25T21:38:25.341132Z","iopub.status.idle":"2022-07-25T21:38:25.353445Z","shell.execute_reply.started":"2022-07-25T21:38:25.341082Z","shell.execute_reply":"2022-07-25T21:38:25.352090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = transformers.BertTokenizer.from_pretrained(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:25.357070Z","iopub.execute_input":"2022-07-25T21:38:25.357701Z","iopub.status.idle":"2022-07-25T21:38:25.417242Z","shell.execute_reply.started":"2022-07-25T21:38:25.357661Z","shell.execute_reply":"2022-07-25T21:38:25.416047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sep = tokenizer.sep_token\nprint(sep)\n\ntrain['inputs'] = train.discourse_type + sep + train.discourse_text\ntrain.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:25.421219Z","iopub.execute_input":"2022-07-25T21:38:25.422012Z","iopub.status.idle":"2022-07-25T21:38:25.458723Z","shell.execute_reply.started":"2022-07-25T21:38:25.421976Z","shell.execute_reply":"2022-07-25T21:38:25.457561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sequence = train['inputs'].iloc[0]\n\ntoken = tokenizer(sample_sequence,\n                 max_length = MAX_LEN,\n                 truncation = True,\n                 padding = 'max_length',\n                 add_special_tokens = True,\n                 return_tensors = \"np\"\n                 )","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:25.460577Z","iopub.execute_input":"2022-07-25T21:38:25.461503Z","iopub.status.idle":"2022-07-25T21:38:25.472139Z","shell.execute_reply.started":"2022-07-25T21:38:25.461446Z","shell.execute_reply":"2022-07-25T21:38:25.471062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def bert_encode(texts, tokenizer, max_len = MAX_LEN):\n    input_ids = np.zeros((len(texts), max_len), dtype = \"int32\")\n    \n    attention_mask = np.zeros((len(texts), max_len), dtype = \"int32\")\n    \n    for i, text in enumerate(texts):\n        token = tokenizer(text,\n                         max_length = max_len,\n                         truncation = True,\n                         padding = \"max_length\",\n                         add_special_tokens = True,\n                         return_tensors = \"np\")\n        \n        input_ids[i] = token['input_ids']\n        \n        attention_mask[i] = token['attention_mask']\n        \n    return input_ids, attention_mask","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:25.474461Z","iopub.execute_input":"2022-07-25T21:38:25.474796Z","iopub.status.idle":"2022-07-25T21:38:25.488479Z","shell.execute_reply.started":"2022-07-25T21:38:25.474755Z","shell.execute_reply":"2022-07-25T21:38:25.486746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import Input\nfrom tensorflow.keras.layers import Dropout\nfrom tensorflow.keras.layers import Dense\nfrom transformers import TFBertModel","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:25.503421Z","iopub.execute_input":"2022-07-25T21:38:25.505337Z","iopub.status.idle":"2022-07-25T21:38:25.513310Z","shell.execute_reply.started":"2022-07-25T21:38:25.505271Z","shell.execute_reply":"2022-07-25T21:38:25.512125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import Model","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:25.515474Z","iopub.execute_input":"2022-07-25T21:38:25.516105Z","iopub.status.idle":"2022-07-25T21:38:25.524441Z","shell.execute_reply.started":"2022-07-25T21:38:25.516059Z","shell.execute_reply":"2022-07-25T21:38:25.522894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.optimizers import Adam","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:25.526902Z","iopub.execute_input":"2022-07-25T21:38:25.527630Z","iopub.status.idle":"2022-07-25T21:38:25.537909Z","shell.execute_reply.started":"2022-07-25T21:38:25.527582Z","shell.execute_reply":"2022-07-25T21:38:25.536678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_ids = Input(shape = (MAX_LEN,), dtype = tf.int32, name = \"input_ids\")\nattention_mask = Input(shape = (MAX_LEN, ), dtype = tf.int32, name = \"attention_mask\")\n\ntransformer_layer = (TFBertModel.from_pretrained(MODEL_PATH))\n\nsequence_output = transformer_layer(input_ids,\n                                   attention_mask = attention_mask)[0]\n\nclf_output = sequence_output[:, 0, :]\nclf_output = Dropout(DROPOUT)(clf_output)\nout = Dense(3, activation = 'softmax')(clf_output)\n\nmodel = Model(inputs = [input_ids, attention_mask],\n             outputs = out)\n\nmodel.compile(Adam(learning_rate = LEARNING_RATE,\n                  ),\n             loss = 'sparse_categorical_crossentropy',\n             metrics = ['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:25.539709Z","iopub.execute_input":"2022-07-25T21:38:25.541310Z","iopub.status.idle":"2022-07-25T21:38:28.040150Z","shell.execute_reply.started":"2022-07-25T21:38:25.541262Z","shell.execute_reply":"2022-07-25T21:38:28.039016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import GroupKFold","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:28.042021Z","iopub.execute_input":"2022-07-25T21:38:28.042401Z","iopub.status.idle":"2022-07-25T21:38:28.048536Z","shell.execute_reply.started":"2022-07-25T21:38:28.042359Z","shell.execute_reply":"2022-07-25T21:38:28.046946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train['inputs']\ny = train['label']\n\nkf = GroupKFold(n_splits = 5)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:28.051072Z","iopub.execute_input":"2022-07-25T21:38:28.051986Z","iopub.status.idle":"2022-07-25T21:38:28.062616Z","shell.execute_reply.started":"2022-07-25T21:38:28.051940Z","shell.execute_reply":"2022-07-25T21:38:28.061017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SEED = 42","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:28.065097Z","iopub.execute_input":"2022-07-25T21:38:28.066941Z","iopub.status.idle":"2022-07-25T21:38:28.076044Z","shell.execute_reply.started":"2022-07-25T21:38:28.066902Z","shell.execute_reply":"2022-07-25T21:38:28.074289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import log_loss\nfrom sklearn.metrics import confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:28.092172Z","iopub.execute_input":"2022-07-25T21:38:28.092971Z","iopub.status.idle":"2022-07-25T21:38:28.106678Z","shell.execute_reply.started":"2022-07-25T21:38:28.092926Z","shell.execute_reply":"2022-07-25T21:38:28.105486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = bert_encode(train['inputs'].astype(str),tokenizer)\npreds = model.predict(X_train, verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:38:28.108644Z","iopub.execute_input":"2022-07-25T21:38:28.109431Z","iopub.status.idle":"2022-07-25T21:47:18.386223Z","shell.execute_reply.started":"2022-07-25T21:38:28.109386Z","shell.execute_reply":"2022-07-25T21:47:18.384709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"label_pred\"] = preds.argmax(axis = 1)\ntrain['Ineffective'] = preds[:,0]\ntrain['Adequate'] = preds[:,1]\ntrain['Effective'] = preds[:,2]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:47:18.389449Z","iopub.execute_input":"2022-07-25T21:47:18.389979Z","iopub.status.idle":"2022-07-25T21:47:18.401752Z","shell.execute_reply.started":"2022-07-25T21:47:18.389930Z","shell.execute_reply":"2022-07-25T21:47:18.400421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = 4\nrows = 2\nfig, ax = plt.subplots(nrows = rows, ncols = cols, figsize = (22,10))            ","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:47:18.408956Z","iopub.execute_input":"2022-07-25T21:47:18.409627Z","iopub.status.idle":"2022-07-25T21:47:18.614168Z","shell.execute_reply.started":"2022-07-25T21:47:18.409591Z","shell.execute_reply":"2022-07-25T21:47:18.612893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i = 0\nfor n, t in enumerate(train.discourse_type.unique()):\n    \n    j = n % (cols)\n    try:\n        y_train = train[train.discourse_type == t].label\n        y_pred = train[train.discourse_type == t].label_pred\n        \n        y_pred_classes = train[train.discourse_type == t][['Ineffective', 'Adequate', 'Effective']].values\n        cf_matrix = confusion_matrix(y_train, y_pred)\n        sns.heatmap(cf_matrix, annot = True, fmt = \".0f\", ax = ax[i,j])\n        ax[i,j].set_title(f\"{t}: Log Loss {log_loss(y_train.values, y_pred_classes):.2f}\")\n        \n    except:\n        pass\n    if j == (cols-1):\n        i = i+1\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:47:18.620562Z","iopub.execute_input":"2022-07-25T21:47:18.624832Z","iopub.status.idle":"2022-07-25T21:47:21.026613Z","shell.execute_reply.started":"2022-07-25T21:47:18.624741Z","shell.execute_reply":"2022-07-25T21:47:21.025174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['text'] = test.discourse_type + sep + test.discourse_text\n\nX_test = bert_encode(test.text.astype(str), tokenizer)\n\ny_pred = model.predict(X_test, verbose = 1)\ny_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:47:21.028880Z","iopub.execute_input":"2022-07-25T21:47:21.029363Z","iopub.status.idle":"2022-07-25T21:47:21.277403Z","shell.execute_reply.started":"2022-07-25T21:47:21.029314Z","shell.execute_reply":"2022-07-25T21:47:21.275940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(\"../input/feedback-prize-effectiveness/sample_submission.csv\")\n\nsubmission['Ineffective'] = y_pred[:,0]\nsubmission['Adequate'] = y_pred[:,1]\nsubmission['Effective'] = y_pred[:,2]\n\nsubmission.to_csv(\"submission.csv\", index = False)\n\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T21:47:21.279305Z","iopub.execute_input":"2022-07-25T21:47:21.280186Z","iopub.status.idle":"2022-07-25T21:47:21.309147Z","shell.execute_reply.started":"2022-07-25T21:47:21.280141Z","shell.execute_reply":"2022-07-25T21:47:21.308031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}