{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div align='center'><font size=\"5\" color='#353B47'>Feedback Prize Effectiveness</font></div>\n<div align='center'><font size=\"4\" color=\"#353B47\">When AI analyses your writing skills</font></div>\n<br>\n<hr>","metadata":{}},{"cell_type":"markdown","source":"## <div id=\"summary\">Table of contents</div>\n\n**<font size=\"2\"><a href=\"#chap1\">1. Data Description</a></font>**\n**<br><font size=\"2\"><a href=\"#chap2\">2. Exploratory Data Analysis</a></font>**\n**<br><font size=\"2\"><a href=\"#chap3\">3. Get some insights from essays and discourses</a></font>**\n**<br><font size=\"2\"><a href=\"#chap4\">4. Getting unannotated part of essays</a></font>**\n**<br><font size=\"2\"><a href=\"#chap5\">5. Create folds</a></font>**\n\n<hr>","metadata":{}},{"cell_type":"code","source":"# Import dependancies\nimport pandas as pd \nimport os\nimport random\nimport numpy as np\nimport time\nimport datetime\nfrom IPython.core.display import HTML, display\n\nimport re\nimport multiprocessing\nfrom joblib import Parallel, delayed\nfrom tqdm import tqdm\n\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom wordcloud import WordCloud, STOPWORDS\n\nfrom sklearn import model_selection\nimport torch\nfrom torch.nn.functional import one_hot\nfrom torch.utils.data import DataLoader, RandomSampler, SequentialSampler, Dataset\nfrom transformers import BertForSequenceClassification, AdamW, BertConfig, BertTokenizer, BertModel, AutoConfig, AutoModel, AutoTokenizer, get_linear_schedule_with_warmup\n\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T13:19:47.770166Z","iopub.execute_input":"2022-08-03T13:19:47.770685Z","iopub.status.idle":"2022-08-03T13:19:59.261303Z","shell.execute_reply.started":"2022-08-03T13:19:47.770638Z","shell.execute_reply":"2022-08-03T13:19:59.259572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# If there's a GPU available...\nif torch.cuda.is_available():    \n\n    # Tell PyTorch to use the GPU.    \n    device = torch.device(\"cuda\")\n\n    print('There are %d GPU(s) available.' % torch.cuda.device_count())\n\n    print('We will use the GPU:', torch.cuda.get_device_name(0))\n\n# If not...\nelse:\n    print('No GPU available, using the CPU instead.')\n    device = torch.device(\"cpu\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:19:59.264135Z","iopub.execute_input":"2022-08-03T13:19:59.265955Z","iopub.status.idle":"2022-08-03T13:19:59.273639Z","shell.execute_reply.started":"2022-08-03T13:19:59.265899Z","shell.execute_reply":"2022-08-03T13:19:59.271971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div id=\"chap1\">Data description</div>","metadata":{}},{"cell_type":"code","source":"# Import train and test\ntrain = pd.read_csv('../input/feedback-prize-effectiveness/train.csv')\ntest = pd.read_csv('../input/feedback-prize-effectiveness/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:20:03.406056Z","iopub.execute_input":"2022-08-03T13:20:03.406521Z","iopub.status.idle":"2022-08-03T13:20:03.777043Z","shell.execute_reply.started":"2022-08-03T13:20:03.406487Z","shell.execute_reply":"2022-08-03T13:20:03.775773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get full path of .txt file as a column\ntrain['essay_id_path'] = train['essay_id'].apply(lambda x: f'../input/feedback-prize-effectiveness/train/{x}.txt')\ntest['essay_id_path'] = test['essay_id'].apply(lambda x: f'../input/feedback-prize-effectiveness/test/{x}.txt')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:20:03.956579Z","iopub.execute_input":"2022-08-03T13:20:03.957810Z","iopub.status.idle":"2022-08-03T13:20:03.991043Z","shell.execute_reply.started":"2022-08-03T13:20:03.957758Z","shell.execute_reply":"2022-08-03T13:20:03.989706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:20:05.208015Z","iopub.execute_input":"2022-08-03T13:20:05.208534Z","iopub.status.idle":"2022-08-03T13:20:05.218015Z","shell.execute_reply.started":"2022-08-03T13:20:05.208493Z","shell.execute_reply":"2022-08-03T13:20:05.217114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking NAs\ntrain.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:20:05.620021Z","iopub.execute_input":"2022-08-03T13:20:05.620813Z","iopub.status.idle":"2022-08-03T13:20:05.646206Z","shell.execute_reply.started":"2022-08-03T13:20:05.620757Z","shell.execute_reply":"2022-08-03T13:20:05.645063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:20:06.646383Z","iopub.execute_input":"2022-08-03T13:20:06.646914Z","iopub.status.idle":"2022-08-03T13:20:06.665957Z","shell.execute_reply.started":"2022-08-03T13:20:06.646871Z","shell.execute_reply":"2022-08-03T13:20:06.664631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many essays contain the trainset ?\nlen(train.essay_id.unique())","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:20:07.145726Z","iopub.execute_input":"2022-08-03T13:20:07.146954Z","iopub.status.idle":"2022-08-03T13:20:07.159837Z","shell.execute_reply.started":"2022-08-03T13:20:07.146905Z","shell.execute_reply":"2022-08-03T13:20:07.158617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div id=\"chap2\">Exploratory Data Analysis</div>","metadata":{}},{"cell_type":"code","source":"def show_examples_for_discourse_type(discourse_type):\n    filter_df = train[train.discourse_type==discourse_type].sample(frac=1, random_state=42)\n    display(HTML(\n        f\"\"\"\n        <h4><code>{discourse_type}</code> examples</h4>\n        <table>\n            <tr>\n              <th width=33% style=\"color:Tomato;\">Ineffective</th>\n              <th width=33% style=\"color:DodgerBlue;\">Adequate</th>\n              <th width=33% style=\"color:MediumSeaGreen;\">Effective</th>\n            </tr>\n            <tr>\n              <td>{filter_df[filter_df.discourse_effectiveness=='Ineffective'].iloc[0].discourse_text}</td>\n              <td>{filter_df[filter_df.discourse_effectiveness=='Adequate'].iloc[0].discourse_text}</td>\n              <td>{filter_df[filter_df.discourse_effectiveness=='Effective'].iloc[0].discourse_text}</td>\n            </tr>\n        </table>\n        \"\"\"\n    ))\n    \ndiscourse_types = train.discourse_type.unique()\nexs = [show_examples_for_discourse_type(dt) for dt in discourse_types]","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T13:23:03.372612Z","iopub.execute_input":"2022-08-03T13:23:03.373118Z","iopub.status.idle":"2022-08-03T13:23:03.479583Z","shell.execute_reply.started":"2022-08-03T13:23:03.373080Z","shell.execute_reply":"2022-08-03T13:23:03.478429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Distribution of number of discourse per essay\ny0 = train.groupby('essay_id').count()['discourse_id']\n\nfig = go.Figure()\nfig.add_trace(\n    go.Box(\n        y=y0, \n        name=\"Train data\",\n        marker_color = '#1e90ff'\n    )\n)\n\n\nfig.update_layout(\n    legend_title_text = \"Discourse effectiveness\",\n    title_text='Number of discourses',\n)\nfig.update_yaxes()\nfig.update_xaxes(showticklabels=False)\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T13:26:39.358391Z","iopub.execute_input":"2022-08-03T13:26:39.358903Z","iopub.status.idle":"2022-08-03T13:26:39.408427Z","shell.execute_reply.started":"2022-08-03T13:26:39.358858Z","shell.execute_reply":"2022-08-03T13:26:39.407538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colors = ['#1e90ff',] * 7 #DodgerBlue color\ncolors[0] = '#ff6347' #Tomato color\n\ndict_discourse_type = dict(train.discourse_type.value_counts())\n\nx = list(dict_discourse_type.keys())\ny = list(dict_discourse_type.values())\n\nfig = go.Figure(\n    data=[\n        go.Bar(\n            x=x,\n            y=y,\n            marker_color=colors\n        )\n    ]\n)\n\nfig.update_layout(\n    title_text='Discourse type distribution',\n)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T13:26:16.929604Z","iopub.execute_input":"2022-08-03T13:26:16.930262Z","iopub.status.idle":"2022-08-03T13:26:16.959725Z","shell.execute_reply.started":"2022-08-03T13:26:16.930201Z","shell.execute_reply":"2022-08-03T13:26:16.957979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colors = ['#1e90ff',] * 3\ncolors[0] = '#ff6347'\n\ndict_discourse_effectiveness = dict(train.discourse_effectiveness.value_counts())\n\nx = list(dict_discourse_effectiveness.keys())\ny = list(dict_discourse_effectiveness.values())\n\nfig = go.Figure(\n    data=[\n        go.Bar(\n            x=x,\n            y=y,\n            marker_color=colors\n        )\n    ]\n)\n\nfig.update_layout(\n    title_text='Discourse effectiveness distribution',\n)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T13:27:28.823695Z","iopub.execute_input":"2022-08-03T13:27:28.824162Z","iopub.status.idle":"2022-08-03T13:27:28.844395Z","shell.execute_reply.started":"2022-08-03T13:27:28.824123Z","shell.execute_reply":"2022-08-03T13:27:28.843196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure()\n\ncolor = {'Adequate': '#1e90ff',\n         'Effective': '#3cb371',\n         'Ineffective': '#ff6347'}\n\nfor discourse_effectiveness, group in train.groupby(\"discourse_effectiveness\"):\n    \n    dict_discourse_effectiveness = dict(group[\"discourse_type\"].value_counts())\n    \n    fig.add_trace(\n        go.Bar(\n            x=list(dict_discourse_effectiveness.keys()), \n            y=list(dict_discourse_effectiveness.values()), \n            name=discourse_effectiveness,\n            marker = dict(color=color[discourse_effectiveness])\n        )\n    )\n    \nfig.update_layout(\n    legend_title_text = \"Discourse effectiveness\",\n    title_text='Discourse effectiveness per discourse type distribution',\n)\nfig.update_xaxes(title_text=\"Discourse type\")\nfig.update_yaxes(title_text=\"Counts\")\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T13:36:48.623330Z","iopub.execute_input":"2022-08-03T13:36:48.624321Z","iopub.status.idle":"2022-08-03T13:36:48.668260Z","shell.execute_reply.started":"2022-08-03T13:36:48.624270Z","shell.execute_reply":"2022-08-03T13:36:48.667029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure()\n\n\nfor discourse_effectiveness, group in train.groupby(\"discourse_effectiveness\"):\n    \n    group_discourse_type = dict(group[\"discourse_type\"].value_counts())\n    \n    fig.add_trace(\n        go.Bar(\n            x=list(group_discourse_type.keys()), \n            y=list(group_discourse_type.values()), \n            name=discourse_effectiveness,\n            marker = dict(color=color[discourse_effectiveness])\n        )\n    )\n    \nfig.update_layout(\n    legend_title_text = \"Discourse effectiveness\",\n    title_text='Discourse effectiveness per discourse type distribution (stacked)',\n    barmode=\"stack\",\n    barnorm='percent'\n)\nfig.update_xaxes(title_text=\"Discourse type\")\nfig.update_yaxes(title_text=\"Counts\")\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T13:37:22.151434Z","iopub.execute_input":"2022-08-03T13:37:22.151944Z","iopub.status.idle":"2022-08-03T13:37:22.194984Z","shell.execute_reply.started":"2022-08-03T13:37:22.151904Z","shell.execute_reply":"2022-08-03T13:37:22.194082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div id=\"chap3\">Get some insights from essays and discourses</div>","metadata":{}},{"cell_type":"code","source":"def parallel_get_essay_info(input_folder):\n\n    manager = multiprocessing.Manager()\n    dicti = manager.dict()\n\n    filenames = os.listdir(input_folder)\n\n    def _get_essay_info(filename):\n        \n        with open(os.path.join(input_folder, filename)) as f:\n            text = f.readlines()\n            text = ''.join(text).lower()\n            text = re.split('\\.|!|\\?|\\n', text)\n            text = [s.strip() for s in text]\n            text = [s for s in text if s != '']\n            \n            total_sentences = len(text)\n            mean_characters_per_sentence = np.mean([len(x) for x in text])\n            total_characters = np.sum([len(x) for x in text])\n            total_words = np.sum([len(x.split(' ')) for x in text])\n            mean_words_per_sentence = np.mean([len(x.split(' ')) for x in text])\n                  \n            dicti[filename] = {\n                'essay_total_sentences':total_sentences,\n                'essay_mean_characters_per_sentence':mean_characters_per_sentence,\n                'essay_total_characters':total_characters,\n                'essay_total_words':total_words,\n                'essay_mean_words_per_sentence':mean_words_per_sentence,\n            }\n            \n    Parallel(n_jobs=20)(\n        delayed(_get_essay_info)(\n            filename=filename,\n        ) for filename in tqdm(filenames)\n    )\n\n    return dict(dicti)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:39:04.685536Z","iopub.execute_input":"2022-08-03T13:39:04.686053Z","iopub.status.idle":"2022-08-03T13:39:04.699097Z","shell.execute_reply.started":"2022-08-03T13:39:04.686011Z","shell.execute_reply":"2022-08-03T13:39:04.697930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"essays_info = parallel_get_essay_info('../input/feedback-prize-effectiveness/train/')\n\ntrain['essay_info'] = train['essay_id'].apply(lambda x: essays_info[x+'.txt'])\ntrain = pd.merge(left = train, right = pd.json_normalize(train['essay_info']), left_index=True, right_index=True)\ntrain = train.drop('essay_info', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:39:05.645773Z","iopub.execute_input":"2022-08-03T13:39:05.646578Z","iopub.status.idle":"2022-08-03T13:39:43.032428Z","shell.execute_reply.started":"2022-08-03T13:39:05.646504Z","shell.execute_reply":"2022-08-03T13:39:43.031148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_discourse(text):\n    text = text.lower()\n    text = re.split('\\.|!|\\?|\\n', text)\n    text = [s.strip() for s in text]\n    text = [s for s in text if s != '']\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:39:43.035174Z","iopub.execute_input":"2022-08-03T13:39:43.035609Z","iopub.status.idle":"2022-08-03T13:39:43.044076Z","shell.execute_reply.started":"2022-08-03T13:39:43.035567Z","shell.execute_reply":"2022-08-03T13:39:43.042652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['cleaned_discourse'] = train['discourse_text'].apply(clean_discourse)\n\ntrain['discourse_total_sentences'] = train['cleaned_discourse'].apply(len) \ntrain['discourse_mean_characters_per_sentence'] = train['cleaned_discourse'].apply(lambda x: np.mean([len(elm) for elm in x]))\ntrain['discourse_total_characters'] = train['cleaned_discourse'].apply(lambda x: np.sum([len(elm) for elm in x]))\ntrain['discourse_total_words'] = train['cleaned_discourse'].apply(lambda x: np.sum([len(elm.split(' ')) for elm in x]))\ntrain['discourse_mean_words_per_sentence'] = train['cleaned_discourse'].apply(lambda x: np.mean([len(elm.split(' ')) for elm in x]))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:39:43.046125Z","iopub.execute_input":"2022-08-03T13:39:43.046947Z","iopub.status.idle":"2022-08-03T13:39:44.937270Z","shell.execute_reply.started":"2022-08-03T13:39:43.046886Z","shell.execute_reply":"2022-08-03T13:39:44.935869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:39:44.940011Z","iopub.execute_input":"2022-08-03T13:39:44.940477Z","iopub.status.idle":"2022-08-03T13:39:44.969025Z","shell.execute_reply.started":"2022-08-03T13:39:44.940437Z","shell.execute_reply":"2022-08-03T13:39:44.967669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y0 = train['discourse_total_sentences']\ny1 = train.loc[train[\"discourse_effectiveness\"]=='Adequate', 'discourse_total_sentences']\ny2 = train.loc[train[\"discourse_effectiveness\"]=='Effective', 'discourse_total_sentences']\ny3 = train.loc[train[\"discourse_effectiveness\"]=='Ineffective', 'discourse_total_sentences']\n\n# dict(color=color[discourse_effectiveness])\nfig = go.Figure()\nfig.add_trace(go.Box(y=y0, name=\"Train data\"))\nfig.add_trace(go.Box(y=y1, name=\"Adequate\", marker_color = color[\"Adequate\"]))\nfig.add_trace(go.Box(y=y2, name=\"Effective\", marker_color = color[\"Effective\"]))\nfig.add_trace(go.Box(y=y3, name=\"Ineffective\", marker_color = color[\"Ineffective\"]))\n\nfig.update_layout(\n    legend_title_text = \"Discourse effectiveness\",\n    title_text='Number of sentences in discourse (logscale)',\n)\nfig.update_yaxes(type=\"log\")\nfig.update_xaxes(showticklabels=False)\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T13:40:10.559520Z","iopub.execute_input":"2022-08-03T13:40:10.561013Z","iopub.status.idle":"2022-08-03T13:40:10.608045Z","shell.execute_reply.started":"2022-08-03T13:40:10.560963Z","shell.execute_reply":"2022-08-03T13:40:10.606667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y0 = train['discourse_mean_characters_per_sentence']\ny1 = train.loc[train[\"discourse_effectiveness\"]=='Adequate', 'discourse_mean_characters_per_sentence']\ny2 = train.loc[train[\"discourse_effectiveness\"]=='Effective', 'discourse_mean_characters_per_sentence']\ny3 = train.loc[train[\"discourse_effectiveness\"]=='Ineffective', 'discourse_mean_characters_per_sentence']\n\n\nfig = go.Figure()\nfig.add_trace(go.Box(y=y0, name=\"Train data\"))\nfig.add_trace(go.Box(y=y1, name=\"Adequate\", marker_color = color[\"Adequate\"]))\nfig.add_trace(go.Box(y=y2, name=\"Effective\", marker_color = color[\"Effective\"]))\nfig.add_trace(go.Box(y=y3, name=\"Ineffective\", marker_color = color[\"Ineffective\"]))\n\nfig.update_layout(\n    legend_title_text = \"Discourse effectiveness\",\n    title_text='Mean size of sentences in discourse (logscale)',\n)\nfig.update_yaxes(type=\"log\")\nfig.update_xaxes(showticklabels=False)\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T12:23:57.388132Z","iopub.execute_input":"2022-08-03T12:23:57.388435Z","iopub.status.idle":"2022-08-03T12:23:57.432558Z","shell.execute_reply.started":"2022-08-03T12:23:57.388406Z","shell.execute_reply":"2022-08-03T12:23:57.429295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y0 = train['discourse_total_characters']\ny1 = train.loc[train[\"discourse_effectiveness\"]=='Adequate', 'discourse_total_characters']\ny2 = train.loc[train[\"discourse_effectiveness\"]=='Effective', 'discourse_total_characters']\ny3 = train.loc[train[\"discourse_effectiveness\"]=='Ineffective', 'discourse_total_characters']\n\n\nfig = go.Figure()\nfig.add_trace(go.Box(y=y0, name=\"Train data\"))\nfig.add_trace(go.Box(y=y1, name=\"Adequate\", marker_color = color[\"Adequate\"]))\nfig.add_trace(go.Box(y=y2, name=\"Effective\", marker_color = color[\"Effective\"]))\nfig.add_trace(go.Box(y=y3, name=\"Ineffective\", marker_color = color[\"Ineffective\"]))\n\nfig.update_layout(\n    legend_title_text = \"Discourse effectiveness\",\n    title_text='Character counts in discourse (logscale)',\n)\nfig.update_yaxes(type=\"log\")\nfig.update_xaxes(showticklabels=False)\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T12:23:57.433987Z","iopub.execute_input":"2022-08-03T12:23:57.434927Z","iopub.status.idle":"2022-08-03T12:23:57.469362Z","shell.execute_reply.started":"2022-08-03T12:23:57.434894Z","shell.execute_reply":"2022-08-03T12:23:57.468513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y0 = train['discourse_total_words']\ny1 = train.loc[train[\"discourse_effectiveness\"]=='Adequate', 'discourse_total_words']\ny2 = train.loc[train[\"discourse_effectiveness\"]=='Effective', 'discourse_total_words']\ny3 = train.loc[train[\"discourse_effectiveness\"]=='Ineffective', 'discourse_total_words']\n\n\nfig = go.Figure()\nfig.add_trace(go.Box(y=y0, name=\"Train data\"))\nfig.add_trace(go.Box(y=y1, name=\"Adequate\", marker_color = color[\"Adequate\"]))\nfig.add_trace(go.Box(y=y2, name=\"Effective\", marker_color = color[\"Effective\"]))\nfig.add_trace(go.Box(y=y3, name=\"Ineffective\", marker_color = color[\"Ineffective\"]))\n\nfig.update_layout(\n    legend_title_text = \"Discourse effectiveness\",\n    title_text='Word counts in discourses (logscale)',\n)\nfig.update_yaxes(type=\"log\")\nfig.update_xaxes(showticklabels=False)\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T12:23:57.470849Z","iopub.execute_input":"2022-08-03T12:23:57.471501Z","iopub.status.idle":"2022-08-03T12:23:57.505063Z","shell.execute_reply.started":"2022-08-03T12:23:57.471467Z","shell.execute_reply":"2022-08-03T12:23:57.504227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y0 = train['discourse_mean_words_per_sentence']\ny1 = train.loc[train[\"discourse_effectiveness\"]=='Adequate', 'discourse_mean_words_per_sentence']\ny2 = train.loc[train[\"discourse_effectiveness\"]=='Effective', 'discourse_mean_words_per_sentence']\ny3 = train.loc[train[\"discourse_effectiveness\"]=='Ineffective', 'discourse_mean_words_per_sentence']\n\n\nfig = go.Figure()\nfig.add_trace(go.Box(y=y0, name=\"Train data\"))\nfig.add_trace(go.Box(y=y1, name=\"Adequate\", marker_color = color[\"Adequate\"]))\nfig.add_trace(go.Box(y=y2, name=\"Effective\", marker_color = color[\"Effective\"]))\nfig.add_trace(go.Box(y=y3, name=\"Ineffective\", marker_color = color[\"Ineffective\"]))\n\nfig.update_layout(\n    legend_title_text = \"Discourse effectiveness\",\n    title_text='Mean number of words per sentence in discourses (logscale)',\n)\nfig.update_yaxes(type=\"log\")\nfig.update_xaxes(showticklabels=False)\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T12:23:57.506353Z","iopub.execute_input":"2022-08-03T12:23:57.506814Z","iopub.status.idle":"2022-08-03T12:23:57.546438Z","shell.execute_reply.started":"2022-08-03T12:23:57.506784Z","shell.execute_reply":"2022-08-03T12:23:57.545700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def wordcloud(df, text):\n    \n    # Join all tweets in one string\n    corpus = \" \".join(str(review) for review in df[text])\n    display(\n        HTML(\n            f\"\"\"\n            <h4>There are: <code> {len(corpus)}</code> words in the combination of all review.</h4>\n            \"\"\"\n        )\n    )\n    \n    wordcloud = WordCloud(max_font_size=50,\n                          max_words=20,\n                          collocations = False,\n                          background_color=\"black\").generate(corpus)\n    \n    plt.figure(figsize=(15,15))\n    plt.imshow(wordcloud, interpolation=\"bilinear\")\n    plt.axis(\"off\")\n    plt.show()\n    return ''\n\ndisplay(\n    HTML(\n        f\"\"\"\n        <h4>Within all discourses</h4>\n        <br>\n        {wordcloud(df = train, text='discourse_text')}\n        \"\"\"\n    )\n)\n\nfor discourse_effectiveness in train.discourse_effectiveness.unique():\n    display(\n        HTML(\n            f\"\"\"\n            <h4>Within all <code style='color:{color[discourse_effectiveness]}'>{discourse_effectiveness}</code> discourses</h4>\n            <br>\n            <p>{wordcloud(df = train[train['discourse_effectiveness']==discourse_effectiveness], text='discourse_text')}</p>\n            \"\"\"\n        )\n    )","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T13:45:40.566169Z","iopub.execute_input":"2022-08-03T13:45:40.566767Z","iopub.status.idle":"2022-08-03T13:45:45.910041Z","shell.execute_reply.started":"2022-08-03T13:45:40.566723Z","shell.execute_reply":"2022-08-03T13:45:45.908632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for discourse_type in train.discourse_type.unique():\n    display(\n        HTML(\n            f\"\"\"\n            <h4>Within all discourses of type: <code>{discourse_type}</code></h4>\n            <br>\n            {wordcloud(df = train[train['discourse_type']==discourse_type], text='discourse_text')}\n            \"\"\"\n        )\n    )","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T13:14:34.072433Z","iopub.execute_input":"2022-08-03T13:14:34.073586Z","iopub.status.idle":"2022-08-03T13:14:38.513809Z","shell.execute_reply.started":"2022-08-03T13:14:34.073537Z","shell.execute_reply":"2022-08-03T13:14:38.512993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encoding discourse type and effectiveness\ntrain['discourse_type_enc'] = train['discourse_type'].astype('category').cat.codes\ntrain['discourse_effectiveness_enc'] = train['discourse_effectiveness'].astype('category').cat.codes","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:49:13.643641Z","iopub.execute_input":"2022-08-03T13:49:13.645151Z","iopub.status.idle":"2022-08-03T13:49:13.667054Z","shell.execute_reply.started":"2022-08-03T13:49:13.645094Z","shell.execute_reply":"2022-08-03T13:49:13.665145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div id=\"chap4\">Getting unannotated part of essays</div>","metadata":{}},{"cell_type":"markdown","source":"There is a discrepancy between total number of words contained in essay and sum of all discourses refering to the same essay, let's explore","metadata":{}},{"cell_type":"code","source":"# # retrieve unannotated part\n# essay_id = 'FFA381E58FC6'\n# essay_1_df = train[train['essay_id']==essay_id]\n\n# with open(os.path.join('../input/feedback-prize-effectiveness/train/', essay_id+'.txt')) as f:\n#     essay = f.readlines()\n#     essay = \"\".join(essay)\n#     for index, discourse_id in enumerate(essay_1_df.discourse_id.values):\n#         discourse = str(essay_1_df.loc[essay_1_df['discourse_id'] == discourse_id, 'discourse_text'].values[0].strip())\n        \n#         if discourse in essay:\n#             print(index)\n#         else:\n#             print(discourse)\n        \n#         essay = essay.replace(discourse, \"\")\n#     print(essay)\n#     print(len(list(set(essay))))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T12:24:06.269771Z","iopub.execute_input":"2022-08-03T12:24:06.270242Z","iopub.status.idle":"2022-08-03T12:24:06.284486Z","shell.execute_reply.started":"2022-08-03T12:24:06.270207Z","shell.execute_reply":"2022-08-03T12:24:06.283388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Store discourses that couldnt have been classified\nunannotated_parts = {}\n\nfor essay_id in tqdm(train['essay_id'].unique()):\n\n    essay_df = train[train['essay_id']==essay_id]\n\n    with open(os.path.join('../input/feedback-prize-effectiveness/train/', essay_id+'.txt')) as f:\n        essay = f.readlines()\n        essay = \"\".join(essay)\n        for index, discourse_id in enumerate(essay_df.discourse_id.values):\n            discourse = str(essay_df.loc[essay_df['discourse_id'] == discourse_id, 'discourse_text'].values[0].strip())\n            essay = essay.replace(discourse, \"\")\n\n        unannotated_parts[essay_id] = str(essay).strip()\n        \ntrain['unannotated_parts'] = train['essay_id'].map(unannotated_parts)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_examples_of_unnanotated():\n    \n    filter_df = train[['essay_id', 'unannotated_parts']].groupby('essay_id').first().sample(frac=1, random_state=5)[:5]\n    \n    display(HTML(\n        f\"\"\"\n        <h4><code>Unnanotated</code> exemples</h4>\n        \"\"\"\n    ))\n    \n    for i in range(1,len(filter_df)):\n        \n        display(HTML(\n            f\"\"\"\n            <table>\n                <tr>\n                  <th width=33% style=\"color:Violet;\">From essay {filter_df.index[i]}</th>\n                </tr>\n                <tr>\n                  <td>{filter_df.iloc[i].unannotated_parts}</td>\n                </tr>\n            </table>\n            \"\"\"\n        ))\n    \nshow_examples_of_unnanotated()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T14:18:46.171379Z","iopub.execute_input":"2022-08-03T14:18:46.172064Z","iopub.status.idle":"2022-08-03T14:18:46.213887Z","shell.execute_reply.started":"2022-08-03T14:18:46.172015Z","shell.execute_reply":"2022-08-03T14:18:46.212452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div id=\"chap5\">Create folds</div>","metadata":{}},{"cell_type":"code","source":"def create_folds(dataframe):\n    \"\"\"\n    Args:\n        dataframe (pd.DataFrame): dataframe which will contain kfold column\n\n    Returns (pd.DataFrame): dataframe with kfold column\n    \"\"\"\n\n    # Create kfold column in dataframe\n    dataframe[\"kfold\"] = -1\n    y = dataframe.discourse_effectiveness.values\n\n    # Chose the number of folds\n    kf = model_selection.KFold(n_splits=5)\n    skf = kf.split(X=dataframe, y=y, groups=dataframe['essay_id'].tolist())\n\n    # Assigning fold for each observation\n    for fold_, (_, val_) in enumerate(skf, start=1):\n        dataframe.loc[val_, \"kfold\"] = fold_\n\n    return dataframe\n\ntrain = create_folds(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:24:31.411241Z","iopub.execute_input":"2022-08-03T12:24:31.411911Z","iopub.status.idle":"2022-08-03T12:24:31.431520Z","shell.execute_reply.started":"2022-08-03T12:24:31.411872Z","shell.execute_reply":"2022-08-03T12:24:31.430396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check if an essay is not contained in several folds to avoid data leakage\nfor essay_id in train.essay_id.unique():\n    if len(train.loc[train['essay_id']==essay_id, 'kfold'].unique()) != 1:\n        print(essay_id)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:24:31.432971Z","iopub.execute_input":"2022-08-03T12:24:31.433598Z","iopub.status.idle":"2022-08-03T12:24:42.778488Z","shell.execute_reply.started":"2022-08-03T12:24:31.433562Z","shell.execute_reply":"2022-08-03T12:24:42.777515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"three essays needs their kfold values to be forced","metadata":{}},{"cell_type":"code","source":"# fixing\ntrain.loc[train['essay_id']=='851B84DCDFCA', 'kfold'] = 1\ntrain.loc[train['essay_id']=='053B8755A4FB', 'kfold'] = 2\ntrain.loc[train['essay_id']=='CA2C96F5B3B7', 'kfold'] = 5","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:24:42.779763Z","iopub.execute_input":"2022-08-03T12:24:42.780143Z","iopub.status.idle":"2022-08-03T12:24:42.796891Z","shell.execute_reply.started":"2022-08-03T12:24:42.780099Z","shell.execute_reply":"2022-08-03T12:24:42.796018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['kfold'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:24:42.798554Z","iopub.execute_input":"2022-08-03T12:24:42.798952Z","iopub.status.idle":"2022-08-03T12:24:42.809113Z","shell.execute_reply.started":"2022-08-03T12:24:42.798912Z","shell.execute_reply":"2022-08-03T12:24:42.808019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for fold in train['kfold'].unique():\n    print(len(train.loc[train['kfold']==fold, 'essay_id'].unique()))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:24:42.810464Z","iopub.execute_input":"2022-08-03T12:24:42.812403Z","iopub.status.idle":"2022-08-03T12:24:42.825473Z","shell.execute_reply.started":"2022-08-03T12:24:42.812367Z","shell.execute_reply":"2022-08-03T12:24:42.824485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<hr>\n\n# References\n\n* https://mccormickml.com/2019/07/22/BERT-fine-tuning/\n* <a href='https://www.kaggle.com/code/anantgupt/pytorch-feedback-eda-train-deberta-v3?scriptVersionId=102108552'>Anant Gupta Notebook</a>","metadata":{}},{"cell_type":"markdown","source":"<hr>\n<br>\n<div align='justify'><font color=\"#353B47\" size=\"4\">Thank you for taking the time to read this notebook. I hope that I was able to answer your questions or your curiosity and that it was quite understandable. <u>any constructive comments are welcome</u>. They help me progress and motivate me to share better quality content. I am above all a passionate person who tries to advance my knowledge but also that of others. If you liked it, feel free to <u>upvote and share my work.</u> </font></div>\n<br>\n<div align='center'><font color=\"#353B47\" size=\"3\">Thank you and may passion guide you.</font></div>","metadata":{}}]}