{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Exploratory Data Analysis !","metadata":{}},{"cell_type":"code","source":"import os\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm\n\nfrom sklearn import feature_extraction, feature_selection\nfrom sklearn.dummy import DummyClassifier\nfrom sklearn.preprocessing import LabelBinarizer\nfrom sklearn import metrics\n\nfrom scipy.stats import f_oneway\nimport statsmodels.api as sm\nfrom statsmodels.formula.api import ols\n\nsns.set(rc={'figure.figsize':(15, 7)})","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:34:02.561177Z","iopub.execute_input":"2022-07-12T11:34:02.561603Z","iopub.status.idle":"2022-07-12T11:34:02.569321Z","shell.execute_reply.started":"2022-07-12T11:34:02.561567Z","shell.execute_reply":"2022-07-12T11:34:02.568081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_path='../input/feedback-prize-effectiveness/'\ntest_path=os.path.join(input_path,'test/')\ntrain_path=os.path.join(input_path,'train/')\n\ndata=pd.read_csv(os.path.join(input_path,'train.csv'))\ntest_df=pd.read_csv(os.path.join(input_path,'test.csv'))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:33:22.173906Z","iopub.execute_input":"2022-07-12T11:33:22.174320Z","iopub.status.idle":"2022-07-12T11:33:22.479733Z","shell.execute_reply.started":"2022-07-12T11:33:22.174284Z","shell.execute_reply":"2022-07-12T11:33:22.478627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:33:22.482688Z","iopub.execute_input":"2022-07-12T11:33:22.483143Z","iopub.status.idle":"2022-07-12T11:33:22.492185Z","shell.execute_reply.started":"2022-07-12T11:33:22.483102Z","shell.execute_reply":"2022-07-12T11:33:22.491411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:33:22.918739Z","iopub.execute_input":"2022-07-12T11:33:22.919480Z","iopub.status.idle":"2022-07-12T11:33:22.937890Z","shell.execute_reply.started":"2022-07-12T11:33:22.919438Z","shell.execute_reply":"2022-07-12T11:33:22.936829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Label and Discourse Type ","metadata":{}},{"cell_type":"code","source":"percentage_each_class = round(100 * data['discourse_effectiveness'].value_counts() / data.shape[0], 1)\nfig = plt.figure(figsize=(15, 7))\nplt.title('discourse_effectiveness distribution')\nax = sns.barplot(x=percentage_each_class.index,\n                 y=percentage_each_class.values,)\nfor i in ax.containers:\n    ax.bar_label(i,)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:33:23.832247Z","iopub.execute_input":"2022-07-12T11:33:23.832657Z","iopub.status.idle":"2022-07-12T11:33:24.073457Z","shell.execute_reply.started":"2022-07-12T11:33:23.832624Z","shell.execute_reply":"2022-07-12T11:33:24.072406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The labels are unbalanced, in favor of the 'Adequate' label.","metadata":{}},{"cell_type":"code","source":"percentage_each_class = round(100 * data['discourse_type'].value_counts() / data.shape[0], 1)\nfig = plt.figure(figsize=(15, 7))\nplt.title('discourse_type distribution')\nax = sns.barplot(x=percentage_each_class.index,\n                 y=percentage_each_class.values,\n                 color='purple')\nfor i in ax.containers:\n    ax.bar_label(i,)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:33:24.503841Z","iopub.execute_input":"2022-07-12T11:33:24.504232Z","iopub.status.idle":"2022-07-12T11:33:24.760616Z","shell.execute_reply.started":"2022-07-12T11:33:24.504185Z","shell.execute_reply":"2022-07-12T11:33:24.759440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Discourse types 'Evidence' and 'Claim' are by far the most common, so their performance should be prioritized.","metadata":{}},{"cell_type":"code","source":"\nx, y = 'discourse_effectiveness', 'discourse_type'\nfig = (data\n.groupby(y)[x]\n.value_counts(normalize=True)\n.mul(100)\n.rename('percent')\n.reset_index()\n.pipe((sns.catplot,'data'), x=x, y='percent', hue=y, kind='bar', height=10, aspect=2))\n\nfig.ax.set_title('label distribution in each discourse_type subgroup', size=15)\nfor p in fig.ax.patches:\n    txt = str(p.get_height().round(2)) + '%'\n    txt_x = p.get_x() \n    txt_y = p.get_height()\n    fig.ax.text(txt_x + 0.02, txt_y + 1, txt)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:33:24.921706Z","iopub.execute_input":"2022-07-12T11:33:24.922950Z","iopub.status.idle":"2022-07-12T11:33:25.544027Z","shell.execute_reply.started":"2022-07-12T11:33:24.922906Z","shell.execute_reply":"2022-07-12T11:33:25.543223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The label distribution is generally consistent accross the differente types of discourse.  \nThe exceptions would be the **Evidence** type with and increased ineffective rate, and reduced Adequate rate.  \nAnd the **Position** and **Claim** types, with increased adequate rate and reduced ineffective rate, reaching almost 70% of Adequates.","metadata":{}},{"cell_type":"markdown","source":"## Loss Baseline","metadata":{}},{"cell_type":"markdown","source":"### Random Predictions","metadata":{}},{"cell_type":"code","source":"dummy_clf = DummyClassifier(strategy='prior')\ndummy_clf.fit(data.drop('discourse_effectiveness', axis=1), data['discourse_effectiveness'])\ndummy_pred = dummy_clf.predict_proba(data.drop('discourse_effectiveness', axis=1))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:33:27.614840Z","iopub.execute_input":"2022-07-12T11:33:27.615250Z","iopub.status.idle":"2022-07-12T11:33:27.647998Z","shell.execute_reply.started":"2022-07-12T11:33:27.615192Z","shell.execute_reply":"2022-07-12T11:33:27.646804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_binarizer = LabelBinarizer()\ny_oh = label_binarizer.fit_transform(data['discourse_effectiveness'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:33:27.844681Z","iopub.execute_input":"2022-07-12T11:33:27.845089Z","iopub.status.idle":"2022-07-12T11:33:27.950906Z","shell.execute_reply.started":"2022-07-12T11:33:27.845054Z","shell.execute_reply":"2022-07-12T11:33:27.949495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metrics.log_loss(y_oh, dummy_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:34:07.392764Z","iopub.execute_input":"2022-07-12T11:34:07.393312Z","iopub.status.idle":"2022-07-12T11:34:07.420523Z","shell.execute_reply.started":"2022-07-12T11:34:07.393268Z","shell.execute_reply":"2022-07-12T11:34:07.419681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**0.97** would be our loss baseline.  \nAny losses higher than that mean that there was probably some mistake during modeling.","metadata":{}},{"cell_type":"markdown","source":"# Features Engineering","metadata":{}},{"cell_type":"markdown","source":"In this section several features are created and their discriminant power explored.","metadata":{}},{"cell_type":"markdown","source":"### Char Length","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(15, 7))\ndata['char_length'] = data['discourse_text'].apply(lambda x: len(x))\ndata['char_length'].plot.hist(bins=30)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:34:10.110892Z","iopub.execute_input":"2022-07-12T11:34:10.111857Z","iopub.status.idle":"2022-07-12T11:34:10.569035Z","shell.execute_reply.started":"2022-07-12T11:34:10.111810Z","shell.execute_reply":"2022-07-12T11:34:10.567885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(15, 7))\nsns.boxplot(x=\"discourse_effectiveness\", y=\"char_length\", data=data)\nplt.title('Aggregate label distribution')\nplt.grid()\nplt.ylim(0, 800)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:38:05.608089Z","iopub.execute_input":"2022-07-12T11:38:05.608488Z","iopub.status.idle":"2022-07-12T11:38:05.864659Z","shell.execute_reply.started":"2022-07-12T11:38:05.608456Z","shell.execute_reply":"2022-07-12T11:38:05.863266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It looks that the mean is a bit bigger for the effective label, but as the size of the text depends greatly on the type of discourse,  \nthe distribution of the labels in each subgroup its more relevant:","metadata":{}},{"cell_type":"code","source":"discourse_types = np.unique(data['discourse_type'])\n\nfig, axes = plt.subplots(2, 4, figsize=(20, 4 * 3))\n\nfor ax, discourse_type in zip(axes.flat, discourse_types):\n    ax.set_title(discourse_type)\n    sns.boxplot(x=\"discourse_effectiveness\",\n                y=\"char_length\",\n                data=data[data['discourse_type'] == discourse_type],\n                palette={'Adequate': 'tab:blue',\n                         'Effective': 'tab:green',\n                         'Ineffective': 'tab:orange'},\n                ax=ax)\n    ax.set_ylim(0, np.quantile(data[data['discourse_type'] == discourse_type]['char_length'], q=0.95))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:42:04.012776Z","iopub.execute_input":"2022-07-12T11:42:04.013142Z","iopub.status.idle":"2022-07-12T11:42:05.202708Z","shell.execute_reply.started":"2022-07-12T11:42:04.013111Z","shell.execute_reply":"2022-07-12T11:42:05.201582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Grouping by discourse type, the difference in distributions is a lot more pronounced.  \nSpecially in some discourse types, such as Concluding Statement, Evidence and Lead, the effective label has a strong correlation with higher char_length.","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(15, 7))\nsns.displot(data, x='char_length', hue='discourse_effectiveness', fill=True,\n            stat=\"density\", common_norm=False,\n            aspect=2, height=6)\nplt.xlim(0, 1250)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:34:12.874742Z","iopub.execute_input":"2022-07-12T11:34:12.875454Z","iopub.status.idle":"2022-07-12T11:34:15.336370Z","shell.execute_reply.started":"2022-07-12T11:34:12.875414Z","shell.execute_reply":"2022-07-12T11:34:15.335204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f_oneway(data['char_length'][data['discourse_effectiveness'] == 'Adequate'],\n         data['char_length'][data['discourse_effectiveness'] == 'Ineffective'],\n         data['char_length'][data['discourse_effectiveness'] == 'Effective'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:34:15.338320Z","iopub.execute_input":"2022-07-12T11:34:15.338971Z","iopub.status.idle":"2022-07-12T11:34:15.357172Z","shell.execute_reply.started":"2022-07-12T11:34:15.338938Z","shell.execute_reply":"2022-07-12T11:34:15.355935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Word Count","metadata":{}},{"cell_type":"markdown","source":"As expected this feature has a really similar behaviour than the char_lenght","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(15, 7))\ndata['word_count'] = data['discourse_text'].apply(lambda x: len(x.split()))\ndata['word_count'].plot.hist(bins=30)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:34:15.579160Z","iopub.execute_input":"2022-07-12T11:34:15.580271Z","iopub.status.idle":"2022-07-12T11:34:15.983088Z","shell.execute_reply.started":"2022-07-12T11:34:15.580225Z","shell.execute_reply":"2022-07-12T11:34:15.981970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(15, 7))\nsns.boxplot(x=\"discourse_effectiveness\", y=\"word_count\", data=data)\nplt.grid()\nplt.ylim(0, 225)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:34:15.984947Z","iopub.execute_input":"2022-07-12T11:34:15.985399Z","iopub.status.idle":"2022-07-12T11:34:16.253787Z","shell.execute_reply.started":"2022-07-12T11:34:15.985356Z","shell.execute_reply":"2022-07-12T11:34:16.252351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 4, figsize=(20, 4 * 3))\n\nfor ax, discourse_type in zip(axes.flat, discourse_types):\n    ax.set_title(discourse_type)\n    sns.boxplot(x=\"discourse_effectiveness\",\n                y=\"word_count\",\n                data=data[data['discourse_type'] == discourse_type],\n                palette={'Adequate': 'tab:blue',\n                         'Effective': 'tab:green',\n                         'Ineffective': 'tab:orange'},\n                ax=ax)\n    ax.set_ylim(0, np.quantile(data[data['discourse_type'] == discourse_type]['word_count'], q=0.95))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:41:28.366641Z","iopub.execute_input":"2022-07-12T11:41:28.367027Z","iopub.status.idle":"2022-07-12T11:41:29.574536Z","shell.execute_reply.started":"2022-07-12T11:41:28.366997Z","shell.execute_reply":"2022-07-12T11:41:29.573503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f_oneway(data['word_count'][data['discourse_effectiveness'] == 'Adequate'],\n         data['word_count'][data['discourse_effectiveness'] == 'Ineffective'],\n         data['word_count'][data['discourse_effectiveness'] == 'Effective'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:42:12.998775Z","iopub.execute_input":"2022-07-12T11:42:12.999192Z","iopub.status.idle":"2022-07-12T11:42:13.019022Z","shell.execute_reply.started":"2022-07-12T11:42:12.999158Z","shell.execute_reply":"2022-07-12T11:42:13.018123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Anova one way test strongly suggest this result isnt spurious and the distributions of each subgroup are indeed different.","metadata":{}},{"cell_type":"markdown","source":"### Average Word Length","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(15, 7))\ndata['avg_word_length'] = data['char_length'] / data['word_count']\ndata['avg_word_length'].plot.hist(bins=30)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:42:20.410241Z","iopub.execute_input":"2022-07-12T11:42:20.411529Z","iopub.status.idle":"2022-07-12T11:42:20.699055Z","shell.execute_reply.started":"2022-07-12T11:42:20.411420Z","shell.execute_reply":"2022-07-12T11:42:20.698286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(15, 7))\nsns.boxplot(x=\"discourse_effectiveness\", y=\"avg_word_length\", data=data)\nplt.grid()\nplt.ylim(3, 8)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:42:21.749857Z","iopub.execute_input":"2022-07-12T11:42:21.750263Z","iopub.status.idle":"2022-07-12T11:42:21.976720Z","shell.execute_reply.started":"2022-07-12T11:42:21.750229Z","shell.execute_reply":"2022-07-12T11:42:21.975909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 4, figsize=(20, 4 * 3))\n\nfor ax, discourse_type in zip(axes.flat, discourse_types):\n    ax.set_title(discourse_type)\n    sns.boxplot(x=\"discourse_effectiveness\",\n                y=\"avg_word_length\",\n                data=data[data['discourse_type'] == discourse_type],\n                palette={'Adequate': 'tab:blue',\n                         'Effective': 'tab:green',\n                         'Ineffective': 'tab:orange'},\n                ax=ax)\n    ax.set_ylim(np.quantile(data[data['discourse_type'] == discourse_type]['avg_word_length'], q=0.05),\n                np.quantile(data[data['discourse_type'] == discourse_type]['avg_word_length'], q=0.95))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:42:35.536082Z","iopub.execute_input":"2022-07-12T11:42:35.536481Z","iopub.status.idle":"2022-07-12T11:42:37.116560Z","shell.execute_reply.started":"2022-07-12T11:42:35.536447Z","shell.execute_reply":"2022-07-12T11:42:37.115496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In Rebuttal type its looks like higher average word count strongly correlates with better calification.","metadata":{}},{"cell_type":"code","source":"sns.displot(data, x='avg_word_length', hue='discourse_effectiveness', fill=True,\n            stat=\"density\", common_norm=False,\n            aspect=2, height=6)\nplt.xlim(4, 7)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:42:37.461112Z","iopub.execute_input":"2022-07-12T11:42:37.461513Z","iopub.status.idle":"2022-07-12T11:42:40.258294Z","shell.execute_reply.started":"2022-07-12T11:42:37.461466Z","shell.execute_reply":"2022-07-12T11:42:40.257493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f_oneway(data['avg_word_length'][data['discourse_effectiveness'] == 'Adequate'],\n         data['avg_word_length'][data['discourse_effectiveness'] == 'Ineffective'],\n         data['avg_word_length'][data['discourse_effectiveness'] == 'Effective'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:42:43.095525Z","iopub.execute_input":"2022-07-12T11:42:43.096373Z","iopub.status.idle":"2022-07-12T11:42:43.115955Z","shell.execute_reply.started":"2022-07-12T11:42:43.096336Z","shell.execute_reply":"2022-07-12T11:42:43.114755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Appearence of 'Source'","metadata":{}},{"cell_type":"code","source":"data['contains_source'] = data['discourse_text'].apply(lambda x: 'source' in x.lower().split())\ndata['contains_source'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:46:27.464578Z","iopub.execute_input":"2022-07-12T11:46:27.465771Z","iopub.status.idle":"2022-07-12T11:46:27.611089Z","shell.execute_reply.started":"2022-07-12T11:46:27.465722Z","shell.execute_reply":"2022-07-12T11:46:27.610029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['contains_source'].apply(lambda x: x)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:46:27.953783Z","iopub.execute_input":"2022-07-12T11:46:27.954521Z","iopub.status.idle":"2022-07-12T11:46:27.971516Z","shell.execute_reply.started":"2022-07-12T11:46:27.954475Z","shell.execute_reply":"2022-07-12T11:46:27.970436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x, y = 'discourse_effectiveness', 'contains_source'\nfig = (data[data[y].notnull()]\n.groupby(y)[x]\n.value_counts(normalize=True)\n.mul(100)\n.rename('percent')\n.reset_index()\n.pipe((sns.catplot,'data'), x=x,y='percent',hue=y,kind='bar'))\n\nfor p in fig.ax.patches:\n    txt = str(p.get_height().round(2)) + '%'\n    txt_x = p.get_x() \n    txt_y = p.get_height()\n    fig.ax.text(txt_x + 0.02, txt_y + 1, txt)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:46:28.624023Z","iopub.execute_input":"2022-07-12T11:46:28.625020Z","iopub.status.idle":"2022-07-12T11:46:29.020946Z","shell.execute_reply.started":"2022-07-12T11:46:28.624977Z","shell.execute_reply":"2022-07-12T11:46:29.020153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x, y = 'discourse_effectiveness', 'contains_source'\nfig = (data[data['discourse_type'] == 'Evidence']\n.groupby(y)[x]\n.value_counts(normalize=True)\n.mul(100)\n.rename('percent')\n.reset_index()\n.pipe((sns.catplot,'data'), x=x,y='percent',hue=y,kind='bar'))\n\nfor p in fig.ax.patches:\n    txt = str(p.get_height().round(2)) + '%'\n    txt_x = p.get_x() \n    txt_y = p.get_height()\n    fig.ax.text(txt_x + 0.02, txt_y + 1, txt)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:46:51.783000Z","iopub.execute_input":"2022-07-12T11:46:51.783377Z","iopub.status.idle":"2022-07-12T11:46:52.180149Z","shell.execute_reply.started":"2022-07-12T11:46:51.783345Z","shell.execute_reply":"2022-07-12T11:46:52.178824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_selection.chi2(np.array(data['contains_source']).reshape(-1, 1), data['discourse_effectiveness'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:49:35.735602Z","iopub.execute_input":"2022-07-12T11:49:35.735962Z","iopub.status.idle":"2022-07-12T11:49:35.846708Z","shell.execute_reply.started":"2022-07-12T11:49:35.735935Z","shell.execute_reply":"2022-07-12T11:49:35.845640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Although not a very common word(389/36765 texts) it seems to have significant discriminant power, specially in the Evidence type.  \nIn this type the probability of being effective raises from 23% to 40% if the word 'source' is present","metadata":{}},{"cell_type":"markdown","source":"### Appearence of 'I'","metadata":{}},{"cell_type":"code","source":"data['contains_I'] = data['discourse_text'].apply(lambda x: 'i' in x.lower().split())\ndata['contains_I'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:34:18.572700Z","iopub.execute_input":"2022-07-12T11:34:18.573067Z","iopub.status.idle":"2022-07-12T11:34:18.709739Z","shell.execute_reply.started":"2022-07-12T11:34:18.573037Z","shell.execute_reply":"2022-07-12T11:34:18.708560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x, y = 'discourse_effectiveness', 'contains_I'\nfig = (data\n.groupby(y)[x]\n.value_counts(normalize=True)\n.mul(100)\n.rename('percent')\n.reset_index()\n.pipe((sns.catplot,'data'), x=x,y='percent',hue=y,kind='bar'))\n\nfor p in fig.ax.patches:\n    txt = str(p.get_height().round(2)) + '%'\n    txt_x = p.get_x() \n    txt_y = p.get_height()\n    fig.ax.text(txt_x + 0.02, txt_y + 1, txt)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:34:18.756753Z","iopub.execute_input":"2022-07-12T11:34:18.757110Z","iopub.status.idle":"2022-07-12T11:34:19.078875Z","shell.execute_reply.started":"2022-07-12T11:34:18.757073Z","shell.execute_reply":"2022-07-12T11:34:19.077379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x, y = 'discourse_effectiveness', 'contains_I'\nfig = (data[data['discourse_type'] == 'Claim']\n.groupby(y)[x]\n.value_counts(normalize=True)\n.mul(100)\n.rename('percent')\n.reset_index()\n.pipe((sns.catplot,'data'), x=x,y='percent',hue=y,kind='bar'))\n\nfor p in fig.ax.patches:\n    txt = str(p.get_height().round(2)) + '%'\n    txt_x = p.get_x() \n    txt_y = p.get_height()\n    fig.ax.text(txt_x + 0.02, txt_y + 1, txt)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:34:19.080799Z","iopub.execute_input":"2022-07-12T11:34:19.081151Z","iopub.status.idle":"2022-07-12T11:34:19.506037Z","shell.execute_reply.started":"2022-07-12T11:34:19.081110Z","shell.execute_reply":"2022-07-12T11:34:19.505102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x, y = 'discourse_effectiveness', 'contains_I'\nfig = (data[data['discourse_type'] == 'Evidence']\n.groupby(y)[x]\n.value_counts(normalize=True)\n.mul(100)\n.rename('percent')\n.reset_index()\n.pipe((sns.catplot,'data'), x=x,y='percent',hue=y,kind='bar'))\n\nfor p in fig.ax.patches:\n    txt = str(p.get_height().round(2)) + '%'\n    txt_x = p.get_x() \n    txt_y = p.get_height()\n    fig.ax.text(txt_x + 0.02, txt_y + 1, txt)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:34:19.508167Z","iopub.execute_input":"2022-07-12T11:34:19.509188Z","iopub.status.idle":"2022-07-12T11:34:19.895615Z","shell.execute_reply.started":"2022-07-12T11:34:19.509144Z","shell.execute_reply":"2022-07-12T11:34:19.894300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_selection.chi2(np.array(data['contains_I']).reshape(-1, 1), data['discourse_effectiveness'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:34:19.937476Z","iopub.execute_input":"2022-07-12T11:34:19.938161Z","iopub.status.idle":"2022-07-12T11:34:20.060426Z","shell.execute_reply.started":"2022-07-12T11:34:19.938120Z","shell.execute_reply":"2022-07-12T11:34:20.059107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In the Claim type the appeareance of the word I, significantly affect the probability of the label(higher chance of ineffective and lower change of effective)","metadata":{}},{"cell_type":"markdown","source":"### Repetead words","metadata":{}},{"cell_type":"code","source":"from collections import Counter\nfrom nltk.corpus import stopwords\nfrom stop_words import get_stop_words\n\nstopwords = list(get_stop_words('en'))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:51:39.329150Z","iopub.execute_input":"2022-07-12T11:51:39.330185Z","iopub.status.idle":"2022-07-12T11:51:39.335849Z","shell.execute_reply.started":"2022-07-12T11:51:39.330143Z","shell.execute_reply":"2022-07-12T11:51:39.334953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def max_repeated_word_count(text):\n    words = [word for word in text.split() if word not in stopwords]\n\n    word_counts = Counter(words)\n    try:\n        return word_counts.most_common(1)[0][1]\n    \n    except IndexError:\n        return 0\n        \n    return max_count\n\n# Counter([word for word in text.split() if word not in stopwords]).most_common(1)[0][1]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:51:40.551184Z","iopub.execute_input":"2022-07-12T11:51:40.551566Z","iopub.status.idle":"2022-07-12T11:51:40.557364Z","shell.execute_reply.started":"2022-07-12T11:51:40.551535Z","shell.execute_reply":"2022-07-12T11:51:40.556275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index = 9\ntext = data['discourse_text'][index]\nprint(Counter([word for word in text.split() if word not in stopwords]).most_common(1)[0])\nprint(data['discourse_effectiveness'][index])\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:51:41.975601Z","iopub.execute_input":"2022-07-12T11:51:41.976667Z","iopub.status.idle":"2022-07-12T11:51:41.983587Z","shell.execute_reply.started":"2022-07-12T11:51:41.976628Z","shell.execute_reply":"2022-07-12T11:51:41.982421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['max_repeated_word_count'] = data['discourse_text'].apply(max_repeated_word_count)\ndata['max_repeated_word_count'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:51:47.044227Z","iopub.execute_input":"2022-07-12T11:51:47.045147Z","iopub.status.idle":"2022-07-12T11:51:50.035241Z","shell.execute_reply.started":"2022-07-12T11:51:47.045084Z","shell.execute_reply":"2022-07-12T11:51:50.034078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['max_repeated_word_count'].hist()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:51:50.037279Z","iopub.execute_input":"2022-07-12T11:51:50.037937Z","iopub.status.idle":"2022-07-12T11:51:50.286905Z","shell.execute_reply.started":"2022-07-12T11:51:50.037892Z","shell.execute_reply":"2022-07-12T11:51:50.285793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(15, 7))\nsns.boxplot(x=\"discourse_effectiveness\", y=\"max_repeated_word_count\", data=data)\nplt.grid()\nplt.ylim(0, 6)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:51:50.289168Z","iopub.execute_input":"2022-07-12T11:51:50.289600Z","iopub.status.idle":"2022-07-12T11:51:50.522111Z","shell.execute_reply.started":"2022-07-12T11:51:50.289566Z","shell.execute_reply":"2022-07-12T11:51:50.521035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 4, figsize=(20, 4 * 3))\n\nfor ax, discourse_type in zip(axes.flat, discourse_types):\n    ax.set_title(discourse_type)\n    sns.boxplot(x=\"discourse_effectiveness\",\n                y=\"max_repeated_word_count\",\n                data=data[data['discourse_type'] == discourse_type],\n                palette={'Adequate': 'tab:blue',\n                         'Effective': 'tab:green',\n                         'Ineffective': 'tab:orange'},\n                ax=ax)\n    ax.set_ylim(np.quantile(data[data['discourse_type'] == discourse_type]['max_repeated_word_count'], q=0.05),\n                np.quantile(data[data['discourse_type'] == discourse_type]['max_repeated_word_count'], q=0.95))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:52:16.404773Z","iopub.execute_input":"2022-07-12T11:52:16.405417Z","iopub.status.idle":"2022-07-12T11:52:17.918346Z","shell.execute_reply.started":"2022-07-12T11:52:16.405367Z","shell.execute_reply":"2022-07-12T11:52:17.917137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Repeating words seems to be relevant only in some types of discourse.  \nIt seems irrelevant in the Rebuttal type for example.  \nBut it looks really promissing for the Concluding Statement, Lead and Evidence types.","metadata":{}},{"cell_type":"markdown","source":"### Sentiment Analysis","metadata":{}},{"cell_type":"code","source":"from nltk.sentiment import SentimentIntensityAnalyzer\n\nsia = SentimentIntensityAnalyzer()\nsia.polarity_scores(\"Wow, NLTK is really powerful!\")['compound']","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:54:01.158749Z","iopub.execute_input":"2022-07-12T11:54:01.159148Z","iopub.status.idle":"2022-07-12T11:54:01.174411Z","shell.execute_reply.started":"2022-07-12T11:54:01.159115Z","shell.execute_reply":"2022-07-12T11:54:01.173583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndata['sentiment'] = data['discourse_text'].apply(lambda text: sia.polarity_scores(text)['compound'])\ndata['sentiment'].hist()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:54:01.878307Z","iopub.execute_input":"2022-07-12T11:54:01.879451Z","iopub.status.idle":"2022-07-12T11:54:19.688406Z","shell.execute_reply.started":"2022-07-12T11:54:01.879404Z","shell.execute_reply":"2022-07-12T11:54:19.687250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(15, 7))\nsns.boxplot(x=\"discourse_effectiveness\", y=\"sentiment\", data=data)\nplt.grid()\nplt.ylim(-1, 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:54:19.690286Z","iopub.execute_input":"2022-07-12T11:54:19.690662Z","iopub.status.idle":"2022-07-12T11:54:19.872097Z","shell.execute_reply.started":"2022-07-12T11:54:19.690629Z","shell.execute_reply":"2022-07-12T11:54:19.871298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 4, figsize=(20, 4 * 3))\n\nfor ax, discourse_type in zip(axes.flat, discourse_types):\n    ax.set_title(discourse_type)\n    sns.boxplot(x=\"discourse_effectiveness\",\n                y=\"sentiment\",\n                data=data[data['discourse_type'] == discourse_type],\n                palette={'Adequate': 'tab:blue',\n                         'Effective': 'tab:green',\n                         'Ineffective': 'tab:orange'},\n                ax=ax)\n    ax.set_ylim(np.quantile(data[data['discourse_type'] == discourse_type]['sentiment'], q=0.05),\n                np.quantile(data[data['discourse_type'] == discourse_type]['sentiment'], q=0.95))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:54:19.873642Z","iopub.execute_input":"2022-07-12T11:54:19.875001Z","iopub.status.idle":"2022-07-12T11:54:21.026314Z","shell.execute_reply.started":"2022-07-12T11:54:19.874953Z","shell.execute_reply":"2022-07-12T11:54:21.024995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Sentiment seems relevant for the Rebuttal and Concluding Statement types specially.","metadata":{}},{"cell_type":"code","source":"f_oneway(data['sentiment'][data['discourse_effectiveness'] == 'Adequate'],\n         data['sentiment'][data['discourse_effectiveness'] == 'Ineffective'],\n         data['sentiment'][data['discourse_effectiveness'] == 'Effective'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:55:04.974488Z","iopub.execute_input":"2022-07-12T11:55:04.974893Z","iopub.status.idle":"2022-07-12T11:55:04.994033Z","shell.execute_reply.started":"2022-07-12T11:55:04.974860Z","shell.execute_reply":"2022-07-12T11:55:04.992871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.displot(data[data['discourse_type'] == 'Rebuttal'], x='sentiment', hue='discourse_effectiveness', fill=True,\n            stat=\"density\", common_norm=False,\n            aspect=2, height=6)\nplt.title('Rebuttal')\nplt.xlim(-1, 1)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:56:37.170175Z","iopub.execute_input":"2022-07-12T11:56:37.170559Z","iopub.status.idle":"2022-07-12T11:56:37.857995Z","shell.execute_reply.started":"2022-07-12T11:56:37.170529Z","shell.execute_reply":"2022-07-12T11:56:37.856882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.displot(data[data['discourse_type'] == 'Concluding Statement'], x='sentiment', hue='discourse_effectiveness', fill=True,\n            stat=\"density\", common_norm=False,\n            aspect=2, height=6)\nplt.title('Concluding Statement')\nplt.xlim(-1, 1)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:56:47.724658Z","iopub.execute_input":"2022-07-12T11:56:47.725043Z","iopub.status.idle":"2022-07-12T11:56:48.417506Z","shell.execute_reply.started":"2022-07-12T11:56:47.725012Z","shell.execute_reply":"2022-07-12T11:56:48.416284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Spelling Mistakes","metadata":{}},{"cell_type":"code","source":"!pip install pyspellchecker","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:56:51.469201Z","iopub.execute_input":"2022-07-12T11:56:51.469587Z","iopub.status.idle":"2022-07-12T11:57:02.556729Z","shell.execute_reply.started":"2022-07-12T11:56:51.469557Z","shell.execute_reply":"2022-07-12T11:57:02.555291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import Counter\nfrom spellchecker import SpellChecker\nimport string\n\nspell = SpellChecker(distance=1)\nspell.word_frequency.load_words(['landform', 'cydonia', 'nondemocratic', '3d', 'dr'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:57:02.558972Z","iopub.execute_input":"2022-07-12T11:57:02.559338Z","iopub.status.idle":"2022-07-12T11:57:02.744563Z","shell.execute_reply.started":"2022-07-12T11:57:02.559306Z","shell.execute_reply":"2022-07-12T11:57:02.743404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ncounter = Counter()\nfor text in tqdm(data['discourse_text']):\n    words = text.translate(str.maketrans('', '', string.punctuation)).split()\n    misspelled = spell.unknown(words)\n    counter.update(misspelled)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:57:02.745776Z","iopub.execute_input":"2022-07-12T11:57:02.746080Z","iopub.status.idle":"2022-07-12T11:57:06.442222Z","shell.execute_reply.started":"2022-07-12T11:57:02.746052Z","shell.execute_reply":"2022-07-12T11:57:06.441222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter.most_common(20)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:57:06.444292Z","iopub.execute_input":"2022-07-12T11:57:06.444620Z","iopub.status.idle":"2022-07-12T11:57:06.454792Z","shell.execute_reply.started":"2022-07-12T11:57:06.444591Z","shell.execute_reply":"2022-07-12T11:57:06.453688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spell_checker = SpellChecker(distance=1)\nspell_checker.word_frequency.load_words(['landform', 'cydonia', 'nondemocratic', '3d', 'dr'])\n\ndef get_num_spelling_mistakes(text, spell_checker):\n    words = text.translate(str.maketrans('', '', string.punctuation)).split()\n    misspelled = spell.unknown(words)\n    return len(misspelled)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:57:06.456837Z","iopub.execute_input":"2022-07-12T11:57:06.457328Z","iopub.status.idle":"2022-07-12T11:57:06.638187Z","shell.execute_reply.started":"2022-07-12T11:57:06.457285Z","shell.execute_reply":"2022-07-12T11:57:06.636932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['n_spelling_mistakes'] = data['discourse_text'].apply(lambda text: get_num_spelling_mistakes(text, spell_checker=spell_checker))\ndata['n_spelling_mistakes'].hist()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:57:06.640143Z","iopub.execute_input":"2022-07-12T11:57:06.640762Z","iopub.status.idle":"2022-07-12T11:57:10.375604Z","shell.execute_reply.started":"2022-07-12T11:57:06.640726Z","shell.execute_reply":"2022-07-12T11:57:10.374282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(15, 7))\nsns.boxplot(x=\"discourse_effectiveness\", y=\"n_spelling_mistakes\", data=data)\nplt.grid()\nplt.ylim(0, 10)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:57:10.378420Z","iopub.execute_input":"2022-07-12T11:57:10.378765Z","iopub.status.idle":"2022-07-12T11:57:10.611411Z","shell.execute_reply.started":"2022-07-12T11:57:10.378736Z","shell.execute_reply":"2022-07-12T11:57:10.610302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 4, figsize=(20, 4 * 3))\n\nfor ax, discourse_type in zip(axes.flat, discourse_types):\n    ax.set_title(discourse_type)\n    sns.boxplot(x=\"discourse_effectiveness\",\n                y=\"n_spelling_mistakes\",\n                data=data[data['discourse_type'] == discourse_type],\n                palette={'Adequate': 'tab:blue',\n                         'Effective': 'tab:green',\n                         'Ineffective': 'tab:orange'},\n                ax=ax)\n    ax.set_ylim(np.quantile(data[data['discourse_type'] == discourse_type]['n_spelling_mistakes'], q=0.05),\n                np.quantile(data[data['discourse_type'] == discourse_type]['n_spelling_mistakes'], q=0.95))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:57:14.136433Z","iopub.execute_input":"2022-07-12T11:57:14.136827Z","iopub.status.idle":"2022-07-12T11:57:15.392317Z","shell.execute_reply.started":"2022-07-12T11:57:14.136792Z","shell.execute_reply":"2022-07-12T11:57:15.391031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It seems more interesting to just consider if the text has a spelling mistake or not, instead of the exact number","metadata":{}},{"cell_type":"code","source":"data['spelling_mistakes'] = data['n_spelling_mistakes'].apply(lambda num: False if num==0 else True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:57:47.420603Z","iopub.execute_input":"2022-07-12T11:57:47.421283Z","iopub.status.idle":"2022-07-12T11:57:47.433706Z","shell.execute_reply.started":"2022-07-12T11:57:47.421241Z","shell.execute_reply":"2022-07-12T11:57:47.432894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x, y = 'discourse_effectiveness', 'spelling_mistakes'\nfig = (data\n.groupby(y)[x]\n.value_counts(normalize=True)\n.mul(100)\n.rename('percent')\n.reset_index()\n.pipe((sns.catplot,'data'), x=x,y='percent',hue=y,kind='bar'))\n\nfor p in fig.ax.patches:\n    txt = str(p.get_height().round(2)) + '%'\n    txt_x = p.get_x() \n    txt_y = p.get_height()\n    fig.ax.text(txt_x + 0.02, txt_y + 1, txt)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:57:48.199977Z","iopub.execute_input":"2022-07-12T11:57:48.200561Z","iopub.status.idle":"2022-07-12T11:57:48.588842Z","shell.execute_reply.started":"2022-07-12T11:57:48.200519Z","shell.execute_reply":"2022-07-12T11:57:48.588095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The presence of spelling mistakes seems to increase the chance of the Ineffective label","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}