{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Evaluation of the dependencies between the binary targets."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# packages\n\n# standard\nimport numpy as np\nimport pandas as pd\n\n# plots\nimport matplotlib.pyplot as plt\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# load data and first glance\ndf_train = pd.read_csv('../input/ranzcr-clip-catheter-line-classification/train.csv')\ndf_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# dimensions of data frame\nn_rows = df_train.shape[0]\ndf_train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# select target columns\ntargets = ['ETT - Abnormal', 'ETT - Borderline',\n           'ETT - Normal', 'NGT - Abnormal', 'NGT - Borderline',\n           'NGT - Incompletely Imaged', 'NGT - Normal', 'CVC - Abnormal',\n           'CVC - Borderline', 'CVC - Normal', 'Swan Ganz Catheter Present']\n\ntargets_ETT = ['ETT - Abnormal', 'ETT - Borderline', 'ETT - Normal']\n\ntargets_NGT = ['NGT - Abnormal', 'NGT - Borderline','NGT - Incompletely Imaged',\n               'NGT - Normal']\n\ntargets_CVC = ['CVC - Abnormal', 'CVC - Borderline', 'CVC - Normal']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# count entries per column\ncol_counts = df_train[targets].sum(axis=0)\n# and plot\ncol_counts.plot(kind='bar')\nplt.title('Absolute frequencies')\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# plot relative counts\n(col_counts / n_rows).plot(kind='bar')\nplt.title('Percentages')\nplt.grid()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### The indicators are not unique in each row (see e. g. the second row). Let's evaluate the multiplicities:"},{"metadata":{"trusted":true},"cell_type":"code","source":"# count\ndf_train['Sum_Indicators'] = df_train[targets].sum(axis=1)\nprint(df_train.Sum_Indicators.value_counts().sort_index())\n# and plot\ndf_train.Sum_Indicators.value_counts().sort_index().plot(kind='bar')\nplt.title('Multiplicities of indicators')\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# let's check the most extreme cases (count=6)\ndf_train[df_train.Sum_Indicators==6]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# we have also 24 rows without any \"1\":\ndf_train[df_train.Sum_Indicators==0]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Correlation of columns"},{"metadata":{},"cell_type":"markdown","source":"#### Pearson correlation is not a really good tool for binary variables, nevertheless let's use it to get a first impression. At the end of this notebook you will find a better (asymmetric) alternative."},{"metadata":{"trusted":true},"cell_type":"code","source":"# correlation of columns\ncorr_pearson = df_train[targets].corr()\n# plot correlation matrix\nfig = plt.figure(figsize = (12,9))\nsns.heatmap(corr_pearson, annot=True, cmap='RdYlGn')\nplt.title('Pearson correlation')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### We can observe some connections between ETT and NGT as well as something happening withing the CVC \"block\"."},{"metadata":{},"cell_type":"markdown","source":"# ETT Targets (endotracheal tube)"},{"metadata":{"trusted":true},"cell_type":"code","source":"# look at ETT targets only\ndf_train['Sum_ETT'] = df_train[targets_ETT].sum(axis=1)\nprint(df_train.Sum_ETT.value_counts().sort_index())\n# and plot\ndf_train.Sum_ETT.value_counts().sort_index().plot(kind='bar')\nplt.title('Multiplicities of ETT indicators')\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Indicators of ETT targets are actually mutually exclusive. So we could in theory convert the ETT columns into only one (multi-class) column:"},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train['ETT'] = 'NONE'\ndf_train.loc[df_train['ETT - Abnormal']==1,'ETT'] = 'ETT_Abnormal'\ndf_train.loc[df_train['ETT - Borderline']==1,'ETT'] = 'ETT_Borderline'\ndf_train.loc[df_train['ETT - Normal']==1,'ETT'] = 'ETT_Normal'\n# evaluate frequencies\ndf_train.ETT.value_counts()\ndf_train.ETT.value_counts().plot(kind='bar')\nplt.title('Frequency of ETT Targets')\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# NGT Targets (nasogastric tube)"},{"metadata":{"trusted":true},"cell_type":"code","source":"# now look at NGT targets only\ndf_train['Sum_NGT'] = df_train[targets_NGT].sum(axis=1)\nprint(df_train.Sum_NGT.value_counts().sort_index())\n# and plot\ndf_train.Sum_NGT.value_counts().sort_index().plot(kind='bar')\nplt.title('Multiplicities of NGT indicators')\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Indicators of NGT targets are NOT mutually exclusive (however, we have \"only\" 45 rows with double entries). This is actually multi-label not multi-class...\n#### Let's check the cross tables to identify common occurrences:"},{"metadata":{"trusted":true},"cell_type":"code","source":"targets_TEMP = targets_NGT\nnn = len(targets_TEMP)\nfor i in range(1,nn+1):\n    for j in range(1,nn+1):\n       if (i<j):\n        f1 = targets_TEMP[i-1]\n        f2 = targets_TEMP[j-1]\n        print(pd.crosstab(df_train[f1], df_train[f2]))\n        print('\\n')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Interesting: In 9 cases we have \"Normal\" and \"Abnormal\" at the same time... No contradiction, we can have more than one object within one image!\n\n#### Let's evaluate the conditional frequencies (e. g. NGT-Normal=1 given NGT-Borderline=1 and vice versa) systematically."},{"metadata":{"trusted":true},"cell_type":"code","source":"# use correlation matrix as container for the frequencies\ncond_NGT = df_train[targets_NGT].corr()\n\n# calc frequency of x given y for all pairs\nfor i in range(1,nn+1):\n    for j in range(1,nn+1):\n       if (i!=j):\n        f1 = targets_TEMP[i-1]\n        f2 = targets_TEMP[j-1]\n        ctab = pd.crosstab(df_train[f1], df_train[f2])\n        n_1 = df_train[f1].sum() # feature 1 = 1\n        n_both = ctab.iloc[1,1]  # both features = 1\n        perc_2_given_1 = n_both / n_1 # feature_2 = 1 given feature_1 = 1\n        print('Percentage ',f2,' given ',f1,':',np.round(perc_2_given_1,4))\n        cond_NGT.loc[f1,f2] = perc_2_given_1 # store value in correlation matrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# plot values as matrix\nsns.heatmap(cond_NGT, annot=True, cmap='RdYlGn')\nplt.title('Conditional Frequencies')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Example how to read this:\n* Relative Frequency of Borderline=1 given Normal=1 can be looked up in \"Normal\"-row: 0.0027.\n* Relative Frequency of Normal=1 given Borderline=1 can be looked up in \"Borderline\"-row: 0.025."},{"metadata":{},"cell_type":"markdown","source":"# CVC Targets (central venous catheter)"},{"metadata":{"trusted":true},"cell_type":"code","source":"# look at CVC targets only\ndf_train['Sum_CVC'] = df_train[targets_CVC].sum(axis=1)\nprint(df_train.Sum_CVC.value_counts().sort_index())\n# and plot\ndf_train.Sum_CVC.value_counts().sort_index().plot(kind='bar')\nplt.title('Multiplicities of CVC indicators')\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Indicators of CVC targets are also NOT mutually exclusive. Here we have quite a few duplicates and even 71 \"triples\".\n#### Let's check the cross tables again:"},{"metadata":{"trusted":true},"cell_type":"code","source":"targets_TEMP = targets_CVC\nnn = len(targets_TEMP)\nfor i in range(1,nn+1):\n    for j in range(1,nn+1):\n       if (i<j):\n        f1 = targets_TEMP[i-1]\n        f2 = targets_TEMP[j-1]\n        print(pd.crosstab(df_train[f1], df_train[f2]))\n        print('\\n')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### We see that there are many (2607) common occurrences of \"Normal\" and \"Borderline\".\n\n#### Let's evaluate the conditional frequencies (e. g. CVC Normal=1 given CVC Borderline=1) again."},{"metadata":{"trusted":true},"cell_type":"code","source":"# use correlation matrix as container for the frequencies\ncond_CVC = df_train[targets_CVC].corr()\n\n# calc frequency of x given y for all pairs\nfor i in range(1,nn+1):\n    for j in range(1,nn+1):\n       if (i!=j):\n        f1 = targets_TEMP[i-1]\n        f2 = targets_TEMP[j-1]\n        ctab = pd.crosstab(df_train[f1], df_train[f2])\n        n_1 = df_train[f1].sum()\n        n_both = ctab.iloc[1,1] \n        perc_2_given_1 = n_both / n_1\n        print('Percentage ',f2,' given ',f1,':',np.round(perc_2_given_1,4))\n        cond_CVC.loc[f1,f2] = perc_2_given_1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# plot matrix of conditional frequencies\nsns.heatmap(cond_CVC, annot=True, cmap='RdYlGn')\nplt.title('Conditional Frequencies - CVC Targets')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Swan Ganz Catheter"},{"metadata":{"trusted":true},"cell_type":"code","source":"# finally let's check the \"Swan Ganz Catheter Present\" target:\ndf_train['Swan Ganz Catheter Present'].value_counts()\ndf_train['Swan Ganz Catheter Present'].value_counts().plot(kind='bar')\nplt.title('Swan Ganz Catheter Present')\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Nothing really interesting here, just a very unbalanced target."},{"metadata":{},"cell_type":"markdown","source":"# Finally let's apply the conditional frequency approach to all the targets"},{"metadata":{},"cell_type":"markdown","source":"### This is the alternative to the correlation matrix promised at the beginning:"},{"metadata":{"trusted":true},"cell_type":"code","source":"# use correlation matrix as container for the frequencies\ncond_ALL = df_train[targets].corr()\n\ntargets_TEMP = targets\nnn = len(targets)\n\n# calc frequency of x given y for all pairs\nfor i in range(1,nn+1):\n    for j in range(1,nn+1):\n       if (i!=j):\n        f1 = targets_TEMP[i-1]\n        f2 = targets_TEMP[j-1]\n        ctab = pd.crosstab(df_train[f1], df_train[f2])\n        n_1 = df_train[f1].sum()\n        n_both = ctab.iloc[1,1] \n        perc_2_given_1 = n_both / n_1\n        # print('Percentage ',f2,' given ',f1,':',np.round(perc_2_given_1,4))\n        cond_ALL.loc[f1,f2] = perc_2_given_1\n        \n# plot matrix of conditional frequencies\nfig = plt.figure(figsize = (12,9))\nsns.heatmap(cond_ALL, annot=True, cmap='RdYlGn')\nplt.title('Conditional Frequencies - All Targets')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Example:\n* Conditional frequency for CVC-Normal given ETT-Normal is 0.73.\n* Conditional frequency for ETT-Normal given CVC-Normal is 0.25. \n\nLet's check that:"},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.crosstab(df_train['ETT - Normal'], df_train['CVC - Normal'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"freq_check_1 = 5302 / (5302+1938)\nprint(freq_check_1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"freq_check_2 = 5302 / (5302+16022)\nprint(freq_check_2)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Different visualization (R corrplot style):"},{"metadata":{"trusted":true},"cell_type":"code","source":"# \"flatten\" matrix to data frame\ncond_ALL_df = cond_ALL.stack().reset_index(name='cond_freq')\n# remove the trivial 1's to get a nicer plot\ncond_ALL_df = cond_ALL_df[cond_ALL_df.cond_freq < 1]\n# show structure\ncond_ALL_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"The following code for plotting is based on this kernel: https://www.kaggle.com/drazen/heatmap-with-sized-markers.\nMany thanks to the author!"},{"metadata":{"trusted":true},"cell_type":"code","source":"# plot matrix in \"corrplot\"-style\n\ncolor_min, color_max = [0, 1] # range of values\nn_colors = 256\npalette = sns.mpl_palette('seismic', n_colors)\n\nsize_scale = 1000\n\n# translate values into color of palette\ndef value_to_color(val):\n    val_position = float((val - color_min)) / (color_max - color_min)\n    ind = int(val_position * (n_colors - 1))\n    return palette[ind]\n\nfig, ax = plt.subplots(figsize=(7,7))\n\nx = cond_ALL_df.level_1 # matrix columns\ny = cond_ALL_df.level_0 # matrix rows\nsize = cond_ALL_df.cond_freq\ncolor = cond_ALL_df.cond_freq\n\n# define mapping between labels and coordinates\nx_labels = y.unique() # intentionally using y here, we want the same (original) order on both axes!\ny_labels = y.unique()\n# reverse y_labels to get diagonal in NW to SE direction\ny_labels = y_labels[::-1]\nx_to_num = {p[1]:p[0] for p in enumerate(x_labels)} \ny_to_num = {p[1]:p[0] for p in enumerate(y_labels)} \n\n# finally the actual plotting\nax.scatter(\n    x=x.map(x_to_num),\n    y=y.map(y_to_num),\n    s=size * size_scale,\n    c=color.apply(value_to_color),\n    marker='o' # use circles as markers\n)\n\n# set labels, title, etc.\nax.set_xticks([x_to_num[v] for v in x_labels])\nax.set_xticklabels(x_labels, rotation=90)\nax.set_yticks([y_to_num[v] for v in y_labels])\nax.set_yticklabels(y_labels)\nax.set(xlabel=\"\", ylabel=\"\", aspect='equal')\nplt.title('Conditional Frequencies - All Targets')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# show top 10 \"dependencies\"\ncond_ALL_df.sort_values('cond_freq', ascending=False)[0:10]","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}