{"cells":[{"metadata":{"_cell_guid":"6af5adcf-3309-464a-b894-e166cd99d26e","_uuid":"fdae6d0c7e24bca6d028a1dbc7604f50830f1e3d"},"cell_type":"markdown","source":"### This Kerenl Found Sensitive User & Group\n[Section] \n\n1. Stacked Bar Group by Columns\n - With columns values satisfied the good quality, create stacked bar\n - Test Independence Between extracted values\n2. Device \n - Who is the early adapter? \n3. App\n - What App motivate people to do Sth?\n4. Channel\n - Where people gathered to make a Good Connection with Producer?"},{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport gc\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\npd.options.mode.chained_assignment = None\n\ndf_train = pd.read_csv('../input/train.csv', nrows=1000000)\nprint('The coulmn List : {}'.format(df_train.columns.tolist()))\n\ncolumn = df_train.columns.tolist()[1:-3]\ny = 'is_attributed'\n#tmp_group = [train_1.groupby([col,y]).size() for col in column]","execution_count":1,"outputs":[]},{"metadata":{"_cell_guid":"7d26b198-66b9-4d73-b47e-c4abce1738c0","_uuid":"b31c31b0c302c86e92263b3ef101323454437943"},"cell_type":"markdown","source":"`### 1. Count Active Bar By Group column"},{"metadata":{"_cell_guid":"261f3c08-b90a-4bc5-b0c2-79bbb2b0d303","_kg_hide-input":true,"_uuid":"456cf68eb0b5da4dda95c6b58d605bcdcd23706d","scrolled":true,"trusted":true},"cell_type":"code","source":"f, ax = plt.subplots(1,4, figsize = (12,4))\nlist_major_group = []\nfor i,col in enumerate(column):\n    tmp = df_train.groupby([col,y]).size().reset_index()\n    tmp.rename(columns = {0:'cnt'}, inplace = True)\n    tmp = tmp.pivot(index = col, columns = 'is_attributed', values = 'cnt')\n    tmp_index = np.full(tmp.shape[0], True, dtype = bool)\n    for num in [0,1]:\n        tmp_index &= ~tmp[num].isnull()\n    tmp = tmp.loc[tmp_index,:]\n    tmp['sum'] = tmp.sum(axis = 1)\n    tmp['ratio_1'] = tmp[1] / tmp['sum']\n    tmp = tmp.loc[(tmp['ratio_1'] > 0.1) & (tmp['sum'] > 50),:]\n    tmp = tmp.sort_values('ratio_1', ascending = False).head(20)\n    tmp.sort_values('sum', inplace = True)\n    list_major_group.append(tmp.index)\n    tmp[[0,1]].plot.barh(stacked = True, ax = ax[i], legend = False)\n    title = str(col)\n    ax[i].set_title(title)\n    ax[i].set_ylabel('')\nax[0].set_ylabel('Column Value')\nax[i].legend()\nplt.subplots_adjust(wspace = 0.2, top = 0.85)\nplt.suptitle('Stacked Bar is_attribued Groupby Column', size = 14)\nplt.show()","execution_count":2,"outputs":[]},{"metadata":{"_cell_guid":"39ce44a0-03c8-45c4-ae37-acc3fef5f11d","_uuid":"245b872bf9b560dcd7f9ffedf48c037fd40fb4c8"},"cell_type":"markdown","source":"- Specially App 35 and os 24, channel 213 and 274 looked at famous as a marketing channel\n- The entire ip couldn't achieve the upper condition. Maybe many people joined one ip so that the total ratio goes almost 0.\n- Who are owner of the device 0 are sensitive to the marketing service.   \n(Which Columns satisfied the condition : (is_attribued_1) / (is_attribued_total) > 0.1 & the number of Appearances of Col > 50) "},{"metadata":{"_cell_guid":"53c0ac78-172f-4993-94a4-998da1ba5f84","_kg_hide-input":true,"_uuid":"4c45ce625ab7f337273f4a976a7b06ea04f5effd","trusted":true},"cell_type":"code","source":"from itertools import combinations\nfor grp1, grp2 in combinations(list_major_group,2):\n    grp1_name, grp2_name = grp1.name, grp2.name\n    grp1_val, grp2_val = grp1.values, grp2.values\n    grp1_tf = df_train[grp1_name].isin(grp1_val)\n    grp2_tf = df_train[grp2_name].isin(grp2_val)\n    grpU_tf = grp1_tf & grp2_tf\n    print('{0} : {1} ({2:0.1f})'.format(grp1_name, grp2_name, (grp1_tf & grp2_tf).sum() / (grp1_tf | grp2_tf).sum()))","execution_count":3,"outputs":[]},{"metadata":{"_cell_guid":"0ed8c072-b940-4dea-8250-4035dd7969c7","_uuid":"7c766eddce662ced063b41bac73efe66bd543079"},"cell_type":"markdown","source":"### Relationship between Extracted Values\n- device-os relationship is strong linear but the other seemed to be independent.\n- As the result of non-relationship, weighted to the high probability \"value\", what we mentioned above(ex: App 35 and os 24, channel 213 and 274 )\n\n----\n\n### 2.  11  <= Device , App, Channel <= 50"},{"metadata":{"_cell_guid":"b24d6f9f-f2d0-4d6e-9db8-c450918339e3","_kg_hide-input":true,"_uuid":"587cfc1c71bde7e30c75e9189b7a628695c9333d","trusted":true},"cell_type":"code","source":"kernel = df_train[['device', 'is_attributed']].copy()\nkernel = kernel.groupby(['device', 'is_attributed']).size()\nkernel = kernel.reset_index()\nkernel.rename(columns = {0:'cnt'}, inplace = True)\nkernel = kernel.pivot(index = 'device', columns = 'is_attributed', values = 'cnt')\nkernel.fillna(0, inplace = True)\nkernel['sum'] = kernel.sum(axis=1)\nkernel['ratio'] = kernel[1].divide(kernel['sum']).round(2)\n\nplt.figure(figsize = (12,12))\nax = plt.subplot2grid((3, 4), (0, 0))\nheight_ratio = [(kernel['ratio']==0).sum(), (kernel['ratio']!=0).sum()]\nax.bar(x = [0, 1], height = height_ratio, color = ['green', 'red'])\nax.set_xticks([0, 1])\n\nfor i, height in enumerate(height_ratio):\n    ax.text(i, height+2, height, ha = 'center', color = 'grey')\nax.set_ylim([0,max(height_ratio)+10])\nax.set_xlabel('Is_attribued')\nax.set_title('The # Device')\n\ntmp_index = kernel.index[(kernel['ratio'] != 0) & (10 < kernel['sum']) & (kernel['sum'] < 50)]\nkernel = kernel.loc[tmp_index, ['sum', 'ratio']]\nkernel = kernel.reset_index()\nkernel.rename(columns = {'index': 'device'})\nkernel.sort_values('ratio', ascending = False, inplace = True)\nkernel.reset_index(drop = True, inplace = True)\nkernel.reset_index(inplace = True)\n\nax = plt.subplot2grid((3, 4), (0, 1), colspan = 3)\nsns.barplot(x=kernel.index, y=\"ratio\", data=kernel, palette = sns.color_palette(\"Blues_d\", kernel.shape[0]), ax = ax)\n#ax.barh(effect_channel.index, effect_channel.ratio,align = 'center', color = 'green')\nax.set_xticks(kernel.index)\nax.set_xticklabels(kernel.device)\n#ax.invert_yaxis()  # labels read top-to-bottom\n#ax.set_xlim([0,0.5])\nax.set_ylabel('Performance')\nax.set_xlabel('Device')\nax.set_title('Devicel Performance')\nax.set_ylim([0,0.5])\nfor i, text in enumerate(kernel.ratio):\n    if text < 0.1: break\n    ax.text(i, text+0.02, text,ha = \"center\", color = 'grey', fontsize = 8)\nax.hlines(0.1, 0, kernel.shape[0], color = 'red')\nax.set_title('Device Performance')\nax.set_ylabel('')\neffect_device = kernel['device'].loc[kernel.ratio >= 0.1].tolist()\n\n\nkernel = df_train[['app', 'is_attributed']].copy()\nkernel = kernel.groupby(['app', 'is_attributed']).size()\nkernel = kernel.reset_index()\nkernel.rename(columns = {0:'cnt'}, inplace = True)\nkernel = kernel.pivot(index = 'app', columns = 'is_attributed', values = 'cnt')\nkernel.fillna(0, inplace = True)\nkernel['sum'] = kernel.sum(axis=1)\nkernel['ratio'] = kernel[1].divide(kernel['sum']).round(2)\n\n\nax = plt.subplot2grid((3, 4), (1, 0))\nheight_ratio = [(kernel['ratio']==0).sum(), (kernel['ratio']!=0).sum()]\nax.bar(x = [0, 1], height = height_ratio, color = ['green', 'red'])\nax.set_xticks([0, 1])\n\nfor i, height in enumerate(height_ratio):\n    ax.text(i, height+2, height, ha = 'center', color = 'grey')\nax.set_ylim([0,max(height_ratio)+10])\nax.set_xlabel('Is_attribued')\nax.set_title('The # App')\n\ntmp_index = kernel.index[(kernel['ratio'] != 0) & (10 < kernel['sum']) & (kernel['sum'] <= 50)]\nkernel = kernel.loc[tmp_index, ['sum', 'ratio']]\nkernel = kernel.reset_index()\nkernel.rename(columns = {'index': 'app'})\nkernel.sort_values('ratio', ascending = False, inplace = True)\nkernel.reset_index(drop = True, inplace = True)\nkernel.reset_index(inplace = True)\n\n\nax = plt.subplot2grid((3, 4), (1, 1), colspan = 3)\nsns.barplot(x=kernel.index, y=\"ratio\", data=kernel, palette = sns.color_palette(\"Blues_d\", kernel.shape[0]), ax = ax)\n#ax.barh(effect_channel.index, effect_channel.ratio,align = 'center', color = 'green')\nax.set_xticks(kernel.index)\nax.set_xticklabels(kernel.app)\n#ax.invert_yaxis()  # labels read top-to-bottom\n#ax.set_xlim([0,0.5])\nax.set_ylabel('Performance')\nax.set_xlabel('App')\nax.set_title('App Performance')\nax.set_ylim([0,0.8])\nfor i, text in enumerate(kernel.ratio):\n    if text < 0.1: break\n    ax.text(i, text+0.02, text,ha = \"center\", color = 'grey', fontsize = 8)\nax.hlines(0.1, 0, kernel.shape[0], color = 'red')\nax.set_ylabel('')\neffect_app = kernel['app'].loc[kernel.ratio >= 0.1].tolist()\n\nkernel = df_train[['channel', 'is_attributed']].copy()\nkernel = kernel.groupby(['channel', 'is_attributed']).size()\nkernel = kernel.reset_index()\nkernel.rename(columns = {0:'cnt'}, inplace = True)\nkernel = kernel.pivot(index = 'channel', columns = 'is_attributed', values = 'cnt')\nkernel.fillna(0, inplace = True)\nkernel['sum'] = kernel.sum(axis=1)\nkernel['ratio'] = kernel[1].divide(kernel['sum']).round(2)\n\n\nax = plt.subplot2grid((3, 4), (2, 0))\nheight_ratio = [(kernel['ratio']==0).sum(), (kernel['ratio']!=0).sum()]\nax.bar(x = [0, 1], height = height_ratio, color = ['green', 'red'])\nax.set_xticks([0, 1])\n\nfor i, height in enumerate(height_ratio):\n    ax.text(i, height+2, height, ha = 'center', color = 'grey')\nax.set_ylim([0,max(height_ratio)+10])\nax.set_title('The # Effectitve channel')\n\neffect_channel = kernel.index[kernel['ratio'] != 0]\neffect_channel = kernel.loc[kernel.index.isin(effect_channel),:]\neffect_channel.sort_values('ratio', ascending = False, inplace = True)\neffect_channel = effect_channel.reset_index()\neffect_channel.rename(columns = {'index':'channel'}, inplace = True)\nax = plt.subplot2grid((3, 4), (2, 1), colspan = 3)\nsns.barplot(x=effect_channel.index, y=\"ratio\", data=effect_channel, label=\"Total\", palette = sns.color_palette(\"Blues_d\", effect_channel.shape[0]), ax = ax)\n#ax.barh(effect_channel.index, effect_channel.ratio,align = 'center', color = 'green')\nax.set_xticks(effect_channel.index)\nax.set_xticklabels(effect_channel.channel)\n#ax.invert_yaxis()  # labels read top-to-bottom\n#ax.set_xlim([0,0.5])\nax.set_ylabel('Performance')\nax.set_xlabel('Channel')\nax.set_title('Channel Performance')\nax.set_ylim([0,1])\nfor i, text in enumerate(effect_channel.ratio):\n    if text < 0.1: break\n    ax.text(i, text+0.02, text,ha = \"center\", color = 'grey', fontsize = 8)\nax.hlines(0.1, 0, effect_channel.shape[0], color = 'red')\nax.set_ylabel('')\nplt.subplots_adjust(wspace = 0.4, hspace = 0.5, top = 0.88)\nplt.suptitle('Performance Issue', size = 14)\nplt.show()\n\nprint('Effect Device: ', effect_device)\nprint('Effect App: ', effect_app)\nprint('Effect Channel: ', effect_channel['channel'].loc[effect_channel.ratio >= 0.1].tolist())","execution_count":4,"outputs":[]},{"metadata":{"_cell_guid":"d3ac0958-df80-4890-94c2-a86b5bae312d","_uuid":"378205476a5b873ecce5183ec0b1dfc420ed3fa3"},"cell_type":"markdown","source":"- 106 Devices, 66 Apps, 26 Channels used to connect customer and producer\n- 14 Devices, 16 Apps, 12 Channels only keep their good qualtiy regard of connection between stakeholder.\n- Device can answer who is an early adapter. Imagaine their behavior. Who bought the new gadget do sth more than the general person to act on the internet.\n- App & Cahnnel are a useful window to communicate with customers. However I doubt of that some of them are made or artifical by their producer.\n\n"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}