{"cells":[{"metadata":{"_uuid":"43aa3464e68e3b1a5779edad1938ba60ac27501b","_cell_guid":"3a7adad8-9214-46e9-9da3-f07cdaff9a07"},"cell_type":"markdown","source":"There are already many versions of EDA kernels done by many Kagglers for Avito competition. This is my first attempt at EDA kernel in Kaggle. I have tried not to repeat few basics which are present in other Kernel.\nI have used Bokeh for the Visualization in Python. I am not an expert in Bokeh, i have started learning it for this kernel. I will continue to refine this kernel with more features and explorations. Please feel free to point out any corrections or improvements."},{"metadata":{"_uuid":"45ec95522b2da74923789da360ede24ebaa93446"},"cell_type":"markdown","source":"# <a id='lib'>1. Load libraries</a>"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_kg_hide-input":true,"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":false,"trusted":true,"collapsed":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom bokeh.models import LinearAxis, Range1d\nfrom bokeh.transform import dodge\nfrom bokeh.core.properties import value\n\nfrom bokeh.plotting import figure\nfrom bokeh.models import ColumnDataSource, HoverTool\nfrom bokeh.io import output_notebook, show\n\nfrom wordcloud import WordCloud\n\nimport os\n# print(os.listdir(\"../input\"))","execution_count":4,"outputs":[]},{"metadata":{"_uuid":"5017a7bc1913108d2a47dc880623219fe95431e6","_kg_hide-input":true,"_cell_guid":"125c221f-4d73-4a27-92f6-0abbf51eb4f4","collapsed":true,"trusted":true},"cell_type":"code","source":"np.set_printoptions(suppress=True)","execution_count":5,"outputs":[]},{"metadata":{"_uuid":"e96a9053d6ba542e3ca88c298dff5fd6ec15bdc1","_kg_hide-input":true,"_cell_guid":"291e670b-9280-48f8-8520-78deeb385ed2","trusted":true},"cell_type":"code","source":"output_notebook()","execution_count":6,"outputs":[]},{"metadata":{"_uuid":"6eab32c7e3b43448b005599f677533b01eb9c4cb","_cell_guid":"aaf7c00f-cead-4464-b2e2-637c1c5b220c"},"cell_type":"markdown","source":"# <a id='data'>2. Import data</a>"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_kg_hide-input":true,"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv', parse_dates = ['activation_date'])\ntest = pd.read_csv('../input/test.csv', parse_dates = ['activation_date'])\nperiods_train = pd.read_csv('../input/periods_train.csv', parse_dates = ['activation_date', 'date_from', 'date_to'])\n# train_active = pd.read_csv('../input/train_active.csv')","execution_count":7,"outputs":[]},{"metadata":{"_uuid":"32b4a4a55c99a6c4a0f8f71f90a0a7837a84120e","_kg_hide-input":true,"_cell_guid":"207197ef-4705-4f94-940b-299610c8346f","trusted":true},"cell_type":"code","source":"print('Number of Observations in train is {0} and number of columns is {1}'.format(train.shape[0], train.shape[1]))\nprint('Number of Observations in periods_train is {0} and number of columns is {1}'.format(periods_train.shape[0], periods_train.shape[1]))","execution_count":8,"outputs":[]},{"metadata":{"_uuid":"6e8b4aef7853aa42f78ee30035d10a00f353ed7d"},"cell_type":"markdown","source":"# <a id='overview'>3. Data Overview</a>"},{"metadata":{"_uuid":"a891da04000b801249871b34e3ecd776a14f97a7","_kg_hide-input":true,"_cell_guid":"fa2db31a-7c20-4902-bb92-e215a7957aaf","trusted":true},"cell_type":"code","source":"print('A sample of train data')\ntrain.head()","execution_count":9,"outputs":[]},{"metadata":{"_uuid":"2c308ea7aeb33e7ceb551c5646ab762a665340be","_kg_hide-input":true,"_cell_guid":"7b3caf07-b8fb-4dab-ac9e-1e32e0ceb206","trusted":true},"cell_type":"code","source":"print('A sample of period_train data')\nperiods_train.head()","execution_count":10,"outputs":[]},{"metadata":{"_uuid":"7000d35a6f279479c970148249c95e92cd62a028","_kg_hide-input":true,"_cell_guid":"9422f196-7f70-4307-8c97-6e45fe9f6c62","collapsed":true,"trusted":true},"cell_type":"code","source":"\"\"\" \nFunction to highlight rows based on data type\n\"\"\"\ndef dtype_highlight(x):\n    if x['type'] == 'object':\n        color = '#2b83ba'\n    elif (x['type'] == 'int64') | (x['type'] == 'int32'):\n        color = '#abdda4'\n    elif (x['type'] == 'float64') | (x['type'] == 'float32'):\n        color = '#ffffbf'\n    elif x['type'] == 'datetime64[ns]':\n        color = '#fdae61'\n    else:\n        color = ''\n    return ['background-color : {}'.format(color) for val in x]\n\ntrain_dtypes = pd.DataFrame(train.dtypes.reset_index())\ntrain_dtypes.columns = ['column', 'type']\nperiods_train_dtypes = pd.DataFrame(periods_train.dtypes.reset_index())\nperiods_train_dtypes.columns = ['column', 'type']\n\ntrain_dtypes.style.apply(dtype_highlight, axis = 1)","execution_count":11,"outputs":[]},{"metadata":{"_uuid":"7ac7049ea9ec79bb8e36cf938c7922eb7e8dc245","_cell_guid":"fe630868-bdcf-425b-bc56-f3bd20d17fd5"},"cell_type":"markdown","source":"Take Aways:\n1. More Categorical independent features than numeric features"},{"metadata":{"_uuid":"423b3cfa3cf067c4e0de69ae59d9419574afc94c","_kg_hide-input":true,"_cell_guid":"3449adc8-c2e6-473c-ba41-29d2e35d4b96","trusted":true},"cell_type":"code","source":"periods_train_dtypes.style.apply(dtype_highlight, axis = 1)","execution_count":14,"outputs":[]},{"metadata":{"_uuid":"6a81a20cf9f9388c9e58021fec0773844117d603","_cell_guid":"d9c1b9ba-e496-4e5e-99bd-ea804cd8a9c3"},"cell_type":"markdown","source":"Since Image_top_1 is a classification code for image, i am considering it as descrete values"},{"metadata":{"_uuid":"76fc1c5aae117511483b5acf4a69b717ff75add7","_kg_hide-input":true,"_cell_guid":"d3c9ef33-3fc8-4f70-b40d-0cf79cb87d72","collapsed":true,"trusted":true},"cell_type":"code","source":"train.image_top_1 = train.image_top_1.astype(object)","execution_count":15,"outputs":[]},{"metadata":{"_uuid":"5ff546f459885f5eb2b9bcfbf701760163da2cf5"},"cell_type":"markdown","source":"# <a id='summary'>4. Basic Summary</a>"},{"metadata":{"_uuid":"578b2ec17501558d39cb0ca341b7eed75787f70b"},"cell_type":"markdown","source":"### <a id='non_num'>4.1 Non-Numeric columns summary</a>"},{"metadata":{"_uuid":"16d91fb26d9e3e63eabdce38b3e73af8c4ada19c","_cell_guid":"56e16f60-4be3-4acd-b247-1266425df672"},"cell_type":"markdown","source":"Let's get basic summary of categorical columns in train. For simplicity, i have ordered the result based on count of unique values in each column."},{"metadata":{"_uuid":"3541e7854b737f8c7a6b99402b681542cca9ca3c","_kg_hide-input":true,"_cell_guid":"bdced94f-4f48-4bc8-9476-c3cbe49f520d","collapsed":true,"trusted":false},"cell_type":"code","source":"desc = train.describe(include=['O']).sort_values('unique', axis = 1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7f648d4f4f8e70277c7896f3094364f77b9ec07f","_kg_hide-input":true,"_cell_guid":"2d52b2ed-f457-4b7c-bde3-73446f7c2b28","collapsed":true,"trusted":false},"cell_type":"code","source":"def highlight_row(x):\n    if x.name == 'unique':\n        color = 'lightblue'\n    else:\n        color = ''\n    return ['background-color: {}'.format(color) for val in x]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"04f941d77903e221d3126fe3a09a613b90539d09","_kg_hide-input":true,"_cell_guid":"436eba27-ee4f-403b-a1c4-f079f008244e","trusted":false,"collapsed":true},"cell_type":"code","source":"desc.style.apply(highlight_row, axis =1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d905a80f978bf428b36861b78f759630bc310497","_cell_guid":"e606a0d1-ec3a-499e-8cde-5221e2e87569"},"cell_type":"markdown","source":"Take Aways:\n    1. Only 2 columns with less than 10 distinct values and rest have 28 to 1733 distinct values - So target encoding with smoothing might help as mentioned in this kernel from Porto Seguro competition- https://www.kaggle.com/aharless/xgboost-cv-lb-284\n    2. Significant missing values in param columns and not all items have description and image"},{"metadata":{"_uuid":"347706bf3c11bd6e4c501b64b87eb6751ca079d3","_kg_hide-input":true,"_cell_guid":"bc4eb4ea-8b87-43c2-849c-700bacadfb39","collapsed":true,"trusted":false},"cell_type":"code","source":"# Since image column is a ID code for image, let's remove it for time being\ntrain.drop('image', axis = 1, inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c97c696d3f245d9482d3938c4eb3644290f7f178"},"cell_type":"markdown","source":"### <a id='num'>4.2 Numeric columns summary</a>"},{"metadata":{"_uuid":"1b91445f21e551e316a81a279145764dd67f7a82","_kg_hide-input":true,"_cell_guid":"28aa3c74-3830-4d37-88b3-99795da078e9","collapsed":true,"trusted":false},"cell_type":"code","source":"def color_zero_red(val):\n    \"\"\"\n    Takes a scalar and returns a string with\n    the css property `'background-color: red'` for negative\n    strings, black otherwise.\n    \"\"\"\n    color = 'red' if val == 0 else ' '\n    return 'background-color: %s' % color","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d87ebedbcedd4bf7c0bd89819295be339d8ee282","_kg_hide-input":true,"_cell_guid":"04c8ddec-9af8-4124-a041-2cc19bb79871","trusted":false,"collapsed":true},"cell_type":"code","source":"# Summary of numeric columns\ntrain.describe().style.applymap(color_zero_red)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"800f146367526407e78f47b21dd9dc01cb7bf38c","_cell_guid":"e328cb2b-3f7b-49d3-9487-86056759f3a2"},"cell_type":"markdown","source":"Missing Values in each column"},{"metadata":{"_uuid":"90451a450b627c9c5bf42f178d80e27be326d543","_kg_hide-input":true,"_cell_guid":"1bd01fc0-0ce5-40a8-bfe2-3b6d965ada1d","trusted":false,"collapsed":true},"cell_type":"code","source":"# Missing values in each column\ntrain.isnull().sum().sort_values(ascending =False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9553e5a74acc13de105bf7c97d385acb5ffef46d","_cell_guid":"8913701c-09ba-4a18-bc5e-eb474ab1b876"},"cell_type":"markdown","source":"# <a id='target'>5. Target column distribution</a>"},{"metadata":{"_uuid":"ce1bdbbbcbcd6209a7f59d6bd9e8e4e0632ed0b7","_kg_hide-input":true,"_cell_guid":"6b798043-8484-4db0-bee3-424b1af94e99","trusted":false,"collapsed":true},"cell_type":"code","source":"train.deal_probability.describe()\n(train.deal_probability ==0).sum()/train.shape[0]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fc8bf9c763fa767a52f02472b33c986d219a5335","_cell_guid":"e470256e-e63d-4d1c-9627-5879ce15ccc3"},"cell_type":"markdown","source":"Exactly 64.8% of the values in deal_probability are zero. Since these zero values skews a lot, let's look at the distribution after removing 0"},{"metadata":{"_uuid":"d72797c27bc9185e029d4f80877887e4121ad40e","_kg_hide-input":true,"_cell_guid":"d18648ee-f8ad-4fbc-a85b-1d6e4cd7f2f5","collapsed":true,"trusted":false},"cell_type":"code","source":"non_zero_probability = train.deal_probability[train.deal_probability !=0]\n\nhist, edges = np.histogram(non_zero_probability, \n                               bins = 50, \n                               range = [0, 1])\n\nhist_edges = pd.DataFrame({'#items': hist, \n                       'left': edges[:-1], \n                       'right': edges[1:]})\nhist_edges['cumulative_items'] = hist_edges['#items'].cumsum()\nhist_edges['p_interval'] = ['%.2f to %.2f' % (left, right) for left, right in zip(hist_edges['left'], hist_edges['right'])]\n\nsrc = ColumnDataSource(hist_edges)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"57d3390544fbb05d13440a153ef0bcfd1e4f29bd","_kg_hide-input":true,"_cell_guid":"b5554129-a447-4bf7-9df3-824a98e0d7f0","trusted":false,"collapsed":true},"cell_type":"code","source":"hover1 = HoverTool(tooltips=[\n    (\"probability interval\", \"@p_interval\"),\n    (\"#Items\", \"$y\")\n])\n\np1 = figure(title=\"deal_probability histogram\",  y_axis_label='No.of.Items', x_axis_label='probability', tools = [hover1], background_fill_color=\"#E8DDCB\")\np1.title.align = 'center'\np1.left[0].formatter.use_scientific = False\np1.below[0].formatter.use_scientific = False\np1.quad(top='#items', bottom=0, left='left', right='right',\n        fill_color=\"#036564\", line_color = \"#2B2626\", source =src)\nshow(p1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1f3e9b8c1008c294f71963d9927c863e5090e704","_kg_hide-input":true,"_cell_guid":"021a81f9-8d25-4d0f-b0a2-486ecb5b5522","trusted":false,"collapsed":true},"cell_type":"code","source":"hover2 = HoverTool(tooltips=[\n    (\"probability <=\", \"@right\"),\n    (\"#Items\", \"$y\")\n])\n\np2 = figure(title=\"deal_probability cumulative\",  y_axis_label='No.of.Items', x_axis_label='probability', tools = [hover2], background_fill_color=\"#E8DDCB\")\np2.title.align = 'center'\np2.left[0].formatter.use_scientific = False\np2.below[0].formatter.use_scientific = False\np2.quad(top='cumulative_items', bottom=0, left='left', right='right',\n        fill_color=\"#036564\", line_color = \"#2B2626\", source =src)\n# p.line('left', 'cumulative_items', line_color=\"#9E3030\", source = src)\nshow(p2)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b8b80b18593dba1d136cc78975995c78d1d39ec1","_cell_guid":"1f678924-bd4a-4f97-8919-239dc200fc52"},"cell_type":"markdown","source":"Take Aways:\n    1. After removing zero probability, we can clearly see that the distribution is bimodal with values concentrated around 0.15 and 0.80"},{"metadata":{"_uuid":"5769a66997c5a863bbf7806cceb37168e4545cb4"},"cell_type":"markdown","source":"# <a id='user_item'>6. user_id and item_id distribution</a>"},{"metadata":{"_uuid":"6c9041df416efade99caf573a502eeed2821b158","_kg_hide-input":true,"_cell_guid":"7c6f5868-495e-4c3c-97d1-7fb9f0fd526b","trusted":true},"cell_type":"code","source":"user_item = train.groupby('user_id')['item_id'].count().sort_values()\nuser_item_dist = user_item.value_counts().reset_index()\nuser_item_dist.columns = ['No.of Ads', 'No.of Users']\n\nuser_item_dist['pct'] = user_item_dist['No.of Users']/user_item_dist['No.of Users'].sum()","execution_count":3,"outputs":[]},{"metadata":{"_uuid":"36394131a8dd8d50d99dcffe61ceac005581a8d7"},"cell_type":"markdown","source":"Distribution of number of Users for each count of Ad post"},{"metadata":{"_uuid":"4c7a3db610fdf383aa272ebbb765e9608c50a9f6","_kg_hide-input":true,"_cell_guid":"7a861501-faa7-44a1-9fc2-0b5db27cdab6","trusted":false,"collapsed":true},"cell_type":"code","source":"user_item_dist.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"de9097f804e06065fdca974fe6eb972620f8038c","_kg_hide-input":true,"_cell_guid":"03746a83-9c59-468d-99e1-2cdea6182bc5","trusted":false,"collapsed":true},"cell_type":"code","source":"user_item_dist.tail()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c78f9aa7ebaf97e653741074ea02e63187ebc900","_kg_hide-input":true,"_cell_guid":"adcc93d9-f402-419b-83d0-421cc9604c31","trusted":false,"collapsed":true},"cell_type":"code","source":"user_item_dist[user_item_dist['No.of Ads'] == 1].loc[:,'pct']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"94cab5450655fc5369485a4decc3bb4187f8a01f","_cell_guid":"969a498d-bfd6-4b15-b134-e976d046009e"},"cell_type":"markdown","source":"Take Aways:\n1. It can be observed that there is very high skew in the distribution. 68.9 % of the users have posted exactly once. Similarly, a exact 1080 Ads have been posted by exactly 1 User."},{"metadata":{"_uuid":"d7ae9988a3e4104db302faf80b9660d2eec5503e","_kg_hide-input":true,"_cell_guid":"030edd89-28b3-4464-90d1-c9f72f2c0932","trusted":false,"collapsed":true},"cell_type":"code","source":"user_item_dist[user_item_dist['No.of Users'] == 1].loc[:,'No.of Ads'].max()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e0197986535e5de1983ff1430a346f2b216500bc"},"cell_type":"markdown","source":"A maximum 1080 of Ads have been posted by exactly one user"},{"metadata":{"_uuid":"aef229c2eeeaeeb27da4a7e508f63d240c9bd585","_kg_hide-input":true,"_cell_guid":"729ddc81-0535-4e57-ae54-02297e7fa950","trusted":false,"collapsed":true},"cell_type":"code","source":"user_item_dist[user_item_dist['No.of Users'] == 1].loc[:,'No.of Ads'].min()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"af5735acf720839de341857ffba9aba77ac99e8b"},"cell_type":"markdown","source":"A minimum of 101 Ads have been posted by exactly one user"},{"metadata":{"_uuid":"599f2748a5e74dd9996497075999009cf987abb1","_cell_guid":"98b3c5b5-fbda-4678-8f97-1e4b57f786e8"},"cell_type":"markdown","source":"Plot the distribution after removing outlier cases"},{"metadata":{"_uuid":"b4571330dfb6c298d3b900992e7d6e8214f332b4","_kg_hide-input":true,"_cell_guid":"9c5817d0-89d1-4df8-9c0e-1bf9f010e3e8","trusted":false,"collapsed":true},"cell_type":"code","source":"user_item_dist2 = user_item_dist[(user_item_dist['No.of Ads'] > 1) & (user_item_dist['No.of Users'] > 1)]\n\nhover3 = HoverTool(tooltips=[\n    (\"No.of Ads ==\", \"@right\"),\n    (\"Percent of Users\", \"$y\")\n])\n\np3 = figure(title=\"Distribution of No. of Ads posted by users\",  y_axis_label='No.of.Users', x_axis_label='No.of Ads posted', tools = [hover3], background_fill_color=\"#E8DDCB\")\np3.title.align = 'center'\np3.left[0].formatter.use_scientific = False\np3.below[0].formatter.use_scientific = False\np3.quad(top= user_item_dist2['pct'], bottom=0, left= user_item_dist2['No.of Ads'][:-1], right= user_item_dist2['No.of Ads'][1:],\n        fill_color=\"#036564\", line_color = \"#2B2626\")\nshow(p3)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c85b5aebb18c5b25ad9f07f2a625a6cde5ed79a8"},"cell_type":"markdown","source":"# <a id='no.ads_prob'>7. Distribution of deal probablity wrt No.of Ads posted by users</a>"},{"metadata":{"_uuid":"a69b54cb2c271419b340e43a8320ff55441da092","_cell_guid":"ed74dba3-cea7-4015-8795-a079df1c4e1c"},"cell_type":"markdown","source":"Let's add the No.of Ads posted feature as along with the train and check how deal probability varies with it. \n\nNote: While creating No.of Ads features i am not taking activation_date into consideration, i am using the entire data and creating the feature. In a way it indicates if one is a spammer/frequent/occasional user or not"},{"metadata":{"_uuid":"867d73f0d3688861fb7ef851f773694c738ed8ee","_kg_hide-input":true,"_cell_guid":"9d565294-3c7c-4a5c-a4ff-e49893c4ddef","collapsed":true,"trusted":false},"cell_type":"code","source":"user_item = user_item.reset_index()\nuser_item.columns = ['user_id', 'No.of Ads']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"48b352d7ed07a703316854723151dd123f84bd17","_kg_hide-input":true,"_cell_guid":"7b2b7cb4-71a9-4431-a4ed-106be7de90c8","collapsed":true,"trusted":false},"cell_type":"code","source":"train = train.merge(user_item, on = 'user_id', how = 'left')\ntrain['No.of Ads bin'] = pd.cut(train['No.of Ads'], 20, labels = range(20))\n\nAds_bin_prob = train.groupby(['No.of Ads bin'])['deal_probability'].mean().reset_index()\nAds_bin_prob.columns = ['No.of Ads bin', 'avg_deal_probability']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f422de10c6e69f3ef1b0ba057b99903637de31bd","_kg_hide-input":true,"_cell_guid":"0e88afee-7530-40b6-a645-0f8ee280faad","trusted":false,"collapsed":true},"cell_type":"code","source":"hover4 = HoverTool(tooltips=[\n    (\"Ads bin \", \"@right\"),\n    (\"avg_deal_probability\", \"$y\")\n])\n\np4 = figure(title=\"Distribution of No. of Ads posted by users\",  y_axis_label='No.of.Users', x_axis_label='No.of Ads posted', tools = [hover4], background_fill_color=\"#E8DDCB\")\np4.title.align = 'center'\np4.left[0].formatter.use_scientific = False\np4.below[0].formatter.use_scientific = False\np4.quad(top= Ads_bin_prob['avg_deal_probability'], bottom=0, left= Ads_bin_prob['No.of Ads bin'][:-1], right= Ads_bin_prob['No.of Ads bin'][1:],\n        fill_color=\"#036564\", line_color = \"#2B2626\")\nshow(p4)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"32a3655a4de70555d0235021e9b0aa6c94d04aad","_cell_guid":"0dde5930-e663-4f6b-97ea-e717821a95a0"},"cell_type":"markdown","source":"Take Aways:\n    1. User belonging to group which posts less than 10 Ads overall has highest average deal_probability followed by users who posts 70-80 Ads\n    \nIt remains to be seen if this features will come out significant in the model"},{"metadata":{"_uuid":"d144bed8023b87fa698b24a213fa692243603b7e","_kg_hide-input":true,"_cell_guid":"2e873d29-7e9e-42a6-b3ee-a287b8863e6d","trusted":false,"collapsed":true},"cell_type":"code","source":"train.region.nunique()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c2103d0e5e9d7fa8bca5272c09df9bb1d6a95eaf","_cell_guid":"c38c6f65-a2ff-44f1-85db-a60227d834db"},"cell_type":"markdown","source":"# <a id='usr_typ_deal'>8. User Type histogram and deal probability</a>"},{"metadata":{"_uuid":"5da2ab92469bf536fd3e1776437719412a1e8efa","_cell_guid":"6f22d9d6-fea5-418b-a186-5374f43fff2d"},"cell_type":"markdown","source":"Only user_type and parent_category_name had less than 10 distinct values"},{"metadata":{"_uuid":"c4f6f403a8e7df184cf937ec3cdbee9ec7acaf1e","_kg_hide-input":true,"_cell_guid":"6d879fa0-44b3-4526-886f-a11db28e162b","collapsed":true,"trusted":false},"cell_type":"code","source":"f = {'deal_probability':['mean'], 'item_id': ['size']}\nuser_hist_prob = train.groupby('user_type').agg(f).reset_index()\nuser_hist_prob.columns = ['user_type','avg_deal_probability', 'user_type_count']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"628bb178b1e2b54b34e623a88018e93da12421bd","_kg_hide-input":true,"_cell_guid":"047f868e-2e24-4017-a333-ddcbc08efae3","trusted":false,"collapsed":true},"cell_type":"code","source":"user_hist_prob","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"850dcbb9ab7f6a0a26375933a2a3f2c68d2cafef","_kg_hide-input":true,"_cell_guid":"efde556c-72a1-4f95-a46e-c8816e4a6cbf","collapsed":true,"trusted":false},"cell_type":"code","source":"user_type_source = ColumnDataSource(data=user_hist_prob)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"27edf295efaa071012977a74c886638f3b15510b","_kg_hide-input":true,"_cell_guid":"3403456e-de17-4913-b6a6-0dd383caccd9","scrolled":false,"trusted":false,"collapsed":true},"cell_type":"code","source":"p5 = figure(x_range = list(user_hist_prob.user_type), plot_width=800, plot_height=400, title = 'user_type distribution and deal_probability mean')\np5.vbar(x = dodge('user_type', -0.20, range=p5.x_range), top = 'user_type_count', width=.4, color='#f45666', source = user_type_source, legend = value('user_type_count'))\np5.y_range =  Range1d(0, user_hist_prob.user_type_count.max())\np5.extra_y_ranges = {\"avg_deal_probability\": Range1d(start=0, end=1)}\np5.xaxis.axis_label = 'user_type'\np5.yaxis.axis_label = 'user_type_count'\np5.add_layout(LinearAxis(y_range_name=\"avg_deal_probability\", axis_label= 'avg deal_probability'), 'right')\np5.vbar(x = dodge('user_type', 0.20, range=p5.x_range), top = 'avg_deal_probability', y_range_name='avg_deal_probability', width = 0.4, color='lightblue', source = user_type_source, legend = value('avg_deal_probability'))\np5.legend.location = \"top_left\"\np5.legend.orientation = \"horizontal\"\nshow(p5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e665674529c6cce17fba832c5a9a1f04aa8b5eca","_cell_guid":"2655e5ad-371d-4663-bd01-6aa67557b527"},"cell_type":"markdown","source":"# <a id='par_cat_deal'>9. parent_category_name histogram and deal probability</a>"},{"metadata":{"_uuid":"ef248ac9b8bd751cdc66ee9720faae84cae67b5f","_kg_hide-input":true,"_cell_guid":"31eed90e-b957-4666-a08d-d46513674b8c","collapsed":true,"trusted":false},"cell_type":"code","source":"f = {'deal_probability':['mean'], 'item_id': ['size']}\nparent_cat_hist_prob = train.groupby('parent_category_name').agg(f).reset_index()\nparent_cat_hist_prob.columns = ['parent_category_name','avg_deal_probability', 'parent_category_count']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3845fc417d9a7a37819e7ec219519e00b9dfe617","_kg_hide-input":true,"_cell_guid":"84e1b035-8a96-4788-8feb-c3e3e1a50070","trusted":false,"collapsed":true},"cell_type":"code","source":"parent_cat_hist_prob","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"751ed74c170632e0e3fb0cdf21a6390e98df38bc","_kg_hide-input":true,"_cell_guid":"fbcf94bb-0273-4c26-8e4b-0450938008ee","collapsed":true,"trusted":false},"cell_type":"code","source":"parent_cat_source = ColumnDataSource(data=parent_cat_hist_prob)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"35f9bc6af2a3476a025b378cf60e4c99864bc366","_kg_hide-input":true,"_cell_guid":"8380cafe-e166-483b-97df-ccdb78852172","trusted":false,"collapsed":true},"cell_type":"code","source":"p6 = figure(x_range = list(parent_cat_hist_prob.parent_category_name), plot_width=800, plot_height=400, title = 'parent_category distribution and deal_probability mean')\np6.vbar(x = dodge('parent_category_name', -0.16, range=p6.x_range), top = 'parent_category_count', width=.3, color='#f45666', source = parent_cat_source, legend = value('parent_category_count'))\np6.y_range =  Range1d(0, parent_cat_hist_prob.parent_category_count.max())\np6.extra_y_ranges = {\"avg_deal_probability\": Range1d(start=0, end=1)}\np6.xaxis.axis_label = 'parent_category_name'\np6.yaxis.axis_label = 'parent_category_count'\np6.add_layout(LinearAxis(y_range_name=\"avg_deal_probability\", axis_label= 'avg deal_probability'), 'right')\np6.vbar(x = dodge('parent_category_name', 0.16, range=p6.x_range), top = 'avg_deal_probability', y_range_name='avg_deal_probability', width = 0.3, color='lightblue', source = parent_cat_source, legend = value('avg_deal_probability'))\np6.legend.location = \"top_left\"\np6.legend.orientation = \"horizontal\"\nshow(p6)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8affdaf8b34cb5832d8012880764c3908111d300","_cell_guid":"577640bb-5472-4c27-83c5-8f619d8808da","trusted":false,"collapsed":true},"cell_type":"code","source":"train.region.value_counts()","execution_count":null,"outputs":[]}],"metadata":{"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":1}