{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5"},"cell_type":"markdown","source":"# Exploratory data analysis"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","collapsed":true},"cell_type":"markdown","source":"# Retrieving the Data"},{"metadata":{"_cell_guid":"a41f17a5-a591-4e92-ad21-86b14791d2be","_uuid":"e1f6ef0b7c5dac586b8d152af4cee134e02adf7a","trusted":true},"cell_type":"code","source":"import pandas as pd # package for high-performance, easy-to-use data structures and data analysis\nimport numpy as np # fundamental package for scientific computing with Python\nimport matplotlib\nimport matplotlib.pyplot as plt # for plotting\nimport seaborn as sns # for making plots with seaborn (statistic library)\n\nimport plotly\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nfrom plotly.offline import init_notebook_mode, iplot\ninit_notebook_mode(connected=True)\nimport plotly.graph_objs as go\nimport plotly.offline as offline\noffline.init_notebook_mode()\nimport plotly.tools as tls\n\n\n\nfrom io import StringIO\n\n","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"9d98cac7-7203-4d40-b41a-e0e73e7e42a7","_uuid":"9a84d0533f26faebc8dac0882181fc3b30942ecd","trusted":true},"cell_type":"code","source":"print(\"Reading Data......\")\n\n#periods_train = pd.read_csv('E:/PROJET/Avito_Demand_Prediction/input/periods_train.csv', parse_dates=[\"activation_date\", \"date_from\", \"date_to\"])\n#periods_test = pd.read_csv('E:/PROJET/Avito_Demand_Prediction/input/periods_test.csv', parse_dates=[\"activation_date\", \"date_from\", \"date_to\"])\n\ntrain = pd.read_csv('../input/train.csv')\ntest = pd.read_csv('../input/test.csv')\n\nprint(\"Reading Done....\")","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"782fc710-6b98-4292-8d96-c07ff3f3911a","_uuid":"a72abd9c3b36935f64e87a5d73f6c86874142260","trusted":true},"cell_type":"code","source":"print(\"size of train data\", train.shape)\nprint(\"size of test data\", test.shape)\n'''print(\"size of periods_train data\", periods_train.shape)\nprint(\"size of periods_test data\", periods_test.shape)'''","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"8df9b007-2b45-4f92-b1aa-628e85e76511","_uuid":"c3e89b5ed4e3e5bd85389e78349a92ba49e6b5f5"},"cell_type":"markdown","source":"# 3- Glimpse of Data\n## 3.1 Overview of tables\n\n### Train data "},{"metadata":{"_cell_guid":"53f1abb6-91f9-4d2e-9b99-3120587840a3","_uuid":"3d3199da3c53c3df58be7bd1a417994f3b560711","trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"803a43cb-e0ae-435e-81e2-57a27506d050","_uuid":"1cb62e9b732261f0fc67636c94ac58ff29d05cd5"},"cell_type":"markdown","source":"### Test data"},{"metadata":{"_cell_guid":"c79ec27e-b82e-4cc1-8838-8cafe2a29a49","_uuid":"0de5b5df73ed8833c23ed5b8275e55da0d667b40","trusted":true},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"8ccab06d-db90-474b-be8d-9cc57107b8ff","_uuid":"158d99563836c6557189cdd80fc61831d2d5542b"},"cell_type":"markdown","source":"### Periods train data"},{"metadata":{"_cell_guid":"e6afa327-d382-41dd-ab9a-367f466516b5","_uuid":"206d1f14ad12f181a16150d39023c3ca85a238a1","trusted":true},"cell_type":"code","source":"periods_train = pd.read_csv('../input/periods_train.csv', parse_dates=[\"activation_date\", \"date_from\", \"date_to\"])\nperiods_test = pd.read_csv('../input/periods_test.csv', parse_dates=[\"activation_date\", \"date_from\", \"date_to\"])","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"0e272c9f-7493-4c54-9649-25f32290262c","_uuid":"d626c3dcfe02d0107b2cf9bfc398935fad97c83d","slideshow":{"slide_type":"notes"},"trusted":true},"cell_type":"code","source":"periods_train.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"ae1f42d9-e1a1-4b3a-8296-c0a60e8dad24","_uuid":"bee589481945eecf06a9d8c4408fe14f524430f5"},"cell_type":"markdown","source":"## 3.2 Statistical overview of the Data\n### Training Data some little info"},{"metadata":{"_cell_guid":"d43c9f98-10e7-4e1b-9993-db5a52546d02","_uuid":"f440ca92710f022361d840d97b09cdb013d6a781","trusted":true},"cell_type":"code","source":"train.info()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"510a171d-7c0c-4655-830a-a04d6beee2aa","_uuid":"6c2930d71efc82a43a55d2417c17fda3cfc3a225","collapsed":true},"cell_type":"markdown","source":"### Little description of training data for numerical features"},{"metadata":{"_cell_guid":"a8d179af-c6a2-4e8a-899a-236ec14b20d3","_uuid":"d03b351e0e641ddb60fc111f3795830224a6e2f1","trusted":true},"cell_type":"code","source":"train.describe()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"952258c8-1425-47ec-8e1f-47eee4719404","_uuid":"5e132737374cc993eb105123ed51add1132c18a2"},"cell_type":"markdown","source":"### Little description of training data for categorical features"},{"metadata":{"_cell_guid":"a11de6b6-1123-493c-975d-7f998789edcf","_uuid":"a48bc0fb6c86cd61b9513cfa1e8b712a73eecce1","scrolled":false,"trusted":true},"cell_type":"code","source":"train.describe(include=[\"O\"])","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"210fbd40-b802-4ae1-bd23-b1a1d85027ca","_uuid":"24be9f53f7b94eff1914a15a6135a3379a016d93"},"cell_type":"markdown","source":"## 4. Data preparation\n### ** I- Train data **\n### checking missing data in training data"},{"metadata":{"_cell_guid":"d47c34f2-30a3-4287-aa17-b77ba5ad86e9","_uuid":"38b3be1dc58d7e307d17110f56f6918add5d7ef5","scrolled":false,"trusted":true},"cell_type":"code","source":"# checking missing data in train data \n# isnull return TRUE if the value NAN, ' ',  exist in dataset\ntotal = train.isnull().sum().sort_values(ascending = False)\npercent = (train.isnull().sum()*100/train.isnull().count()).sort_values(ascending = False)\nmissing_train_data =pd.concat([total, percent], axis = 1, keys=['total', 'percent'])\nmissing_train_data.head(10)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"a9d8e65f-ebd8-477e-9947-ba97f808f793","_uuid":"8c43c02242cf8381ef6588c3b2725cfeec88ccb2"},"cell_type":"markdown","source":"### checking missing data in periods training data"},{"metadata":{"_cell_guid":"c16aa404-6cf4-4e30-b837-9d46e09baae8","_uuid":"538aff8c71fb3f3e5e6b1855b303c627c3688446","trusted":true},"cell_type":"code","source":"total = periods_train.isnull().sum().sort_values(ascending = False)\npercent = (periods_train.isnull().sum()*100/periods_train.isnull().count()).sort_values(ascending = False)\nmissing_periods_train = pd.concat([total, percent], axis='columns', keys=['total', 'percent'])\nmissing_periods_train\n","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"6bd2fcb3-8512-49e0-a54a-ae3e974349f1","_uuid":"8f209a10fc437e672b6352a1f9aa7981b4b51459"},"cell_type":"markdown","source":"### ** Test data **\n### Checking missing data in test data "},{"metadata":{"_cell_guid":"ae82b45b-5902-466d-b122-a4ce6067daf6","_uuid":"90ef7d74c041344df78fbbd3f2007ab4560dce4f","trusted":true},"cell_type":"code","source":"total = test.isnull().sum().sort_values(ascending=False)\npercent = (test.isnull().sum()/test.isnull().count()*100).sort_values(ascending=False)\nmissing_test = pd.concat([total, percent], axis = 1, keys = ['total', 'percent'])\nmissing_test","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"4141e204-811b-4718-af01-ff9daf8ea7f8","_uuid":"6e13fc57694f3a95aebeb30a7426f5ed0c2b235c","collapsed":true},"cell_type":"markdown","source":"### Checking missing data in periods test data "},{"metadata":{"_cell_guid":"f785d960-0fbf-4a10-8404-0d3a9d78006b","_uuid":"7f846a83729d1535e1c0fef0d64f2fc70d8ed5b4","trusted":true},"cell_type":"code","source":"total = periods_test.isnull().sum().sort_values(ascending=False)\npercent = (periods_test.isnull().sum()/periods_test.isnull().count()*100).sort_values(ascending=False)\nmissing_periods_test = pd.concat([total, percent], axis = 1, keys = ['total', 'percent'])\nmissing_periods_test","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"785c5218-603d-445f-ab4c-7b3d2f3200f3","_uuid":"a1b84ca4fae4cc05d108738769146af7fb6e68d5","collapsed":true},"cell_type":"markdown","source":" ## 5. Data Exploration\n### 5.1 Histogram and distribution of deal probability"},{"metadata":{"_cell_guid":"fbbdf2ea-bd9b-46f2-8c4e-c27fa5b32716","_uuid":"52667800c416b3a228d9d5e3c444e18481029ec6","trusted":true},"cell_type":"code","source":"plt.figure(figsize = (12, 8)) #figsize = (12, 8)\nsns.distplot(train['deal_probability'])\nplt.xlabel('likelihood that an ad sold something', fontsize = 12)\nplt.title(\"Histogram of probability that an ad actually sold something\")\nplt.show()\n\nplt.figure(figsize = (12, 8))\nplt.scatter(range(train.shape[0]), np.sort(train.deal_probability.values))\nplt.xlabel('likelihood that an ad actually sold something', fontsize=12)\nplt.title(\"Distribution of likelihood that an ad actually sold something\")","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"d4a32234-5822-4e31-b208-a40e8996bb71","_uuid":"f20515c51737b41c9ae1b76c7bcc5f4b1d76d478","collapsed":true},"cell_type":"markdown","source":"### 5.2 Histogram and distribution of Ad price"},{"metadata":{"_cell_guid":"70964c5e-c889-4ce4-8aa4-ad7dc3b4d619","_uuid":"56f9284bb14bfa78b7823503316dfee55e6d046a","trusted":true},"cell_type":"code","source":"plt.figure()\nsns.distplot(train['price'].dropna())\nplt.xlabel('Advertisement Price')\nplt.title(\"Histogram of Ad price\")\n\nplt.figure()\nplt.scatter(range(train.shape[0]), np.sort(train.price.values))\nplt.xlabel('Ad price', fontsize=12)\nplt.title(\"Distribution of Ad price\")\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"6bed1c02-f760-4fb0-b026-8326db3d94b2","_uuid":"b5ace871bb7f17940783e1f188cfe40e60f34d20","trusted":true},"cell_type":"code","source":"train['deal_class'] = train['deal_probability'].apply(lambda x:'>= 0.5' if x >= 0.5 else '<0.5')\ntemp = train['deal_class'].value_counts()\nlabels = temp.index\nsizes = (temp/temp.sum()*100)\ntrace = go.Pie(labels = labels, values = sizes, hoverinfo = 'label+percent')\nlayout = go.Layout(title='Distribution of deal class')\nfig = go.Figure(data=[trace], layout=layout)\npy.iplot(fig)\n\ndel train['deal_class']","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"7786ebe0-5a10-45ec-b952-0e748f7baf0d","_uuid":"24f27af00186ecfca533030f1c176e491353ef1e"},"cell_type":"markdown","source":"* ** we notice that 88% of training data have less than 0.5 deal probabilty and 12% having deal probabilty more or equal than 0.5**"},{"metadata":{"_cell_guid":"52f70e1a-0cce-4db2-accf-dcce884f27c4","_uuid":"84a6108bc6713e61687eef5fc558172d04cc4ebd"},"cell_type":"markdown","source":"***to make our data set more comprehonsive we will translate russian region into english \n the function remove_duplicates serve to remove all duplicates rows excisting in a colum and display a list contain without duplication *****"},{"metadata":{"_cell_guid":"4f86f4e1-add1-4523-8a80-c8c5018b6219","_uuid":"3a66b221a286da49e2c4c94f52ad467eda2425cb","trusted":true},"cell_type":"code","source":"'''def remove_duplicates(column):\n    newlist = []\n    for row in column:\n       if row not in newlist:\n           newlist.append(row)\n    return newlist\n\nremove_duplicates(train['region'])'''\n\n# without function remove_duplicates\nnewlist = []\nfor row in train['region']:\n    if row not in newlist:\n        newlist.append(row)\nprint(newlist)\nprint(len(newlist)) #count elements in list ","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"eeb6ded8-6b05-4fb5-bf3a-898dcb3174b2","_uuid":"519710cdb14949a2781f4ad944a39c035c6bec94","trusted":true},"cell_type":"code","source":"from io import StringIO\n\nconversion = StringIO(\"\"\"\nregion,region_english\nСвердловская область, Sverdlovsk oblast\nСамарская область, Samara oblast\nРостовская область, Rostov oblast\nТатарстан, Tatarstan\nВолгоградская область, Volgograd oblast\nНижегородская область, Nizhny Novgorod oblast\nПермский край, Perm Krai\nОренбургская область, Orenburg oblast\nХанты-Мансийский АО, Khanty-Mansi Autonomous Okrug\nТюменская область, Tyumen oblast\nБашкортостан, Bashkortostan\nКраснодарский край, Krasnodar Krai\nНовосибирская область, Novosibirsk oblast\nОмская область, Omsk oblast\nБелгородская область, Belgorod oblast\nЧелябинская область, Chelyabinsk oblast\nВоронежская область, Voronezh oblast\nКемеровская область, Kemerovo oblast\nСаратовская область, Saratov oblast\nВладимирская область, Vladimir oblast\nКалининградская область, Kaliningrad oblast\nКрасноярский край, Krasnoyarsk Krai\nЯрославская область, Yaroslavl oblast\nУдмуртия, Udmurtia\nАлтайский край, Altai Krai\nИркутская область, Irkutsk oblast\nСтавропольский край, Stavropol Krai\nТульская область, Tula oblast\n\"\"\")\n\nconversion = pd.read_csv(conversion)\ntrain = pd.merge(train, conversion, how=\"left\", on=\"region\")\n#del train['region_english_x', 'region_english_y']\n\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"a5d386b9-decc-49e4-aaa5-7bea01c10fb4","_uuid":"ffff112a107c13cb07397b9c22ae2ee75913880c","trusted":true},"cell_type":"code","source":"#columns = ['region_english_x', 'region_english_y']\n#train.drop(columns, inplace=True, axis = 1)\ntrain['region_english'].head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"300f00ff-149d-4047-a80e-5d6078c098e2","_uuid":"0d6a7fa844ce83b5e1efad6c459c95c92e6a3ee8","collapsed":true},"cell_type":"markdown","source":"### 5.3 Distribution of different Ad regions"},{"metadata":{"_cell_guid":"cb4d4998-382c-48cf-8e18-16a939c5940c","_uuid":"03164e099c2c3a6d8ffd5b970a0de5d1ced769c9","trusted":true},"cell_type":"code","source":"temp = train['region_english'].value_counts()\nlabels = temp.index\nsizes = (temp / temp.sum())*100\ntrace = go.Pie(labels=labels, values=sizes, hoverinfo='label+percent')\nlayout = go.Layout(title='Distribution of differnet Ad regions')\ndata = [trace]\nfig = go.Figure(data=data, layout=layout)\npy.iplot(fig)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"39b58a3923bcb5c8f8ac4171dab8aea050513c61"},"cell_type":"markdown","source":"### Distribution of different Ad parent_category_name"},{"metadata":{"trusted":true,"_uuid":"63cbefa9cc8d327d28f31db1f70aeaebd6bd8380"},"cell_type":"code","source":"from io import StringIO\n\nconversion = StringIO(\"\"\"\nparent_category_name,parent_category_name_en\nЛичные вещи,Personal belongings\nДля дома и дачи,For the home and garden\nБытовая электроника,Consumer electronics\nНедвижимость,Real estate\nХобби и отдых,Hobbies & leisure\nТранспорт,Transport\nУслуги,Services\nЖивотные,Animals\nДля бизнеса,For business\n\"\"\")\n\nconversion = pd.read_csv(conversion)\ntrain = pd.merge(train, conversion, how=\"left\", on=\"parent_category_name\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d314344f2ec6c17cab1c7dbda9e2e8b9e4f25415"},"cell_type":"code","source":"temp = train['parent_category_name_en'].value_counts()\nlabels = temp.index\nsizes = (temp / temp.sum())*100\ntrace = go.Pie(labels=labels, values=sizes, hoverinfo='label+percent')\nlayout = go.Layout(title='Distribution of differnet Ad parent_category_name_en')\ndata = [trace]\nfig = go.Figure(data=data, layout=layout)\npy.iplot(fig)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"76f7fe25-cb6c-44f4-aa02-77fbbcf27839","_uuid":"44f8353cc7274fd43f033a0da68b695c972015d5","collapsed":true},"cell_type":"markdown","source":"## TOP FIVE\n### Top 5 Ad titles"},{"metadata":{"_cell_guid":"eb4a1e43-0c6a-4bff-8de7-1a0446c8955d","_uuid":"1bbeb3c34cabb0154eee4d16e23cdde573563a13","trusted":true},"cell_type":"code","source":"temp = train[\"title\"].value_counts().head(20)\nprint(\"Top 5 Ad titles :\\n\", temp.head(5))\nprint(\"Total Ad titles : \",len(train[\"title\"]))\ntrace = go.Bar(\n    x = temp.index,\n    y = temp.values,\n)\ndata = [trace]\nlayout = go.Layout(\n    title = \"Top Ad titles\", xaxis=dict( title='', tickfont=dict( size=14,color='rgb(107, 107, 107)')),\n    yaxis=dict(title='Count of Ad titles', titlefont=dict(size=16, color='rgb(107, 107, 107)'),\n        tickfont=dict(\n            size=14,\n            color='rgb(107, 107, 107)'\n        )))\nfig = go.Figure(data=data, layout=layout)\npy.iplot(fig)\n","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"bf223d0e-9b2a-4d7a-bfc1-eb6c1bab14de","_uuid":"6fe8c1fdd254fde6f600633d27bc7e667c0f80cf"},"cell_type":"markdown","source":"**Top 5 Ad titles are :**\n1. Платье(Dress)\n1. Туфли (Shoes)\n1. Куртка(Jacket)\n1. Пальто (Coat)\n1. Джинсы(Jeans)"},{"metadata":{"_cell_guid":"a08b73a1-b800-458c-89ef-450a7c24b911","_uuid":"5becca3fbeb6eae794d8ed494cb2ca07186423ff"},"cell_type":"markdown","source":"### Top 5 Ad city"},{"metadata":{"_cell_guid":"eaabd967-3cec-47ca-954c-d66525618310","_uuid":"1efe0a5d7bbafb267a27f9805d77b50a6004174d","trusted":true},"cell_type":"code","source":"temp = train[\"city\"].value_counts().head(20)\nprint('Top 5 Ad cities :\\n', temp.head(5))\nprint(\"Total Ad cities : \",len(train[\"title\"]))\ntrace = go.Bar(\n    x = temp.index,\n    y = temp.values,\n)\ndata = [trace]\nlayout = go.Layout(\n    title = \"Top Ad city\",\n    xaxis=dict( title='', tickfont=dict( size=14, color='rgb(107, 107, 107)')\n    ),\n    yaxis=dict( title='Count of Ad cities', titlefont=dict( size=16, color='rgb(107, 107, 107)'),\n        tickfont=dict(size=14, color='rgb(107, 107, 107)')\n)\n)\nfig = go.Figure(data=data, layout=layout)\npy.iplot(fig)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"434e5ec3-3eb2-4a10-992e-029fbc5a7827","_uuid":"bd53723a2a816b3cc053c0be7d476d1b4ddd6ce5"},"cell_type":"markdown","source":"**Top 5 Ad cities :\n1. Краснодар (Krasnodar)\n1. Екатеринбург (Yekaterinburg)\n1. Новосибирск (Novosibirsk)\n1. Ростов-на-Дону (Rostov-on-don)\n1. Нижний Новгород (Nizhny Novgorod)"},{"metadata":{"_cell_guid":"6b32dffe-b16e-43a3-8e51-a5afb2c12bd8","_uuid":"7342d69fd67d3a9484c84f0129338192568e14cc"},"cell_type":"markdown","source":"### Top 5 Ad regions"},{"metadata":{"_cell_guid":"94f99e84-7c1e-4824-ae3f-ada26eec13e1","_uuid":"3b05d3fad2956b81b86ad652eb9a360c12467043","trusted":true},"cell_type":"code","source":"temp = train[\"region_english\"].value_counts().head(20)\nprint('Top 5 Ad regions :\\n',temp.head(5))\nprint(\"Total Ad regions : \",len(train[\"title\"]))\ntrace = go.Bar(\n    x = temp.index,\n    y = temp.values,\n)\ndata = [trace]\nlayout = go.Layout(\n    title = \"Top Ad regions\", xaxis=dict( title='',\n        tickfont=dict( size=14, color='rgb(107, 107, 107)') ),\n    yaxis=dict( title='Count of Ad regions', titlefont=dict(size=16, color='rgb(107, 107, 107)'),\n        tickfont=dict(size=14, color='rgb(107, 107, 107)')\n)\n)\nfig = go.Figure(data=data, layout=layout)\npy.iplot(fig)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"00728238-1caf-4fc9-a427-8ac71153fa0b","_uuid":"c4d0b9d0356d4b488bf57131fa5d6e61e1fb5934"},"cell_type":"markdown","source":"**Top 5 Ad regions :**\n1. Krasnodar Krai\n1. Sverdlovsk oblast\n1. Rostov oblast\n1. Tatarstan\n1. Chelyabinsk oblast"},{"metadata":{"_cell_guid":"34ee0999-f5dc-48a3-89af-562a7e62cf5a","_uuid":"f7851fa3ad359fa9a4f1dc19930e5eda7a635e63"},"cell_type":"markdown","source":"### Top 5  ad category as classified by Avito's ad mode"},{"metadata":{"_cell_guid":"c188d9d9-8972-4881-ac4e-01a319b18123","_uuid":"0e21dcab866d85993e2e60c7b3ef5380d071210a","trusted":true},"cell_type":"code","source":"temp = train[\"category_name\"].value_counts().head(20)\nprint(\"Top 5 Fine grain ad category as classified by Avito's ad mode : \\n\", temp.head(5))\nprint(\"Total ad category as classified by Avito's ad mode : \",len(train[\"title\"]))\ntrace = go.Bar(x = temp.index,y = temp.values,)\ndata = [trace]\nlayout = go.Layout(\n    title = \"Top ad category as classified by Avito's ad mode\",\n    xaxis=dict(\ntitle='ad category as classified by Avitos ad mode',tickfont=dict(size=14,color='rgb(107, 107, 107)')),\n    yaxis=dict(title='Count of ad category',titlefont=dict(size=16,color='rgb(107, 107, 107)'),\n        tickfont=dict(size=14,color='rgb(107, 107, 107)')))\nfig = go.Figure(data=data, layout=layout)\npy.iplot(fig)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"00ac446e-3acd-4123-a16d-e5b9d657aaad","_uuid":"33190f95148e5a0b6f1333fbc8917d2866c83038"},"cell_type":"markdown","source":"**Top 5 ad category as classified by Avito's ad mode :**\n1. Clothing, shoes and accessories\n1. Children clothing and shoes\n1. Childrens product and toys\n1. Apartments\n1. Phones"},{"metadata":{"_cell_guid":"cf09d7c7-1921-49fa-8676-ba35302d8164","_uuid":"6945dce4e60f6350be5995391505f087f19229b6"},"cell_type":"markdown","source":"### Top 5 Top level (parent)  ad category as classified by Avito's ad model"},{"metadata":{"_cell_guid":"779f515f-2f21-4322-9ef7-ed03033628ea","_uuid":"98bb7de896b462a3edd4c938cb1b64c46146da13","trusted":true},"cell_type":"code","source":"conversion = StringIO(\"\"\"\nparent_category_name,parent_category_name_english\nЛичные вещи,Personal belongings\nДля дома и дачи,For the home and garden\nБытовая электроника,Consumer electronics\nНедвижимость,Real estate\nХобби и отдых,Hobbies & leisure\nТранспорт,Transport\nУслуги,Services\nЖивотные,Animals\nДля бизнеса,For business\n\"\"\")\n\nconversion = pd.read_csv(conversion)\ntrain = pd.merge(train, conversion, on=\"parent_category_name\", how=\"left\")\n\n\ntemp = train[\"parent_category_name_english\"].value_counts()\nprint(\"Total Top level ad category as classified by Avito's ad model : \",len(train[\"title\"]))\ntrace = go.Bar(x = temp.index,y = (temp / temp.sum())*100,)\ndata = [trace]\nlayout = go.Layout(title = \"Top level ad category as classified by Avito's ad model\",\n    xaxis=dict(title='Top level ad category as classified by Avitos ad model',\n        tickfont=dict(size=14,color='rgb(107, 107, 107)')),\n    yaxis=dict(title='Count of Top level ad category in %',titlefont=dict(size=16,color='rgb(107, 107, 107)'),\n        tickfont=dict(size=14,color='rgb(107, 107, 107)')))\nfig = go.Figure(data=data, layout=layout)\npy.iplot(fig)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"1404e87b-c885-46a7-aaa7-32eb113e85db","_uuid":"7d0a481c4fb884f5e41942c9ecce1fd674246d21","collapsed":true},"cell_type":"markdown","source":"**Top 5 Top level ad category as classified by Avito's ad model :\n\n1. Personal belongings - 46 %\n1. For the home and garden - 12 %\n1. Consumer electronics - 12 %\n1. Real estate - 10 %\n1. Hobbies & leisure - 6 %"},{"metadata":{"_cell_guid":"90a6003b-b8d2-488d-81ab-0a59bc954685","_uuid":"2ccdd937c746946fe97e5ba08090df41487dfafc"},"cell_type":"markdown","source":"> ## Price price in relation to Deal probability"},{"metadata":{"_cell_guid":"4ebf229f-44ca-4ab1-bc38-5a63a1720750","_uuid":"4debc91a42b974e1c499d1392bfc94d753e01443","trusted":true},"cell_type":"code","source":"plt.figure(figsize=(15,6))\nplt.scatter(np.log(train.price), train.deal_probability)\nplt.xlabel('Ad price')\nplt.ylabel('deal probability')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"cd051078-b5a3-4926-9ec3-4812363616a5","_uuid":"f9ea9f98cb8bf9688791791c51bdad39d0cb6ffc"},"cell_type":"markdown","source":"## Distribution of user type"},{"metadata":{"_cell_guid":"6ffd5d9e-242f-46ee-b675-396dc36869f0","_uuid":"644b8e9a44db2e927ccaa13a0f0c3435db474364","trusted":true},"cell_type":"code","source":"temp = train['user_type'].value_counts()\nlabels = temp.index\nsizes = (temp / temp.sum())*100\ntrace = go.Pie(labels=labels, values=sizes, hoverinfo='label+percent')\nlayout = go.Layout(title='Distribution of user type')\ndata = [trace]\nfig = go.Figure(data=data, layout=layout)\npy.iplot(fig)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"72b994fe-e797-41d1-b47a-c6a2b5fdab6d","_uuid":"cfa6b46d9b42cbbbf1d9c497be3ee99387906e5d"},"cell_type":"markdown","source":"**Distribution of user types :**\n1. Private users constitutes 71.6 % data\n1. Comapny users constitutes 23.1 % data\n1. Shop users constitutes 5.35 % data"},{"metadata":{"_cell_guid":"27a95395-2629-4d75-803d-24de887e08df","_uuid":"7b714782b7e32ab9b43423c0858426cf623b15ae"},"cell_type":"markdown","source":"## Monthly distribution of Ad prices in different regions "},{"metadata":{"_cell_guid":"8307e52c-1984-4b06-9737-1491a0190122","_uuid":"9a7f20e1a3979bdbc340f14dcac945a9bc7f458b","trusted":true},"cell_type":"code","source":"train['activation_date'] = pd.to_datetime(train['activation_date'])\ntrain['month'] = train.activation_date.dt.month\npr = train.groupby(['region_english', 'month'])['price'].mean().unstack()\n#pr = pr.sort_values([12], ascending=False)\nf, ax = plt.subplots(figsize=(15, 20)) \npr = pr.fillna(0)\ntemp = sns.heatmap(pr, cmap='Reds')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"ac48c2d3-13bc-4d79-9c3d-a078c6026fd9","_uuid":"1e3aa014fbf5337ae4ef8e8d30ea41721994f267"},"cell_type":"markdown","source":"**Highest Ad prices is in Irkutsk oblast region followed by Krasnodar Krai region**"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}