{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-03T15:28:52.825111Z","iopub.execute_input":"2021-08-03T15:28:52.825592Z","iopub.status.idle":"2021-08-03T15:28:52.844917Z","shell.execute_reply.started":"2021-08-03T15:28:52.825491Z","shell.execute_reply":"2021-08-03T15:28:52.843621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\ncolor = sns.color_palette()\n%matplotlib inline\n\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\nimport plotly.tools as tls\n\npd.options.mode.chained_assignment = None\npd.options.display.max_columns = 999","metadata":{"execution":{"iopub.status.busy":"2021-08-03T15:28:57.975743Z","iopub.execute_input":"2021-08-03T15:28:57.976137Z","iopub.status.idle":"2021-08-03T15:28:59.239545Z","shell.execute_reply.started":"2021-08-03T15:28:57.976105Z","shell.execute_reply":"2021-08-03T15:28:59.238343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. EDA","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/avito-demand-prediction/train.csv\", parse_dates=[\"activation_date\"])\ntest_df = pd.read_csv(\"/kaggle/input/avito-demand-prediction/test.csv\", parse_dates=[\"activation_date\"])\nprint(\"Train file rows and columns are : \", train_df.shape)\nprint(\"Test file rows and columns are : \", test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T15:28:59.241203Z","iopub.execute_input":"2021-08-03T15:28:59.241598Z","iopub.status.idle":"2021-08-03T15:29:51.044653Z","shell.execute_reply.started":"2021-08-03T15:28:59.241556Z","shell.execute_reply":"2021-08-03T15:29:51.043263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T15:29:51.046634Z","iopub.execute_input":"2021-08-03T15:29:51.046962Z","iopub.status.idle":"2021-08-03T15:29:51.085506Z","shell.execute_reply.started":"2021-08-03T15:29:51.046931Z","shell.execute_reply":"2021-08-03T15:29:51.08424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 1.1 Deal Probablity Distributions","metadata":{}},{"cell_type":"code","source":"# Let see how the probablities are distributed with Deal_Probablity on a Histogram plot\nplt.figure(figsize=(12,8))\nsns.displot(train_df[\"deal_probability\"].values, bins=100, kde=False)\nplt.xlabel('Deal Probility', fontsize=12)\nplt.title(\"Deal Probability Histogram\", fontsize=14)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T15:29:51.087526Z","iopub.execute_input":"2021-08-03T15:29:51.087941Z","iopub.status.idle":"2021-08-03T15:29:52.82113Z","shell.execute_reply.started":"2021-08-03T15:29:51.087894Z","shell.execute_reply":"2021-08-03T15:29:52.82026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let see how the probablities are distributed with Deal_Probablity on a scatter plot\nplt.figure(figsize=(8,6))\nplt.scatter(range(train_df.shape[0]), np.sort(train_df['deal_probability'].values))\nplt.xlabel('index', fontsize=12)\nplt.ylabel('deal probability', fontsize=12)\nplt.title(\"Deal Probability Distribution\", fontsize=14)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T15:29:52.82252Z","iopub.execute_input":"2021-08-03T15:29:52.822958Z","iopub.status.idle":"2021-08-03T15:29:56.211267Z","shell.execute_reply.started":"2021-08-03T15:29:52.822913Z","shell.execute_reply":"2021-08-03T15:29:56.210206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<b>From Above plots, Its clear that approx 100K Ads have sold nothing, Few ads has probablity of 1 and rest are between 0 to 1.</b> ","metadata":{}},{"cell_type":"markdown","source":"#### 1.2 Region wise distribution of Ads:","metadata":{}},{"cell_type":"code","source":"# Let see how the probablities are distributed with Deal_Probablity on a Histogram plot\nplt.figure(figsize=(12,8))\nregions =  train_df[\"region\"].value_counts().index\nsns.barplot(x=train_df[\"region\"].value_counts(), y=regions, data=train_df)\nplt.xlabel('Regions_Counts', fontsize=15)\nplt.ylabel('Regions_Name', fontsize=15)\nplt.title(\"Regions Distributions\", fontsize=15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T15:31:30.363166Z","iopub.execute_input":"2021-08-03T15:31:30.363602Z","iopub.status.idle":"2021-08-03T15:31:31.56819Z","shell.execute_reply.started":"2021-08-03T15:31:30.363552Z","shell.execute_reply":"2021-08-03T15:31:31.566556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The Distributions of region with ads shows that some regions have high number of ads than other**","metadata":{}},{"cell_type":"markdown","source":"#### 1.3 Region wise Deal Probablity distributions","metadata":{}},{"cell_type":"code","source":"# Let see with region wise deal_probablity\nplt.figure(figsize=(12,8))\nsns.boxplot(y=train_df[\"region\"], x=train_df[\"deal_probability\"], data=train_df)\nplt.xlabel('Deal probability', fontsize=12)\nplt.ylabel('Region', fontsize=12)\nplt.title(\"Deal probability by region\")\nplt.xticks(rotation='vertical')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T16:03:20.695048Z","iopub.execute_input":"2021-08-03T16:03:20.695508Z","iopub.status.idle":"2021-08-03T16:03:24.86004Z","shell.execute_reply.started":"2021-08-03T16:03:20.69547Z","shell.execute_reply":"2021-08-03T16:03:24.859213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Box Plot with deal probablity shows that some region has minor advantages on probablity score**","metadata":{}},{"cell_type":"markdown","source":"#### 1.4 City wise distribution of Ads:","metadata":{}},{"cell_type":"code","source":"# Let see how th ads are distrubuted with respect to city \nplt.figure(figsize=(12,8))\ncities =  train_df[\"city\"].value_counts()[:25].index\nsns.barplot(x=train_df[\"city\"].value_counts()[:25], y=cities, data=train_df)\nplt.xlabel('Cities_Counts', fontsize=15)\nplt.ylabel('Cities_Name', fontsize=15)\nplt.title(\"Cities Ads Distributions\", fontsize=15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T16:06:25.628332Z","iopub.execute_input":"2021-08-03T16:06:25.628811Z","iopub.status.idle":"2021-08-03T16:06:26.783084Z","shell.execute_reply.started":"2021-08-03T16:06:25.62877Z","shell.execute_reply":"2021-08-03T16:06:26.781863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Top 25 cities ads distributions, the top cities are good cities in russia**","metadata":{}},{"cell_type":"markdown","source":"#### 1.5 Parent Category wise Ads Distribution","metadata":{}},{"cell_type":"code","source":"# Let see how how th ads are distributed with respect to parent_category\nplt.figure(figsize=(10,5))\nparent_cat =  train_df[\"parent_category_name\"].value_counts().index\nsns.barplot(x=train_df[\"parent_category_name\"].value_counts(), y=parent_cat, data=train_df)\nplt.xlabel('Parent Category Counts', fontsize=15)\nplt.ylabel('Parent Category Name', fontsize=15)\nplt.title(\"Parent Categories Distributions\", fontsize=15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T16:08:43.482655Z","iopub.execute_input":"2021-08-03T16:08:43.48306Z","iopub.status.idle":"2021-08-03T16:08:44.4596Z","shell.execute_reply.started":"2021-08-03T16:08:43.483029Z","shell.execute_reply":"2021-08-03T16:08:44.458345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The parent category 'Personal Belongings' is highly dominated in the dataset.**","metadata":{}},{"cell_type":"markdown","source":"#### 1.6 Deal Probablity are Distribution with respect to the parent category name","metadata":{}},{"cell_type":"code","source":"# Let see how the deal probablity are distributed with respect to the parent category name\nplt.figure(figsize=(12,8))\nsns.boxplot(y=train_df[\"parent_category_name\"], x=train_df[\"deal_probability\"], data=train_df)\nplt.xlabel('Deal probability', fontsize=12)\nplt.ylabel('Region', fontsize=12)\nplt.title(\"Deal probability by region\")\nplt.xticks(rotation='vertical')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T16:12:03.138704Z","iopub.execute_input":"2021-08-03T16:12:03.139114Z","iopub.status.idle":"2021-08-03T16:12:06.524879Z","shell.execute_reply.started":"2021-08-03T16:12:03.139076Z","shell.execute_reply":"2021-08-03T16:12:06.523644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The parent category 'Services' has a better deal probablity than others.**","metadata":{}},{"cell_type":"markdown","source":"#### 1.7 Category Name Wise distributions of Ads","metadata":{}},{"cell_type":"code","source":"# Let see how how th ads are distrubuted with respect to parent_category\nplt.figure(figsize=(10,20))\ncategory_name =  train_df[\"category_name\"].value_counts().index\nsns.barplot(x=train_df[\"category_name\"].value_counts(), y=category_name, data=train_df)\nplt.xlabel('Counts', fontsize=15)\nplt.ylabel('Category Name', fontsize=15)\nplt.title(\"Ads Distributions with Category Name\", fontsize=15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T16:14:18.280218Z","iopub.execute_input":"2021-08-03T16:14:18.280623Z","iopub.status.idle":"2021-08-03T16:14:19.859258Z","shell.execute_reply.started":"2021-08-03T16:14:18.28059Z","shell.execute_reply":"2021-08-03T16:14:19.858226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The distributions of ads clearly shows the domination of two category over other category. Those category are Clothes, shoes, accessories & Children's clothing and footwear**","metadata":{}},{"cell_type":"markdown","source":"#### 1.8 Deal Probablity distribution according to the category name","metadata":{}},{"cell_type":"code","source":"# Let see how the deal probablity are distributed with respect to the category name\nplt.figure(figsize=(8,20))\nsns.boxplot(y=train_df[\"category_name\"], x=train_df[\"deal_probability\"], data=train_df)\nplt.xlabel('Deal probability', fontsize=12)\nplt.ylabel('Category Name', fontsize=12)\nplt.title(\"Deal probability by Category Name\")\nplt.xticks(rotation='vertical')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T16:14:19.861011Z","iopub.execute_input":"2021-08-03T16:14:19.861338Z","iopub.status.idle":"2021-08-03T16:14:24.54517Z","shell.execute_reply.started":"2021-08-03T16:14:19.861307Z","shell.execute_reply":"2021-08-03T16:14:24.544192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The box plot shows that some categories are having a good deal probablity than others.**","metadata":{}},{"cell_type":"markdown","source":"#### 1.9 Ads Distributions by User Type ","metadata":{"execution":{"iopub.status.busy":"2021-08-03T04:50:04.021268Z","iopub.execute_input":"2021-08-03T04:50:04.021833Z","iopub.status.idle":"2021-08-03T04:50:04.025471Z","shell.execute_reply.started":"2021-08-03T04:50:04.021797Z","shell.execute_reply":"2021-08-03T04:50:04.024764Z"}}},{"cell_type":"code","source":"# Let see how how th ads are distrubuted with respect to parent_category\nplt.figure(figsize=(10,5))\nuser_type =  train_df[\"user_type\"].value_counts().index\nsns.barplot(x=train_df[\"user_type\"].value_counts(), y=user_type, data=train_df)\nplt.xlabel('Counts', fontsize=15)\nplt.ylabel('User Type', fontsize=15)\nplt.title(\"Ads Distributions with user_type\", fontsize=15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T16:14:24.546873Z","iopub.execute_input":"2021-08-03T16:14:24.547362Z","iopub.status.idle":"2021-08-03T16:14:25.43427Z","shell.execute_reply.started":"2021-08-03T16:14:24.547312Z","shell.execute_reply":"2021-08-03T16:14:25.433353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Dataset having more private users followed by company and shops**","metadata":{}},{"cell_type":"markdown","source":"#### 1.10 Deal Probablity Distributions by User type","metadata":{}},{"cell_type":"code","source":"# Let see how the deal probablity are distributed with respect to the User Type\nplt.figure(figsize=(10,5))\nsns.boxplot(y=train_df[\"user_type\"], x=train_df[\"deal_probability\"], data=train_df)\nplt.xlabel('Deal probability', fontsize=12)\nplt.ylabel('User Type', fontsize=12)\nplt.title(\"Deal probability by user type\")\nplt.xticks(rotation='vertical')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T16:14:25.435875Z","iopub.execute_input":"2021-08-03T16:14:25.436356Z","iopub.status.idle":"2021-08-03T16:14:28.567866Z","shell.execute_reply.started":"2021-08-03T16:14:25.436304Z","shell.execute_reply":"2021-08-03T16:14:28.566516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 1.11 Distributions of Ads Price","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(20,12))\nsns.displot(np.log1p(train_df[\"price\"].values), bins=100, kde=True)\nplt.xlabel('Log of price', fontsize=12)\nplt.title(\"Log of Price Histogram\", fontsize=14)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T16:14:28.569705Z","iopub.execute_input":"2021-08-03T16:14:28.570147Z","iopub.status.idle":"2021-08-03T16:14:36.819368Z","shell.execute_reply.started":"2021-08-03T16:14:28.570099Z","shell.execute_reply":"2021-08-03T16:14:36.817076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The Log of Price shows not a fully gussian distribution**","metadata":{}},{"cell_type":"markdown","source":"#### 1.12 Words length in Title","metadata":{}},{"cell_type":"code","source":"#How to calculate number of words in a string in DataFrame: https://stackoverflow.com/a/37483537/4084039\nword_count = train_df['title'].str.split().apply(len).value_counts()\nword_dict = dict(word_count)\nword_dict = dict(sorted(word_dict.items(), key=lambda kv: kv[1]))\n\n\nind = np.arange(len(word_dict))\nplt.figure(figsize=(20,10))\np1 = plt.bar(ind, list(word_dict.values()))\n\nplt.xlabel('Length of words in the Title')\nplt.ylabel('Number of Titles')\nplt.title('Title Word Length Distributions')\nplt.xticks(ind, list(word_dict.keys()))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T16:14:36.820684Z","iopub.execute_input":"2021-08-03T16:14:36.821009Z","iopub.status.idle":"2021-08-03T16:14:42.676617Z","shell.execute_reply.started":"2021-08-03T16:14:36.820978Z","shell.execute_reply":"2021-08-03T16:14:42.675549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The maximum title having 1 to 6 words in their title**","metadata":{}},{"cell_type":"code","source":"# Let see how how th ads are distrubuted with respect to parent_category\nplt.figure(figsize=(10,5))\nact_date_train =  train_df[\"activation_date\"].value_counts().index\nsns.barplot(x=train_df[\"activation_date\"].value_counts(), y=act_date_train, data=train_df)\nplt.xlabel('Counts', fontsize=15)\nplt.ylabel('Activation Date', fontsize=15)\nplt.title(\"Activatation Date Distribution of Train Set\", fontsize=15)\nplt.show()\n\nact_date_test =  test_df[\"activation_date\"].value_counts().index\nsns.barplot(x=test_df[\"activation_date\"].value_counts(), y=act_date_test, data=test_df)\nplt.xlabel('Counts', fontsize=15)\nplt.ylabel('Activations Date', fontsize=15)\nplt.title(\"Activation Date Distribution of Test Set\", fontsize=15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T16:14:42.678746Z","iopub.execute_input":"2021-08-03T16:14:42.679066Z","iopub.status.idle":"2021-08-03T16:14:43.300807Z","shell.execute_reply.started":"2021-08-03T16:14:42.679035Z","shell.execute_reply":"2021-08-03T16:14:43.299658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The dates are different between train and test sets.Given dataset has data for training from March 15 to March 28 and for testing April 12 to April 18, 2017. There is a gap of two weeks in between training and testing data.**","metadata":{}}]}