{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.feature_selection import RFE\nfrom sklearn.linear_model import LogisticRegression\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":42,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"#Loading the dataset in a dataframe and printing the head of the values\n\ndataset=pd.read_csv(\"../input/train_sample.csv\")\nprint(dataset.head())","execution_count":3,"outputs":[]},{"metadata":{"_cell_guid":"9ea4dfb2-9836-47d4-ac78-625d1a144002","_uuid":"90b92636e0042070f406bb7c4a4a1e6455580fe1","trusted":true},"cell_type":"code","source":"#Dropping the IP column for now as dont feel the need in the actual dataset\n\ndata=dataset.drop(\"ip\",axis=1)\nprint(data.head())","execution_count":4,"outputs":[]},{"metadata":{"_cell_guid":"3087b192-5275-48c9-aa74-5e1a2fb8b46f","_uuid":"870b7a6de6ec003e43a598e83112d74a53d519cf","trusted":true},"cell_type":"code","source":"#Converting the click_time and attributed_time to date time \n\ndata[\"click_time\"]=pd.to_datetime(data[\"click_time\"])\ndata[\"attributed_time\"]=pd.to_datetime(data[\"attributed_time\"])\nprint(data.head(200))","execution_count":27,"outputs":[]},{"metadata":{"_cell_guid":"70bd65c8-22f4-460b-bcb1-a6c3612c79d2","_uuid":"34eec31324597df4f10cb9686429f18f814c476c","trusted":true},"cell_type":"code","source":"#Checking and plotting the table form of is_attributed variable in order to check the distribution of data\n\ndata[\"is_attributed\"].value_counts()\nsns.countplot(x='is_attributed',data=data,palette='hls')\nplt.show()","execution_count":6,"outputs":[]},{"metadata":{"_cell_guid":"e0262da4-7962-47ce-a213-8a8b5f5eb691","_uuid":"d63f7192050a448566ccc37af7426b490e6d924c","trusted":true},"cell_type":"code","source":"# Checking the count of all the values on basis of \"OS,APP,DEVICE,CHANNEL\"\n\ndata.groupby(\"os\").count()\ndata.groupby(\"app\").count()\ndata.groupby(\"device\").count()\ndata.groupby(\"channel\").count()","execution_count":7,"outputs":[]},{"metadata":{"_cell_guid":"9681c1e5-9fe2-4ec7-b8de-f4e90773cb0e","_uuid":"f6819ba431fef14eb22ff8ebc3eee201fc5be746","trusted":true},"cell_type":"code","source":"#Plotting the Device vs Is_Attributed and checking at the Distribution on basis of Device\n\n%matplotlib inline\npd.crosstab(data.device,data.is_attributed).plot(kind='bar')\nplt.title('Device with Is_Attribted')\nplt.xlabel('Device')\nplt.ylabel('Is_Attributed')","execution_count":8,"outputs":[]},{"metadata":{"_cell_guid":"3a65930f-5988-4677-97d6-1ee09bd6c174","_uuid":"71a1fc46c5a6304f5cab185036ceec9c95c9ab6b","trusted":true},"cell_type":"code","source":"#Encoding the variables according to the one hot encoding.\nencoded=pd.get_dummies(data,columns=[\"app\",\"device\",\"os\",\"channel\"],prefix=[\"app_encoded\",\"device_encoded\",\"os_encoded\",\"channel_encoded\"])\nprint(encoded.head())\nprint(encoded.columns.values)","execution_count":35,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"175033c25daf52879a23ee1b878cd6153cc27fb0"},"cell_type":"code","source":"#Adding a new variable which calculates the time difference between click_time and attributed_time\nnewencoded=encoded.fillna(0)\n#newencoded['attributed_time'].astype('datetime64[ns]')\n#newencoded[\"attributed_time\"]=pd.to_datetime(newencoded[\"attributed_time\"])\nprint(newencoded.dtypes)\n#newencoded['time_difference']=newencoded['click_time']-newencoded['attributed_time']\n#final_frame=encoded.drop(['click_time','attributed_time'],axis=1)\nprint(newencoded.head())\n","execution_count":45,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"14e66b1763c5362065eaadd258f00676e35977d1"},"cell_type":"code","source":"logreg = LogisticRegression()\nrfe = RFE(logreg, 20)\nrfe = rfe.fit(encoded[X], encoded[y] )\nprint(rfe.support_)\nprint(rfe.ranking_)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}