{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n\n# Any results you write to the current directory are saved as output.","execution_count":14,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7fed5bc0dd45d90d2d618b18ef19a5567cd5a356"},"cell_type":"code","source":"# reading train data\n\n#import first 10,000,000 rows of train and all test data\ndf_train = pd.read_csv('../input/train.csv', nrows=10000000) #Reading the dataset in a dataframe using Pandas\ndf_test = pd.read_csv('../input/test.csv')\ndisplay(df_train.head(5))\ndisplay('==========================================================================')\ndisplay(df_test.head(5))","execution_count":13,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"display(df_train.shape)","execution_count":16,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7419ca00797fa60c78f8a2bd8148e68bd573d5d1"},"cell_type":"code","source":"display(df_train.describe())","execution_count":17,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"313cafa5b8122e836e4ee07f98d50ec4bc1b0ae8"},"cell_type":"code","source":"print(\"Finding the count of null values in the police dataset\")\npd.isnull(df_train).sum()","execution_count":18,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2f7f8fb8f4d37842b0177fa144380801c4ce20f6"},"cell_type":"code","source":"# Exploring APP ID \napp=df_train['app']\nprint(\"Most App : \", len(app.unique()))\nprint(app.value_counts().head(4))\napp_name = app.value_counts().head(4)","execution_count":19,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ffb7dad27567dbf14741c35fbfec0433898ccb14"},"cell_type":"code","source":"# OS  Distribution \nos=df_train['os']\nprint(\"Most os : \", len(os.unique()))\nprint(os.value_counts().head(4))\nos = os.value_counts().head(4)","execution_count":20,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7ade5805cfb49060b09d04f62c2da76d52ef3f66"},"cell_type":"code","source":"plt.figure(figsize=(15,8))\nsns.barplot(x=app_name.index,y=app_name.values,alpha=0.9)\nplt.xticks(rotation='vertical')\nplt.xlabel('APP ID ', fontsize=12)\nplt.ylabel('Most APP ID USED FOR MARKETING', fontsize=12)\nplt.title(\"Distribution of APP ID \", fontsize=16)\nplt.show()","execution_count":21,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b390cc1d1004a062aa3a2618f0e8cadba390c5eb"},"cell_type":"code","source":"plt.figure(figsize=(8,10))\ndf_train.groupby([df_train['os'].head(10)]).size().sort_values(ascending=True).plot(kind='barh')\nplt.title('Most OS ')\nplt.ylabel('OS used ')\nplt.xlabel('Distribution of OS Version used by various user')\nplt.show()","execution_count":23,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"013fbe32be62979579bbf9fa5cdb9ab1c898e83a"},"cell_type":"code","source":"channel=df_train['channel']\nprint(\"Most channel : \", len(app.unique()))\nprint(channel.value_counts().head(4))\nchannel_id = channel.value_counts().head(4)","execution_count":24,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bce598d82f73739f1761c3510e965b69843bdba2"},"cell_type":"code","source":"plt.figure(figsize=(15,8))\nsns.barplot(x=channel_id.index,y=channel_id.values,alpha=0.9)\nplt.xticks(rotation='vertical')\nplt.xlabel('Channel Id ', fontsize=12)\nplt.ylabel('Most Channel Id USED', fontsize=12)\nplt.title(\"Distribution of Channel Id \", fontsize=16)\nplt.show()","execution_count":25,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0cd6472068639c7ebf383b87de1abd6d05dea04b"},"cell_type":"code","source":"df_num = df_train.select_dtypes(include = ['float64', 'int64'])\ndf_num.head(4)","execution_count":26,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2cd2d01695f5cf67bbb2848fdf3d76f4b674c163"},"cell_type":"code","source":"print('Numerical distribution of the dataset')\nplt.figure(figsize=(9, 8))\ndf_num.hist(figsize=(16, 20), bins=50, xlabelsize=8, ylabelsize=8); # ; avoid having the matplotlib verbose informations\nplt.show()","execution_count":27,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0bdeb86e8d04b90af527d106aa0f79449fe48d14"},"cell_type":"code","source":"df_train['year']    = pd.to_datetime(df_train.click_time).dt.year.astype('uint8')\ndf_train['hour']    = pd.to_datetime(df_train.click_time).dt.hour.astype('uint8')\ndf_train['day']     = pd.to_datetime(df_train.click_time).dt.day.astype('uint8')\ndf_train['wday']    = pd.to_datetime(df_train.click_time).dt.dayofweek.astype('uint8')\nprint(df_train.dtypes) ","execution_count":29,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"63a4d2e64e89550e66aaa8423413b28d70bd09e4"},"cell_type":"code","source":"plt.figure(figsize=(8,10))\ndf_train.groupby([df_train['day']]).size().sort_values(ascending=True).plot(kind='barh')\nplt.title('Most day ')\nplt.ylabel('day used ')\nplt.xlabel('Distribution of days used by various user')\nplt.show()","execution_count":30,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3a899d9dd06dc890158500b7f627197296f6b3e1"},"cell_type":"code","source":"# Understanding the correlation between the features and the labels\nplt.figure(figsize=(15, 5))\nplt.hist(app[df_train['is_attributed'] == 0], bins=20, normed=True, label='App Downloaded')\nplt.hist(app[df_train['is_attributed'] == 1], bins=20, normed=True, alpha=0.7, label='App Not downloaded')\nplt.legend()\nplt.title('Label distribution', fontsize=15)\nplt.xlabel('App ', fontsize=15)\nplt.show()\n","execution_count":33,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"46131d42b18a4a5727a6980301032b998d7d719e"},"cell_type":"code","source":"hour=df_train['hour']\nplt.figure(figsize=(15, 5))\nplt.hist(hour[df_train['is_attributed'] == 0], bins=20, normed=True, label='App Downloaded')\nplt.hist(hour[df_train['is_attributed'] == 1], bins=20, normed=True, alpha=0.7, label='App Not downloaded')\nplt.legend()\nplt.title('Label distribution', fontsize=15)\nplt.xlabel('Hour ', fontsize=15)\nplt.show()","execution_count":34,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c8ac8a06b7eea903370274c862245f70be681cb6"},"cell_type":"code","source":"wday=df_train['wday']\nplt.figure(figsize=(15, 5))\nplt.hist(wday[df_train['is_attributed'] == 0], bins=20, normed=True, label='App Downloaded')\nplt.hist(wday[df_train['is_attributed'] == 1], bins=20, normed=True, alpha=0.7, label='App Not downloaded')\nplt.legend()\nplt.title('Label distribution', fontsize=15)\nplt.xlabel('wday ', fontsize=15)\nplt.show()","execution_count":35,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a56306d96b714c44295229856b882549e11e5e06"},"cell_type":"code","source":"device=df_train['device']\nplt.figure(figsize=(15, 5))\nplt.hist(device[df_train['is_attributed'] == 0], bins=20, normed=True, label='App Downloaded')\nplt.hist(device[df_train['is_attributed'] == 1], bins=20, normed=True, alpha=0.7, label='App Not downloaded')\nplt.legend()\nplt.title('Label distribution', fontsize=15)\nplt.xlabel('Device', fontsize=15)\nplt.show()","execution_count":36,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2d1f69e5178c745a395c1caefab88c8263ece4ba"},"cell_type":"code","source":"channel=df_train['channel']\nplt.figure(figsize=(15, 5))\nplt.hist(channel[df_train['is_attributed'] == 0], bins=20, normed=True, label='App Downloaded')\nplt.hist(channel[df_train['is_attributed'] == 1], bins=20, normed=True, alpha=0.7, label='App Not downloaded')\nplt.legend()\nplt.title('Label distribution', fontsize=15)\nplt.xlabel('channel', fontsize=15)\nplt.show()","execution_count":37,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"71777d09e5788151dce9746b8eaf132096e57139"},"cell_type":"code","source":"os=df_train['os']\nplt.figure(figsize=(15, 5))\nplt.hist(os[df_train['is_attributed'] == 0], bins=20, normed=True, label='App Downloaded')\nplt.hist(os[df_train['is_attributed'] == 1], bins=20, normed=True, alpha=0.7, label='App Not downloaded')\nplt.legend()\nplt.title('Label distribution', fontsize=15)\nplt.xlabel('os', fontsize=15)\nplt.show()","execution_count":38,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d348044a5409ddb821c8985d43aac286e9b3e67c"},"cell_type":"code","source":"display(df_train.shape)","execution_count":68,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"897c2ea06ebd26d2c0de3a1fbeae5ee3496e72cd"},"cell_type":"code","source":"display(df_train.columns)","execution_count":66,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"ef3dc05098e0352b7f51ebdceddd509409c9b71a"},"cell_type":"code","source":"cols = list(df_train.columns.values) #Make a list of all of the columns in the df\ncols.pop(cols.index('is_attributed')) #Remove b from list\ndf_train = df_train[cols+['is_attributed']]","execution_count":69,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"5c84f870cba5c2738155d098b95b0ad9e8ad21ad"},"cell_type":"code","source":"display(df_train.columns)\ndisplay(df_train.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"59302bd5e7b25d50e51a23ff678a50f19ce71539"},"cell_type":"code","source":"from sklearn.feature_selection import SelectKBest\nfrom sklearn.feature_selection import chi2\n\nX=df_train.loc[:, : 'wday']\nX.drop('click_time',axis=1,inplace=True)\nX.drop('attributed_time',axis=1,inplace=True)\nprint(X.columns)\nY =df_train['is_attributed']","execution_count":71,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8e68ff31f047a6154d20f0c88e9d2af4bcbf3906"},"cell_type":"code","source":"# feature extraction\ntest = SelectKBest(score_func=chi2, k=4)\nfit = test.fit(X, Y)","execution_count":72,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"56170a6d8461ff7eae53d7296bceabb3f758da08"},"cell_type":"code","source":"# summarize scores\nnp.set_printoptions(precision=3)\nscores=fit.scores_\nprint(scores)\nprint('====================================================================')\nprint(np.sort(scores))","execution_count":73,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"d4fbb4003587a9208b4b17536f3114b701d16eb0"},"cell_type":"code","source":"# The best features which are describing the data are :\n# IP , APP , Channel , OS , Device , hour , wday , day , year ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3e522086f73699c437ada54a00d0a519f470006f"},"cell_type":"code","source":"features = fit.transform(X)\nprint(features.shape)\nprint(features[0:5,:])","execution_count":74,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"25dd5852f8fdd1acf43321387493d42575876c73"},"cell_type":"code","source":"from sklearn.ensemble import ExtraTreesClassifier\nmodel = ExtraTreesClassifier()\nmodel.fit(X, Y)\nprint(model.feature_importances_)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}