{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\nimport os\nimport seaborn as sns\nimport plotly.graph_objs as go\nimport matplotlib.pyplot as plt\ncolor = sns.color_palette()\nprint(os.listdir(\"../input\"))\nimport csv\n# Any results you write to the current directory are saved as output.","execution_count":67,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"dataset=pd.read_csv(\"../input/train_sample.csv\")\nprint(\"\\n\",dataset.head())","execution_count":68,"outputs":[]},{"metadata":{"_cell_guid":"bd6cea0a-24c4-44d9-b734-c3f2e9699760","_uuid":"7ff90bb0a85e8142424f8794feda193e6abd6eae","trusted":true},"cell_type":"code","source":"print(dataset.shape)","execution_count":14,"outputs":[]},{"metadata":{"_cell_guid":"f1ff4d14-c92e-4a3a-8fe7-670c8ec28aa5","_uuid":"493700ae65cc81349d0eccac0a6ee3dbdda5c62f","trusted":true},"cell_type":"code","source":"dataset.isnull().sum().sort_values(ascending = False)\n","execution_count":69,"outputs":[]},{"metadata":{"_cell_guid":"80bbc3e1-691c-4222-bde6-e2ac7127b59f","_uuid":"42921830377e3bccb8c5b2cf4adbfa8ea7ce9de1","trusted":true},"cell_type":"code","source":"dataset.fillna(0,inplace = True)\ndataset.isnull().sum().sort_values(ascending = False)","execution_count":70,"outputs":[]},{"metadata":{"_cell_guid":"f8d45a2a-724e-403d-91be-bf0e82ce1fe2","_uuid":"59e8e48470fce6534502dcd62289b1e00936cfe4","collapsed":true,"trusted":true},"cell_type":"code","source":"def normalizeData(dataset):\n    # Instantiate Scaler Object \n    ss = StandardScaler()\n    # Fit and transform \n    ss.fit_transform(dataset)\n    dataset.head(100000)\n    #country.head(10)\n    return dataset\ndef univariateAnalysis(dataset):\n    sns.distplot(dataset['ip'])\n\n#heatmapAnaly(testdata)","execution_count":72,"outputs":[]},{"metadata":{"_cell_guid":"3e009067-6f33-41e1-8519-5829d614ecee","_uuid":"1a37b7b66720402bc3197a51de92f34bf9bc8ff1","trusted":true},"cell_type":"code","source":"univariateAnalysis(dataset)","execution_count":73,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e60b292c902c19c1f930d1f07ac4e51bb517b417"},"cell_type":"code","source":"dataset.groupby(\"os\").count()\ndataset.groupby(\"app\").count()\ndataset.groupby(\"device\").count()\ndataset.groupby(\"channel\").count()","execution_count":74,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dd7179c537f014029575fa29ae5080e76e7e5f5e"},"cell_type":"code","source":"\ndataset['is_attributed'].fillna(0,inplace=True)\ndataset['is_attributed'].value_counts()","execution_count":75,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cb21adf3a882f8d48291743e74cf1e9d12fd09e4"},"cell_type":"code","source":"%matplotlib inline\npd.crosstab(dataset.device,dataset.is_attributed).plot(kind='bar')\nplt.title('Device with Is_Attribted')\nplt.xlabel('Device')\nplt.ylabel('Is_Attributed')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}