{"cells":[{"metadata":{"trusted":true,"_uuid":"e4b8d37fa256bff37a5140a62de2bde78d87dca8"},"cell_type":"code","source":"from IPython.display import Image\nImage(url='https://upload.wikimedia.org/wikipedia/commons/thumb/4/48/3-phase_flow.gif/357px-3-phase_flow.gif')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import pandas as pd","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7a47b60fe3c1f84574e2894b70c7036ff6e488c6"},"cell_type":"code","source":"metadata_train_ = pd.read_csv(\"../input/metadata_train.csv\")\nmetadata_test_ = pd.read_csv(\"../input/metadata_test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aa28da5d8b7169dc93fff906f189428d860515ca"},"cell_type":"code","source":"metadata_test_.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"795d476805f4676cf3f2f15901544596bc2cd2cf"},"cell_type":"code","source":"metadata_train_.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"79f55d27bf9f106ebc0d9d4f83992d6f3efbd60f"},"cell_type":"code","source":"metadata_train_.dtypes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ae04508567de53b9983855ebd13300ed5b82c412"},"cell_type":"code","source":"metadata_train_.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5ef59b7c30c180805d3525340e3781ed3ff91875"},"cell_type":"code","source":"for col in ['id_measurement', 'phase', 'target']:\n    metadata_train_[col] = metadata_train_[col].astype('category')\n    \n    \nfor col in ['id_measurement', 'phase']:\n    metadata_test_[col] = metadata_test_[col].astype('category')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"411e03d3a5cb24a8e8c0aa3faf7cb52beaad107e"},"cell_type":"code","source":"metadata_train_.dtypes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2566fb3c460dc420c432dfc520b8a6f746edf6cc"},"cell_type":"code","source":"metadata_test_.dtypes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c6bbfe900de5cc29ae6dfd1ababa8d6eaa4cbd5f"},"cell_type":"code","source":"stats = []\nfor col in metadata_train_.columns:\n    stats.append((col, metadata_train_[col].nunique(), metadata_train_[col].isnull().sum() * 100 / metadata_train_.shape[0], metadata_train_[col].value_counts(normalize=True, dropna=False).values[0] * 100, metadata_train_[col].dtype))\n    \nstats_df = pd.DataFrame(stats, columns=['Feature', 'Unique_values', 'Percentage of missing values', 'Percentage of values in the biggest category', 'type'])\nstats_df.sort_values('Percentage of missing values', ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"53b4b1e19e7f95a1dfcf0aa982ef3d0d1241cbd1"},"cell_type":"code","source":"import plotly\nfrom plotly.offline import download_plotlyjs, init_notebook_mode, plot, iplot\nimport plotly.graph_objs as go\ninit_notebook_mode(connected=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bc4767c96fbccc89d09e62fc78b69189f989f228"},"cell_type":"code","source":"data = [go.Bar(x=stats_df.Feature,\n            y=stats_df.Unique_values)]\n\niplot(data, filename='jupyter-basic_bar')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c22525dbd5bb81c1703a7dacfc4e2fcea9a77e5f"},"cell_type":"code","source":"stats_ = []\nfor col in metadata_test_.columns:\n    stats_.append((col, metadata_test_[col].nunique(), metadata_test_[col].isnull().sum() * 100 / metadata_test_.shape[0], metadata_test_[col].value_counts(normalize=True, dropna=False).values[0] * 100, metadata_test_[col].dtype))\n    \nstats_df_ = pd.DataFrame(stats_, columns=['Feature', 'Unique_values', 'Percentage of missing values', 'Percentage of values in the biggest category', 'type'])\nstats_df_.sort_values('Percentage of missing values', ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a12db16d4f588e2660034ae20f05d6d7b95ffdc1"},"cell_type":"code","source":"data_ = [go.Bar(x=stats_df_.Feature,\n            y=stats_df_.Unique_values)]\n\niplot(data_, filename='jupyter-basic_bar')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"50b3127c75fe9c4cfef26dbdff959ebab9dcd77b"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"64e8bcd55c4c871528c5746042eb7bea89b24948"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}