{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":1,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"path = \"../input/\"\n\nprint('load train...')\ntrain_df = pd.read_csv(path+\"train.csv\",usecols=[ 'click_time'])\nprint('load test...')\ntest_df = pd.read_csv(path+\"test.csv\", usecols=['click_time'])\n","execution_count":4,"outputs":[]},{"metadata":{"_uuid":"a2ee8c64eaf4e1488604d8fcd449e052a83b66e4"},"cell_type":"markdown","source":"## Let's have a look at the hour distribution of click_time"},{"metadata":{"trusted":true,"_uuid":"7df55c59a2271e75ea4196a4b099b141625ca62d"},"cell_type":"code","source":"train_df.click_time.str[11:13].value_counts().sort_index()","execution_count":6,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"87834db82c34e8f43770e215d4c1f0b252bbd70e"},"cell_type":"code","source":"test_df.click_time.str[11:13].value_counts().sort_index()","execution_count":7,"outputs":[]},{"metadata":{"_uuid":"6a0eda06c8e3c735e29dfd5c4a87cb9b069ee9cd"},"cell_type":"markdown","source":"## You can find that only 4,5,9,10,13,14 occur  in the  test set  and ignore 6,11,15. Just try this：\n\n\n\n"},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"5ba77f3c8ff629d302700d8bbc306f1917aa41cf"},"cell_type":"code","source":"train_df['hour'] = pd.to_datetime(train_df.click_time).dt.hour.astype('uint8')\ntrain_df = train_df[(train_df.hour == 4)|(train_df.hour == 5)|(train_df.hour == 9)|(train_df.hour == 10)|(train_df.hour == 13)|(train_df.hour == 14)]","execution_count":8,"outputs":[]},{"metadata":{"_uuid":"d2c6a97ceaaf11da305a789d9d042ea549e65d59"},"cell_type":"markdown","source":""}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}