{"cells":[{"metadata":{"_uuid":"1ab1c87c68b1474f3173b5793db4b8eb8466eca9"},"cell_type":"markdown","source":"## 1. Imports and settings"},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"e6fa89e134c1c5a2bcd6e3d0858702752348e616"},"cell_type":"code","source":"# This is how I built my validation set. I'm curious if you use the same? If not, say what set you use.","execution_count":12,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"6a1304cc19bc3edfcbad5deff4fc8df7262b3cf7"},"cell_type":"code","source":"# IMPORTS\n\nimport numpy as np\nimport pandas as pd\nimport os\nfrom datetime import datetime\nimport gc","execution_count":13,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b9d912d1ba9126b409a06db8dc83665e953cee8f"},"cell_type":"code","source":"# SETTINGS\n\n# Path to train.csv and valid.csv\npath_in = '../input/train.csv'\npath_out = '../input/valid.csv'\n\n# Names of columns to load from train set\ntrain_columns = ['ip', 'app', 'device', 'os', 'channel', 'click_time', 'is_attributed']\n\n# Types of columns to load from trains set\ndtypes = {\n    'ip'            : 'uint32',\n    'app'           : 'uint16',\n    'device'        : 'uint16',\n    'os'            : 'uint16',\n    'channel'       : 'uint16',\n    'is_attributed' : 'uint8',\n    'click_id'      : 'uint32'\n}\n\n# Limit dates for the training and test set\ntest_start_date = \"2017-11-09 04:00:00\"\ntest_end_date = \"2017-11-09 16:00:00\"\n\ntest_start_date = datetime.strptime(test_start_date, '%Y-%m-%d %H:%M:%S')\ntest_end_date = datetime.strptime(test_end_date, '%Y-%m-%d %H:%M:%S')","execution_count":14,"outputs":[]},{"metadata":{"_uuid":"acfc77dc0b26a966857130149a34bf2c33b1d516"},"cell_type":"markdown","source":"## 2. Building filtering function"},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"1f5480307086bdf241af89f5df5b344ad43c2565"},"cell_type":"code","source":"def buildValidationSet(df):\n    print(\"Converting to datetime...\")\n    df['click_time'] = pd.to_datetime(df['click_time'])\n    print(\"Filtration of dataset...\")\n    df = df[(df['click_time'] < test_end_date) & (df['click_time'] >= test_start_date)]\n    print(\"Number of unique entries per each hour:\")\n    print(df.click_time.dt.hour.value_counts())\n    return df","execution_count":15,"outputs":[]},{"metadata":{"_uuid":"e6a39c43801fbd0f654fbbddb6d50afacc7c293f"},"cell_type":"markdown","source":"## 3. Make validation set"},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"d808121bccbd6cdf7e5ad23375497a579abd61cb"},"cell_type":"code","source":"# LOAD DATA\ndf_train = pd.read_csv(path_in, usecols=train_columns, dtype=dtypes)\ndf_train = buildValidationSet(df_train)\ngc.collect()\n# this line is, of course, to be uncommented\n# df_train.to_csv(path_out)\ngc.collect()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.5.2"}},"nbformat":4,"nbformat_minor":1}