{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","collapsed":true,"trusted":false},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n#import matplotlib.pyplot as plt\nimport pickle\nimport os\nimport gc\nfrom sklearn.model_selection import train_test_split\nfrom catboost import CatBoostClassifier,Pool\nimport lightgbm as lgb\nimport xgboost as xgb\nimport matplotlib.pyplot as plt\n\nlocal_path='../input/'\n\nimport pandas as pd\n\ndef load_data(name,skip=None,rows=None):\n    ''' Load the csv files into a TimeSeries dataframe with minimal data types to reduce the used RAM space. \n    It also saves the files in parquet file to reduce loading time by a factor of ~10.\n\n    Arg:\n    \n        -name (str): ante_day, last_day, train, train_sample or test\n\n    Returns:\n        pd.DataFrame, with int index equal to 'click_id'\n    '''\n\n    # Setting file path\n    file_path='{}{}'.format(local_path,name)\n    if skip!=None:\n        skip=range(1,skip)\n\n    # Defining dtypes\n    types = {\n            'ip':np.uint32,\n            'app': np.uint16,\n            'os': np.uint16,\n            'device': np.uint16,\n            'channel':np.uint16,\n            'click_time': object\n            }\n\n    if name=='test':\n        types['click_id']= np.uint32\n    elif name=='test_supplement':\n        types['click_id']= np.uint32\n    else:\n        types['is_attributed']='bool'\n\n    # Defining csv file reading parameters\n    read_args={\n        'nrows':rows,\n        'skiprows': skip,\n        'parse_dates':['click_time'],\n        'infer_datetime_format':True,\n        'index_col':'click_time',\n        'usecols':list(types.keys()),\n        'dtype':types,\n        'engine':'c',\n        'sep':','\n        }\n\n    print('Loading {}.csv'.format(file_path))\n    with open('{}.csv'.format(file_path),'rb') as File:\n        data=(pd\n            .read_csv(File,**read_args)\n            .tz_localize('UTC')\n            .tz_convert('Asia/Shanghai')\n        )\n\n    return data\n\ndef force_list(*arg):\n    ''' Takes a list of arguments and returns the same, \n    but where all items were forced to a list.\n\n    example : list_1,list_2=force_list(item1,item2)\n    '''\n    Gen=(x if isinstance(x,list) else [x] for x in arg)\n    if len(arg)>1:\n        return Gen\n    else:\n        return next(Gen)","execution_count":22,"outputs":[]},{"metadata":{"_uuid":"ad45f26d1d380d8eff61b1ced38af496d46bfd8f"},"cell_type":"markdown","source":"Let's try to use the whole day of the 10th for feature engineering and still be able to extract the test set for sudmission."},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","scrolled":false,"trusted":false},"cell_type":"code","source":"test=load_data('test')\nprint('The test set has {} observations'.format(len(test)))\ntest.resample('20T').app.count().plot.bar(figsize=(15,7))\nplt.gcf().autofmt_xdate()","execution_count":27,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"a747307153be4863f12c0bf45a0384732936e5e3"},"cell_type":"code","source":"test.head()","execution_count":16,"outputs":[]},{"metadata":{"_cell_guid":"9c3a22bd-8aea-46be-811e-9402dc3a69b5","_uuid":"e87016cb2056ce19e9b167dc656ab91640579356","scrolled":false,"trusted":false},"cell_type":"code","source":"test_plus=load_data('test_supplement')\nprint('The test supplement set has {} observations'.format(len(test_plus)))\ntest_plus.resample('20T').app.count().plot.bar(figsize=(15,7))\nplt.gcf().autofmt_xdate()","execution_count":28,"outputs":[]},{"metadata":{"_cell_guid":"1188fddc-588e-46ef-bbaa-32d90c61b73f","_uuid":"5e80908ff3642b7a3b0053e5ec96b4c1ef51915f","collapsed":true,"trusted":false},"cell_type":"code","source":"def combine(test,test_plus):\n    test_plus=test_plus.assign(click_id=-1)\n    test1=test.loc[test.index.hour.isin([12,13,14]),:]\n    test2=test.loc[test.index.hour.isin([17,18,19]),:]\n    test3=test.loc[test.index.hour.isin([21,22,23]),:]\n    one_sec=pd.Timedelta('1s')\n    test_plus1=test_plus.loc[:test1.index.min()-one_sec,:]\n    test_plus2=test_plus.loc[test1.index.max()+one_sec:test2.index.min()-one_sec,:]\n    test_plus3=test_plus.loc[test2.index.max()+one_sec:test3.index.min()-one_sec,:]\n    test_plus4=test_plus.loc[test3.index.max()+one_sec:,:]\n    del(test,test_plus)\n    gc.collect()\n    new_test=pd.concat([test_plus1,test1,test_plus2,test2,test_plus3,test3,test_plus4])\n    return new_test.reset_index()","execution_count":31,"outputs":[]},{"metadata":{"_cell_guid":"e991a362-42ca-4e12-8a3b-36acb6c9e6b6","_uuid":"bdce8862dd0c9094507abfd41a118c6706fa9bb4","scrolled":false,"trusted":false},"cell_type":"code","source":"test_whole=combine(test,test_plus)\nprint('The test supplement set has {} observations'.format(len(test_whole)))\ntest_plus.resample('20T').app.count().plot.bar(figsize=(15,7))\nplt.gcf().autofmt_xdate()","execution_count":32,"outputs":[]},{"metadata":{"_uuid":"e9551ef2c8e5857d7e3a18d3655460377d5be021"},"cell_type":"markdown","source":"It seems that the swap did not change the distribution of the click during the day, which shows it "},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"5bc587ff6ae7f309a5dc2e39ac691618ee88e4b9"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}