{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","collapsed":true,"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import gc\nimport numpy as np\nimport pandas as pd\nfrom sklearn.metrics import roc_auc_score, mean_absolute_error","execution_count":1,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","collapsed":true,"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"def add_noise(series, noise_level):\n    return series * (1 + noise_level * np.random.randn(len(series)))\n\ndef target_encode(trn_series=None, \n                  tst_series=None, \n                  target=None, \n                  min_samples_leaf=1, \n                  smoothing=1,\n                  noise_level=0):\n    assert len(trn_series) == len(target)\n    assert trn_series.name == tst_series.name\n    temp = pd.concat([trn_series, target], axis=1)\n    # Compute target mean \n    averages = temp.groupby(by=trn_series.name)[target.name].agg([\"mean\", \"count\"])\n    # Compute smoothing\n    smoothing = 1 / (1 + np.exp(-(averages[\"count\"] - min_samples_leaf) / smoothing))\n    # Apply average function to all target data\n    prior = target.mean()\n    # The bigger the count the less full_avg is taken into account\n    averages[target.name] = prior * (1 - smoothing) + averages[\"mean\"] * smoothing\n    averages.drop([\"mean\", \"count\"], axis=1, inplace=True)\n    # Apply averages to trn and tst series\n    ft_trn_series = pd.merge(\n        trn_series.to_frame(trn_series.name),\n        averages.reset_index().rename(columns={'index': target.name, target.name: 'average'}),\n        on=trn_series.name,\n        how='left')['average'].rename(trn_series.name + '_mean').fillna(prior)\n    # pd.merge does not keep the index so restore it\n    ft_trn_series.index = trn_series.index \n    ft_tst_series = pd.merge(\n        tst_series.to_frame(tst_series.name),\n        averages.reset_index().rename(columns={'index': target.name, target.name: 'average'}),\n        on=tst_series.name,\n        how='left')['average'].rename(trn_series.name + '_mean').fillna(prior)\n    # pd.merge does not keep the index so restore it\n    ft_tst_series.index = tst_series.index\n    return add_noise(ft_trn_series, noise_level), add_noise(ft_tst_series, noise_level)","execution_count":2,"outputs":[]},{"metadata":{"_uuid":"342f60015559249616d2e790b31acdad87052ea5","collapsed":true,"_cell_guid":"93470ab8-f9de-4df3-a8f6-1cbb0aac69ea","trusted":true},"cell_type":"code","source":"dtypes = {\n        'ip'            : 'uint32',\n        'app'           : 'uint16',\n        'device'        : 'uint16',\n        'os'            : 'uint16',\n        'channel'       : 'uint16',\n        'is_attributed' : 'uint8'}\ntrain_cols = ['ip', 'app', 'device', 'os', 'channel', 'click_time', 'is_attributed']\ntest_cols = ['click_id','ip', 'app', 'device', 'os', 'channel', 'click_time']","execution_count":3,"outputs":[]},{"metadata":{"_uuid":"327752e85a21583d87fdc087654161e2edb2d27e","collapsed":true,"_cell_guid":"b0f68277-1bc9-4cf4-8635-a482f640249e","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv',skiprows=range(1,131886954),dtype=dtypes, usecols=train_cols) # Just use day \ntest = pd.read_csv('../input/test.csv',dtype=dtypes, usecols=test_cols)","execution_count":4,"outputs":[]},{"metadata":{"_uuid":"bcacd54294818b39a93ce90cd4cbf211a36f7e25","collapsed":true,"_cell_guid":"cad064a4-3359-48cc-a9e3-f9d83e5db395","trusted":true},"cell_type":"code","source":"train['click_time'] = pd.to_datetime(train.click_time)\ntrain['hour'] = train.click_time.dt.hour.astype('uint8')\ntest['click_time'] = pd.to_datetime(test.click_time)\ntest['hour'] = test.click_time.dt.hour.astype('uint8')\n","execution_count":5,"outputs":[]},{"metadata":{"_uuid":"4514c740b567264b28ad6ad44b7098a33cd9bf9d","_cell_guid":"b8fc2f76-9866-4b5f-999d-b20669edf358","trusted":true},"cell_type":"code","source":"train['dummy'] = (train.app.astype(str)+'_' +train.channel.astype(str)+'_'+train.hour.astype(str)).apply(hash) % 2**26\ntest['dummy'] = (test.app.astype(str)+'_'+ test.channel.astype(str)+'_'+test.hour.astype(str)).apply(hash) % 2**26\ntrain['app_channel'], test['app_channel'] = target_encode(train['dummy'], \n                                                          test['dummy'], \n                                                          target=train.is_attributed, \n                                                          min_samples_leaf=100,\n                                                          smoothing=0,\n                                                          noise_level=0.0)\ntest.drop('dummy',inplace=True,axis=1)\ngc.collect()\ntrain.drop('dummy',inplace=True,axis=1)\ngc.collect()","execution_count":8,"outputs":[]},{"metadata":{"_uuid":"e4657b36ca02effbc4c37c4d6d959150b82cfe07","_cell_guid":"172fcb11-9431-4618-b9de-08ae740f6310","trusted":true},"cell_type":"code","source":"train['dummy'] = (train.ip.astype(str)+'_'+train.hour.astype(str)).apply(hash) % 2**26\ntest['dummy'] = (test.ip.astype(str)+'_'+test.hour.astype(str)).apply(hash) % 2**26\ntrain['ip'], test['ip'] = target_encode(train['dummy'], \n                                       test['dummy'], \n                                       target=train.is_attributed, \n                                       min_samples_leaf=100,\n                                       smoothing=0,\n                                       noise_level=0.0)\ntest.drop('dummy',inplace=True,axis=1)\ngc.collect()\ntrain.drop('dummy',inplace=True,axis=1)\ngc.collect()","execution_count":9,"outputs":[]},{"metadata":{"_uuid":"0fc443fd8ba3a3adbd3d8054ab9d5c6872d70c05","_cell_guid":"43b74dc3-cf49-4834-8dcc-57b9ccf84d36","trusted":true},"cell_type":"code","source":"train['dummy'] = (train.app.astype(str)+'_'+train.hour.astype(str)).apply(hash) % 2**26\ntest['dummy'] = (test.app.astype(str)+'_'+test.hour.astype(str)).apply(hash) % 2**26\ntrain['app'], test['app'] = target_encode(train['dummy'], \n                                       test['dummy'], \n                                       target=train.is_attributed, \n                                       min_samples_leaf=100,\n                                       smoothing=0,\n                                       noise_level=0.0)\ntest.drop('dummy',inplace=True,axis=1)\ngc.collect()\ntrain.drop('dummy',inplace=True,axis=1)\ngc.collect()","execution_count":10,"outputs":[]},{"metadata":{"_uuid":"d4bffc5ba3fa0b7d2c6915dc36685083c66d4314","_cell_guid":"216b906a-0af7-4b17-84dc-d780b3570858","trusted":true},"cell_type":"code","source":"train['dummy'] = (train.device.astype(str)+'_'+train.hour.astype(str)).apply(hash) % 2**26\ntest['dummy'] = (test.device.astype(str)+'_'+test.hour.astype(str)).apply(hash) % 2**26\ntrain['device'], test['device'] = target_encode(train['dummy'], \n                                               test['dummy'], \n                                               target=train.is_attributed, \n                                               min_samples_leaf=100,\n                                               smoothing=0,\n                                               noise_level=0.0)\ntest.drop('dummy',inplace=True,axis=1)\ngc.collect()\ntrain.drop('dummy',inplace=True,axis=1)\ngc.collect()","execution_count":11,"outputs":[]},{"metadata":{"_uuid":"9b4954fad4fbc727ff09ffaac4d31db7e2969191","_cell_guid":"a965fb14-d741-4e1e-a1b4-006ff70050e9","trusted":true},"cell_type":"code","source":"train['dummy'] = (train.os.astype(str)+'_'+train.hour.astype(str)).apply(hash) % 2**26\ntest['dummy'] = (test.os.astype(str)+'_'+test.hour.astype(str)).apply(hash) % 2**26\ntrain['os'], test['os'] = target_encode(train['dummy'], \n                                       test['dummy'], \n                                       target=train.is_attributed, \n                                       min_samples_leaf=100,\n                                       smoothing=0,\n                                       noise_level=0.0)\ntest.drop('dummy',inplace=True,axis=1)\ngc.collect()\ntrain.drop('dummy',inplace=True,axis=1)\ngc.collect()","execution_count":12,"outputs":[]},{"metadata":{"_uuid":"4a4821484d4e5b4f73ae04da54e6b2a85b23cc3e","_cell_guid":"e58c79fc-65a2-4e00-9b9f-89607336879e","trusted":true},"cell_type":"code","source":"train['dummy'] = (train.channel.astype(str)+'_'+train.hour.astype(str)).apply(hash) % 2**26\ntest['dummy'] = (test.channel.astype(str)+'_'+test.hour.astype(str)).apply(hash) % 2**26\ntrain['channel'], test['channel'] = target_encode(train['dummy'], \n                                                   test['dummy'], \n                                                   target=train.is_attributed, \n                                                   min_samples_leaf=100,\n                                                   smoothing=0,\n                                                   noise_level=0.0)\ntest.drop('dummy',inplace=True,axis=1)\ngc.collect()\ntrain.drop('dummy',inplace=True,axis=1)\ngc.collect()","execution_count":13,"outputs":[]},{"metadata":{"_uuid":"625cc5d17af25c9c39870871319dcf74acfb75de","_cell_guid":"37a0c565-016e-4c27-9e3b-4eaa65cff751","trusted":true},"cell_type":"code","source":"train['dummy'] = (train.hour.astype(str)).apply(hash) % 2**26\ntest['dummy'] = (test.hour.astype(str)).apply(hash) % 2**26\ntrain['hour'], test['hour'] = target_encode(train['dummy'], \n                                           test['dummy'], \n                                           target=train.is_attributed, \n                                           min_samples_leaf=100,\n                                           smoothing=0,\n                                           noise_level=0.0)\ntest.drop('dummy',inplace=True,axis=1)\ngc.collect()\ntrain.drop('dummy',inplace=True,axis=1)\ngc.collect()","execution_count":14,"outputs":[]},{"metadata":{"_uuid":"3b4418eb79ac930620a1daa6d9dcb6c4f23145b0","collapsed":true,"_cell_guid":"6537191b-c5d6-48a6-895b-5e25744bc6d3","trusted":true},"cell_type":"code","source":"def Output(p):\n    return 1./(1.+np.exp(-p))\n\ndef GP(data):\n    return Output(np.tanh((((-1.0) + (((np.where(data[\"os\"]>0, (((((data[\"app_channel\"]) > (np.tanh((data[\"app_channel\"]))))*1.)) * 2.0), -1.0 )) * 2.0)))/2.0)) +\n                  np.tanh(((np.where(data[\"app\"]>0, np.where(data[\"os\"]>0, (((((data[\"channel\"]) * 2.0)) > (data[\"hour\"]))*1.), -2.0 ), -2.0 )) * 2.0)) +\n                  np.tanh((((((((((((data[\"channel\"]) + (((((data[\"app\"]) * 2.0)) * 2.0)))/2.0)) * 2.0)) * 2.0)) * 2.0)) * 2.0)))","execution_count":15,"outputs":[]},{"metadata":{"_uuid":"e14acaa2275a0d8b5ef0f3bce0882bfc0b4ada98","_cell_guid":"7fe57c1c-0e01-42a2-89bf-3c083021634a","trusted":true},"cell_type":"code","source":"roc_auc_score(train.is_attributed,GP(train))","execution_count":16,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"25f636ecf0f3260a27088a5f4479be5617bcdb8c"},"cell_type":"code","source":"del train\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1d041e1d1886cd800d5a9bce7a2df01242174342","collapsed":true,"_cell_guid":"fb9db83b-51c5-44b8-ae25-00bdcbdd87a7","trusted":true},"cell_type":"code","source":"sub = pd.DataFrame()\nsub['click_id'] = test.click_id.values\nsub['is_attributed'] = GP(test).values\nsub.to_csv('xxx.csv.gz',compression='gzip',index=False)","execution_count":17,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}