{"cells":[{"metadata":{"collapsed":true,"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom sklearn.model_selection import train_test_split \nimport lightgbm as lgb\n","execution_count":1,"outputs":[]},{"metadata":{"_cell_guid":"dcda61fb-1efb-4e3e-a475-f47d3841551a","_uuid":"75485bf7cb60c3f45e6f266fabba902584759aca","trusted":true},"cell_type":"code","source":"!ls -lh ../input","execution_count":2,"outputs":[]},{"metadata":{"scrolled":true,"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"dtypes = {\n        'ip'            : 'uint32',\n        'app'           : 'uint16',\n        'device'        : 'uint16',\n        'os'            : 'uint16',\n        'channel'       : 'uint16',\n        'is_attributed' : 'uint8',\n        'click_id'      : 'uint32'\n}\ndf_train = pd.read_csv('../input/train.csv', nrows=10**7, dtype=dtypes)\ndf_test = pd.read_csv('../input/test.csv', dtype=dtypes)\ndf_train, df_val = train_test_split(df_train, train_size=.95, shuffle=False)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_cell_guid":"aadf2936-5294-4de2-92b3-768de6695c31","_uuid":"a8acf7132a1835274b572bf445661b1d4968a3f0","trusted":false},"cell_type":"code","source":"def do_feature_engineering(df):\n    df['hour'] = pd.to_datetime(df.click_time).dt.hour.astype('uint8')\n    #df['day'] = pd.to_datetime(df.click_time).dt.day.astype('uint8')\n    df.drop(['ip', 'click_time'], axis=1, inplace=True)\n    return df\n    \ndf_train = do_feature_engineering(df_train)\ndf_val = do_feature_engineering(df_val)\ndf_test = do_feature_engineering(df_test)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_cell_guid":"af8f46b6-9f4e-4782-b8ef-a4c67134d9fd","_uuid":"e06c580cb6e64a65680533abf8f0010878a9b3a6","trusted":false},"cell_type":"code","source":"target = 'is_attributed'\npredictors = ['app','device','os', 'channel', 'hour']\nxgtrain = lgb.Dataset(df_train[predictors].values, label=df_train[target].values,\n                      feature_name=predictors,\n                      categorical_feature=predictors\n)\nxgvalid = lgb.Dataset(df_val[predictors].values, label=df_val[target].values,\n                     feature_name=predictors,\n                     categorical_feature=predictors\n)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_cell_guid":"e72c5006-0ef9-4efb-af0e-3ad402890b7b","_uuid":"3f61688644f236a1dd90900697a82f4189209656","trusted":false},"cell_type":"code","source":"evals_results = {}\nlgb_params = {  # credit for https://www.kaggle.com/aharless/try-pranav-s-r-lgbm-in-python\n        'boosting_type': 'gbdt',\n        'objective': 'binary',\n        'metric': 'auc',\n        'learning_rate': 0.1,\n        'num_leaves': 7,  # we should let it be smaller than 2^(max_depth)\n        'max_depth': 4,  # -1 means no limit\n        'min_child_samples': 100,  # Minimum number of data need in a child(min_data_in_leaf)\n        'max_bin': 100,  # Number of bucketed bin for feature values\n        'subsample': 0.7,  # Subsample ratio of the training instance.\n        'subsample_freq': 1,  # frequence of subsample, <=0 means no enable\n        'colsample_bytree': 0.7,  # Subsample ratio of columns when constructing each tree.\n        'min_child_weight': 0,  # Minimum sum of instance weight(hessian) needed in a child(leaf)\n        'min_split_gain': 0,  # lambda_l1, lambda_l2 and min_gain_to_split to regularization\n        'nthread': 8,\n        'verbose': 0,\n        'scale_pos_weight':99.7, # because training data is extremely unbalanced \n}\n\nbst = lgb.train(lgb_params, \n                xgtrain, \n                valid_sets= [xgvalid], \n                valid_names=['valid'], \n                evals_result=evals_results, \n                num_boost_round=1000,\n                early_stopping_rounds=50,\n                verbose_eval=10, \n                feval=None\n)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_cell_guid":"1d23783f-c58f-463e-b536-477c7e351460","_uuid":"da9a8f56f60d6ffe465acb4b4d218f83b82444d6","trusted":false},"cell_type":"code","source":"n_estimators = bst.best_iteration\nprint(\"n_estimators: \", n_estimators)\nprint(\"best auc: \", evals_results['valid']['auc'][n_estimators-1])","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_cell_guid":"f2efa1b6-e15f-4bc9-80c5-0574898bec27","_uuid":"9f3802a50d7787867e46b0b3c3dc3fce393fa682","trusted":false},"cell_type":"code","source":"df_output = pd.DataFrame()\ndf_output['click_id'] = df_test['click_id']\ndf_output['is_attributed'] = bst.predict(df_test[predictors])\ndf_output.to_csv('output.csv', index=False, float_format='%.9f')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}