{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","collapsed":true},"cell_type":"markdown","source":"A different way to do this problem is to use cuts, train a bunch of binary classifiers and then feed them into linalg or scipy optimize.  Note this is not optimized; I just slapped it together for learning! ;)"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import gc\nimport numpy as np\nimport pandas as pd\nimport lightgbm as lgbm\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_absolute_error","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"afcd2797f99183cc301a8297f10fce8c80b6d9aa","trusted":true},"cell_type":"code","source":"X = pd.read_csv('../input/andrews-new-stuff/train_features.csv')\nX_test = pd.read_csv('../input/andrews-new-stuff/test_features.csv')\ny = pd.read_csv('../input/andrews-new-stuff/y.csv')\nsubmission = pd.read_csv('../input/andrews-new-stuff/submission.csv')\n\nX['target'] = y.target\nX_test['target'] = -1","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"35adb7e4bef4afde911a7daa059da3896cd91fb3","trusted":true},"cell_type":"code","source":"x = pd.qcut(X.target, 7, labels=[0, 1, 2, 3, 4, 5, 6])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e1071b263d41998ff0342fd6a0b11e7f0d792a82","trusted":true},"cell_type":"code","source":"for a in [0, 1, 2, 3, 4, 5]:\n    print(a,X.target[x==a].min(),X.target[x==a].max())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e929d432bec93a6c1b538f439c520e933210fcdd","trusted":true},"cell_type":"code","source":"lgbm_params =  {\n    'task': 'train',\n    'boosting_type': 'gbdt',\n    'objective': 'binary',\n    'metric': 'binary_logloss',\n    \"learning_rate\": 0.01,\n    \"num_leaves\": 100,\n    \"feature_fraction\": .5,\n    \"bagging_fraction\": .5,\n    #'bagging_freq': 4,\n    \"max_depth\": -1,\n    \"reg_alpha\": 0.3,\n    \"reg_lambda\": 0.1,\n    \"min_child_weight\":10,\n    \"n_jobs\":4\n}","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6dcee1964b0dcfacca2f95801d41cd0ce0322174","trusted":true},"cell_type":"code","source":"feats = X.columns[:-1]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4ee52ad4a2439449f451f36618f73133cb0a0e11"},"cell_type":"markdown","source":"Use the block below - kernels are slow so I am giving you the \"best\" params"},{"metadata":{"_uuid":"b8d97691aa828c247ee31b866b454dc3152c1543","trusted":false},"cell_type":"code","source":"# bestparams = {}\n# for i in [0, 1, 2, 3, 4, 5]:\n#     lgtrain = lgbm.Dataset(X[feats],x>i)\n#     lgb_cv = lgbm.cv(\n#         params = lgbm_params,\n#         train_set = lgtrain,\n#         num_boost_round=2000,\n#         stratified=False,\n#         nfold = 5,\n#         verbose_eval=0,\n#         seed = 42,\n#         early_stopping_rounds=75)\n\n#     optimal_rounds = np.argmin(lgb_cv['binary_logloss-mean'])\n#     best_cv_score = min(lgb_cv['binary_logloss-mean'])\n#     bestparams[i] = (optimal_rounds,best_cv_score)\n#     del lgtrain\n#     gc.collect()\n# bestparams","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"adf7b49736393d41aec2677976cc46c32f7f23f1","trusted":true},"cell_type":"code","source":"bestparams = {0: (343, 0.33300775159479323),\n             1: (377, 0.3937832899637623),\n             2: (381, 0.43248276195861485),\n             3: (368, 0.4567765830007994),\n             4: (279, 0.4364333328398183),\n             5: (301, 0.3130615907253661)}","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b76d5a0fa60cb6d38103bade7e3b94902d4dbd3b","trusted":true},"cell_type":"code","source":"folds = KFold(n_splits=5, shuffle=True, random_state=42)\noof_preds = np.zeros((X.shape[0],6))\nsub_preds = np.zeros((X_test.shape[0],6))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9d55f4c85c12a5149435eab689efb0db0432f2c8","trusted":true},"cell_type":"code","source":"for i in [0, 1, 2, 3, 4, 5]:\n    optimal_rounds, best_cv_score = bestparams[i]\n    print(i, optimal_rounds, best_cv_score)\n    for n_fold, (trn_idx, val_idx) in enumerate(folds.split(X)):\n        print(n_fold)\n        trn_x, trn_y = X[feats].iloc[trn_idx], x[trn_idx]>i\n        val_x, val_y = X[feats].iloc[val_idx], x[val_idx]>i\n        \n        clf = lgbm.train(lgbm_params,\n                         lgbm.Dataset(trn_x,trn_y),\n                         num_boost_round = optimal_rounds + 1,\n                         verbose_eval=200)\n\n        oof_preds[val_idx,i] = clf.predict(val_x, num_iteration=optimal_rounds + 1)\n        sub_preds[:,i] += clf.predict(X_test[feats], num_iteration=optimal_rounds + 1) / folds.n_splits\n\n        del clf\n        del trn_x, trn_y, val_x, val_y\n        gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"afd0238960c0c7489c9f7364797ef160beb43abd","trusted":true},"cell_type":"code","source":"off_preds_withbias = np.hstack([oof_preds,np.ones(shape=(oof_preds.shape[0],1))])\nsub_preds_withbias = np.hstack([sub_preds,np.ones(shape=(sub_preds.shape[0],1))])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a3246d4415e84427a81c1c39af900cd354292dcf","trusted":true},"cell_type":"code","source":"params = np.linalg.lstsq(off_preds_withbias, y.target,rcond=-1)[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"params","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"412d0e18d464779fbd74728364302d4ce1c8ad41","trusted":true},"cell_type":"code","source":"trainpreds = np.dot(off_preds_withbias,params)\nprint(mean_absolute_error(y.target,trainpreds))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"36ec120521837962230f4f0fc34d55f902ebf6a9","trusted":true},"cell_type":"code","source":"testpreds = np.dot(sub_preds_withbias,params)\nsub = pd.DataFrame({'seg_id':submission.seg_id, 'time_to_failure':testpreds})\nsub.to_csv('cut.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.time_to_failure.min()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.time_to_failure.max()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.7.1"}},"nbformat":4,"nbformat_minor":1}