{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import lightgbm as lgbm\nfrom scipy import sparse as ssp\nfrom sklearn.model_selection import StratifiedKFold\nimport numpy as np\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-07T23:52:47.780888Z","iopub.execute_input":"2022-07-07T23:52:47.781391Z","iopub.status.idle":"2022-07-07T23:52:50.118204Z","shell.execute_reply.started":"2022-07-07T23:52:47.781293Z","shell.execute_reply":"2022-07-07T23:52:50.116973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# gini계수 함수\ndef Gini(y_true, y_pred):\n    # Check and get number of \n    assert y_true.shape == y_pred.shape\n    n_samples = y_true.shape[0]\n    \n    # sort rows on prediction column\n    # (from largest to smallest)\n    arr = np.array([y_true, y_pred]).transpose()\n    true_order = arr[arr[:, 0].argsort()][::-1, 0]\n    pred_order = arr[arr[:, 1].argsort()][::-1, 0]\n    \n    # get Lorenz curves\n    L_true = np.cumsum(true_order) * 1. / np.sum(true_order)\n    L_pred = np.cumsum(pred_order) * 1. / np.sum(pred_order)\n    L_ones = np.linspace(1 / n_samples, 1, n_samples)\n    \n    # get Gini coefficients (area between curves)\n    G_true = np.sum(L_ones - L_true)\n    G_pred = np.sum(L_ones - L_pred)\n    \n    # normalize to true Gini coefficient\n    return G_pred * 1. / G_true","metadata":{"execution":{"iopub.status.busy":"2022-07-08T01:04:24.328085Z","iopub.execute_input":"2022-07-08T01:04:24.328659Z","iopub.status.idle":"2022-07-08T01:04:24.344556Z","shell.execute_reply.started":"2022-07-08T01:04:24.328617Z","shell.execute_reply":"2022-07-08T01:04:24.343423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 디버깅할때 편하게\ncv_only = True\nsave_cv = True\nfull_train = False","metadata":{"execution":{"iopub.status.busy":"2022-07-07T23:52:50.131961Z","iopub.execute_input":"2022-07-07T23:52:50.132966Z","iopub.status.idle":"2022-07-07T23:52:50.150563Z","shell.execute_reply.started":"2022-07-07T23:52:50.132918Z","shell.execute_reply":"2022-07-07T23:52:50.149417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evalerror(preds, dtrain):\n    labels = dtrain.get_label()\n    return 'gini', Gini(labels, preds), True","metadata":{"execution":{"iopub.status.busy":"2022-07-07T23:52:50.153617Z","iopub.execute_input":"2022-07-07T23:52:50.154230Z","iopub.status.idle":"2022-07-07T23:52:50.163020Z","shell.execute_reply.started":"2022-07-07T23:52:50.154194Z","shell.execute_reply":"2022-07-07T23:52:50.162136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '../input/porto-seguro-safe-driver-prediction/'","metadata":{"execution":{"iopub.status.busy":"2022-07-07T23:52:50.165814Z","iopub.execute_input":"2022-07-07T23:52:50.166243Z","iopub.status.idle":"2022-07-07T23:52:50.173623Z","shell.execute_reply.started":"2022-07-07T23:52:50.166199Z","shell.execute_reply":"2022-07-07T23:52:50.172591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain = pd.read_csv(path+'train.csv')\ntrain_label = train['target']\ntrain_id = train['id']\n\ntest = pd.read_csv(path+'test.csv')\ntest_id = test['id']","metadata":{"execution":{"iopub.status.busy":"2022-07-07T23:52:50.175093Z","iopub.execute_input":"2022-07-07T23:52:50.176258Z","iopub.status.idle":"2022-07-07T23:53:00.663713Z","shell.execute_reply.started":"2022-07-07T23:52:50.176209Z","shell.execute_reply":"2022-07-07T23:53:00.662453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NFOLDS = 5\nkfold = StratifiedKFold(n_splits= NFOLDS, shuffle= True, random_state= 218)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T23:53:00.665180Z","iopub.execute_input":"2022-07-07T23:53:00.665625Z","iopub.status.idle":"2022-07-07T23:53:00.671831Z","shell.execute_reply.started":"2022-07-07T23:53:00.665581Z","shell.execute_reply":"2022-07-07T23:53:00.670709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train['target'].values\ndrop_feature = ['id', 'target']\n\nX = train.drop(drop_feature, axis= 1)\nfeature_name = X.columns.tolist()\ncat_features = [c for c in feature_name if ('cat' in c and 'count' not in c)]\nnum_features = [c for c in feature_name if ('cat' not in c and 'calc' not in c)]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T23:53:00.673274Z","iopub.execute_input":"2022-07-07T23:53:00.673562Z","iopub.status.idle":"2022-07-07T23:53:00.776016Z","shell.execute_reply.started":"2022-07-07T23:53:00.673536Z","shell.execute_reply":"2022-07-07T23:53:00.774947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"train['missing'] = (train== -1).sum(axis= 1).astype(float)\ntest['missing'] = (test== -1).sum(axis= 1).astype(float)\nnum_features.append('missing')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T23:53:00.777686Z","iopub.execute_input":"2022-07-07T23:53:00.778101Z","iopub.status.idle":"2022-07-07T23:53:01.021713Z","shell.execute_reply.started":"2022-07-07T23:53:00.778057Z","shell.execute_reply":"2022-07-07T23:53:01.020528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for c in cat_features:\n    le = LabelEncoder()\n    le.fit(train[c])\n    train[c] = le.transform(train[c])\n    test[c] = le.transform(test[c])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T23:53:01.024411Z","iopub.execute_input":"2022-07-07T23:53:01.024745Z","iopub.status.idle":"2022-07-07T23:53:02.216520Z","shell.execute_reply.started":"2022-07-07T23:53:01.024713Z","shell.execute_reply":"2022-07-07T23:53:02.215545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"enc = OneHotEncoder()\nenc.fit(train[cat_features])\nX_cat = enc.transform(train[cat_features])\nX_t_cat = enc.transform(test[cat_features])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T23:53:02.217873Z","iopub.execute_input":"2022-07-07T23:53:02.218303Z","iopub.status.idle":"2022-07-07T23:53:04.141145Z","shell.execute_reply.started":"2022-07-07T23:53:02.218272Z","shell.execute_reply":"2022-07-07T23:53:04.140302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ind_feature = [c for c in feature_name if 'ind' in c]\ncount= 0\nfor c in ind_feature:\n    if count == 0:\n        train['new_ind'] = train[c].astype(str)+ '_'\n        test['new_ind'] = test[c].astype(str)+ '_'\n        count+= 1\n    else:\n        train['new_ind'] += train[c].astype(str)+ '_'\n        test['new_ind'] += test[c].astype(str)+ '_'","metadata":{"execution":{"iopub.status.busy":"2022-07-07T23:53:04.142230Z","iopub.execute_input":"2022-07-07T23:53:04.142927Z","iopub.status.idle":"2022-07-07T23:53:39.201496Z","shell.execute_reply.started":"2022-07-07T23:53:04.142896Z","shell.execute_reply":"2022-07-07T23:53:39.200474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- ind feature 들을 하나로 묶어서, 새로운 카테고리를 만들어 낸거다.","metadata":{}},{"cell_type":"code","source":"# frequency encoding 과 유사함\ncat_count_features = []\nfor c in cat_features+['new_ind']:\n    d = pd.concat([train[c], test[c]]).value_counts().to_dict()\n    train['%s_count'%c] = train[c].apply(lambda x: d.get(x, 0))\n    test['%s_count'%c] = test[c].apply(lambda x: d.get(x, 0))\n    cat_count_features.append('%s_count'%c)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T23:53:39.203108Z","iopub.execute_input":"2022-07-07T23:53:39.203660Z","iopub.status.idle":"2022-07-07T23:53:54.397769Z","shell.execute_reply.started":"2022-07-07T23:53:39.203613Z","shell.execute_reply":"2022-07-07T23:53:54.396726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_list = [train[num_features + cat_count_features].values, X_cat, ]\ntest_list = [test[num_features + cat_count_features]. values, X_t_cat, ]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T23:53:54.399083Z","iopub.execute_input":"2022-07-07T23:53:54.399463Z","iopub.status.idle":"2022-07-07T23:53:55.869292Z","shell.execute_reply.started":"2022-07-07T23:53:54.399435Z","shell.execute_reply":"2022-07-07T23:53:55.868308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- sparse matrix를 풀어줘야한다","metadata":{}},{"cell_type":"code","source":"X = ssp.hstack(train_list).tocsr()\nX_test = ssp.hstack(test_list).tocsr()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T00:31:07.439351Z","iopub.execute_input":"2022-07-08T00:31:07.439802Z","iopub.status.idle":"2022-07-08T00:31:11.341370Z","shell.execute_reply.started":"2022-07-08T00:31:07.439765Z","shell.execute_reply":"2022-07-08T00:31:11.340360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model development","metadata":{}},{"cell_type":"code","source":"learning_rate = 0.1\nnum_leaves = 15\nmin_data_in_leaf = 2000\nfeature_fraction = 0.6\nnum_boost_round = 10000\nparams = {'objcetive': 'binary',\n         'boosting_type': 'gbdt',\n         'learning_rate': learning_rate,\n         'min_leaves': num_leaves,\n         'max_bin': 256,\n         'feature_fraction': feature_fraction,\n         'verbosity': 0,\n         'drop_rate': 0.1,\n         'is_unbalance': False,\n         'max_drop': 50,\n         'min_child_samples': 10,\n         'min_child_weight': 150,\n         'min_split_gain': 0,\n         'subsample': 0.9\n         }","metadata":{"execution":{"iopub.status.busy":"2022-07-08T00:36:44.163285Z","iopub.execute_input":"2022-07-08T00:36:44.163705Z","iopub.status.idle":"2022-07-08T00:36:44.170713Z","shell.execute_reply.started":"2022-07-08T00:36:44.163650Z","shell.execute_reply":"2022-07-08T00:36:44.169847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_score = []\nfinal_cv_train = np.zeros(len(train_label))\nfinal_cv_pred = np.zeros(len(test_id))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T00:39:55.738748Z","iopub.execute_input":"2022-07-08T00:39:55.739550Z","iopub.status.idle":"2022-07-08T00:39:55.745921Z","shell.execute_reply.started":"2022-07-08T00:39:55.739509Z","shell.execute_reply":"2022-07-08T00:39:55.744905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for s in np.arange(16):\n    print('#'* 30, 'random number outer iteration: {}'. format(s))\n    cv_train = np.zeros(len(train_label))\n    cv_pred = np.zeros(len(test_id))\n    \n    params['seed'] = s\n    \n    if cv_only:\n        kf = kfold.split(X, train_label)\n        \n        best_trees= []\n        fold_scores= []\n        \n        for i, (train_fold, validate) in enumerate(kf):\n            print('#'* 10, 'inner crossvalidation system: {}'.format(i))\n            X_train, X_validate, label_train, label_validate = \\\n              X[train_fold, :], X[validate, :], train_label[train_fold], train_label[validate]\n            \n            dtrain = lgbm.Dataset(X_train, label_train)\n            dvalid = lgbm.Dataset(X_validate, label_validate, reference = dtrain)\n            bst = lgbm.train(params, dtrain, num_boost_round, valid_sets= dvalid,\n                            feval= evalerror, verbose_eval= 100, early_stopping_rounds= 100)\n            \n            best_trees.append(bst.best_iteration)\n            cv_pred += bst.predict(X_test, num_iteration= bst.best_iteration)\n            cv_train[validate] += bst.predict(X_validate)\n            \n            score = Gini(label_validate, cv_train[validate])\n            print(score)\n            fold_scores.append(score)\n            \n        cv_pred /= NFOLDS\n        final_cv_train += cv_train\n        final_cv_pred += cv_pred\n        \n        print('cv score:')\n        print(Gini(train_label, cv_train))\n        print('curren score:', Gini(train_label, final_cv_train / (s + 1.)), s+1)\n        print(fold_scores)\n        print(best_trees, np.mean(best_trees))\n        \n        x_score.append(Gini(train_label, cv_train))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T01:12:13.693159Z","iopub.execute_input":"2022-07-08T01:12:13.693642Z","iopub.status.idle":"2022-07-08T01:34:46.757560Z","shell.execute_reply.started":"2022-07-08T01:12:13.693602Z","shell.execute_reply":"2022-07-08T01:34:46.756439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x_score)\npd.DataFrame({'id': test_id, 'target': final_cv_pred / 16.}).to_csv('./porto_lgbg3_pred_avg.csv', index= False)\npd.DataFrame({'id': train_id, 'target': final_cv_train / 16.}).to_csv('./porto_lgbm3_cv_avg.csv', index= False)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T01:41:12.763382Z","iopub.execute_input":"2022-07-08T01:41:12.763819Z","iopub.status.idle":"2022-07-08T01:41:18.015108Z","shell.execute_reply.started":"2022-07-08T01:41:12.763785Z","shell.execute_reply":"2022-07-08T01:41:18.013907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}