{"cells":[{"metadata":{"_uuid":"2fe66343a5cc066eb6e7bfa28bde1cf171d019f6"},"cell_type":"markdown","source":"# LGB Parameter_SimpleVersion <section id=\"section_top\" />\n\n- [Loading Libraries](#section_LL)\n- [Defining Loss](#section_DL)\n- [Extracting Useful Features](#section_EUF)\n- [Parameter_tuning](#section_pt)\n- [Examination](#section_ex)\n- [Convergence Plot](#section_CPlot)\n- [Training LGB Classifier with tuned Parameters](#section_train)\n- [Making Predictions](#section_pred)\n\n---------------------\n"},{"metadata":{"_uuid":"79700cc0f69c9f6ea59e1466a679af30e7d3a677"},"cell_type":"markdown","source":"This kernel was created with reference to the following.  \n\nref.    \nhttps://www.kaggle.com/meaninglesslives/lgb-parameter-tuning  \nhttps://www.kaggle.com/ogrellier/plasticc-in-a-kernel-meta-and-data  \nhttps://www.kaggle.com/ashishpatel26/can-this-make-sense-of-the-universe-tuned  \n\n- In this kernel, passband and object_id are used as the grouping key.  \n- aggs: Deleted except flux, flux_err   \n- etc...\n"},{"metadata":{"_uuid":"bce67b503f67cbce3b3c79200c700034684440d9"},"cell_type":"markdown","source":"# Loading Libraries  <section id=\"section_LL\" />\n\n[return](#section_top)"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.ensemble import ExtraTreesClassifier\nfrom sklearn.metrics import log_loss\nfrom sklearn.model_selection import StratifiedKFold\nimport gc\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns \nimport lightgbm as lgb\n\n%matplotlib inline\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom skopt import gp_minimize, forest_minimize\nfrom skopt.space import Real, Categorical, Integer\nfrom skopt.plots import plot_convergence\nfrom skopt.plots import plot_objective, plot_evaluations\nfrom skopt.utils import use_named_args\nfrom sklearn.preprocessing import MinMaxScaler, StandardScaler\nimport time\nnotebookstart= time.time()\npd.set_option(\"display.max_rows\", 101)\nisDataCheck=False","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5e8d9c3d8180cc175b0f5f4fb203d15d8f0f1baf"},"cell_type":"markdown","source":"# Defining Loss <section id=\"section_DL\" />\n\n[return](#section_top)"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"def lgb_multi_weighted_logloss(y_true, y_preds):#use by eval_metric\n    \"\"\"\n    @author olivier https://www.kaggle.com/ogrellier\n    multi logloss for PLAsTiCC challenge\n    \"\"\"\n    classes = [6, 15, 16, 42, 52, 53, 62, 64, 65, 67, 88, 90, 92, 95]\n    class_weight = {6: 1, 15: 2, 16: 1, 42: 1, 52: 1, 53: 1, 62: 1, 64: 2, 65: 1, 67: 1, 88: 1, 90: 1, 92: 1, 95: 1}\n    if len(np.unique(y_true)) > 14:\n        classes.append(99)\n        class_weight[99] = 2\n    y_p = y_preds.reshape(y_true.shape[0], len(classes), order='F')\n    \n    # Trasform y_true in dummies\n    y_ohe = pd.get_dummies(y_true)\n    # Normalize rows and limit y_preds to 1e-15, 1-1e-15\n    y_p = np.clip(a=y_p, a_min=1e-15, a_max=1-1e-15)\n    # Transform to log\n    y_p_log = np.log(y_p)\n    # Get the log for ones, .values is used to drop the index of DataFrames\n    # Exclude class 99 for now, since there is no class99 in the training set \n    # we gave a special process for that class\n    y_log_ones = np.sum(y_ohe.values * y_p_log, axis=0)\n    # Get the number of positives for each class\n    nb_pos = y_ohe.sum(axis=0).values.astype(float)\n    # Weight average and divide by the number of positives\n    class_arr = np.array([class_weight[k] for k in sorted(class_weight.keys())])\n    y_w = y_log_ones * class_arr / nb_pos\n    \n    loss = - np.sum(y_w) / np.sum(class_arr)\n    return 'wloss', loss, False\n\ndef multi_weighted_logloss(y_true, y_preds):\n    \"\"\"\n    @author olivier https://www.kaggle.com/ogrellier\n    multi logloss for PLAsTiCC challenge\n    \"\"\"\n    classes = [6, 15, 16, 42, 52, 53, 62, 64, 65, 67, 88, 90, 92, 95]\n    class_weight = {6: 1, 15: 2, 16: 1, 42: 1, 52: 1, 53: 1, 62: 1, 64: 2, 65: 1, 67: 1, 88: 1, 90: 1, 92: 1, 95: 1}\n    if len(np.unique(y_true)) > 14:\n        classes.append(99)\n        class_weight[99] = 2\n    y_p = y_preds\n    # Trasform y_true in dummies\n    y_ohe = pd.get_dummies(y_true)\n    # Normalize rows and limit y_preds to 1e-15, 1-1e-15\n    y_p = np.clip(a=y_p, a_min=1e-15, a_max=1-1e-15)\n    # Transform to log\n    y_p_log = np.log(y_p)\n    # Get the log for ones, .values is used to drop the index of DataFrames\n    # Exclude class 99 for now, since there is no class99 in the training set \n    # we gave a special process for that class\n    y_log_ones = np.sum(y_ohe.values * y_p_log, axis=0)\n    # Get the number of positives for each class\n    nb_pos = y_ohe.sum(axis=0).values.astype(float)\n    # Weight average and divide by the number of positives\n    class_arr = np.array([class_weight[k] for k in sorted(class_weight.keys())])\n    y_w = y_log_ones * class_arr / nb_pos\n    \n    loss = - np.sum(y_w) / np.sum(class_arr)\n    return loss","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6b10269d3c0f0024596ce80fec38c63ccdd4dd6a"},"cell_type":"markdown","source":"# Extracting Useful Features <section id=\"section_EUF\" />\n\n[return](#section_top)"},{"metadata":{"trusted":true,"_uuid":"f96c012354e695d1b37b2b557591a9c87bde84f2"},"cell_type":"code","source":"#grouping(object_id,passband)\naggs = {\n    'flux': ['min', 'max', 'mean', 'median', 'std'],\n    'flux_err': ['median', 'std'],\n} \n\ngrp_col=['object_id','passband']#'object_id'    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"631c418da4275fba3070f2a6cff3873d11591f95","scrolled":true},"cell_type":"code","source":"%%time\ngc.enable()\n\n# train = pd.read_csv('../input/training_set.csv')\ntrain = pd.read_csv('../input/training_set.csv',\n                   dtype = {\n                       'object_id':np.int32,\n                       'mjd':np.float64,\n                       'passband':np.int8,\n                       'flux':np.float32,\n                       'flux_err':np.float32,\n                       'detected':np.int32})\n\n# agg_train = train.groupby('object_id').agg(aggs)\nagg_train = train.groupby(grp_col).agg(aggs)\nnew_columns = [k + '_' + agg for k in aggs.keys() for agg in aggs[k]]\nagg_train.columns = new_columns\nagg_train=pd.pivot_table(agg_train, index='object_id', columns='passband')\ndel train\n\ndisplay(agg_train.head(10))\nprint(\"gc.collect:\",gc.collect())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1c6de093646141210e58362f9ea3b97f29fb170a"},"cell_type":"code","source":"%%time\n\n# meta_train = pd.read_csv('../input/training_set_metadata.csv')\nmeta_train = pd.read_csv('../input/training_set_metadata.csv',\n                         dtype = {\n                             'object_id':np.int32,\n                             'ra':np.float32,\n                             'decl':np.float32,                 \n                             'gal_l':np.float32,           \n                             'gal_b':np.float32,           \n                             'ddf':np.int8,#bool\n                             'hostgal_specz':np.float32,         \n                             'hostgal_photoz':np.float32,        \n                             'hostgal_photoz_err':np.float32,    \n                             'distmod':np.float32,          \n                             'mwebv':np.float32,            \n                             'target':np.int8})\n\ndisplay(meta_train.head())\n\nfull_train = agg_train.reset_index().merge(\n    right=meta_train,\n    how='outer',\n    on='object_id'\n)\nfull_train=full_train.drop( columns=[('object_id', '')])\n# print(full_train.columns)\n\nif 'target' in full_train:\n    y = full_train['target']\n    del full_train['target']\nclasses = sorted(y.unique())\n\n# Taken from Giba's topic : https://www.kaggle.com/titericz\n# https://www.kaggle.com/c/PLAsTiCC-2018/discussion/67194\n# with Kyle Boone's post https://www.kaggle.com/kyleboone\nclass_weight = {\n    c: 1 for c in classes\n}\nfor c in [64, 15]:\n    class_weight[c] = 2\n\nprint('Unique classes : ', classes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6e62d509f8ec429b854961dd8691bb389e56e7c1","scrolled":true},"cell_type":"code","source":"del agg_train\nprint(\"gc.collect:\",gc.collect())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2b6f79b71d0e9b62266a52d9e7ac209c56e14aff"},"cell_type":"code","source":"%%time\n\nisfillNaN=False#True\n\nif 'object_id' in full_train:\n    oof_df = full_train[['object_id']]\n    del full_train['object_id'], full_train['hostgal_specz'],full_train['ddf']\n    \nif isfillNaN:    \n    train_mean = full_train.mean(axis=0)\n    full_train.fillna(train_mean, inplace=True)\n\nfolds = StratifiedKFold(n_splits=5, shuffle=True, random_state=1)\nclfs = []\nimportances = pd.DataFrame()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7b06db11cf25b64b14a0065f75d964560ccb1103"},"cell_type":"markdown","source":" "},{"metadata":{"_uuid":"f86a8f970532b50956807173795c04691db110c9"},"cell_type":"markdown","source":"# Parameter Tuning <section id=\"section_pt\" />\n\n[return](#section_top)"},{"metadata":{"trusted":true,"_uuid":"b2e76e21e2b4a5c86b6f624098ab43f6dd99131d"},"cell_type":"code","source":"%%time\ndim_learning_rate = Real(low=1e-6, high=1e-1, prior='log-uniform',name='learning_rate')\ndim_estimators = Integer(low=800, high=2000,name='n_estimators')\ndim_max_depth = Integer(low=3, high=6,name='max_depth')\n\ndimensions = [dim_learning_rate,\n              dim_estimators,\n              dim_max_depth]\n\ndefault_parameters = [0.03,1000,3]\n\nlgb_params = {\n    'boosting_type': 'gbdt',\n    'objective': 'multiclass',\n    'num_class': 14,\n    'metric': 'multi_logloss',\n    'subsample': .9,\n    'colsample_bytree': .7,\n    'reg_alpha': .01,#L1\n    'reg_lambda': .02,#01,#L2\n#     'num_leaves': 31,#63,# Add 2^(max_depth) > num_leaves warning\n    'min_split_gain': 0.01,\n    'min_child_weight': 10,\n    'silent':True,\n    'verbosity':-1,\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ac29bffcab6c94715687f2effe16b0f2ad575aae"},"cell_type":"code","source":"%%time\ndef createModel(learning_rate,n_estimators,max_depth):       \n\n    oof_preds = np.zeros((len(full_train), len(classes)))\n    for fold_, (trn_, val_) in enumerate(folds.split(y, y)):\n        trn_x, trn_y = full_train.iloc[trn_], y.iloc[trn_]\n        val_x, val_y = full_train.iloc[val_], y.iloc[val_]\n\n        clf = lgb.LGBMClassifier(**lgb_params,learning_rate=learning_rate,\n                                n_estimators=n_estimators,max_depth=max_depth)\n        clf.fit(\n            trn_x, trn_y,\n            eval_set=[(trn_x, trn_y), (val_x, val_y)],\n            eval_metric=lgb_multi_weighted_logloss,\n            verbose=False,#True,\n            early_stopping_rounds=50\n        )\n        oof_preds[val_, :] = clf.predict_proba(val_x, num_iteration=clf.best_iteration_)\n        print('fold',fold_+1,multi_weighted_logloss(val_y, clf.predict_proba(val_x, num_iteration=clf.best_iteration_)))\n\n        clfs.append(clf)\n    \n    loss = multi_weighted_logloss(y_true=y, y_preds=oof_preds)\n    print('MULTI WEIGHTED LOG LOSS : %.5f ' % loss)\n    \n    return loss","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1e693c12f9a69b7bd23f38598979012aa257fb31"},"cell_type":"code","source":"%%time\n@use_named_args(dimensions=dimensions)\ndef fitness(learning_rate,n_estimators,max_depth):\n    \"\"\"\n    Hyper-parameters:\n    learning_rate:     Learning-rate for the optimizer.\n    n_estimators:      Number of estimators.\n    max_depth:         Maximum Depth of tree.\n    \"\"\"\n\n    # Print the hyper-parameters.\n    print('learning rate: {0:.2e}'.format(learning_rate))\n    print('estimators:', n_estimators)\n    print('max depth:', max_depth)\n    \n    lv= createModel(learning_rate=learning_rate,\n                    n_estimators=n_estimators,\n                    max_depth = max_depth)\n    return lv","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a35941ed8d697977e2dd39e63a14f8e65dc30033"},"cell_type":"markdown","source":"## Examination\n<section id=\"section_ex\" />\n\n[return](#section_top)\n    "},{"metadata":{"trusted":true,"_uuid":"c17b738185e7d049111d9352d68101dceef34a80","scrolled":true},"cell_type":"code","source":"%%time\nfolds = StratifiedKFold(n_splits=5, shuffle=True, random_state=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ff28c53b564ac1092c60f2dd4fe1f9f45607a8e1"},"cell_type":"markdown","source":"----------------------------------------"},{"metadata":{"trusted":true,"_uuid":"b0cb3ae266b3dde83fc282581de0628c60fae36a","scrolled":true},"cell_type":"code","source":"%%time\n\nisSearchForHyperparameters=False\n\nif isSearchForHyperparameters:\n    search_result = gp_minimize(func=fitness,\n                                dimensions=dimensions,\n                                acq_func='EI', \n                                n_calls=20,\n                                x0=default_parameters,n_jobs=-1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8305dd7744e94ef2dedb08484dd278c9f21b39eb"},"cell_type":"markdown","source":"# Convergence Plot\n<section id=\"section_CPlot\" />\n\n[return](#section_top)"},{"metadata":{"trusted":true,"_uuid":"d91f31721a5a4674c093c701d9add7d922727302"},"cell_type":"code","source":"if isSearchForHyperparameters:\n    plot_convergence(search_result)\n    plt.show()\n\n# optimal parameters found using scikit optimize. use these parameter to initialize the 2nd level model.\nif isSearchForHyperparameters:\n    print(search_result.x)\n    learning_rate = search_result.x[0]\n    n_estimators = search_result.x[1]\n    max_depth = search_result.x[2]\nelse:\n    learning_rate = default_parameters[0]\n    n_estimators = default_parameters[1]\n    max_depth = default_parameters[2] \nprint(\"learning_rate:\",learning_rate)\nprint(\"n_estimators:\",n_estimators)\nprint(\"max_depth:\",max_depth)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"756c85719a9d944ee26e8722ef126c76aa236efa"},"cell_type":"code","source":"if isSearchForHyperparameters:\n    del search_result,plot_convergence","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d413e67b3a8735ceeec25b2c3fc6b911d8646464"},"cell_type":"markdown","source":"# Training LGB Classifier with tuned Parameters <section id=\"section_train\" />\n\n[return](#section_top)"},{"metadata":{"trusted":true,"_uuid":"4e5e1ed34e0887b3c10b8e649abef15f44a34853"},"cell_type":"code","source":"%%time\n\nfolds = StratifiedKFold(n_splits=5, \n                        shuffle=True, random_state=1)\nclfs = []\nimportances = pd.DataFrame()\n\noof_preds = np.zeros((len(full_train), len(classes)))\nfor fold_, (trn_, val_) in enumerate(folds.split(y, y)):\n    trn_x, trn_y = full_train.iloc[trn_], y.iloc[trn_]\n    val_x, val_y = full_train.iloc[val_], y.iloc[val_]\n    \n    clf = lgb.LGBMClassifier(\n        **lgb_params,\n        learning_rate=learning_rate,\n        n_estimators=n_estimators,max_depth=max_depth)\n    clf.fit(\n        trn_x, trn_y,\n        eval_set=[(trn_x, trn_y), (val_x, val_y)],\n        eval_metric=lgb_multi_weighted_logloss,\n        verbose=100,\n        early_stopping_rounds=50)\n    oof_preds[val_, :] = clf.predict_proba(val_x, num_iteration=clf.best_iteration_)\n    print(multi_weighted_logloss(val_y, clf.predict_proba(val_x, num_iteration=clf.best_iteration_)))\n    \n    imp_df = pd.DataFrame()\n    imp_df['feature'] = full_train.columns\n    imp_df['gain'] = clf.feature_importances_\n    imp_df['fold'] = fold_ + 1\n    importances = pd.concat([importances, imp_df], axis=0, sort=False)\n    \n    clfs.append(clf)\n\nprint('MULTI WEIGHTED LOG LOSS : %.5f ' % multi_weighted_logloss(y_true=y, y_preds=oof_preds))\n\n\nmean_gain = importances[['gain', 'feature']].groupby('feature').mean()\nimportances['mean_gain'] = importances['feature'].map(mean_gain['gain'])\n# plt.figure(figsize=(8, 12))\n# sns.barplot(x='gain', y='feature', data=importances.sort_values('mean_gain', ascending=False))\n# plt.tight_layout()\n# plt.savefig('importances.png')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"508ad431b59895821cbe77fc1f4e75f1f37eaed6"},"cell_type":"code","source":"importances.loc[:,['feature','mean_gain']].groupby(\n    'feature').mean().sort_values('mean_gain',ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c1aa714713cd4706c81677abd01a98a3a626b83c"},"cell_type":"code","source":"# lgb.plot_tree(clf,figsize=(18,10))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5ebf8951ad55f25e24e2d216c36091f04d6ca9b7"},"cell_type":"code","source":"importances.loc[importances.fold==1].sort_values('gain',ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bcbdd561056bd849732daee33491ed842fe4336d"},"cell_type":"code","source":"importances.loc[importances.fold==2].sort_values('gain',ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"247859f670473140e4f1f53797b0077bdaa1b310"},"cell_type":"code","source":"importances.loc[importances.fold==3].sort_values('gain',ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"789dfd0e1a43f285ce1c678a18fbfdc41e18d84b"},"cell_type":"code","source":"importances.loc[importances.fold==4].sort_values('gain',ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"129f47533aa10117f3a7daaef09c8fbd12d3b29c"},"cell_type":"code","source":"importances.loc[importances.fold==5].sort_values('gain',ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0ffd5e0cf7c4b64c348becb860d6839ba31ae848"},"cell_type":"code","source":"importances.loc[:,['feature','mean_gain']].groupby('feature').mean().sort_values('mean_gain',ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a023348a6124f4bacb3368e4bb4623328a7af253"},"cell_type":"code","source":"del oof_preds,importances,mean_gain\nprint(\"gc.collect:\",gc.collect())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0caa1fc29c48dc0b35967cd9cad1eb3a306b7864"},"cell_type":"markdown","source":"# Making Predictions <section id=\"section_pred\" />\n\n[return](#section_top)"},{"metadata":{"trusted":true,"_uuid":"4dbf05b2e41acb69616bffcc9bee08f270cb1b8c"},"cell_type":"code","source":"%%time\nprint(\"read test_set_metadata.csv\")\n# meta_test = pd.read_csv('../input/test_set_metadata.csv')\nmeta_test = pd.read_csv('../input/test_set_metadata.csv',\n                        dtype = {'object_id':np.int32,\n                                 'ra':np.float32,\n                                 'decl':np.float32,                 \n                                 'gal_l':np.float32,           \n                                 'gal_b':np.float32,           \n                                 'ddf':np.int8,#bool\n                                 'hostgal_specz':np.float32,         \n                                 'hostgal_photoz':np.float32,        \n                                 'hostgal_photoz_err':np.float32,    \n                                 'distmod':np.float32,          \n                                 'mwebv':np.float32, } )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1744237f1696ca45a697fb16b02515bdd49340dd","scrolled":true},"cell_type":"code","source":"%%time\n\n# isDebug=False\n\nimport time\n\nstart = time.time()\n# chunks = 20_000_000\nchunks = 5_000_000\n\npreds_1 = 14\nfrom tqdm import tqdm\nprint(\"read test_set.csv\")\ncolumnslist=full_train.columns\ndel full_train\nprint(\"gc.collect\",gc.collect())\nfor i_c, df in enumerate(tqdm(pd.read_csv('../input/test_set.csv', \n                                     chunksize=chunks, iterator=True,\n                                     dtype = {'object_id':np.int32,\n                                              'mjd':np.float64,\n                                              'passband':np.int8,\n                                              'flux':np.float32,\n                                              'flux_err':np.float32,\n                                              'detected':np.int32}))):\n    agg_test = df.groupby(grp_col).agg(aggs)\n    agg_test.columns = new_columns \n    agg_test=pd.pivot_table(agg_test, index='object_id', columns='passband')#.reset_index()\n    full_test = agg_test.reset_index().merge(right=meta_test, how='left', on='object_id')\n    if isfillNaN:\n        full_test = full_test.fillna(train_mean)\n\n    # Make predictions\n    preds = None\n    for clf in clfs:\n        if preds is None:\n#             preds = clf.predict_proba(full_test[full_train.columns]) / folds.n_splits\n            preds = clf.predict_proba(full_test[columnslist]) / folds.n_splits\n        else:\n#             preds += clf.predict_proba(full_test[full_train.columns]) / folds.n_splits\n            preds += clf.predict_proba(full_test[columnslist]) / folds.n_splits\n\n    # preds_99 = 0.1 gives 1.769\n    preds_99 = np.ones(preds.shape[0])\n    #     for i in range(preds.shape[1]):\n    #         preds_99 *= (1 - preds[:, i])\n    #     preds_1 = preds.shape[1]\n    for i in range(preds_1):\n        preds_99 *= (1 - preds[:, i])\n\n    # Store predictions\n    preds_df = pd.DataFrame(preds, columns=['class_' + str(s) for s in clfs[0].classes_])\n    preds_df['object_id'] = full_test['object_id']\n    #     https://www.kaggle.com/ogrellier/plasticc-in-a-kernel-meta-and-data/code\n    # https://www.kaggle.com/c/PLAsTiCC-2018/discussion/68943\n    #     preds_df['class_99'] = preds_99\n    preds_df['class_99'] = 0.14 * preds_99 / np.mean(preds_99) \n#     if isDebug:\n#         print(preds_df['class_99'].mean(),np.mean(preds_99))\n#         print(np.mean(0.14 * preds_99))\n    \n    if i_c == 0:\n        preds_df.to_csv('predictions.csv',  header=True, mode='a', index=False)\n    else: \n        preds_df.to_csv('predictions.csv',  header=False, mode='a', index=False)\n        \n    del agg_test, full_test, preds_df, preds\n    gc.collect()\n    \n    if (i_c + 1) % 10 == 0:\n        print('%15d done in %5.1f' % (chunks * (i_c + 1), (time.time() - start) / 60))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ab549b5d9fc196d47d2a5d7b73f241dda07cdd8c"},"cell_type":"code","source":"%%time\nz = pd.read_csv('predictions.csv')\n\nprint(z.groupby('object_id').size().max())\nprint((z.groupby('object_id').size() > 1).sum())\n\nz = z.groupby('object_id').mean()\n\nz.to_csv('single_predictions.csv', index=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dd511175977c05b8cd3c6a725858722055104a50"},"cell_type":"code","source":"print(\"Notebook Runtime: %0.2f Minutes\"%((time.time() - notebookstart)/60))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}