{"cells":[{"metadata":{"trusted":true,"_uuid":"d0c8ec64403664c12d9c53d102e2ae82a3517dd9"},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold\nimport gc\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport lightgbm as lgb\nimport logging\ngc.enable()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f15c810fc525eb105a165f71bf6f8583374217a6"},"cell_type":"code","source":"train = pd.read_csv('../input/training_set.csv')\nprint(train.shape)\nmeta_train = pd.read_csv('../input/training_set_metadata.csv')\nprint(meta_train.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eacf37ca32ba8ac1e26e4a3f45eaf6a388e14bc0"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e2bae55be331888ac699c42e072637c4e109961b"},"cell_type":"code","source":"meta_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1ca2c2bf15ba9d76703e337bd74c934724f43bfe"},"cell_type":"code","source":"from astropy.time import Time\ntrain['mjd'] = Time(train.mjd.values, format='mjd').iso\ntrain['mjd'] = pd.to_datetime(train['mjd'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b6cb71e46a9dde957732a2d674df53dc75d64d0"},"cell_type":"code","source":"x = train.copy()\nx.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e04185b9a5ad229b4f033d28289a1712dd4763f0"},"cell_type":"code","source":"x.mjd = x.groupby(['object_id','passband']).diff().transform(lambda x: x.fillna(x.mean()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1f399d79b97df4c27883f01e6f191342b8985aae"},"cell_type":"code","source":"x.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3c1f362717b8b88675400a2c86c0a61ff346b9c9"},"cell_type":"code","source":"x['cc'] = x.groupby(['object_id','passband'])['mjd'].cumcount()\nx.drop('mjd',inplace=True,axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8a8ed8c15b94b18ab0d4c2dd9309efe891f61fba"},"cell_type":"code","source":"x = x.set_index(['object_id','passband','cc'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6ac2f415aaeae52312558c1cf7d6143f2d97eddd"},"cell_type":"code","source":"x.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6d11af792195fa556f7b942ad238aa1cb61fcddd"},"cell_type":"code","source":"x = x.unstack()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c24a9c8171d1a674d4860d4bebe631a2e37277c3"},"cell_type":"code","source":"x.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8f87cc3b41f6f6952720575c701abf3b8f67b819"},"cell_type":"code","source":"x.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0b32b2aa5a870618b8689667632d534d22ae2621"},"cell_type":"code","source":"meta_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6d812d6a7a3b850d760c7abf6b8d95827ff18333"},"cell_type":"code","source":"meta_train = meta_train.set_index('object_id')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0a914f2fe9dc68fa5ee9ddf73b79b93e1071e720"},"cell_type":"code","source":"x = x.join(meta_train,on='object_id',how='left')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a2d8d3bc73d8762728f77e6e20c86898cfd67621"},"cell_type":"code","source":"x.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c7e2b0afb82b4de795738fa38f61ae3cb83afcef"},"cell_type":"code","source":"x = x.reset_index(drop=False)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ce201a5838931d6535f699cf69321372dccc83f8"},"cell_type":"code","source":"cols = ['_'.join(str(s).strip() for s in col if s) if len(col)==2 else col for col in x.columns ]\ncols\nx.columns = cols","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e0941349e07ee818d2f55f71950959d7072999b9"},"cell_type":"code","source":"x.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c79650bf72ea496f4549d3167fd88e94e7578d08"},"cell_type":"code","source":"x = x.replace(-np.inf,np.nan)\nx = x.replace(np.inf,np.nan)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ca8ecbe8c19629d40c6371bbcc1fe30cbad9fa02"},"cell_type":"code","source":"x.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4844d28cfe95eebafcbadfef00ff1578e3da3aa7"},"cell_type":"code","source":"fluxcolumns = [a  for a in x.columns if a.startswith('flux') and not a.startswith('flux_err')]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"831f1171ad3c72f49af99909d91200aae9fde84f"},"cell_type":"code","source":"uniques = sorted(x.target.unique())\nf, ax = plt.subplots(len(uniques),6)\nf.set_figheight(15)\nf.set_figwidth(15)\nfor a in range(len(uniques)):\n    for b in range(6):\n        ax[a,b].set_xticks([])\n   \nfor i in range(len(uniques)):\n    ax[i][0].plot(x[(x.target==uniques[i])&(x.passband==0)][fluxcolumns].mean())\n    ax[i][1].plot(x[(x.target==uniques[i])&(x.passband==1)][fluxcolumns].mean())\n    ax[i][2].plot(x[(x.target==uniques[i])&(x.passband==2)][fluxcolumns].mean())\n    ax[i][3].plot(x[(x.target==uniques[i])&(x.passband==3)][fluxcolumns].mean())\n    ax[i][4].plot(x[(x.target==uniques[i])&(x.passband==4)][fluxcolumns].mean())\n    ax[i][5].plot(x[(x.target==uniques[i])&(x.passband==5)][fluxcolumns].mean())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f5efc5e8a473e192f810dfd1a3c7dcc9eef6f57"},"cell_type":"code","source":"def create_logger():\n    logger_ = logging.getLogger('main')\n    logger_.setLevel(logging.DEBUG)\n    fh = logging.FileHandler('simple_lightgbm.log')\n    fh.setLevel(logging.DEBUG)\n    ch = logging.StreamHandler()\n    ch.setLevel(logging.DEBUG)\n    formatter = logging.Formatter('[%(levelname)s]%(asctime)s:%(name)s:%(message)s')\n    fh.setFormatter(formatter)\n    ch.setFormatter(formatter)\n    # add the handlers to the logger\n    logger_.addHandler(fh)\n    logger_.addHandler(ch)\n\n\ndef get_logger():\n    return logging.getLogger('main')\n\n\ndef lgb_multi_weighted_logloss(y_true, y_preds):\n    \"\"\"\n    @author olivier https://www.kaggle.com/ogrellier\n    multi logloss for PLAsTiCC challenge\n    \"\"\"\n    # class_weights taken from Giba's topic : https://www.kaggle.com/titericz\n    # https://www.kaggle.com/c/PLAsTiCC-2018/discussion/67194\n    # with Kyle Boone's post https://www.kaggle.com/kyleboone\n    classes = [6, 15, 16, 42, 52, 53, 62, 64, 65, 67, 88, 90, 92, 95]\n    class_weight = {6: 1, 15: 2, 16: 1, 42: 1, 52: 1, 53: 1, 62: 1, 64: 2, 65: 1, 67: 1, 88: 1, 90: 1, 92: 1, 95: 1}\n    if len(np.unique(y_true)) > 14:\n        classes.append(99)\n        class_weight[99] = 2\n    y_p = y_preds.reshape(y_true.shape[0], len(classes), order='F')\n\n    # Trasform y_true in dummies\n    y_ohe = pd.get_dummies(y_true)\n    # Normalize rows and limit y_preds to 1e-15, 1-1e-15\n    y_p = np.clip(a=y_p, a_min=1e-15, a_max=1 - 1e-15)\n    # Transform to log\n    y_p_log = np.log(y_p)\n    # Get the log for ones, .values is used to drop the index of DataFrames\n    # Exclude class 99 for now, since there is no class99 in the training set\n    # we gave a special process for that class\n    y_log_ones = np.sum(y_ohe.values * y_p_log, axis=0)\n    # Get the number of positives for each class\n    nb_pos = y_ohe.sum(axis=0).values.astype(float)\n    # Weight average and divide by the number of positives\n    class_arr = np.array([class_weight[k] for k in sorted(class_weight.keys())])\n    y_w = y_log_ones * class_arr / nb_pos\n\n    loss = - np.sum(y_w) / np.sum(class_arr)\n    return 'wloss', loss, False\n\n\ndef multi_weighted_logloss(y_true, y_preds):\n    \"\"\"\n    @author olivier https://www.kaggle.com/ogrellier\n    multi logloss for PLAsTiCC challenge\n    \"\"\"\n    # class_weights taken from Giba's topic : https://www.kaggle.com/titericz\n    # https://www.kaggle.com/c/PLAsTiCC-2018/discussion/67194\n    # with Kyle Boone's post https://www.kaggle.com/kyleboone\n    classes = [6, 15, 16, 42, 52, 53, 62, 64, 65, 67, 88, 90, 92, 95]\n    class_weight = {6: 1, 15: 2, 16: 1, 42: 1, 52: 1, 53: 1, 62: 1, 64: 2, 65: 1, 67: 1, 88: 1, 90: 1, 92: 1, 95: 1}\n    if len(np.unique(y_true)) > 14:\n        classes.append(99)\n        class_weight[99] = 2\n    y_p = y_preds\n    # Trasform y_true in dummies\n    y_ohe = pd.get_dummies(y_true)\n    # Normalize rows and limit y_preds to 1e-15, 1-1e-15\n    y_p = np.clip(a=y_p, a_min=1e-15, a_max=1 - 1e-15)\n    # Transform to log\n    y_p_log = np.log(y_p)\n    # Get the log for ones, .values is used to drop the index of DataFrames\n    # Exclude class 99 for now, since there is no class99 in the training set\n    # we gave a special process for that class\n    y_log_ones = np.sum(y_ohe.values * y_p_log, axis=0)\n    # Get the number of positives for each class\n    nb_pos = y_ohe.sum(axis=0).values.astype(float)\n    # Weight average and divide by the number of positives\n    class_arr = np.array([class_weight[k] for k in sorted(class_weight.keys())])\n    y_w = y_log_ones * class_arr / nb_pos\n\n    loss = - np.sum(y_w) / np.sum(class_arr)\n    return loss\n\n\ndef predict_chunk(df_, clfs_, meta_, features, train_mean):\n\n    df_['flux_ratio_sq'] = np.power(df_['flux'] / df_['flux_err'], 2.0)\n    df_['flux_by_flux_ratio_sq'] = df_['flux'] * df_['flux_ratio_sq']\n\n    # Group by object id\n    aggs = get_aggregations()\n\n    aggs = get_aggregations()\n    aggs['flux_ratio_sq'] = ['sum']\n    aggs['flux_by_flux_ratio_sq'] = ['sum']\n\n    new_columns = get_new_columns(aggs)\n\n    agg_ = df_.groupby('object_id').agg(aggs)\n    agg_.columns = new_columns\n\n    agg_ = add_features_to_agg(df=agg_)\n\n    # Merge with meta data\n    full_test = agg_.reset_index().merge(\n        right=meta_,\n        how='left',\n        on='object_id'\n    )\n\n    full_test = full_test.fillna(train_mean)\n    # Make predictions\n    preds_ = None\n    for clf in clfs_:\n        if preds_ is None:\n            preds_ = clf.predict_proba(full_test[features]) / len(clfs_)\n        else:\n            preds_ += clf.predict_proba(full_test[features]) / len(clfs_)\n\n    # Compute preds_99 as the proba of class not being any of the others\n    # preds_99 = 0.1 gives 1.769\n    preds_99 = np.ones(preds_.shape[0])\n    for i in range(preds_.shape[1]):\n        preds_99 *= (1 - preds_[:, i])\n\n    # Create DataFrame from predictions\n    preds_df_ = pd.DataFrame(preds_, columns=['class_' + str(s) for s in clfs_[0].classes_])\n    preds_df_['object_id'] = full_test['object_id']\n    preds_df_['class_99'] = 0.14 * preds_99 / np.mean(preds_99) \n\n    print(preds_df_['class_99'].mean())\n\n    del agg_, full_test, preds_\n    gc.collect()\n\n    return preds_df_\n\n\ndef save_importances(importances_):\n    mean_gain = importances_[['gain', 'feature']].groupby('feature').mean()\n    importances_['mean_gain'] = importances_['feature'].map(mean_gain['gain'])\n    plt.figure(figsize=(8, 12))\n    sns.barplot(x='gain', y='feature', data=importances_.sort_values('mean_gain', ascending=False))\n    plt.tight_layout()\n    #plt.savefig('importances.png')\n\n\ndef train_classifiers(full_train=None, y=None):\n\n    folds = StratifiedKFold(n_splits=5, shuffle=True, random_state=1)\n    clfs = []\n    importances = pd.DataFrame()\n    lgb_params = {\n        'boosting_type': 'gbdt',\n        'objective': 'multiclass',\n        'num_class': 14,\n        'metric': 'multi_logloss',\n        'learning_rate': 0.03,\n        'subsample': .9,\n        'colsample_bytree': .7,\n        'reg_alpha': .01,\n        'reg_lambda': .01,\n        'min_split_gain': 0.01,\n        'min_child_weight': 10,\n        'n_estimators': 1000,\n        'silent': -1,\n        'verbose': -1,\n        'max_depth': 3\n    }\n    oof_preds = np.zeros((len(full_train), np.unique(y).shape[0]))\n    for fold_, (trn_, val_) in enumerate(folds.split(y, y)):\n        trn_x, trn_y = full_train.iloc[trn_], y.iloc[trn_]\n        val_x, val_y = full_train.iloc[val_], y.iloc[val_]\n\n        clf = lgb.LGBMClassifier(**lgb_params)\n        clf.fit(\n            trn_x, trn_y,\n            eval_set=[(trn_x, trn_y), (val_x, val_y)],\n            eval_metric=lgb_multi_weighted_logloss,\n            verbose=100,\n            early_stopping_rounds=50\n        )\n        oof_preds[val_, :] = clf.predict_proba(val_x, num_iteration=clf.best_iteration_)\n        get_logger().info(multi_weighted_logloss(val_y, clf.predict_proba(val_x, num_iteration=clf.best_iteration_)))\n\n        imp_df = pd.DataFrame()\n        imp_df['feature'] = full_train.columns\n        imp_df['gain'] = clf.feature_importances_\n        imp_df['fold'] = fold_ + 1\n        importances = pd.concat([importances, imp_df], axis=0, sort=False)\n\n        clfs.append(clf)\n\n    get_logger().info('MULTI WEIGHTED LOG LOSS : %.5f ' % multi_weighted_logloss(y_true=y, y_preds=oof_preds))\n\n    return clfs, importances, oof_preds, y","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5cf72c3299cc78d662b9ee43f9dc7931a5bada4c"},"cell_type":"code","source":"x.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"862442672e6c1a7cae2a87146ae561f1a4d91272"},"cell_type":"code","source":"clfs, importances, oof_preds, y = train_classifiers(full_train=x[x.columns[1:-1]], y=x.target)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ffaac224d46578572bca61e495c5f17bcc539cb"},"cell_type":"code","source":"save_importances(importances[:50])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d2eb4038715550d51612175a62b85e6c31aacfea"},"cell_type":"code","source":"oof_preds.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"45e6c8bbe6c6044af0fb1bb64e9faf5dba9c415b"},"cell_type":"code","source":"preds = pd.DataFrame(oof_preds)\npreds.columns = ['class_'+str(a) for a in [6, 15, 16, 42, 52, 53, 62, 64, 65, 67, 88, 90, 92, 95]]\ntargets = pd.DataFrame(y)\ntargets.columns = ['target']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8e20c20cbf2a4eebbc5531738464dd2e39f7693e"},"cell_type":"code","source":"x.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"942c6e9aebe12b04311d4d957aa1713daf137445"},"cell_type":"code","source":"preds.insert(0,'object_id',x.object_id)\ntargets.insert(0,'object_id',x.object_id)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fb508100d0904e06f96b8e8d62ee40f8c44a8817"},"cell_type":"code","source":"grppreds = preds.groupby('object_id').mean()\ngpptargets = targets.groupby('object_id').mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fbf1efaed0c46574b272702153cb10a688ffa06f"},"cell_type":"code","source":"from sklearn.manifold import TSNE\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b95cb52559fd927792259c70e6d2521cb17ce0d2"},"cell_type":"code","source":"model = TSNE(n_components=2, random_state=0)\ntsnedata = model.fit_transform(grppreds)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f1ba85ffbaec2e79b3fe577188513d481391bcbd"},"cell_type":"code","source":"cm = plt.cm.get_cmap('RdYlBu')\nfig, axes = plt.subplots(1, 1, figsize=(15, 15))\nsc = axes.scatter(tsnedata[:,0], tsnedata[:,1], alpha=.5, c=(gpptargets.target), cmap=cm, s=30)\ncbar = fig.colorbar(sc, ax=axes)\ncbar.set_label('Log1p(target)')\n_ = axes.set_title(\"Clustering colored by target\")\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f653935e7bce731adedce0e0b86192bc15c6c95a"},"cell_type":"code","source":"gpptargets.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6c6753312a98c7b4fab419698bc7e3777aca7469"},"cell_type":"code","source":"grppreds.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0bbb612b1fa5883888345201c6618468a223f388"},"cell_type":"code","source":"multi_weighted_logloss(y_true=pd.get_dummies(gpptargets.target), y_preds=grppreds)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"94a7c2c34a5f9a30305b4974de7644e0eafa3cfa"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}