{"cells":[{"metadata":{"_uuid":"609006646970a83c89ff8a332459ddceb2e02e4c"},"cell_type":"markdown","source":"# Loading Libraries"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.metrics import log_loss\nfrom sklearn.model_selection import StratifiedKFold\nimport gc\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns \nimport lightgbm as lgb\nfrom catboost import Pool, CatBoostClassifier\nimport itertools\nimport pickle, gzip\nimport glob\nfrom sklearn.preprocessing import StandardScaler","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7dbccfa992f29644a47a31341b2c68c6b42b835d"},"cell_type":"markdown","source":"# Extracting Features from train set"},{"metadata":{"trusted":true,"_uuid":"631c418da4275fba3070f2a6cff3873d11591f95"},"cell_type":"code","source":"gc.enable()\n\ntrain = pd.read_csv('../input/training_set.csv')\ntrain['flux_ratio_sq'] = np.power(train['flux'] / train['flux_err'], 2.0)\ntrain['flux_by_flux_ratio_sq'] = train['flux'] * train['flux_ratio_sq']\n\naggs = {\n    'mjd': ['min', 'max', 'size'],\n    'passband': ['min', 'max', 'mean', 'median', 'std'],\n    'flux': ['min', 'max', 'mean', 'median', 'std','skew'],\n    'flux_err': ['min', 'max', 'mean', 'median', 'std','skew'],\n    'detected': ['mean'],\n    'flux_ratio_sq':['sum','skew'],\n    'flux_by_flux_ratio_sq':['sum','skew'],\n}\n\nagg_train = train.groupby('object_id').agg(aggs)\nnew_columns = [\n    k + '_' + agg for k in aggs.keys() for agg in aggs[k]\n]\nagg_train.columns = new_columns\nagg_train['mjd_diff'] = agg_train['mjd_max'] - agg_train['mjd_min']\nagg_train['flux_diff'] = agg_train['flux_max'] - agg_train['flux_min']\nagg_train['flux_dif2'] = (agg_train['flux_max'] - agg_train['flux_min']) / agg_train['flux_mean']\nagg_train['flux_w_mean'] = agg_train['flux_by_flux_ratio_sq_sum'] / agg_train['flux_ratio_sq_sum']\nagg_train['flux_dif3'] = (agg_train['flux_max'] - agg_train['flux_min']) / agg_train['flux_w_mean']\n\ndel agg_train['mjd_max'], agg_train['mjd_min']\nagg_train.head()\n\ndel train\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7ed21e25fb5678c78bb19a8297bf83db00ccae01"},"cell_type":"markdown","source":"# Merging extracted features with meta data"},{"metadata":{"trusted":true,"_uuid":"696ee870837c23375bc7f0388de1b07510351123"},"cell_type":"code","source":"meta_train = pd.read_csv('../input/training_set_metadata.csv')\nmeta_train.head()\n\nfull_train = agg_train.reset_index().merge(\n    right=meta_train,\n    how='outer',\n    on='object_id'\n)\n\nif 'target' in full_train:\n    y = full_train['target']\n    del full_train['target']\nclasses = sorted(y.unique())\n\n# Taken from Giba's topic : https://www.kaggle.com/titericz\n# https://www.kaggle.com/c/PLAsTiCC-2018/discussion/67194\n# with Kyle Boone's post https://www.kaggle.com/kyleboone\nclass_weight = {\n    c: 1 for c in classes\n}\nfor c in [64, 15]:\n    class_weight[c] = 2\n\nprint('Unique classes : ', classes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2b6f79b71d0e9b62266a52d9e7ac209c56e14aff"},"cell_type":"code","source":"if 'object_id' in full_train:\n    oof_df = full_train[['object_id']]\n    del full_train['object_id'], full_train['distmod'], full_train['hostgal_specz']\n    del full_train['ra'], full_train['decl'], full_train['gal_l'],full_train['gal_b'],full_train['ddf']\n    \n    \ntrain_mean = full_train.mean(axis=0)\nfull_train.fillna(train_mean, inplace=True)\n\nfolds = StratifiedKFold(n_splits=5, shuffle=True, random_state=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7aaadd7fb68f26be1aa78e0721332689365f2322"},"cell_type":"markdown","source":"# Standard Scaling the input (imp.)"},{"metadata":{"trusted":true,"_uuid":"c1aaa4b97bec65047182097d08d33541aaf9a057"},"cell_type":"code","source":"full_train_new = full_train.copy()\nss = StandardScaler()\nfull_train_ss = ss.fit_transform(full_train_new)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"92b0ae5a7f0b581b1f4e971a5c62eb97d2397455"},"cell_type":"markdown","source":"# Deep Learning Begins..."},{"metadata":{"trusted":true,"_uuid":"7b7424590dc985a79354838aa99ed5cb94ebc3f7"},"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Dense,BatchNormalization,Dropout\nfrom keras.callbacks import ReduceLROnPlateau,ModelCheckpoint\nfrom keras.utils import to_categorical\nimport tensorflow as tf\nfrom keras import backend as K\nimport keras\nfrom keras import regularizers\nfrom collections import Counter\nfrom sklearn.metrics import confusion_matrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"59889ae9a9041a1c9efa8c42c59b75aa58249bf1"},"cell_type":"code","source":"# https://www.kaggle.com/c/PLAsTiCC-2018/discussion/69795\ndef mywloss(y_true,y_pred):  \n    yc=tf.clip_by_value(y_pred,1e-15,1-1e-15)\n    loss=-(tf.reduce_mean(tf.reduce_mean(y_true*tf.log(yc),axis=0)/wtable))\n    return loss","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4c35b11a75cdc6aa926109ae876a1f2bb9f4078c"},"cell_type":"code","source":"def multi_weighted_logloss(y_ohe, y_p):\n    \"\"\"\n    @author olivier https://www.kaggle.com/ogrellier\n    multi logloss for PLAsTiCC challenge\n    \"\"\"\n    classes = [6, 15, 16, 42, 52, 53, 62, 64, 65, 67, 88, 90, 92, 95]\n    class_weight = {6: 1, 15: 2, 16: 1, 42: 1, 52: 1, 53: 1, 62: 1, 64: 2, 65: 1, 67: 1, 88: 1, 90: 1, 92: 1, 95: 1}\n    # Normalize rows and limit y_preds to 1e-15, 1-1e-15\n    y_p = np.clip(a=y_p, a_min=1e-15, a_max=1-1e-15)\n    # Transform to log\n    y_p_log = np.log(y_p)\n    # Get the log for ones, .values is used to drop the index of DataFrames\n    # Exclude class 99 for now, since there is no class99 in the training set \n    # we gave a special process for that class\n    y_log_ones = np.sum(y_ohe * y_p_log, axis=0)\n    # Get the number of positives for each class\n    nb_pos = y_ohe.sum(axis=0).astype(float)\n    # Weight average and divide by the number of positives\n    class_arr = np.array([class_weight[k] for k in sorted(class_weight.keys())])\n    y_w = y_log_ones * class_arr / nb_pos    \n    loss = - np.sum(y_w) / np.sum(class_arr)\n    return loss","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6601f669629d8e1bf9182054206ef316341e0e3e"},"cell_type":"markdown","source":"# Defining simple model in keras"},{"metadata":{"trusted":true,"_uuid":"4a411f6097378ab0c3199064478f4ed4e064b6fb"},"cell_type":"code","source":"K.clear_session()\ndef build_model(dropout_rate=0.25,activation='relu'):\n    start_neurons = 512\n    # create model\n    model = Sequential()\n    model.add(Dense(start_neurons, input_dim=full_train_ss.shape[1], activation=activation))\n    model.add(BatchNormalization())\n    model.add(Dropout(dropout_rate))\n    \n    model.add(Dense(start_neurons//2,activation=activation))\n    model.add(BatchNormalization())\n    model.add(Dropout(dropout_rate))\n    \n    model.add(Dense(start_neurons//4,activation=activation))\n    model.add(BatchNormalization())\n    model.add(Dropout(dropout_rate))\n    \n    model.add(Dense(start_neurons//8,activation=activation))\n    model.add(BatchNormalization())\n    model.add(Dropout(dropout_rate/2))\n    \n    model.add(Dense(len(classes), activation='softmax'))\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9235d7edc1520d10784f0f14b88e35651d7c9d5b"},"cell_type":"code","source":"unique_y = np.unique(y)\nclass_map = dict()\nfor i,val in enumerate(unique_y):\n    class_map[val] = i\n        \ny_map = np.zeros((y.shape[0],))\ny_map = np.array([class_map[val] for val in y])\ny_categorical = to_categorical(y_map)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4b915ef7ac26164d0b83e265a277ea50e8557eac"},"cell_type":"markdown","source":"# Calculating the class weights"},{"metadata":{"trusted":true,"_uuid":"fe95383168c5bd60e4349d6bce4999ffdff02471"},"cell_type":"code","source":"y_count = Counter(y_map)\nwtable = np.zeros((len(unique_y),))\nfor i in range(len(unique_y)):\n    wtable[i] = y_count[i]/y_map.shape[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"449a0c6c0f3baa75ef4719ef8e76ac5fa62d094b"},"cell_type":"code","source":"def plot_loss_acc(history):\n    plt.plot(history.history['loss'][1:])\n    plt.plot(history.history['val_loss'][1:])\n    plt.title('model loss')\n    plt.ylabel('val_loss')\n    plt.xlabel('epoch')\n    plt.legend(['train','Validation'], loc='upper left')\n    plt.show()\n    \n    plt.plot(history.history['acc'][1:])\n    plt.plot(history.history['val_acc'][1:])\n    plt.title('model Accuracy')\n    plt.ylabel('val_acc')\n    plt.xlabel('epoch')\n    plt.legend(['train','Validation'], loc='upper left')\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6417136e31687df52d5d9f85a83df00aaa6cf2eb"},"cell_type":"code","source":"clfs = []\noof_preds = np.zeros((len(full_train_ss), len(classes)))\nepochs = 600\nbatch_size = 100\nfor fold_, (trn_, val_) in enumerate(folds.split(y_map, y_map)):\n    checkPoint = ModelCheckpoint(\"./keras.model\",monitor='val_loss',mode = 'min', save_best_only=True, verbose=0)\n    x_train, y_train = full_train_ss[trn_], y_categorical[trn_]\n    x_valid, y_valid = full_train_ss[val_], y_categorical[val_]\n    \n    model = build_model(dropout_rate=0.5,activation='tanh')    \n    model.compile(loss=mywloss, optimizer='adam', metrics=['accuracy'])\n    history = model.fit(x_train, y_train,\n                    validation_data=[x_valid, y_valid], \n                    epochs=epochs,\n                    batch_size=batch_size,shuffle=True,verbose=0,callbacks=[checkPoint])       \n    \n    plot_loss_acc(history)\n    \n    print('Loading Best Model')\n    model.load_weights('./keras.model')\n    # # Get predicted probabilities for each class\n    oof_preds[val_, :] = model.predict_proba(x_valid,batch_size=batch_size)\n    print(multi_weighted_logloss(y_valid, model.predict_proba(x_valid,batch_size=batch_size)))\n    clfs.append(model)\n    \nprint('MULTI WEIGHTED LOG LOSS : %.5f ' % multi_weighted_logloss(y_categorical,oof_preds))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d8d149b6795271bdce59ee091118bd1681cb3462"},"cell_type":"code","source":"# http://scikit-learn.org/stable/modules/generated/sklearn.metrics.confusion_matrix.html\ndef plot_confusion_matrix(cm, classes,\n                          normalize=False,\n                          title='Confusion matrix',\n                          cmap=plt.cm.Blues):\n    \"\"\"\n    This function prints and plots the confusion matrix.\n    Normalization can be applied by setting `normalize=True`.\n    \"\"\"\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n        print(\"Normalized confusion matrix\")\n    else:\n        print('Confusion matrix, without normalization')\n\n    print(cm)\n\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45)\n    plt.yticks(tick_marks, classes)\n\n    fmt = '.2f' if normalize else 'd'\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, format(cm[i, j], fmt),\n                 horizontalalignment=\"center\",\n                 color=\"white\" if cm[i, j] > thresh else \"black\")\n\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    plt.tight_layout()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc5c23f2780db663e813e796b811f3831ef42686"},"cell_type":"code","source":"# Compute confusion matrix\ncnf_matrix = confusion_matrix(y_map, np.argmax(oof_preds,axis=-1))\nnp.set_printoptions(precision=2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"161c3e5e83a2ee3927df2c782fcb53fb7a379292"},"cell_type":"code","source":"sample_sub = pd.read_csv('../input/sample_submission.csv')\nclass_names = list(sample_sub.columns[1:-1])\ndel sample_sub;gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f9416f8a1b8b43e49f54642e03793ffce83adedd","scrolled":true},"cell_type":"code","source":"# Plot non-normalized confusion matrix\nplt.figure(figsize=(12,12))\nfoo = plot_confusion_matrix(cnf_matrix, classes=class_names,normalize=True,\n                      title='Confusion matrix')\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f65492856178f1692de22989481d7dde8afd1e84"},"cell_type":"markdown","source":"# Test Set Predictions"},{"metadata":{"trusted":true,"_uuid":"2922fdae392d8b9e71882e809b479311e6db1844"},"cell_type":"code","source":"meta_test = pd.read_csv('../input/test_set_metadata.csv')\n\nimport time\n\nstart = time.time()\nchunks = 5000000\nfor i_c, df in enumerate(pd.read_csv('../input/test_set.csv', chunksize=chunks, iterator=True)):\n    df['flux_ratio_sq'] = np.power(df['flux'] / df['flux_err'], 2.0)\n    df['flux_by_flux_ratio_sq'] = df['flux'] * df['flux_ratio_sq']\n    # Group by object id\n    agg_test = df.groupby('object_id').agg(aggs)\n    agg_test.columns = new_columns\n    agg_test['mjd_diff'] = agg_test['mjd_max'] - agg_test['mjd_min']\n    agg_test['flux_diff'] = agg_test['flux_max'] - agg_test['flux_min']\n    agg_test['flux_dif2'] = (agg_test['flux_max'] - agg_test['flux_min']) / agg_test['flux_mean']\n    agg_test['flux_w_mean'] = agg_test['flux_by_flux_ratio_sq_sum'] / agg_test['flux_ratio_sq_sum']\n    agg_test['flux_dif3'] = (agg_test['flux_max'] - agg_test['flux_min']) / agg_test['flux_w_mean']\n\n    del agg_test['mjd_max'], agg_test['mjd_min']\n#     del df\n#     gc.collect()\n    \n    # Merge with meta data\n    full_test = agg_test.reset_index().merge(\n        right=meta_test,\n        how='left',\n        on='object_id'\n    )\n    full_test[full_train.columns] = full_test[full_train.columns].fillna(train_mean)\n    full_test_ss = ss.transform(full_test[full_train.columns])\n    # Make predictions\n    preds = None\n    for clf in clfs:\n        if preds is None:\n            preds = clf.predict_proba(full_test_ss) / folds.n_splits\n        else:\n            preds += clf.predict_proba(full_test_ss) / folds.n_splits\n    \n   # Compute preds_99 as the proba of class not being any of the others\n    # preds_99 = 0.1 gives 1.769\n    preds_99 = np.ones(preds.shape[0])\n    for i in range(preds.shape[1]):\n        preds_99 *= (1 - preds[:, i])\n    \n    # Store predictions\n    preds_df = pd.DataFrame(preds, columns=class_names)\n    preds_df['object_id'] = full_test['object_id']\n    preds_df['class_99'] = 0.14 * preds_99 / np.mean(preds_99) \n    \n    if i_c == 0:\n        preds_df.to_csv('predictions.csv',  header=True, mode='a', index=False)\n    else: \n        preds_df.to_csv('predictions.csv',  header=False, mode='a', index=False)\n        \n    del agg_test, full_test, preds_df, preds\n#     print('done')\n    if (i_c + 1) % 10 == 0:\n        print('%15d done in %5.1f' % (chunks * (i_c + 1), (time.time() - start) / 60))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ab549b5d9fc196d47d2a5d7b73f241dda07cdd8c"},"cell_type":"code","source":"z = pd.read_csv('predictions.csv')\n\nprint(z.groupby('object_id').size().max())\nprint((z.groupby('object_id').size() > 1).sum())\n\nz = z.groupby('object_id').mean()\n\nz.to_csv('single_predictions.csv', index=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e99166b5c98de7f41722d6319dd121d445828ba3"},"cell_type":"code","source":"z.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9e58e8719ced4eb61b9af02838e91340799cff37"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}