{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10384,"databundleVersionId":120379,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Loading Libraries","metadata":{"_uuid":"609006646970a83c89ff8a332459ddceb2e02e4c"}},{"cell_type":"code","source":"# decision tree classifier \n# cross validation\n","metadata":{"execution":{"iopub.status.busy":"2024-04-20T02:22:40.721761Z","iopub.execute_input":"2024-04-20T02:22:40.722139Z","iopub.status.idle":"2024-04-20T02:22:40.726491Z","shell.execute_reply.started":"2024-04-20T02:22:40.722107Z","shell.execute_reply":"2024-04-20T02:22:40.725481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.metrics import log_loss\nfrom sklearn.model_selection import StratifiedKFold\nimport gc\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport lightgbm as lgb\nfrom catboost import Pool, CatBoostClassifier\nimport itertools\nimport pickle, gzip\nimport glob\nfrom sklearn.preprocessing import StandardScaler","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-20T01:42:22.598787Z","iopub.execute_input":"2024-04-20T01:42:22.599206Z","iopub.status.idle":"2024-04-20T01:42:28.026825Z","shell.execute_reply.started":"2024-04-20T01:42:22.599153Z","shell.execute_reply":"2024-04-20T01:42:28.025971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# data=pd.read_csv(\"/kaggle/input/PLAsTiCC-2018/training_set.csv\")\n# data.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-19T19:30:09.252512Z","iopub.execute_input":"2024-04-19T19:30:09.252870Z","iopub.status.idle":"2024-04-19T19:30:09.269086Z","shell.execute_reply.started":"2024-04-19T19:30:09.252846Z","shell.execute_reply":"2024-04-19T19:30:09.268039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Extracting Features from train set","metadata":{"_uuid":"7dbccfa992f29644a47a31341b2c68c6b42b835d"}},{"cell_type":"code","source":"gc.enable()\n\ntrain = pd.read_csv('/kaggle/input/PLAsTiCC-2018/training_set.csv')\ntrain['flux_ratio_sq'] = np.power(train['flux'] / train['flux_err'], 2.0)\ntrain['flux_by_flux_ratio_sq'] = train['flux'] * train['flux_ratio_sq']\n\naggs = {\n    'mjd': ['min', 'max', 'size'],\n    'passband': ['min', 'max', 'mean', 'median', 'std'],\n    'flux': ['min', 'max', 'mean', 'median', 'std','skew'],\n    'flux_err': ['min', 'max', 'mean', 'median', 'std','skew'],\n    'detected': ['mean'],\n    'flux_ratio_sq':['sum','skew'],\n    'flux_by_flux_ratio_sq':['sum','skew'],\n}\n\nagg_train = train.groupby('object_id').agg(aggs)\nnew_columns = [\n    k + '_' + agg for k in aggs.keys() for agg in aggs[k]\n]\nagg_train.columns = new_columns\nagg_train['mjd_diff'] = agg_train['mjd_max'] - agg_train['mjd_min']\nagg_train['flux_diff'] = agg_train['flux_max'] - agg_train['flux_min']\nagg_train['flux_dif2'] = (agg_train['flux_max'] - agg_train['flux_min']) / agg_train['flux_mean']\nagg_train['flux_w_mean'] = agg_train['flux_by_flux_ratio_sq_sum'] / agg_train['flux_ratio_sq_sum']\nagg_train['flux_dif3'] = (agg_train['flux_max'] - agg_train['flux_min']) / agg_train['flux_w_mean']\n\ndel agg_train['mjd_max'], agg_train['mjd_min']\nagg_train.head()\n\ndel train\ngc.collect()","metadata":{"_uuid":"631c418da4275fba3070f2a6cff3873d11591f95","execution":{"iopub.status.busy":"2024-04-20T02:22:40.748952Z","iopub.execute_input":"2024-04-20T02:22:40.749287Z","iopub.status.idle":"2024-04-20T02:22:42.287444Z","shell.execute_reply.started":"2024-04-20T02:22:40.749247Z","shell.execute_reply":"2024-04-20T02:22:42.286562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Merging extracted features with meta data","metadata":{"_uuid":"7ed21e25fb5678c78bb19a8297bf83db00ccae01"}},{"cell_type":"code","source":"meta_train = pd.read_csv('/kaggle/input/PLAsTiCC-2018/training_set_metadata.csv')\nmeta_train.head()\n\nfull_train = agg_train.reset_index().merge(\n    right=meta_train,\n    how='outer',\n    on='object_id'\n)\n\nif 'target' in full_train:\n    y = full_train['target']\n    del full_train['target']\nclasses = sorted(y.unique())\n\n# Taken from Giba's topic : https://www.kaggle.com/titericz\n# https://www.kaggle.com/c/PLAsTiCC-2018/discussion/67194\n# with Kyle Boone's post https://www.kaggle.com/kyleboone\nclass_weight = {\n    c: 1 for c in classes\n}\nfor c in [64, 15]:\n    class_weight[c] = 2\n\nprint('Unique classes : ', classes)","metadata":{"_uuid":"696ee870837c23375bc7f0388de1b07510351123","execution":{"iopub.status.busy":"2024-04-20T02:22:42.289423Z","iopub.execute_input":"2024-04-20T02:22:42.290068Z","iopub.status.idle":"2024-04-20T02:22:42.316378Z","shell.execute_reply.started":"2024-04-20T02:22:42.290034Z","shell.execute_reply":"2024-04-20T02:22:42.315433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'object_id' in full_train:\n    oof_df = full_train[['object_id']]\n    del full_train['object_id'], full_train['distmod'], full_train['hostgal_specz']\n    del full_train['ra'], full_train['decl'], full_train['gal_l'],full_train['gal_b'],full_train['ddf']\n    \n    \n","metadata":{"_uuid":"2b6f79b71d0e9b62266a52d9e7ac209c56e14aff","execution":{"iopub.status.busy":"2024-04-20T02:22:42.317377Z","iopub.execute_input":"2024-04-20T02:22:42.317648Z","iopub.status.idle":"2024-04-20T02:22:42.326840Z","shell.execute_reply.started":"2024-04-20T02:22:42.317626Z","shell.execute_reply":"2024-04-20T02:22:42.325758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Start","metadata":{"execution":{"iopub.status.busy":"2024-04-19T19:31:27.340044Z","iopub.execute_input":"2024-04-19T19:31:27.340414Z","iopub.status.idle":"2024-04-19T19:31:27.345805Z","shell.execute_reply.started":"2024-04-19T19:31:27.340385Z","shell.execute_reply":"2024-04-19T19:31:27.344926Z"}}},{"cell_type":"code","source":"\nunique_classes = sorted(set(y))  # Get unique class labels and sort them\nclass_counts = [sum(y == cls) for cls in unique_classes]  # Count occurrences of each class\n\n# Plot class distribution\nplt.figure(figsize=(10, 6))\nplt.bar(unique_classes, class_counts, color='skyblue')\nplt.xlabel('Class Label')\nplt.ylabel('Count')\nplt.title('Class Distribution')\nplt.xticks(unique_classes)\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-20T03:00:48.623014Z","iopub.execute_input":"2024-04-20T03:00:48.623798Z","iopub.status.idle":"2024-04-20T03:00:48.881675Z","shell.execute_reply.started":"2024-04-20T03:00:48.623765Z","shell.execute_reply":"2024-04-20T03:00:48.880697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from imblearn.over_sampling import ADASYN\n\n# Apply ADASYN oversampling\nadasyn = ADASYN(random_state=42)\nX_resampled, y_resampled = adasyn.fit_resample(full_train, y)\n\n# Print the resampled data\n#print(\"Resampled X:\")\n#print(X_resampled)\n#print(\"Resampled y:\")\n#print(y_resampled)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-20T03:01:56.796591Z","iopub.execute_input":"2024-04-20T03:01:56.797332Z","iopub.status.idle":"2024-04-20T03:01:56.998626Z","shell.execute_reply.started":"2024-04-20T03:01:56.797301Z","shell.execute_reply":"2024-04-20T03:01:56.997833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom collections import Counter\n\n\n# Calculate class distribution after resampling\nclass_distribution = Counter(y_resampled)\n\n# Extract classes and their corresponding counts\nclasses = list(class_distribution.keys())\ncounts = list(class_distribution.values())\n\n# Plot class distribution\nplt.figure(figsize=(10, 6))\nplt.bar(classes, counts, color='skyblue')\nplt.xlabel('Class Label')\nplt.ylabel('Count')\nplt.title('Class Distribution after Resampling')\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-20T03:02:20.495292Z","iopub.execute_input":"2024-04-20T03:02:20.495650Z","iopub.status.idle":"2024-04-20T03:02:20.711071Z","shell.execute_reply.started":"2024-04-20T03:02:20.495624Z","shell.execute_reply":"2024-04-20T03:02:20.710157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_train.shape #(7848, 31)\nfull_train.head()\n#full_train.info()\n#full_train.isnull().sum()\n#import matplotlib.pyplot as plt\n#full_train.hist()\n#plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-20T02:22:42.329120Z","iopub.execute_input":"2024-04-20T02:22:42.329468Z","iopub.status.idle":"2024-04-20T02:22:42.358193Z","shell.execute_reply.started":"2024-04-20T02:22:42.329438Z","shell.execute_reply":"2024-04-20T02:22:42.357223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\ncorrelation=full_train.corr()\n#print(correlation)\n\nplt.figure(figsize=(10, 8))\nsns.heatmap(correlation, annot=True, cmap='coolwarm', fmt=\".2f\")\nplt.title('Correlation Matrix')\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-20T02:22:42.359204Z","iopub.execute_input":"2024-04-20T02:22:42.359588Z","iopub.status.idle":"2024-04-20T02:22:44.600867Z","shell.execute_reply.started":"2024-04-20T02:22:42.359565Z","shell.execute_reply":"2024-04-20T02:22:44.599859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import log_loss","metadata":{"execution":{"iopub.status.busy":"2024-04-20T02:22:44.612491Z","iopub.execute_input":"2024-04-20T02:22:44.613151Z","iopub.status.idle":"2024-04-20T02:22:44.619585Z","shell.execute_reply.started":"2024-04-20T02:22:44.613116Z","shell.execute_reply":"2024-04-20T02:22:44.618653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nX_train_resampled, X_validation, y_train_resampled, y_validation = train_test_split(X_resampled, y_resampled, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-20T03:10:41.375316Z","iopub.execute_input":"2024-04-20T03:10:41.376120Z","iopub.status.idle":"2024-04-20T03:10:41.396995Z","shell.execute_reply.started":"2024-04-20T03:10:41.376090Z","shell.execute_reply":"2024-04-20T03:10:41.396144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rfmodel = RandomForestClassifier(random_state=42)\nrfmodel.fit(X_train_resampled, y_train_resampled)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T03:10:45.840031Z","iopub.execute_input":"2024-04-20T03:10:45.840789Z","iopub.status.idle":"2024-04-20T03:10:59.651104Z","shell.execute_reply.started":"2024-04-20T03:10:45.840756Z","shell.execute_reply":"2024-04-20T03:10:59.650019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define hyperparameters to tune\nparam_grid = {\n    'n_estimators': [100, 200, 300],\n    'max_depth': [None, 10, 20],\n    'min_samples_split': [2, 5, 10],\n    'min_samples_leaf': [1, 2, 4]\n}\n","metadata":{"execution":{"iopub.status.busy":"2024-04-20T03:11:05.484144Z","iopub.execute_input":"2024-04-20T03:11:05.485088Z","iopub.status.idle":"2024-04-20T03:11:05.489770Z","shell.execute_reply.started":"2024-04-20T03:11:05.485055Z","shell.execute_reply":"2024-04-20T03:11:05.488840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Perform grid search cross-validation to find the best hyperparameters\ngrid_search = GridSearchCV(estimator=rfmodel, param_grid=param_grid, cv=5, scoring='accuracy')\ngrid_search.fit(X_train_resampled, y_train_resampled)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T03:11:08.237944Z","iopub.execute_input":"2024-04-20T03:11:08.238345Z","iopub.status.idle":"2024-04-20T05:18:53.021076Z","shell.execute_reply.started":"2024-04-20T03:11:08.238312Z","shell.execute_reply":"2024-04-20T05:18:53.020322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the best hyperparameters\nbest_params = grid_search.best_params_","metadata":{"execution":{"iopub.status.busy":"2024-04-20T06:21:07.298144Z","iopub.execute_input":"2024-04-20T06:21:07.299034Z","iopub.status.idle":"2024-04-20T06:21:07.302928Z","shell.execute_reply.started":"2024-04-20T06:21:07.299002Z","shell.execute_reply":"2024-04-20T06:21:07.301948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize Random Forest classifier with best hyperparameters\nbest_rf_classifier = RandomForestClassifier(**best_params, random_state=42)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the classifier on resampled training data\nbest_rf_classifier.fit(X_train_resampled, y_train_resampled)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Now you can use the fitted classifier for prediction\ny_validation_pred = best_rf_classifier.predict(X_validation)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T06:43:27.223168Z","iopub.execute_input":"2024-04-20T06:43:27.224139Z","iopub.status.idle":"2024-04-20T06:43:27.704650Z","shell.execute_reply.started":"2024-04-20T06:43:27.224103Z","shell.execute_reply":"2024-04-20T06:43:27.703850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Validation Set Classification Report:\")\nprint(classification_report(y_validation, y_validation_pred))","metadata":{"execution":{"iopub.status.busy":"2024-04-20T06:43:36.986559Z","iopub.execute_input":"2024-04-20T06:43:36.987054Z","iopub.status.idle":"2024-04-20T06:43:37.005003Z","shell.execute_reply.started":"2024-04-20T06:43:36.987016Z","shell.execute_reply":"2024-04-20T06:43:37.004089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the predicted probabilities for each class\ny_validation_pred_proba = best_rf_classifier.predict_proba(X_validation)\n\n# Convert y_validation to binary format for each class\ny_validation_bin = label_binarize(y_validation, classes=np.unique(y_validation))\n\n# Compute ROC curve and ROC area for each class\nfpr = dict()\ntpr = dict()\nroc_auc = dict()\nfor i in range(len(best_rf_classifier.classes_)):\n    fpr[i], tpr[i], _ = roc_curve(y_validation_bin[:, i], y_validation_pred_proba[:, i])\n    roc_auc[i] = auc(fpr[i], tpr[i])\n\nplt.figure(figsize=(8, 6))\nfor i in range(len(best_rf_classifier.classes_)):\n    plt.plot(fpr[i], tpr[i],\n             label='ROC curve of class {0} (area = {1:0.2f})'\n             ''.format(i, roc_auc[i]))\n\nplt.plot([0, 1], [0, 1], 'k--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('ROC Curve for Random Forest Classifier on Validation Set (Multiclass)')\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T07:09:15.381718Z","iopub.execute_input":"2024-04-20T07:09:15.382475Z","iopub.status.idle":"2024-04-20T07:09:16.275150Z","shell.execute_reply.started":"2024-04-20T07:09:15.382437Z","shell.execute_reply":"2024-04-20T07:09:16.274340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test with the actual test datasets","metadata":{}},{"cell_type":"code","source":"#y_test_pred = best_rf_classifier.predict(X_test)\n\n# Print classification report for test set\n#print(\"Test Set Classification Report:\")\n#print(classification_report(y_test_original, y_test_pred))","metadata":{"execution":{"iopub.status.busy":"2024-04-20T06:44:01.799399Z","iopub.execute_input":"2024-04-20T06:44:01.800020Z","iopub.status.idle":"2024-04-20T06:44:01.804283Z","shell.execute_reply.started":"2024-04-20T06:44:01.799991Z","shell.execute_reply":"2024-04-20T06:44:01.803142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission (do it for test data set)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\n# Make predictions on the test set\ntest_predictions = best_rf_classifier.predict_proba(X_validation)\n\n# Define the class labels\nclass_labels = best_rf_classifier.classes_\n\n# Clip probabilities to avoid extremes of the log function\ntest_predictions = np.clip(test_predictions, 1e-15, 1 - 1e-15)\n\n# Rescale probabilities to ensure each row sums up to one\ntest_predictions = test_predictions / test_predictions.sum(axis=1, keepdims=True)\n\n# Create a DataFrame to hold the predictions\nsubmission_df = pd.DataFrame(test_predictions, columns=['class_' + str(class_label) for class_label in class_labels])\nsubmission_df.insert(0, 'object_id', X_validation.index)  # Replace test_object_ids with the actual object IDs\n\n# Write the predictions to a CSV file\n#submission_df.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-20T06:46:25.277248Z","iopub.execute_input":"2024-04-20T06:46:25.277747Z","iopub.status.idle":"2024-04-20T06:46:25.751545Z","shell.execute_reply.started":"2024-04-20T06:46:25.277706Z","shell.execute_reply":"2024-04-20T06:46:25.750729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2024-04-20T01:50:19.618704Z","iopub.execute_input":"2024-04-20T01:50:19.619385Z","iopub.status.idle":"2024-04-20T01:50:19.653986Z","shell.execute_reply.started":"2024-04-20T01:50:19.619352Z","shell.execute_reply":"2024-04-20T01:50:19.653033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# find logloss according to your requirement","metadata":{}},{"cell_type":"markdown","source":"# end","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_mean = full_train.mean(axis=0)\nfull_train.fillna(train_mean, inplace=True)\n\nfolds = StratifiedKFold(n_splits=5, shuffle=True, random_state=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Standard Scaling the input (imp.)","metadata":{"_uuid":"7aaadd7fb68f26be1aa78e0721332689365f2322"}},{"cell_type":"code","source":"full_train_new = full_train.copy()\nss = StandardScaler()\nfull_train_ss = ss.fit_transform(full_train_new)","metadata":{"_uuid":"c1aaa4b97bec65047182097d08d33541aaf9a057","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Deep Learning Begins...","metadata":{"_uuid":"92b0ae5a7f0b581b1f4e971a5c62eb97d2397455"}},{"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Dense,BatchNormalization,Dropout\nfrom keras.callbacks import ReduceLROnPlateau,ModelCheckpoint\nfrom keras.utils import to_categorical\nimport tensorflow as tf\nfrom keras import backend as K\nimport keras\nfrom keras import regularizers\nfrom collections import Counter\nfrom sklearn.metrics import confusion_matrix","metadata":{"_uuid":"7b7424590dc985a79354838aa99ed5cb94ebc3f7","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/c/PLAsTiCC-2018/discussion/69795\ndef mywloss(y_true,y_pred):  \n    yc=tf.clip_by_value(y_pred,1e-15,1-1e-15)\n    loss=-(tf.reduce_mean(tf.reduce_mean(y_true*tf.math.log(yc),axis=0)/wtable))\n    return loss","metadata":{"_uuid":"59889ae9a9041a1c9efa8c42c59b75aa58249bf1","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def multi_weighted_logloss(y_ohe, y_p):\n    \"\"\"\n    @author olivier https://www.kaggle.com/ogrellier\n    multi logloss for PLAsTiCC challenge\n    \"\"\"\n    classes = [6, 15, 16, 42, 52, 53, 62, 64, 65, 67, 88, 90, 92, 95]\n    class_weight = {6: 1, 15: 2, 16: 1, 42: 1, 52: 1, 53: 1, 62: 1, 64: 2, 65: 1, 67: 1, 88: 1, 90: 1, 92: 1, 95: 1}\n    # Normalize rows and limit y_preds to 1e-15, 1-1e-15\n    y_p = np.clip(a=y_p, a_min=1e-15, a_max=1-1e-15)\n    # Transform to log\n    y_p_log = np.log(y_p)\n    # Get the log for ones, .values is used to drop the index of DataFrames\n    # Exclude class 99 for now, since there is no class99 in the training set \n    # we gave a special process for that class\n    y_log_ones = np.sum(y_ohe * y_p_log, axis=0)\n    # Get the number of positives for each class\n    nb_pos = y_ohe.sum(axis=0).astype(float)\n    # Weight average and divide by the number of positives\n    class_arr = np.array([class_weight[k] for k in sorted(class_weight.keys())])\n    y_w = y_log_ones * class_arr / nb_pos    \n    loss = - np.sum(y_w) / np.sum(class_arr)\n    return loss","metadata":{"_uuid":"4c35b11a75cdc6aa926109ae876a1f2bb9f4078c","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Defining simple model in keras","metadata":{"_uuid":"6601f669629d8e1bf9182054206ef316341e0e3e"}},{"cell_type":"code","source":"K.clear_session()\ndef build_model(dropout_rate=0.25,activation='relu'):\n    start_neurons = 512\n    # create model\n    model = Sequential()\n    model.add(Dense(start_neurons, input_dim=full_train_ss.shape[1], activation=activation))\n    model.add(BatchNormalization())\n    model.add(Dropout(dropout_rate))\n    \n    model.add(Dense(start_neurons//2,activation=activation))\n    model.add(BatchNormalization())\n    model.add(Dropout(dropout_rate))\n    \n    model.add(Dense(start_neurons//4,activation=activation))\n    model.add(BatchNormalization())\n    model.add(Dropout(dropout_rate))\n    \n    model.add(Dense(start_neurons//8,activation=activation))\n    model.add(BatchNormalization())\n    model.add(Dropout(dropout_rate/2))\n    \n    model.add(Dense(len(classes), activation='softmax'))\n    return model","metadata":{"_uuid":"4a411f6097378ab0c3199064478f4ed4e064b6fb","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_y = np.unique(y)\nclass_map = dict()\nfor i,val in enumerate(unique_y):\n    class_map[val] = i\n        \ny_map = np.zeros((y.shape[0],))\ny_map = np.array([class_map[val] for val in y])\ny_categorical = to_categorical(y_map)","metadata":{"_uuid":"9235d7edc1520d10784f0f14b88e35651d7c9d5b","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calculating the class weights","metadata":{"_uuid":"4b915ef7ac26164d0b83e265a277ea50e8557eac"}},{"cell_type":"code","source":"y_count = Counter(y_map)\nwtable = np.zeros((len(unique_y),))\nfor i in range(len(unique_y)):\n    wtable[i] = y_count[i]/y_map.shape[0]","metadata":{"_uuid":"fe95383168c5bd60e4349d6bce4999ffdff02471","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_loss_acc(history):\n    plt.plot(history.history['loss'][1:])\n    plt.plot(history.history['val_loss'][1:])\n    plt.title('model loss')\n    plt.ylabel('val_loss')\n    plt.xlabel('epoch')\n    plt.legend(['train','Validation'], loc='upper left')\n    plt.show()\n    \n    plt.plot(history.history['accuracy'][1:])\n    plt.plot(history.history['val_accuracy'][1:])\n    plt.title('model Accuracy')\n    plt.ylabel('val_acc')\n    plt.xlabel('epoch')\n    plt.legend(['train','Validation'], loc='upper left')\n    plt.show()","metadata":{"_uuid":"449a0c6c0f3baa75ef4719ef8e76ac5fa62d094b","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clfs = []\noof_preds = np.zeros((len(full_train_ss), len(classes)))\nepochs = 600\nbatch_size = 100\nfor fold_, (trn_, val_) in enumerate(folds.split(y_map, y_map)):\n    checkPoint = ModelCheckpoint(\"/tmp/ckpt/checkpoint.model.keras\",monitor='val_loss',mode = 'min', save_best_only=True, verbose=0)\n    x_train, y_train = full_train_ss[trn_], y_categorical[trn_]\n    x_valid, y_valid = full_train_ss[val_], y_categorical[val_]\n    \n    model = build_model(dropout_rate=0.5,activation='tanh')    \n    model.compile(loss=mywloss, optimizer='adam', metrics=['accuracy'])\n    history = model.fit(x_train, y_train,\n                    validation_data=[x_valid, y_valid], \n                    epochs=epochs,\n                    batch_size=batch_size,shuffle=True,verbose=0,callbacks=[checkPoint])       \n    \n    plot_loss_acc(history)\n    \n    print('Loading Best Model')\n    model.load_weights('/tmp/ckpt/checkpoint.model.keras')\n    \n    # Compute accuracy on the validation set\n    y_valid_pred = np.argmax(model.predict(x_valid, batch_size=batch_size), axis=1)\n    acc = np.mean(y_valid_pred == np.argmax(y_valid, axis=1))\n    print(f\"Validation Accuracy: {acc:.4f}\")\n    \n    \n    # # Get predicted probabilities for each class\n    oof_preds[val_, :] = model.predict(x_valid,batch_size=batch_size)\n    print(multi_weighted_logloss(y_valid, model.predict(x_valid,batch_size=batch_size)))\n    clfs.append(model)\n    \n    \nprint('MULTI WEIGHTED LOG LOSS : %.5f ' % multi_weighted_logloss(y_categorical,oof_preds))\n\n    ","metadata":{"_uuid":"6417136e31687df52d5d9f85a83df00aaa6cf2eb","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\n# Train the random forest classifier\nrf_clf = RandomForestClassifier(n_estimators=100, random_state=42)\n\noof_rf_preds = np.zeros((len(full_train_ss), len(classes)))\nfor fold_, (trn_, val_) in enumerate(folds.split(y_map, y_map)):\n    x_train, y_train = full_train_ss[trn_], y_categorical[trn_]\n    x_valid, y_valid = full_train_ss[val_], y_categorical[val_]\n    \n    rf_clf.fit(x_train, np.argmax(y_train, axis=1))\n    oof_rf_preds[val_] = rf_clf.predict_proba(x_valid)\n    \nprint('MULTI WEIGHTED LOG LOSS (Random Forest): %.5f ' % multi_weighted_logloss(y_categorical, oof_rf_preds))\n    \n    \n    \n\n\n# Combine the neural network and random forest predictions\noof_preds = (oof_preds + oof_rf_preds) / 2\nprint('MULTI WEIGHTED LOG LOSS (Combined): %.5f ' % multi_weighted_logloss(y_categorical, oof_preds))","metadata":{"_uuid":"6417136e31687df52d5d9f85a83df00aaa6cf2eb","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# http://scikit-learn.org/stable/modules/generated/sklearn.metrics.confusion_matrix.html\ndef plot_confusion_matrix(cm, classes,\n                          normalize=False,\n                          title='Confusion matrix',\n                          cmap=plt.cm.Blues):\n    \"\"\"\n    This function prints and plots the confusion matrix.\n    Normalization can be applied by setting `normalize=True`.\n    \"\"\"\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n        print(\"Normalized confusion matrix\")\n    else:\n        print('Confusion matrix, without normalization')\n\n    print(cm)\n\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45)\n    plt.yticks(tick_marks, classes)\n\n    fmt = '.2f' if normalize else 'd'\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, format(cm[i, j], fmt),\n                 horizontalalignment=\"center\",\n                 color=\"white\" if cm[i, j] > thresh else \"black\")\n\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    plt.tight_layout()","metadata":{"_uuid":"d8d149b6795271bdce59ee091118bd1681cb3462","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute confusion matrix\ncnf_matrix = confusion_matrix(y_map, np.argmax(oof_preds,axis=-1))\nnp.set_printoptions(precision=2)","metadata":{"_uuid":"cc5c23f2780db663e813e796b811f3831ef42686","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv('/kaggle/input/PLAsTiCC-2018/sample_submission.csv')\nclass_names = list(sample_sub.columns[1:-1])\ndel sample_sub;gc.collect()","metadata":{"_uuid":"161c3e5e83a2ee3927df2c782fcb53fb7a379292","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot non-normalized confusion matrix\nplt.figure(figsize=(12,12))\nfoo = plot_confusion_matrix(cnf_matrix, classes=class_names,normalize=True,\n                      title='Confusion matrix')\n","metadata":{"_uuid":"f9416f8a1b8b43e49f54642e03793ffce83adedd","scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate accuracy\naccuracy = np.trace(cnf_matrix) / np.sum(cnf_matrix)\nprint(f\"Accuracy: {accuracy:.4f}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.metrics import roc_curve, auc\n\n# # Assuming you have your predicted probabilities and true labels\n# y_pred_prob = model.predict(X_test)[:, 1]  # Probability of the positive class\n# y_true = y_test\n\n# # Calculate the ROC curve and AUC\n# fpr, tpr, _ = roc_curve(y_true, y_pred_prob)\n# roc_auc = auc(fpr, tpr)\n\n# print(f'AUC: {roc_auc:.2f}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test Set Predictions","metadata":{"_uuid":"f65492856178f1692de22989481d7dde8afd1e84"}},{"cell_type":"code","source":"meta_test = pd.read_csv('/kaggle/input/PLAsTiCC-2018/training_set_metadata.csv')\n\nimport time\n\nstart = time.time()\nchunks = 5000000\nfor i_c, df in enumerate(pd.read_csv('/kaggle/input/PLAsTiCC-2018/test_set.csv', chunksize=chunks, iterator=True)):\n    df['flux_ratio_sq'] = np.power(df['flux'] / df['flux_err'], 2.0)\n    df['flux_by_flux_ratio_sq'] = df['flux'] * df['flux_ratio_sq']\n    # Group by object id\n    agg_test = df.groupby('object_id').agg(aggs)\n    agg_test.columns = new_columns\n    agg_test['mjd_diff'] = agg_test['mjd_max'] - agg_test['mjd_min']\n    agg_test['flux_diff'] = agg_test['flux_max'] - agg_test['flux_min']\n    agg_test['flux_dif2'] = (agg_test['flux_max'] - agg_test['flux_min']) / agg_test['flux_mean']\n    agg_test['flux_w_mean'] = agg_test['flux_by_flux_ratio_sq_sum'] / agg_test['flux_ratio_sq_sum']\n    agg_test['flux_dif3'] = (agg_test['flux_max'] - agg_test['flux_min']) / agg_test['flux_w_mean']\n\n    del agg_test['mjd_max'], agg_test['mjd_min']\n#     del df\n#     gc.collect()\n    \n    # Merge with meta data\n    full_test = agg_test.reset_index().merge(\n        right=meta_test,\n        how='left',\n        on='object_id'\n    )\n    full_test[full_train.columns] = full_test[full_train.columns].fillna(train_mean)\n    full_test_ss = ss.transform(full_test[full_train.columns])\n#     # Make predictions\n#     preds = None\n#     for clf in clfs:\n#         if preds is None:\n#             preds = clf.predict_proba(full_test_ss) / folds.n_splits\n#         else:\n#             preds += clf.predict_proba(full_test_ss) / folds.n_splits\n    \n    \n    # Make predictions using the neural network and random forest models\n    nn_preds = None\n    for clf in clfs:\n        if nn_preds is None:\n            nn_preds = clf.predict(full_test_ss) / folds.n_splits\n        else:\n            nn_preds += clf.predict(full_test_ss) / folds.n_splits\n    \n    rf_preds = rf_clf.predict_proba(full_test_ss)\n    \n    # Combine the predictions\n    preds = (nn_preds + rf_preds) / 2\n    \n    \n    \n    \n    \n   # Compute preds_99 as the proba of class not being any of the others\n    # preds_99 = 0.1 gives 1.769\n    preds_99 = np.ones(preds.shape[0])\n    for i in range(preds.shape[1]):\n        preds_99 *= (1 - preds[:, i])\n    \n    # Store predictions\n    preds_df = pd.DataFrame(preds, columns=class_names)\n    preds_df['object_id'] = full_test['object_id']\n    preds_df['class_99'] = 0.14 * preds_99 / np.mean(preds_99) \n    \n    if i_c == 0:\n        preds_df.to_csv('predictions.csv',  header=True, mode='a', index=False)\n    else: \n        preds_df.to_csv('predictions.csv',  header=False, mode='a', index=False)\n        \n    del agg_test, full_test, preds_df, preds\n#     print('done')\n    if (i_c + 1) % 10 == 0:\n        print('%15d done in %5.1f' % (chunks * (i_c + 1), (time.time() - start) / 60))","metadata":{"_uuid":"2922fdae392d8b9e71882e809b479311e6db1844","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"z = pd.read_csv('predictions.csv')\n\nprint(z.groupby('object_id').size().max())\nprint((z.groupby('object_id').size() > 1).sum())\n\nz = z.groupby('object_id').mean()\n\nz.to_csv('single_predictions.csv', index=True)","metadata":{"_uuid":"ab549b5d9fc196d47d2a5d7b73f241dda07cdd8c","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"z.head()","metadata":{"_uuid":"e99166b5c98de7f41722d6319dd121d445828ba3","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.ensemble import RandomForestClassifier\n# from sklearn.metrics import log_loss\n# from sklearn.model_selection import StratifiedKFold\n# import numpy as np\n# import pandas as pd\n# import matplotlib.pyplot as plt\n# import seaborn as sns\n# from sklearn.preprocessing import StandardScaler\n# from sklearn.metrics import confusion_matrix\n# import itertools\n# import gc\n\n# gc.enable()\n\n# train = pd.read_csv('../input/training_set.csv')\n# train['flux_ratio_sq'] = np.power(train['flux'] / train['flux_err'], 2.0)\n# train['flux_by_flux_ratio_sq'] = train['flux'] * train['flux_ratio_sq']\n\n# aggs = {\n#     'mjd': ['min', 'max', 'size'],\n#     'passband': ['min', 'max', 'mean', 'median', 'std'],\n#     'flux': ['min', 'max', 'mean', 'median', 'std', 'skew'],\n#     'flux_err': ['min', 'max', 'mean', 'median', 'std', 'skew'],\n#     'detected': ['mean'],\n#     'flux_ratio_sq': ['sum', 'skew'],\n#     'flux_by_flux_ratio_sq': ['sum', 'skew'],\n# }\n\n# agg_train = train.groupby('object_id').agg(aggs)\n# new_columns = [k + '_' + agg for k in aggs.keys() for agg in aggs[k]]\n# agg_train.columns = new_columns\n# agg_train['mjd_diff'] = agg_train['mjd_max'] - agg_train['mjd_min']\n# agg_train['flux_diff'] = agg_train['flux_max'] - agg_train['flux_min']\n# agg_train['flux_dif2'] = (agg_train['flux_max'] - agg_train['flux_min']) / agg_train['flux_mean']\n# agg_train['flux_w_mean'] = agg_train['flux_by_flux_ratio_sq_sum'] / agg_train['flux_ratio_sq_sum']\n# agg_train['flux_dif3'] = (agg_train['flux_max'] - agg_train['flux_min']) / agg_train['flux_w_mean']\n\n# del agg_train['mjd_max'], agg_train['mjd_min']\n# agg_train.head()\n\n# del train\n# gc.collect()\n\n# meta_train = pd.read_csv('../input/training_set_metadata.csv')\n# full_train = agg_train.reset_index().merge(right=meta_train, how='outer', on='object_id')\n\n# if 'target' in full_train:\n#     y = full_train['target']\n#     del full_train['target']\n# classes = sorted(y.unique())\n\n# class_weight = {c: 1 for c in classes}\n# for c in [64, 15]:\n#     class_weight[c] = 2\n\n# print('Unique classes : ', classes)\n\n# if 'object_id' in full_train:\n#     oof_df = full_train[['object_id']]\n#     del full_train['object_id'], full_train['distmod'], full_train['hostgal_specz']\n#     del full_train['ra'], full_train['decl'], full_train['gal_l'], full_train['gal_b'], full_train['ddf']\n\n# train_mean = full_train.mean(axis=0)\n# full_train.fillna(train_mean, inplace=True)\n\n# folds = StratifiedKFold(n_splits=5, shuffle=True, random_state=1)\n\n# full_train_new = full_train.copy()\n# ss = StandardScaler()\n# full_train_ss = ss.fit_transform(full_train_new)\n\n# clfs = []\n# oof_preds = np.zeros((len(full_train_ss), len(classes)))\n\n# for fold_, (trn_, val_) in enumerate(folds.split(y, y)):\n#     x_train, y_train = full_train_ss[trn_], y.iloc[trn_]\n#     x_valid, y_valid = full_train_ss[val_], y.iloc[val_]\n\n#     rf_classifier = RandomForestClassifier(n_estimators=100, class_weight=class_weight, random_state=42)\n#     rf_classifier.fit(x_train, y_train)\n\n#     oof_preds[val_, :] = rf_classifier.predict_proba(x_valid)\n\n#     clfs.append(rf_classifier)\n\n# # Calculate and print multi-weighted log loss\n# print('MULTI WEIGHTED LOG LOSS : %.5f ' % log_loss(to_categorical(y), oof_preds))\n\n# # Plot confusion matrix\n# cnf_matrix = confusion_matrix(y, np.argmax(oof_preds, axis=-1))\n# plot_confusion_matrix(cnf_matrix, classes=classes, normalize=True)\n\n# # Define the plot_confusion_matrix function\n# def plot_confusion_matrix(cm, classes, normalize=False, title='Confusion matrix', cmap=plt.cm.Blues):\n#     if normalize:\n#         cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n#         print(\"Normalized confusion matrix\")\n#     else:\n#         print('Confusion matrix, without normalization')\n\n#     print(cm)\n\n#     plt.imshow(cm, interpolation='nearest', cmap=cmap)\n#     plt.title(title)\n#     plt.colorbar()\n#     tick_marks = np.arange(len(classes))\n#     plt.xticks(tick_marks, classes, rotation=45)\n#     plt.yticks(tick_marks, classes)\n\n#     fmt = '.2f' if normalize else 'd'\n#     thresh = cm.max() / 2.\n#     for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n#         plt.text(j, i, format(cm[i, j], fmt), horizontalalignment=\"center\", color=\"white\" if cm[i, j] > thresh else \"black\")\n\n#     plt.ylabel('True label')\n#     plt.xlabel('Predicted label')\n#     plt.tight_layout()\n#     plt.show()\n","metadata":{"_uuid":"9e58e8719ced4eb61b9af02838e91340799cff37","trusted":true},"execution_count":null,"outputs":[]}]}