{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\n\nimport plotly.graph_objects as go\n\nimport pandas as pd\n\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Input, Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2022-06-16T09:56:03.136808Z","iopub.status.busy":"2022-06-16T09:56:03.136359Z","iopub.status.idle":"2022-06-16T09:56:10.930649Z","shell.execute_reply":"2022-06-16T09:56:10.929387Z"},"papermill":{"duration":7.803896,"end_time":"2022-06-16T09:56:10.933590","exception":false,"start_time":"2022-06-16T09:56:03.129694","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Data","metadata":{"papermill":{"duration":0.003728,"end_time":"2022-06-16T09:56:10.943964","exception":false,"start_time":"2022-06-16T09:56:10.940236","status":"completed"},"tags":[]}},{"cell_type":"code","source":"df = pd.read_csv('../input/hyperspectral-classification-ii/train_without_leak.csv')\ndf = df.drop(columns=['pixel_id'])","metadata":{"execution":{"iopub.execute_input":"2022-06-16T09:56:10.953763Z","iopub.status.busy":"2022-06-16T09:56:10.953069Z","iopub.status.idle":"2022-06-16T09:56:22.631273Z","shell.execute_reply":"2022-06-16T09:56:22.630301Z"},"papermill":{"duration":11.68592,"end_time":"2022-06-16T09:56:22.633814","exception":false,"start_time":"2022-06-16T09:56:10.947894","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df[df['target']!=0].iloc[:, :-1].values\n\ny = tf.keras.utils.to_categorical(df[df['target']!=0].iloc[:, -1].values , \n                                  num_classes = np.unique(df['target']).shape[0], \n                                  dtype='float32') \n\nX_train, X_test, y_train, y_test = train_test_split(X, y, train_size = 0.7, stratify = y)\n\nprint(f\"Train Data: {X_train.shape}\\nTest Data: {X_test.shape}\")","metadata":{"execution":{"iopub.execute_input":"2022-06-16T09:56:22.644029Z","iopub.status.busy":"2022-06-16T09:56:22.643644Z","iopub.status.idle":"2022-06-16T09:56:22.665318Z","shell.execute_reply":"2022-06-16T09:56:22.664382Z"},"papermill":{"duration":0.030454,"end_time":"2022-06-16T09:56:22.668732","exception":false,"start_time":"2022-06-16T09:56:22.638278","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{"papermill":{"duration":0.003878,"end_time":"2022-06-16T09:56:22.676903","exception":false,"start_time":"2022-06-16T09:56:22.673025","status":"completed"},"tags":[]}},{"cell_type":"code","source":"model = Sequential([\n    Input(shape = X_train[0].shape),\n    BatchNormalization(),\n    Dense(units = 128, activation = 'relu'),\n    Dense(units = 128, activation = 'relu'),\n    Dense(units = 128, activation = 'relu'),\n    Dense(units = 128, activation = 'relu'),\n    Dropout(rate = 0.2),\n    Dense(units = 64, activation = 'relu'),\n    Dense(units = 64, activation = 'relu'),\n    Dense(units = 64, activation = 'relu'),\n    Dense(units = 64, activation = 'relu'),\n    Dropout(rate = 0.2),\n    Dense(units = 32, activation = 'relu'),\n    Dense(units = 32, activation = 'relu'),\n    Dense(units = 32, activation = 'relu'),\n    Dense(units = 32, activation = 'relu'),\n    Dense(units = y_train.shape[1], activation = 'softmax')  \n])\n\nmodel.summary()","metadata":{"execution":{"iopub.execute_input":"2022-06-16T09:56:22.686602Z","iopub.status.busy":"2022-06-16T09:56:22.686217Z","iopub.status.idle":"2022-06-16T09:56:22.915452Z","shell.execute_reply":"2022-06-16T09:56:22.914216Z"},"papermill":{"duration":0.237881,"end_time":"2022-06-16T09:56:22.918835","exception":false,"start_time":"2022-06-16T09:56:22.680954","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer = 'adam', loss = 'categorical_crossentropy', metrics = ['accuracy'])\n\nes = EarlyStopping(monitor = 'val_loss',\n                   min_delta = 0,\n                   patience = 15,\n                   verbose = 1,\n                   restore_best_weights = True)\n\ncheckpoint = ModelCheckpoint(filepath = 'SPIE_model.h5', \n                             monitor = 'val_loss', \n                             mode ='min', \n                             save_best_only = True,\n                             verbose = 1)\n\nhistory = model.fit(x = X_train, \n          y = y_train,\n          validation_data = (X_test, y_test), \n          epochs = 50,\n          callbacks = [es, checkpoint])","metadata":{"execution":{"iopub.execute_input":"2022-06-16T09:56:22.930246Z","iopub.status.busy":"2022-06-16T09:56:22.929283Z","iopub.status.idle":"2022-06-16T09:56:34.448052Z","shell.execute_reply":"2022-06-16T09:56:34.447028Z"},"papermill":{"duration":11.526807,"end_time":"2022-06-16T09:56:34.450527","exception":false,"start_time":"2022-06-16T09:56:22.923720","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hist = pd.DataFrame(data= history.history)\n\nfig = go.Figure()\n\nfig.add_trace(go.Scatter(x = hist.index, y = hist.loss.values,\n                    mode='lines+markers',\n                    name='Train Loss'))\n\nfig.add_trace(go.Scatter(x = hist.index, y = hist.accuracy.values,\n                    mode='lines+markers',\n                    name='Train Accuracy'))\n\nfig.add_trace(go.Scatter(x = hist.index, y = hist.val_loss.values,\n                    mode='lines+markers', name='Test loss'))\n\nfig.add_trace(go.Scatter(x = hist.index, y = hist.val_accuracy.values,\n                    mode='lines+markers', name='Test Accuracy'))\n\nfig.show()","metadata":{"execution":{"iopub.execute_input":"2022-06-16T09:56:34.483622Z","iopub.status.busy":"2022-06-16T09:56:34.482849Z","iopub.status.idle":"2022-06-16T09:56:34.612958Z","shell.execute_reply":"2022-06-16T09:56:34.611708Z"},"papermill":{"duration":0.149705,"end_time":"2022-06-16T09:56:34.615954","exception":false,"start_time":"2022-06-16T09:56:34.466249","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction","metadata":{"papermill":{"duration":0.012834,"end_time":"2022-06-16T09:56:34.642083","exception":false,"start_time":"2022-06-16T09:56:34.629249","status":"completed"},"tags":[]}},{"cell_type":"code","source":"test = pd.read_csv('../input/hyperspectral-classification-ii/test_without_leak.csv')\ntest = test.drop(columns=['pixel_id'])\n\nprint(f\"Test Data: {test.shape}\")","metadata":{"execution":{"iopub.execute_input":"2022-06-16T09:56:34.670324Z","iopub.status.busy":"2022-06-16T09:56:34.669349Z","iopub.status.idle":"2022-06-16T09:56:36.055929Z","shell.execute_reply":"2022-06-16T09:56:36.054874Z"},"papermill":{"duration":1.402996,"end_time":"2022-06-16T09:56:36.058214","exception":false,"start_time":"2022-06-16T09:56:34.655218","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(test)\npredictions = [np.argmax(pred) for pred in predictions]","metadata":{"execution":{"iopub.execute_input":"2022-06-16T09:56:36.086603Z","iopub.status.busy":"2022-06-16T09:56:36.085684Z","iopub.status.idle":"2022-06-16T09:56:37.477994Z","shell.execute_reply":"2022-06-16T09:56:37.476821Z"},"papermill":{"duration":1.409085,"end_time":"2022-06-16T09:56:37.480476","exception":false,"start_time":"2022-06-16T09:56:36.071391","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(predictions).rename(columns={0: 'target'}).to_csv('submission.csv', index_label='pixel_id')","metadata":{"execution":{"iopub.execute_input":"2022-06-16T09:56:37.509243Z","iopub.status.busy":"2022-06-16T09:56:37.508550Z","iopub.status.idle":"2022-06-16T09:56:37.550639Z","shell.execute_reply":"2022-06-16T09:56:37.549565Z"},"papermill":{"duration":0.058749,"end_time":"2022-06-16T09:56:37.553101","exception":false,"start_time":"2022-06-16T09:56:37.494352","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}