{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":41880,"databundleVersionId":5677426,"sourceType":"competition"}],"dockerImageVersionId":30627,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\n\nfrom matplotlib import pyplot as plt\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras import layers","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-04T08:19:19.457849Z","iopub.execute_input":"2024-04-04T08:19:19.458264Z","iopub.status.idle":"2024-04-04T08:19:19.463300Z","shell.execute_reply.started":"2024-04-04T08:19:19.458235Z","shell.execute_reply":"2024-04-04T08:19:19.462338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install numpy","metadata":{"execution":{"iopub.status.busy":"2024-04-04T08:19:19.465295Z","iopub.execute_input":"2024-04-04T08:19:19.465564Z","iopub.status.idle":"2024-04-04T08:19:51.250812Z","shell.execute_reply.started":"2024-04-04T08:19:19.465533Z","shell.execute_reply":"2024-04-04T08:19:51.249652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# If TPU is available\ntry:\n    # Detect TPU\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() \n    print('Running on TPU ', tpu.master())\n\nexcept:\n    tpu = None\n    \n\n# If TPU is defined\nif tpu:\n    # Initialise TPU\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    # Instantiate a TPU distribution strategy\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\n\nelse:\n    # Or instanstiate available strategy\n    strategy = tf.distribute.get_strategy() \n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T08:19:51.252196Z","iopub.execute_input":"2024-04-04T08:19:51.252498Z","iopub.status.idle":"2024-04-04T08:19:51.260191Z","shell.execute_reply.started":"2024-04-04T08:19:51.252471Z","shell.execute_reply":"2024-04-04T08:19:51.259298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_dir = \"../input/tlvmc-parkinsons-freezing-gait-prediction/\"\ntdcsfog_dir = base_dir + \"train/tdcsfog/\"\n\nsamples = []\n\n# Read in samples\nfor file in os.listdir(tdcsfog_dir):\n    samples.append(pd.read_csv(tdcsfog_dir + file, index_col=0))\n\nm = len(samples)\nprint(f\"Number of sessions: {m}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-04T08:19:51.261290Z","iopub.execute_input":"2024-04-04T08:19:51.261544Z","iopub.status.idle":"2024-04-04T08:19:59.713056Z","shell.execute_reply.started":"2024-04-04T08:19:51.261521Z","shell.execute_reply":"2024-04-04T08:19:59.711966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patches = []\nwindow = 3000\nstep = 1500\nstride = 1\n\nfor sample in samples:\n    \n    pad_idx = step - (sample.shape[0] - window) % step\n    padded_sample = pd.concat([sample, sample[:pad_idx]])\n\n    for i in range((padded_sample.shape[0] - window) // step):\n        patch = padded_sample[i * step:i * step + window:stride]\n        patches.append(patch)\n\nraw_ds = np.array(patches)\n\nprint(f\"Size of train set: {raw_ds.shape[0]}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-04T08:19:59.716039Z","iopub.execute_input":"2024-04-04T08:19:59.716408Z","iopub.status.idle":"2024-04-04T08:20:01.543758Z","shell.execute_reply.started":"2024-04-04T08:19:59.716376Z","shell.execute_reply":"2024-04-04T08:20:01.542601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Store feature columns into X array\nraw_X = raw_ds[:, :, :3]\n# Store target columns into y array\ny = raw_ds[:, :, 3:]\n\ninput_shape = raw_X[0].shape\nprint(f\"Input Shape: {input_shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-04T08:20:01.545073Z","iopub.execute_input":"2024-04-04T08:20:01.545841Z","iopub.status.idle":"2024-04-04T08:20:01.552796Z","shell.execute_reply.started":"2024-04-04T08:20:01.545800Z","shell.execute_reply":"2024-04-04T08:20:01.551909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Find triplette of minima for every sample\nmin_arr = np.min(raw_X, axis=1, keepdims=True)\n# Find triplette of maxima for every sample\nmax_arr = np.max(raw_X, axis=1, keepdims=True)\n# Normalise with MinMax rescaling\nX = (raw_X - min_arr) / (max_arr - min_arr)\n\nprint(f\"Min and Max before Norm: {np.min(raw_X):.1f} and {np.max(raw_X):.1f}\")\nprint(f\"Min and Max after Norm: {np.min(X)} and {np.max(X)}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-04T08:20:01.554178Z","iopub.execute_input":"2024-04-04T08:20:01.554530Z","iopub.status.idle":"2024-04-04T08:20:03.184263Z","shell.execute_reply.started":"2024-04-04T08:20:01.554498Z","shell.execute_reply":"2024-04-04T08:20:03.183303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Input Example:\")\nprint(\"\\n\")\n\nprint(\"Feature series:\")\nprint(X[0])\nprint(\"\\n\")\n\nprint(\"Target series:\")\nprint(y[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-04T08:20:03.185673Z","iopub.execute_input":"2024-04-04T08:20:03.186057Z","iopub.status.idle":"2024-04-04T08:20:03.193467Z","shell.execute_reply.started":"2024-04-04T08:20:03.186020Z","shell.execute_reply":"2024-04-04T08:20:03.192444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"devel_fraction = 0.1\n\ndevel_idx = np.random.choice(len(X), int(devel_fraction * len(X)), replace=False)\ntrain_idx = np.setdiff1d(np.arange(len(X)), devel_idx)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T08:20:03.194597Z","iopub.execute_input":"2024-04-04T08:20:03.194867Z","iopub.status.idle":"2024-04-04T08:20:03.206100Z","shell.execute_reply.started":"2024-04-04T08:20:03.194846Z","shell.execute_reply":"2024-04-04T08:20:03.205314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = tf.data.Dataset.from_tensor_slices((X[train_idx], y[train_idx])).batch(32)\ndevel_ds = tf.data.Dataset.from_tensor_slices((X[devel_idx], y[devel_idx])).batch(32)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T08:20:03.207286Z","iopub.execute_input":"2024-04-04T08:20:03.207568Z","iopub.status.idle":"2024-04-04T08:20:04.543438Z","shell.execute_reply.started":"2024-04-04T08:20:03.207546Z","shell.execute_reply":"2024-04-04T08:20:04.542626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set up model within scope of tpu strategy\nwith strategy.scope():\n\n    # Design LSTM model with one recurrent layer\n    model = keras.Sequential([\n        layers.Bidirectional(layers.LSTM(32, return_sequences=True, input_shape=input_shape), merge_mode=\"ave\"),\n        layers.Dense(3, activation=\"softmax\")\n    ])\n\n    # Compile model\n    model.compile(\n        optimizer=keras.optimizers.Adam(learning_rate=0.01),\n        loss=\"categorical_crossentropy\",\n        metrics=[\"categorical_accuracy\"]\n    )","metadata":{"execution":{"iopub.status.busy":"2024-04-04T08:20:04.544736Z","iopub.execute_input":"2024-04-04T08:20:04.545071Z","iopub.status.idle":"2024-04-04T08:20:04.576620Z","shell.execute_reply.started":"2024-04-04T08:20:04.545041Z","shell.execute_reply":"2024-04-04T08:20:04.575918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learning_scheduler = keras.callbacks.ReduceLROnPlateau(\n    monitor=\"val_loss\",\n    factor=0.1,\n    patience=10\n)\n\nearly_stopping = keras.callbacks.EarlyStopping(\n    monitor=\"val_loss\",\n    patience=20,\n    restore_best_weights=True,\n    mode=\"min\")\n\n\n# Train model\nhistory = model.fit(\n    train_ds,\n    validation_data=devel_ds,\n    epochs=2,\n    callbacks=[learning_scheduler, early_stopping],\n    verbose=True\n)\n\n# Store training history as a dataframe\nhistory_df = pd.DataFrame(history.history)\n\nprint(f\"Train loss: {history_df['loss'].iloc[-1]:.3f}\")\nprint(f\"Train accuracy: {history_df['categorical_accuracy'].iloc[-1]:.3f}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-04T08:20:04.577881Z","iopub.execute_input":"2024-04-04T08:20:04.578256Z","iopub.status.idle":"2024-04-04T08:20:37.568659Z","shell.execute_reply.started":"2024-04-04T08:20:04.578223Z","shell.execute_reply":"2024-04-04T08:20:37.567647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2)\n\nfig.set_figheight(3)\nfig.set_figwidth(10)\n\n# Visualise performance\nhistory_df.loc[:, [\"loss\", \"val_loss\"]].plot(ax=axes[0])\naxes[0].set_xlabel(\"Number of Epochs\")\naxes[0].set_ylabel(\"Crossentropy Loss\")\n\nhistory_df.loc[:, [\"categorical_accuracy\", \"val_categorical_accuracy\"]].plot(ax=axes[1])\naxes[1].set_xlabel(\"Number of Epochs\")\naxes[1].set_ylabel(\"Accuracy\")\n\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T08:20:37.569918Z","iopub.execute_input":"2024-04-04T08:20:37.570306Z","iopub.status.idle":"2024-04-04T08:20:38.271540Z","shell.execute_reply.started":"2024-04-04T08:20:37.570276Z","shell.execute_reply":"2024-04-04T08:20:38.270625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Store test id\ntest_id = \"003f117e14\"\n# Read in test session\ntest_sample = pd.read_csv(base_dir + \"test/tdcsfog/\" + test_id + \".csv\")\n\n# Predict labels\ny_pred = model.predict(test_sample.drop(\"Time\", axis=1).to_numpy().reshape(1, -1, 3)).reshape(-1, 3)\n\nprint(f\"Predicted time points: {y_pred.shape[1]}\")\nprint(\"Example Predictions:\")\nprint(y_pred)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T08:20:38.275376Z","iopub.execute_input":"2024-04-04T08:20:38.275776Z","iopub.status.idle":"2024-04-04T08:20:40.002435Z","shell.execute_reply.started":"2024-04-04T08:20:38.275749Z","shell.execute_reply":"2024-04-04T08:20:40.001471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Collect predictions into dataframe\nsubmission_df = pd.DataFrame(\n    y_pred,\n    columns=[\"StartHesitation\", \"Turn\", \"Walking\"]\n)\n\n# Produce Id column\nid_col = [test_id + \"_\" + str(i) for i in range(submission_df.shape[0])]\n# Add Id column to dataframe\nsubmission_df.insert(0, \"Id\", id_col)\n\n# Write submission file\nsubmission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T08:20:40.003735Z","iopub.execute_input":"2024-04-04T08:20:40.004223Z","iopub.status.idle":"2024-04-04T08:20:40.044997Z","shell.execute_reply.started":"2024-04-04T08:20:40.004185Z","shell.execute_reply":"2024-04-04T08:20:40.044263Z"},"trusted":true},"execution_count":null,"outputs":[]}]}