{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1.: Make Predictions on the Test Data\n## 1.1.: Import Packages and Specify File Paths","metadata":{}},{"cell_type":"code","source":"# Install tsflex and seglearn package (offline)\n!pip install tsflex --no-index --find-links=file:///kaggle/input/fog-packages\n!pip install seglearn --no-index --find-links=file:///kaggle/input/fog-packages\n\n# Import packages and functions    \nimport numpy as np\nimport pathlib\nimport pandas as pd\nfrom tqdm.auto import tqdm\nfrom sklearn import *\nimport glob\nimport os\nimport gc\nimport pickle\nfrom IPython.display import FileLink\nfrom sklearn.model_selection import  train_test_split\nfrom sklearn.multioutput import MultiOutputRegressor\nfrom seglearn.feature_functions import base_features, emg_features\nfrom tsflex.features import FeatureCollection, MultipleFeatureDescriptors\nfrom tsflex.features.integrations import seglearn_feature_dict_wrapper\n\n# Define important paths\nos.chdir(r'/kaggle/working') \nfog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/'\nmetadata_path = '/kaggle/input/fog-metadata/'\nmodel_path = '/kaggle/input/fog-ensemble-regressors/' #'/kaggle/input/ensemble-regressors/'\n\n# Retrieve the paths of the testing data files\ntest_files = glob.glob(fog_path + 'test/**/**')","metadata":{"execution":{"iopub.status.busy":"2023-06-11T06:13:41.069539Z","iopub.execute_input":"2023-06-11T06:13:41.069974Z","iopub.status.idle":"2023-06-11T06:13:58.667750Z","shell.execute_reply.started":"2023-06-11T06:13:41.069940Z","shell.execute_reply":"2023-06-11T06:13:58.666596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.2.: Initialize Functions","metadata":{}},{"cell_type":"code","source":"# MultiOutputRegressor\nclass CustomMultiOutputRegressor(MultiOutputRegressor):\n    def fit(self, X, y, eval_set=None, **fit_params):\n        self.estimators_ = [clone(self.estimator) for _ in range(y.shape[1])]\n        \n        for i, estimator in enumerate(self.estimators_):\n            if eval_set:\n                fit_params['eval_set'] = [(eval_set[0], eval_set[1][:, i])]\n            estimator.fit(X, y[:, i], **fit_params)\n        \n        return self\n\n\n# Read and preprocess data + extract features\n# Adapted from : XXX\ndef reader(f):\n    \"\"\"\n    Reads a CSV file, performs data processing operations, and returns a DataFrame.\n    \n    Args:\n        f (str): File path of the CSV file.\n        tasks (pd.DataFrame): DataFrame containing tasks data.\n        metadata_complex (pd.DataFrame): DataFrame containing complex metadata.\n        fc (FeatureCollection): The feature collection object used for calculating features.\n        \n    Returns:\n        pd.DataFrame: The processed DataFrame.\n    \"\"\"\n    try:\n        # Read the CSV file with the specified index column as 'Time'\n        df = pd.read_csv(f, index_col=\"Time\")\n\n        # Extract the ID and module information from the file path\n        df['Id'] = f.split('/')[-1].split('.')[0]\n        df['Module'] = pathlib.Path(f).parts[-2]\n\n        # Calculate the fractional time index\n        df['Time_frac'] = (df.index / df.index.max()).values\n\n        # Merge with the tasks and metadata_complex DataFrames\n        df = pd.merge(df, tasks[['Id','t_kmeans']], how='left', on='Id').fillna(-1)\n        df = pd.merge(df, metadata_complex[['Id','Subject']+['Visit','Test','Medication','s_kmeans']], how='left', on='Id').fillna(-1)\n\n        # Calculate features using the provided FeatureCollection object (fc)\n        df_feats = fc.calculate(df, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\").astype(np.float32)\n\n        # Merge the calculated features with the original DataFrame\n        df = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\n\n        # Fill missing values with the previous values\n        df.fillna(method=\"ffill\", inplace=True)\n\n        # Modify the ID column to include the index values\n        df['Id'] = df['Id'].astype(str) + '_' + df.index.astype(str)\n\n        return df\n    except:\n        pass ","metadata":{"execution":{"iopub.status.busy":"2023-06-11T06:13:58.669979Z","iopub.execute_input":"2023-06-11T06:13:58.670298Z","iopub.status.idle":"2023-06-11T06:13:58.682781Z","shell.execute_reply.started":"2023-06-11T06:13:58.670268Z","shell.execute_reply":"2023-06-11T06:13:58.681387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.3.: Load the Data and Extract the Features from the Test Files\n### 1.3.1: Load the preprocessed Metadata and Ensemble Regressors","metadata":{}},{"cell_type":"code","source":"# Load metadata\nsubjects = pd.read_csv(metadata_path + 'subjects.csv')\ntasks = pd.read_csv(metadata_path + 'tasks.csv')\nmetadata_complex = pd.read_csv(metadata_path + 'metadata_complex.csv')\n\n# Load ensemble regressors with corresponding regressor weights\nestimators = pickle.load(open(model_path + 'ensemble_regressors.sav', 'rb'))\nregressor_weights = np.load(model_path + 'regressor_weights.npy')","metadata":{"execution":{"iopub.status.busy":"2023-06-11T06:13:58.686702Z","iopub.execute_input":"2023-06-11T06:13:58.687045Z","iopub.status.idle":"2023-06-11T06:14:01.127150Z","shell.execute_reply.started":"2023-06-11T06:13:58.687016Z","shell.execute_reply":"2023-06-11T06:14:01.125918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.3.2.: Create Feature Collection","metadata":{}},{"cell_type":"code","source":"# Specify the stride and window size\nwindow_size = 5_000\nstride = 5_000\n\n# Define basic_feats feature descriptors\nbasic_feats = MultipleFeatureDescriptors(\n    functions=seglearn_feature_dict_wrapper(base_features()),\n    series_names=['AccV', 'AccML', 'AccAP'],\n    windows=[window_size],\n    strides=[stride]\n)\n\n# Define emg_feats feature descriptors\nemg_feats = emg_features()\ndel emg_feats['simple square integral']  # Remove 'simple square integral' from emg_feats\n\nemg_feats = MultipleFeatureDescriptors(\n    functions=seglearn_feature_dict_wrapper(emg_feats),\n    series_names=['AccV', 'AccML', 'AccAP'],\n    windows=[window_size],\n    strides=[stride]\n)\n\n# Create a FeatureCollection by combining basic_feats and emg_feats\nfc = FeatureCollection([basic_feats, emg_feats])\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-11T06:14:01.130040Z","iopub.execute_input":"2023-06-11T06:14:01.130379Z","iopub.status.idle":"2023-06-11T06:14:01.346453Z","shell.execute_reply.started":"2023-06-11T06:14:01.130350Z","shell.execute_reply":"2023-06-11T06:14:01.345403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.3.3.: Extract Features from the Test Files","metadata":{}},{"cell_type":"code","source":"# Extract features from test set\ntest_set = pd.concat([reader(file) for file in tqdm(test_files)]).fillna(0)\ntest_set.reset_index(drop=True, inplace=True)\n\n# Select columns for modeling, prediction and submission excluding unnecessary columns\nfeature_cols = [c for c in test_set.columns if c not in ['Id', 'Subject', 'Module', 'Time', 'StartHesitation', 'Turn', 'Walking', 'Valid', 'Task', 'Event']]\nprediction_cols = ['StartHesitation', 'Turn', 'Walking']\nsubmission_cols = ['Id', 'StartHesitation', 'Turn', 'Walking']\ntest_set","metadata":{"execution":{"iopub.status.busy":"2023-06-11T06:14:01.347603Z","iopub.execute_input":"2023-06-11T06:14:01.347881Z","iopub.status.idle":"2023-06-11T06:14:04.819902Z","shell.execute_reply.started":"2023-06-11T06:14:01.347859Z","shell.execute_reply":"2023-06-11T06:14:04.818834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#estimators = [estimators[0], estimators[3], estimators[4]]\nestimators","metadata":{"execution":{"iopub.status.busy":"2023-06-11T06:14:04.821357Z","iopub.execute_input":"2023-06-11T06:14:04.822141Z","iopub.status.idle":"2023-06-11T06:14:04.840598Z","shell.execute_reply.started":"2023-06-11T06:14:04.822107Z","shell.execute_reply":"2023-06-11T06:14:04.839182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#regressor_weights = [regressor_weights[0], regressor_weights[3], regressor_weights[4]]\nregressor_weights","metadata":{"execution":{"iopub.status.busy":"2023-06-11T06:14:04.842001Z","iopub.execute_input":"2023-06-11T06:14:04.843304Z","iopub.status.idle":"2023-06-11T06:14:04.850259Z","shell.execute_reply.started":"2023-06-11T06:14:04.843248Z","shell.execute_reply":"2023-06-11T06:14:04.849113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.2.: Make final predictions with the Averaging - Ensemble and Create Submission","metadata":{}},{"cell_type":"code","source":"weighted = True\n\nestimator_predictions = []\nensemble_predictions = []\naverage_precision = []\n\n# Iterate over the folds\nfor fold in tqdm(range(len(estimators[0]))):\n    estimator_predictions = []\n    \n    # Iterate over the different regressors\n    for estimator in estimators:\n        estimator_prediction = estimator[fold].predict(test_set[feature_cols]).clip(0.0, 1.0)\n        estimator_predictions.append(estimator_prediction)\n    \n    # Compute the (weighted) average predictions over the different regressors\n    if weighted == True:\n        ensemble_prediction = np.average(estimator_predictions, axis = 0, weights = regressor_weights)\n        ensemble_predictions.append(np.average(estimator_predictions, axis = 0, weights = regressor_weights))\n        \n    else:\n        ensemble_prediction = np.average(estimator_predictions, axis = 0)\n    ensemble_predictions.append(ensemble_prediction)\n\n# Create final prediction by averaging over the different folds\nfinal_predictions = np.mean(ensemble_predictions, axis = 0)\n\n# Create final submission\nresults = pd.DataFrame(final_predictions, columns=prediction_cols)\nresults = pd.concat([test_set, results], axis=1)\nsubmission = results[submission_cols]\nsubmission.to_csv('submission.csv', index=False)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2023-06-11T06:14:23.256266Z","iopub.execute_input":"2023-06-11T06:14:23.256649Z","iopub.status.idle":"2023-06-11T06:15:02.498340Z","shell.execute_reply.started":"2023-06-11T06:14:23.256623Z","shell.execute_reply":"2023-06-11T06:15:02.497458Z"},"trusted":true},"execution_count":null,"outputs":[]}]}