{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-13T17:30:46.921088Z","iopub.execute_input":"2023-05-13T17:30:46.922054Z","iopub.status.idle":"2023-05-13T17:30:46.955053Z","shell.execute_reply.started":"2023-05-13T17:30:46.922014Z","shell.execute_reply":"2023-05-13T17:30:46.953793Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nfrom imblearn.over_sampling import RandomOverSampler\nimport glob\nimport gc\n\npath = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/\"\ntrain_defog = glob.glob(path + 'train/defog/*.csv')\ntrain_tdcsfog = glob.glob(path + 'train/tdcsfog/*.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:30:46.957564Z","iopub.execute_input":"2023-05-13T17:30:46.958239Z","iopub.status.idle":"2023-05-13T17:30:46.971334Z","shell.execute_reply.started":"2023-05-13T17:30:46.958198Z","shell.execute_reply":"2023-05-13T17:30:46.970457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_defog)\nprint(train_tdcsfog)","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:30:46.972952Z","iopub.execute_input":"2023-05-13T17:30:46.973914Z","iopub.status.idle":"2023-05-13T17:30:46.981297Z","shell.execute_reply.started":"2023-05-13T17:30:46.973873Z","shell.execute_reply":"2023-05-13T17:30:46.980093Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_data(f):\n    # Read the CSV file into a DataFrame\n    df = pd.read_csv(f)\n    \n    # Extract the file name without the extension and set it as the 'Id' column\n    df['Id'] = f.split('/')[-1].split('.')[0]\n    \n    # Extract the parent directory name and set it as the 'data_type' column\n    df['data_type'] = f.split('/')[-2]\n    \n    # Return the modified DataFrame\n    return df\n","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:30:46.983062Z","iopub.execute_input":"2023-05-13T17:30:46.983898Z","iopub.status.idle":"2023-05-13T17:30:46.994559Z","shell.execute_reply.started":"2023-05-13T17:30:46.983866Z","shell.execute_reply":"2023-05-13T17:30:46.993527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_train_defog = pd.concat([get_data(f) for f in train_defog])\n","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:30:46.998404Z","iopub.execute_input":"2023-05-13T17:30:46.999422Z","iopub.status.idle":"2023-05-13T17:31:15.820113Z","shell.execute_reply.started":"2023-05-13T17:30:46.999389Z","shell.execute_reply":"2023-05-13T17:31:15.819079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train_defog)","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:31:15.821793Z","iopub.execute_input":"2023-05-13T17:31:15.822237Z","iopub.status.idle":"2023-05-13T17:31:15.835414Z","shell.execute_reply.started":"2023-05-13T17:31:15.822193Z","shell.execute_reply":"2023-05-13T17:31:15.834344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_tdcsfog = pd.concat([get_data(f) for f in train_tdcsfog])\nprint(df_train_tdcsfog)","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:31:15.836990Z","iopub.execute_input":"2023-05-13T17:31:15.837379Z","iopub.status.idle":"2023-05-13T17:31:31.219835Z","shell.execute_reply.started":"2023-05-13T17:31:15.837349Z","shell.execute_reply":"2023-05-13T17:31:31.218761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.concat([df_train_defog, df_train_tdcsfog])\nprint(df_train.columns)","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:31:31.221434Z","iopub.execute_input":"2023-05-13T17:31:31.221878Z","iopub.status.idle":"2023-05-13T17:31:35.032314Z","shell.execute_reply.started":"2023-05-13T17:31:31.221837Z","shell.execute_reply":"2023-05-13T17:31:35.031285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['Task']","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:31:35.033437Z","iopub.execute_input":"2023-05-13T17:31:35.033760Z","iopub.status.idle":"2023-05-13T17:31:35.041544Z","shell.execute_reply.started":"2023-05-13T17:31:35.033733Z","shell.execute_reply":"2023-05-13T17:31:35.040657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# avoid object-dtype columns with all-bool values may not be included in reductions when bool_only=True is specified.\ndf_train['Valid'] = df_train['Valid'].astype(bool)\ndf_train['Task'] = df_train['Task'].astype(bool)","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:31:35.042856Z","iopub.execute_input":"2023-05-13T17:31:35.043193Z","iopub.status.idle":"2023-05-13T17:31:37.730508Z","shell.execute_reply.started":"2023-05-13T17:31:35.043163Z","shell.execute_reply":"2023-05-13T17:31:37.729249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.fillna(0, inplace=True)\nprint(df_train)","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:31:37.732129Z","iopub.execute_input":"2023-05-13T17:31:37.732465Z","iopub.status.idle":"2023-05-13T17:31:53.071031Z","shell.execute_reply.started":"2023-05-13T17:31:37.732435Z","shell.execute_reply":"2023-05-13T17:31:53.069724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfeatures = ['Time', 'AccV', 'AccML', 'AccAP']\ntargets = ['StartHesitation', 'Turn', 'Walking']\n\nX_train, X_valid, y_train, y_valid = train_test_split(df_train[features], df_train[targets], test_size=0.3, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:31:53.072271Z","iopub.execute_input":"2023-05-13T17:31:53.072605Z","iopub.status.idle":"2023-05-13T17:31:58.855096Z","shell.execute_reply.started":"2023-05-13T17:31:53.072564Z","shell.execute_reply":"2023-05-13T17:31:58.854135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_train, df_train_defog, df_train_tdcsfog\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:31:58.856552Z","iopub.execute_input":"2023-05-13T17:31:58.856999Z","iopub.status.idle":"2023-05-13T17:31:59.942411Z","shell.execute_reply.started":"2023-05-13T17:31:58.856956Z","shell.execute_reply":"2023-05-13T17:31:59.941189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check target variable values\nprint(np.isnan(y_train).sum())\nprint(np.isnan(y_valid).sum())\n\n# Check for class imbalance\nprint(\"---------------\")\nprint(y_train.sum())\nprint(y_valid.sum())","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:31:59.946208Z","iopub.execute_input":"2023-05-13T17:31:59.946586Z","iopub.status.idle":"2023-05-13T17:32:00.053120Z","shell.execute_reply.started":"2023-05-13T17:31:59.946554Z","shell.execute_reply":"2023-05-13T17:32:00.051991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# oversampler = RandomOverSampler(random_state=42)\n# X_train_resampled, y_train_resampled = oversampler.fit_resample(X_train, y_train)\n# X_valid_resampled, y_valid_resampled = oversampler.fit_resample(X_valid, y_valid)","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:32:00.054435Z","iopub.execute_input":"2023-05-13T17:32:00.054788Z","iopub.status.idle":"2023-05-13T17:32:00.060754Z","shell.execute_reply.started":"2023-05-13T17:32:00.054757Z","shell.execute_reply":"2023-05-13T17:32:00.059445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model \nmodel_reg = RandomForestRegressor(n_estimators=10, max_depth=2, random_state=42, n_jobs=-1) \nmodel_reg.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:32:00.062219Z","iopub.execute_input":"2023-05-13T17:32:00.062649Z","iopub.status.idle":"2023-05-13T17:33:20.579874Z","shell.execute_reply.started":"2023-05-13T17:32:00.062592Z","shell.execute_reply":"2023-05-13T17:33:20.578734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### TODO\nGradient Boosting Regressor: Gradient Boosting models, such as XGBoost, LightGBM, or CatBoost\nSupport Vector Regressor (SVR)\nNeural Network Regressor: Deep learning models, such as Multi-Layer Perceptron (MLP) or Recurrent Neural Networks (RNN)\nDecision Tree Regressor with randomforest or gradient boost\nBayesian Regression: Bayesian regression models, such as Bayesian Ridge Regression or Gaussian Process Regression","metadata":{}},{"cell_type":"code","source":"# Evaluate the model\ny_pred = np.clip(model_reg.predict(X_valid), 0.0, 1.0)\naverage_precision = metrics.average_precision_score(y_valid, y_pred)\nprint(\"Average Precision Score:\", average_precision)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:33:20.581333Z","iopub.execute_input":"2023-05-13T17:33:20.581655Z","iopub.status.idle":"2023-05-13T17:33:25.053899Z","shell.execute_reply.started":"2023-05-13T17:33:20.581623Z","shell.execute_reply":"2023-05-13T17:33:25.052674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Generate submission file for test data\ntest_files = glob.glob(path + 'test/tdcsfog/*.csv') \nif len(test_files) == 0:\n    raise ValueError(\"No test files found\")\n\nsubmission = []\nfor file in test_files:\n    df_test = pd.read_csv(file).fillna(0)\n    file_id = file.split('/')[-1].split('.')[0]\n    predictions = np.round(model_reg.predict(df_test[features]), 3)\n    df_submission = pd.DataFrame({'Id': file_id + '_' + df_test['Time'].astype(str)})\n    df_submission[targets] = predictions\n    submission.append(df_submission)\n\nsubmission = pd.concat(submission)\nsubmission.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:37:46.527041Z","iopub.execute_input":"2023-05-13T17:37:46.527474Z","iopub.status.idle":"2023-05-13T17:37:46.589890Z","shell.execute_reply.started":"2023-05-13T17:37:46.527436Z","shell.execute_reply":"2023-05-13T17:37:46.588762Z"},"trusted":true},"execution_count":null,"outputs":[]}]}