{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport glob\nfrom sklearn.utils import shuffle\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import classification_report, confusion_matrix\nfrom sklearn.preprocessing import StandardScaler\nimport joblib","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:26:53.157388Z","iopub.execute_input":"2023-05-28T15:26:53.157834Z","iopub.status.idle":"2023-05-28T15:26:54.375855Z","shell.execute_reply.started":"2023-05-28T15:26:53.157799Z","shell.execute_reply":"2023-05-28T15:26:54.374338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is a function that reduces the memory occupied by a file by converting float to int\nIt's taken from the link: https://www.kaggle.com/code/arjanso/reducing-dataframe-memory-size-by-65","metadata":{}},{"cell_type":"code","source":"def reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024 ** 2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024 ** 2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:26:54.378488Z","iopub.execute_input":"2023-05-28T15:26:54.378879Z","iopub.status.idle":"2023-05-28T15:26:54.393015Z","shell.execute_reply.started":"2023-05-28T15:26:54.378851Z","shell.execute_reply":"2023-05-28T15:26:54.391993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is a function for downsampling:","metadata":{}},{"cell_type":"code","source":"def downsample(features, target, fraction):\n    # Dividing the training sample into negative and positive objects\n    features_zeros = features[target == 0]\n    features_ones = features[target == 1]\n    target_zeros = target[target == 0]\n    target_ones = target[target == 1]\n    \n    #Discarding part from negative objects and creating a new training sample\n    features_downsampled = pd.concat(\n        [features_zeros.sample(frac=fraction, random_state=12345)] + [features_ones])\n    target_downsampled = pd.concat(\n        [target_zeros.sample(frac=fraction, random_state=12345)] + [target_ones])\n    \n    #Shuffling the data\n    features_downsampled, target_downsampled = shuffle(\n        features_downsampled, target_downsampled, random_state=12345)\n    \n    return features_downsampled, target_downsampled","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:26:54.394084Z","iopub.execute_input":"2023-05-28T15:26:54.395207Z","iopub.status.idle":"2023-05-28T15:26:54.414923Z","shell.execute_reply.started":"2023-05-28T15:26:54.395173Z","shell.execute_reply":"2023-05-28T15:26:54.413391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA: \nThe only file from which I was able to observe anything is subjects.","metadata":{}},{"cell_type":"code","source":"subjects = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:26:54.418315Z","iopub.execute_input":"2023-05-28T15:26:54.419055Z","iopub.status.idle":"2023-05-28T15:26:54.450968Z","shell.execute_reply.started":"2023-05-28T15:26:54.419015Z","shell.execute_reply":"2023-05-28T15:26:54.449832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:26:54.454768Z","iopub.execute_input":"2023-05-28T15:26:54.455338Z","iopub.status.idle":"2023-05-28T15:26:54.498335Z","shell.execute_reply.started":"2023-05-28T15:26:54.455304Z","shell.execute_reply":"2023-05-28T15:26:54.496722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since the data are duplicated due to repeated visits, I leave only the data of the first visit. Through the age and number of years since the diagnosis of the disease, I counted at what age the patient's disease was diagnosed.","metadata":{}},{"cell_type":"code","source":"subjects = subjects.loc[subjects['Visit']==1]\nsubjects['YearDx'] = subjects['Age'] - subjects['YearsSinceDx']\nsubjects_male = subjects.query('Sex == \"M\"')\nsubjects_female = subjects.query('Sex == \"F\"')","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:26:54.499748Z","iopub.execute_input":"2023-05-28T15:26:54.500560Z","iopub.status.idle":"2023-05-28T15:26:54.524041Z","shell.execute_reply.started":"2023-05-28T15:26:54.500508Z","shell.execute_reply":"2023-05-28T15:26:54.522752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The number of records is low, and there are more men among the patients","metadata":{}},{"cell_type":"code","source":"M = subjects_male.shape[0]\nF = subjects_female.shape[0]\nplt.bar(x=['Male', 'Female'], height=[M,F], edgecolor='black')","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:26:54.525254Z","iopub.execute_input":"2023-05-28T15:26:54.525963Z","iopub.status.idle":"2023-05-28T15:26:54.669381Z","shell.execute_reply.started":"2023-05-28T15:26:54.525930Z","shell.execute_reply":"2023-05-28T15:26:54.668628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that among the patients described in the file, the disease was most often diagnosed at the age of 50 years and older","metadata":{}},{"cell_type":"code","source":"subjects['YearDx'].hist(edgecolor='black', grid=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:26:54.673342Z","iopub.execute_input":"2023-05-28T15:26:54.676776Z","iopub.status.idle":"2023-05-28T15:26:54.929111Z","shell.execute_reply.started":"2023-05-28T15:26:54.676735Z","shell.execute_reply":"2023-05-28T15:26:54.927454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects_male['YearDx'].hist(edgecolor='black', grid=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:26:54.930870Z","iopub.execute_input":"2023-05-28T15:26:54.931242Z","iopub.status.idle":"2023-05-28T15:26:55.109964Z","shell.execute_reply.started":"2023-05-28T15:26:54.931204Z","shell.execute_reply":"2023-05-28T15:26:55.108865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects_female['YearDx'].hist(edgecolor='black', grid=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:26:55.111068Z","iopub.execute_input":"2023-05-28T15:26:55.111509Z","iopub.status.idle":"2023-05-28T15:26:55.283416Z","shell.execute_reply.started":"2023-05-28T15:26:55.111480Z","shell.execute_reply":"2023-05-28T15:26:55.281781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"NFOGQ is higher on average in men","metadata":{}},{"cell_type":"code","source":"male_NFOGQ = subjects_male['NFOGQ'].mean()\nfemale_NFOGQ = subjects_female['NFOGQ'].mean()\nplt.bar(x=['male', 'female'],height=[male_NFOGQ, female_NFOGQ], edgecolor='black')","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:26:55.284999Z","iopub.execute_input":"2023-05-28T15:26:55.285369Z","iopub.status.idle":"2023-05-28T15:26:55.600451Z","shell.execute_reply.started":"2023-05-28T15:26:55.285341Z","shell.execute_reply":"2023-05-28T15:26:55.598912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"UPDRSIII is, on average, higher in patients not taking medications","metadata":{}},{"cell_type":"code","source":"on = subjects['UPDRSIII_On'].mean()\noff = subjects['UPDRSIII_Off'].mean()\nplt.bar(x=['on', 'off'], height=[on, off], edgecolor='black')","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:26:55.602628Z","iopub.execute_input":"2023-05-28T15:26:55.603042Z","iopub.status.idle":"2023-05-28T15:26:55.718586Z","shell.execute_reply.started":"2023-05-28T15:26:55.603006Z","shell.execute_reply":"2023-05-28T15:26:55.717831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is some correlation between the number of years since diagnosis and the NFOGQ score","metadata":{}},{"cell_type":"code","source":"sns.heatmap(subjects.corr())","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:26:55.719977Z","iopub.execute_input":"2023-05-28T15:26:55.720494Z","iopub.status.idle":"2023-05-28T15:26:55.998648Z","shell.execute_reply.started":"2023-05-28T15:26:55.720467Z","shell.execute_reply":"2023-05-28T15:26:55.997830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model:\nI decided to teach by defog, so I combined all the files from the folder","metadata":{}},{"cell_type":"code","source":"defog = [item for item in glob.glob(r'/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/*{}'.format('.csv'))]","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:26:56.003889Z","iopub.execute_input":"2023-05-28T15:26:56.004750Z","iopub.status.idle":"2023-05-28T15:26:56.024148Z","shell.execute_reply.started":"2023-05-28T15:26:56.004719Z","shell.execute_reply":"2023-05-28T15:26:56.023362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_combined = pd.DataFrame()\nfor file in defog:\n    file = pd.read_csv(file)\n    file.Time = file.Time/(len(file) - 1)\n    defog_combined = pd.concat([defog_combined, file])\n    ","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:26:56.025677Z","iopub.execute_input":"2023-05-28T15:26:56.026353Z","iopub.status.idle":"2023-05-28T15:27:20.186412Z","shell.execute_reply.started":"2023-05-28T15:26:56.026319Z","shell.execute_reply":"2023-05-28T15:27:20.184983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Then I applied a function that reduces the amount of memory occupied","metadata":{}},{"cell_type":"code","source":"defog_combined = reduce_memory_usage(defog_combined)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:27:20.187672Z","iopub.execute_input":"2023-05-28T15:27:20.187949Z","iopub.status.idle":"2023-05-28T15:27:20.848700Z","shell.execute_reply.started":"2023-05-28T15:27:20.187926Z","shell.execute_reply":"2023-05-28T15:27:20.847062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next I left only the true data ('Valid' = True and 'Task' = True)","metadata":{}},{"cell_type":"code","source":"defog_combined = defog_combined.loc[(defog_combined['Valid'] == 1) & \n                                    (defog_combined['Task'] == 1)]","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:27:20.850756Z","iopub.execute_input":"2023-05-28T15:27:20.851128Z","iopub.status.idle":"2023-05-28T15:27:21.304428Z","shell.execute_reply.started":"2023-05-28T15:27:20.851095Z","shell.execute_reply":"2023-05-28T15:27:21.303471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checked the signs for correlation, no obvious correlations","metadata":{}},{"cell_type":"code","source":"sns.heatmap(defog_combined.corr())","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:27:21.306415Z","iopub.execute_input":"2023-05-28T15:27:21.306725Z","iopub.status.idle":"2023-05-28T15:27:22.371445Z","shell.execute_reply.started":"2023-05-28T15:27:21.306696Z","shell.execute_reply":"2023-05-28T15:27:22.370442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Divided the data into target attribute and other attributes","metadata":{}},{"cell_type":"code","source":"features_hesitation = defog_combined[['Time','AccV', 'AccML', 'AccAP']]\nfeatures_turn = defog_combined[['Time','AccV', 'AccML', 'AccAP']]\nfeatures_walking = defog_combined[['Time','AccV', 'AccML', 'AccAP']]","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:27:22.372602Z","iopub.execute_input":"2023-05-28T15:27:22.372928Z","iopub.status.idle":"2023-05-28T15:27:22.423297Z","shell.execute_reply.started":"2023-05-28T15:27:22.372898Z","shell.execute_reply":"2023-05-28T15:27:22.422208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_hesitation = defog_combined['StartHesitation']\ntarget_turn = defog_combined['Turn']\ntarget_walking = defog_combined['Walking']","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:27:22.425489Z","iopub.execute_input":"2023-05-28T15:27:22.425867Z","iopub.status.idle":"2023-05-28T15:27:22.431888Z","shell.execute_reply.started":"2023-05-28T15:27:22.425839Z","shell.execute_reply":"2023-05-28T15:27:22.430536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Divided the samples into training and validation samples","metadata":{}},{"cell_type":"code","source":"features_train_hesitation, features_valid_hesitation, target_train_hesitation, target_valid_hesitation = train_test_split(features_hesitation, target_hesitation, \n                                                                              test_size=0.25, random_state=12345, stratify=target_hesitation)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:27:22.433111Z","iopub.execute_input":"2023-05-28T15:27:22.433436Z","iopub.status.idle":"2023-05-28T15:27:24.050129Z","shell.execute_reply.started":"2023-05-28T15:27:22.433409Z","shell.execute_reply":"2023-05-28T15:27:24.049032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_train_turn, features_valid_turn, target_train_turn, target_valid_turn = train_test_split(features_turn, target_turn, \n                                                                              test_size=0.25, random_state=12345, stratify=target_turn)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:27:24.051621Z","iopub.execute_input":"2023-05-28T15:27:24.051954Z","iopub.status.idle":"2023-05-28T15:27:25.738835Z","shell.execute_reply.started":"2023-05-28T15:27:24.051925Z","shell.execute_reply":"2023-05-28T15:27:25.737812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_train_walking, features_valid_walking, target_train_walking, target_valid_walking = train_test_split(features_walking, target_walking, \n                                                                              test_size=0.25, random_state=12345, stratify=target_walking)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:27:25.740044Z","iopub.execute_input":"2023-05-28T15:27:25.740446Z","iopub.status.idle":"2023-05-28T15:27:27.432762Z","shell.execute_reply.started":"2023-05-28T15:27:25.740416Z","shell.execute_reply":"2023-05-28T15:27:27.431865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since there is a lot of data, I decided to use downsampling to train a random forest, and the degree of reduction selected manually so that the model is not trained too long","metadata":{}},{"cell_type":"code","source":"features_train_dwwalking, target_train_dwwalking = downsample(features_train_walking, target_train_walking, 0.3)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:34:03.225184Z","iopub.execute_input":"2023-05-28T15:34:03.225653Z","iopub.status.idle":"2023-05-28T15:34:03.886356Z","shell.execute_reply.started":"2023-05-28T15:34:03.225623Z","shell.execute_reply":"2023-05-28T15:34:03.885145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_train_dwwalking.hist()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:34:03.889075Z","iopub.execute_input":"2023-05-28T15:34:03.889508Z","iopub.status.idle":"2023-05-28T15:34:04.080276Z","shell.execute_reply.started":"2023-05-28T15:34:03.889477Z","shell.execute_reply":"2023-05-28T15:34:04.078922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_walking = RandomForestClassifier(random_state=12345, n_estimators = 40)\nmodel_walking.fit(features_train_dwwalking, target_train_dwwalking)\npredicted_valid_dwwalking = model_walking.predict(features_valid_walking)\nf1 = f1_score(target_valid_walking, predicted_valid_dwwalking)\nprint('f1 score', f1)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:34:04.081793Z","iopub.execute_input":"2023-05-28T15:34:04.082655Z","iopub.status.idle":"2023-05-28T15:36:37.482461Z","shell.execute_reply.started":"2023-05-28T15:34:04.082601Z","shell.execute_reply":"2023-05-28T15:36:37.481550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(target_valid_walking, predicted_valid_dwwalking))","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:36:37.484162Z","iopub.execute_input":"2023-05-28T15:36:37.484426Z","iopub.status.idle":"2023-05-28T15:36:38.758495Z","shell.execute_reply.started":"2023-05-28T15:36:37.484405Z","shell.execute_reply":"2023-05-28T15:36:38.757556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_train_dwhesitate, target_train_dwhesitate = downsample(features_train_hesitation, target_train_hesitation, 0.3)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:36:38.759506Z","iopub.execute_input":"2023-05-28T15:36:38.759763Z","iopub.status.idle":"2023-05-28T15:36:39.312286Z","shell.execute_reply.started":"2023-05-28T15:36:38.759742Z","shell.execute_reply":"2023-05-28T15:36:39.311034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_train_dwhesitate.hist()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:36:39.313729Z","iopub.execute_input":"2023-05-28T15:36:39.314034Z","iopub.status.idle":"2023-05-28T15:36:39.502841Z","shell.execute_reply.started":"2023-05-28T15:36:39.314011Z","shell.execute_reply":"2023-05-28T15:36:39.500883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_hesitate = RandomForestClassifier(random_state=12345, n_estimators = 100)\nmodel_hesitate.fit(features_train_dwhesitate, target_train_dwhesitate)\npredicted_valid_dwhesitate = model_hesitate.predict(features_valid_hesitation)\nf1 = f1_score(target_valid_hesitation, predicted_valid_dwhesitate)\nprint('f1 score', f1)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:36:39.505546Z","iopub.execute_input":"2023-05-28T15:36:39.505875Z","iopub.status.idle":"2023-05-28T15:39:33.907392Z","shell.execute_reply.started":"2023-05-28T15:36:39.505842Z","shell.execute_reply":"2023-05-28T15:39:33.905351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(target_valid_hesitation, predicted_valid_dwhesitate))","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:39:33.911227Z","iopub.execute_input":"2023-05-28T15:39:33.911490Z","iopub.status.idle":"2023-05-28T15:39:35.150142Z","shell.execute_reply.started":"2023-05-28T15:39:33.911469Z","shell.execute_reply":"2023-05-28T15:39:35.148922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_train_dwturn, target_train_dwturn = downsample(features_train_turn, target_train_turn, 0.25)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:39:35.151411Z","iopub.execute_input":"2023-05-28T15:39:35.152244Z","iopub.status.idle":"2023-05-28T15:39:35.709866Z","shell.execute_reply.started":"2023-05-28T15:39:35.152211Z","shell.execute_reply":"2023-05-28T15:39:35.708953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_train_dwturn.hist()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:39:35.710989Z","iopub.execute_input":"2023-05-28T15:39:35.711216Z","iopub.status.idle":"2023-05-28T15:39:35.923075Z","shell.execute_reply.started":"2023-05-28T15:39:35.711195Z","shell.execute_reply":"2023-05-28T15:39:35.921894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_turn = RandomForestClassifier(random_state=12345, n_estimators = 40)\nmodel_turn.fit(features_train_dwturn, target_train_dwturn)\npredicted_valid_dwturn = model_turn.predict(features_valid_turn)\nf1 = f1_score(target_valid_turn, predicted_valid_dwturn)\nprint('Модель: лес, f1', f1)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:39:35.924433Z","iopub.execute_input":"2023-05-28T15:39:35.924738Z","iopub.status.idle":"2023-05-28T15:43:06.138319Z","shell.execute_reply.started":"2023-05-28T15:39:35.924706Z","shell.execute_reply":"2023-05-28T15:43:06.136560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(target_valid_turn, predicted_valid_dwturn))","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:43:06.139835Z","iopub.execute_input":"2023-05-28T15:43:06.140193Z","iopub.status.idle":"2023-05-28T15:43:07.587716Z","shell.execute_reply.started":"2023-05-28T15:43:06.140164Z","shell.execute_reply":"2023-05-28T15:43:07.586555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The results are not bad, but the main problem with the model is that it takes time into account for prediction. I think this will skew the data where events are in other time periods.","metadata":{}},{"cell_type":"markdown","source":"# Test: \nI opened tdcsfog and defog data from the test folder. They have different units for the sensors: g and m/s^2","metadata":{}},{"cell_type":"code","source":"tdcsfog_test = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog/003f117e14.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:43:07.588999Z","iopub.execute_input":"2023-05-28T15:43:07.589321Z","iopub.status.idle":"2023-05-28T15:43:07.608462Z","shell.execute_reply.started":"2023-05-28T15:43:07.589300Z","shell.execute_reply":"2023-05-28T15:43:07.607551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_test = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog/02ab235146.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:43:07.609816Z","iopub.execute_input":"2023-05-28T15:43:07.610066Z","iopub.status.idle":"2023-05-28T15:43:07.906369Z","shell.execute_reply.started":"2023-05-28T15:43:07.610045Z","shell.execute_reply":"2023-05-28T15:43:07.904922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_test","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:43:07.910518Z","iopub.execute_input":"2023-05-28T15:43:07.911202Z","iopub.status.idle":"2023-05-28T15:43:07.924054Z","shell.execute_reply.started":"2023-05-28T15:43:07.911167Z","shell.execute_reply":"2023-05-28T15:43:07.922928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So, I converted the data from tdcsfog to a suitable format and normalized the time as in training the model","metadata":{}},{"cell_type":"code","source":"tdcsfog_test['AccV'] = tdcsfog_test['AccV'] / 9.80665\ntdcsfog_test['AccML'] = tdcsfog_test['AccML'] / 9.80665\ntdcsfog_test['AccAP'] = tdcsfog_test['AccAP'] / 9.80665\ntdcsfog_test.Time = tdcsfog_test.Time/(len(tdcsfog_test) - 1)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:50:59.629816Z","iopub.execute_input":"2023-05-28T15:50:59.631188Z","iopub.status.idle":"2023-05-28T15:50:59.639899Z","shell.execute_reply.started":"2023-05-28T15:50:59.631132Z","shell.execute_reply":"2023-05-28T15:50:59.638618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_test","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:50:59.965116Z","iopub.execute_input":"2023-05-28T15:50:59.965554Z","iopub.status.idle":"2023-05-28T15:50:59.979649Z","shell.execute_reply.started":"2023-05-28T15:50:59.965501Z","shell.execute_reply":"2023-05-28T15:50:59.978511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_test","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:51:01.677145Z","iopub.execute_input":"2023-05-28T15:51:01.677546Z","iopub.status.idle":"2023-05-28T15:51:01.694831Z","shell.execute_reply.started":"2023-05-28T15:51:01.677500Z","shell.execute_reply":"2023-05-28T15:51:01.693273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_test.Time = defog_test.Time/(len(defog_test) - 1)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:51:02.297027Z","iopub.execute_input":"2023-05-28T15:51:02.297439Z","iopub.status.idle":"2023-05-28T15:51:02.304362Z","shell.execute_reply.started":"2023-05-28T15:51:02.297411Z","shell.execute_reply":"2023-05-28T15:51:02.302817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_test","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:51:03.196862Z","iopub.execute_input":"2023-05-28T15:51:03.197277Z","iopub.status.idle":"2023-05-28T15:51:03.214158Z","shell.execute_reply.started":"2023-05-28T15:51:03.197248Z","shell.execute_reply":"2023-05-28T15:51:03.212286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Combined the files into a common test file","metadata":{}},{"cell_type":"code","source":"features_test = pd.concat([tdcsfog_test, defog_test], axis = 0).reset_index(drop = True)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:51:04.817015Z","iopub.execute_input":"2023-05-28T15:51:04.817386Z","iopub.status.idle":"2023-05-28T15:51:04.829475Z","shell.execute_reply.started":"2023-05-28T15:51:04.817358Z","shell.execute_reply":"2023-05-28T15:51:04.828077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_test","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:51:05.668083Z","iopub.execute_input":"2023-05-28T15:51:05.668503Z","iopub.status.idle":"2023-05-28T15:51:05.683785Z","shell.execute_reply.started":"2023-05-28T15:51:05.668474Z","shell.execute_reply":"2023-05-28T15:51:05.681910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Predicted events using the model","metadata":{}},{"cell_type":"code","source":"prediction_walking = model_walking.predict(features_test)\nprediction_turn = model_turn.predict(features_test)\nprediction_hesitate = model_hesitate.predict(features_test)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:51:09.316980Z","iopub.execute_input":"2023-05-28T15:51:09.317329Z","iopub.status.idle":"2023-05-28T15:51:12.788977Z","shell.execute_reply.started":"2023-05-28T15:51:09.317303Z","shell.execute_reply":"2023-05-28T15:51:12.788020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_test['StartHesitation'] = prediction_hesitate\nfeatures_test['Turn'] = prediction_turn\nfeatures_test['Walking'] = prediction_walking","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:51:12.790483Z","iopub.execute_input":"2023-05-28T15:51:12.790825Z","iopub.status.idle":"2023-05-28T15:51:12.796134Z","shell.execute_reply.started":"2023-05-28T15:51:12.790794Z","shell.execute_reply":"2023-05-28T15:51:12.795555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = features_test\nsubmission.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:51:44.880209Z","iopub.execute_input":"2023-05-28T15:51:44.880892Z","iopub.status.idle":"2023-05-28T15:51:46.205265Z","shell.execute_reply.started":"2023-05-28T15:51:44.880858Z","shell.execute_reply":"2023-05-28T15:51:46.204144Z"},"trusted":true},"execution_count":null,"outputs":[]}]}