{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Here I will consider only tdsfog, I will look at the balance of the two binary components of the target trait. I will also combine them with the available information about the subjects of the study and draw conclusions about their impact on features.**","metadata":{}},{"cell_type":"code","source":"import random\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nimport seaborn as sns\nimport warnings\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, precision_score, average_precision_score\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestRegressor\n\nwarnings.simplefilter(action='ignore')","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:17:38.427335Z","iopub.execute_input":"2023-06-08T06:17:38.428586Z","iopub.status.idle":"2023-06-08T06:17:41.528284Z","shell.execute_reply.started":"2023-06-08T06:17:38.428527Z","shell.execute_reply":"2023-06-08T06:17:41.526881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = Path(r\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog\")","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:17:42.258154Z","iopub.execute_input":"2023-06-08T06:17:42.259460Z","iopub.status.idle":"2023-06-08T06:17:42.265662Z","shell.execute_reply.started":"2023-06-08T06:17:42.259402Z","shell.execute_reply":"2023-06-08T06:17:42.263915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_df=[]\nfor file in path.glob('*.csv'):\n    df = pd.read_csv(file, index_col=[0])\n    df['Id'] = file.name\n    all_df.append(df)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:17:44.035469Z","iopub.execute_input":"2023-06-08T06:17:44.036036Z","iopub.status.idle":"2023-06-08T06:18:06.367467Z","shell.execute_reply.started":"2023-06-08T06:17:44.035992Z","shell.execute_reply":"2023-06-08T06:18:06.366298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.DataFrame()\nfor x in all_df:\n    data = pd.concat([data, x])","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:18:06.369500Z","iopub.execute_input":"2023-06-08T06:18:06.370154Z","iopub.status.idle":"2023-06-08T06:20:41.680059Z","shell.execute_reply.started":"2023-06-08T06:18:06.370117Z","shell.execute_reply":"2023-06-08T06:20:41.678675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.Id = data.Id.str.replace('.csv', '')","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:20:41.681973Z","iopub.execute_input":"2023-06-08T06:20:41.682424Z","iopub.status.idle":"2023-06-08T06:20:53.832742Z","shell.execute_reply.started":"2023-06-08T06:20:41.682388Z","shell.execute_reply":"2023-06-08T06:20:53.830930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('The ratio of positive and negative reactions in the target')\nfig, axs = plt.subplots(ncols=3, figsize=(10, 3))\nautopct='%.0f'\nfont_size=15\ndata.StartHesitation.value_counts().plot.pie(autopct=autopct,fontsize=font_size, ax=axs[0])\ndata.Turn.value_counts().plot.pie(autopct=autopct, fontsize=font_size, ax=axs[1])\ndata.Walking.value_counts().plot.pie(autopct=autopct, fontsize=font_size, ax=axs[2])\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:20:53.835828Z","iopub.execute_input":"2023-06-08T06:20:53.836251Z","iopub.status.idle":"2023-06-08T06:20:54.794535Z","shell.execute_reply.started":"2023-06-08T06:20:53.836217Z","shell.execute_reply":"2023-06-08T06:20:54.792980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:20:54.796177Z","iopub.execute_input":"2023-06-08T06:20:54.796739Z","iopub.status.idle":"2023-06-08T06:20:54.814556Z","shell.execute_reply.started":"2023-06-08T06:20:54.796688Z","shell.execute_reply":"2023-06-08T06:20:54.812754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata.Medication = np.where(metadata.Medication=='on', 1, 0)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:20:54.817182Z","iopub.execute_input":"2023-06-08T06:20:54.818694Z","iopub.status.idle":"2023-06-08T06:20:54.829266Z","shell.execute_reply.started":"2023-06-08T06:20:54.818642Z","shell.execute_reply":"2023-06-08T06:20:54.827153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subject = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:20:54.831338Z","iopub.execute_input":"2023-06-08T06:20:54.831913Z","iopub.status.idle":"2023-06-08T06:20:54.856585Z","shell.execute_reply.started":"2023-06-08T06:20:54.831876Z","shell.execute_reply":"2023-06-08T06:20:54.855392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subject = subject.drop('Visit', axis=1)\nsubject.UPDRSIII_On = subject.UPDRSIII_On.fillna(subject.UPDRSIII_On.median())\nsubject.UPDRSIII_Off = subject.UPDRSIII_Off.fillna(subject.UPDRSIII_Off.median())\nsubject.Sex = np.where(subject.Sex=='M', 1, 0)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:20:54.858747Z","iopub.execute_input":"2023-06-08T06:20:54.859235Z","iopub.status.idle":"2023-06-08T06:20:54.876209Z","shell.execute_reply.started":"2023-06-08T06:20:54.859196Z","shell.execute_reply":"2023-06-08T06:20:54.874962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_id = metadata.merge(subject, on='Subject')\ndata_td = data.merge(data_id, on='Id')","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:20:54.877899Z","iopub.execute_input":"2023-06-08T06:20:54.878889Z","iopub.status.idle":"2023-06-08T06:20:58.939511Z","shell.execute_reply.started":"2023-06-08T06:20:54.878824Z","shell.execute_reply":"2023-06-08T06:20:58.938244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(font_scale=1.15)\nplt.figure(figsize=(15,4))\nsns.heatmap(\n    data_td.corr(),        \n    cmap='RdBu_r', \n    annot=True,\n    vmin=-1, vmax=1);","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:20:58.947781Z","iopub.execute_input":"2023-06-08T06:20:58.948303Z","iopub.status.idle":"2023-06-08T06:21:07.693545Z","shell.execute_reply.started":"2023-06-08T06:20:58.948269Z","shell.execute_reply":"2023-06-08T06:21:07.692192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We need to determine which features of (AccV, AccML, AccAP) have the greatest impact on the appearance of a positive reaction in the signs of StartHesitation, Turn, Walking, but to clarify the issue, we will consider the effect of other features on these parameters (+ - positive correlation b - - negative ).\nBased on the heat map, we will select the features that have the greatest impact on the parameters under study and also see what affects them:\n* on StartHesitation:\n   +NFOGO(+)\n   + YearsSinceDx(-)\n   + AccAP(+)\n   +Visit(+)\n* on Turn:\n   +NFOGO(+)\n   + AccAP(+)\n   + YearsSinceDx(-)\n   + age\n   +UPDRSIII_On(+)\n   + Medication(-)\n* on Walking:\n   +Visit(+)\n   +NFOGO(+)\n   + AccAP(+)\n\nNow let's see what has the biggest impact on (AccV, AccML, AccAP):\n* on AccV:\n   + AccAP(+)\n* on AccML:\n   +Visit(-)\n   +NFOGO(-)\n   + UPDRSIII_On(-)\n* on AccAP:\n   +NFOGO(+)\n   +Visit(+)\n   + YearsSinceDx(-)","metadata":{}},{"cell_type":"markdown","source":"Conclusions:\n- the main problem is the imbalance of the predicted classes, a positive class appears extremely rarely;\n- it is necessary to study all the available data in more detail, determine the degree of their influence on the features, and then assign each of the features its own weight when training the models.","metadata":{}},{"cell_type":"code","source":"random.seed(42)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:21:07.695168Z","iopub.execute_input":"2023-06-08T06:21:07.695770Z","iopub.status.idle":"2023-06-08T06:21:07.703252Z","shell.execute_reply.started":"2023-06-08T06:21:07.695717Z","shell.execute_reply":"2023-06-08T06:21:07.701864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = data_td[['AccV', 'AccML', 'AccAP']]\nstart = data.StartHesitation\nturn = data.Turn\nwalk = data.Walking","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:21:07.706207Z","iopub.execute_input":"2023-06-08T06:21:07.706694Z","iopub.status.idle":"2023-06-08T06:21:09.655776Z","shell.execute_reply.started":"2023-06-08T06:21:07.706657Z","shell.execute_reply":"2023-06-08T06:21:09.654257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(X, start, stratify=start,\n                                                      test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:21:09.657477Z","iopub.execute_input":"2023-06-08T06:21:09.657896Z","iopub.status.idle":"2023-06-08T06:21:15.408623Z","shell.execute_reply.started":"2023-06-08T06:21:09.657863Z","shell.execute_reply":"2023-06-08T06:21:15.407217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr = LogisticRegression()\nlr.fit(X_train, y_train)\npred_lr = lr.predict(X_valid)\nprint(accuracy_score(y_valid, pred_lr))\nprint(precision_score(y_valid, pred_lr))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:21:15.410343Z","iopub.execute_input":"2023-06-08T06:21:15.410774Z","iopub.status.idle":"2023-06-08T06:21:27.948626Z","shell.execute_reply.started":"2023-06-08T06:21:15.410738Z","shell.execute_reply":"2023-06-08T06:21:27.947281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We see that with excellent indicators of overall accuracy, the level of predicting the precision of the positive class is extremely low.\nI tried to increase the level of precision of the positive reaction of the target by applying several classification models, it takes a very long time and the maximum that I could achieve was 0.6 And this is not much! And then I saw that there are very close relationships between all the features in the training sample, and what if we try to solve this classification problem using regression.\n","metadata":{}},{"cell_type":"code","source":"sns.pairplot(X, height=1.5, aspect=1);","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:21:27.950237Z","iopub.execute_input":"2023-06-08T06:21:27.950730Z","iopub.status.idle":"2023-06-08T06:24:40.892538Z","shell.execute_reply.started":"2023-06-08T06:21:27.950685Z","shell.execute_reply":"2023-06-08T06:24:40.891164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's try to use the Random Forest model in regression and look at the resulting evaluation metric, such as MAE.","metadata":{}},{"cell_type":"code","source":"rf = RandomForestRegressor(n_estimators=40, max_depth=10, random_state=42)\nrf.fit(X_train, y_train)\npreds_start = rf.predict(X_valid)\nprint('MAE: ', mean_absolute_error(y_valid, preds_start).round(3))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:24:40.894702Z","iopub.execute_input":"2023-06-08T06:24:40.895203Z","iopub.status.idle":"2023-06-08T06:46:21.387094Z","shell.execute_reply.started":"2023-06-08T06:24:40.895163Z","shell.execute_reply":"2023-06-08T06:46:21.385609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Predictions of the Random Forest model in regression by target - StartHesitation')\nplt.figure(figsize=(20, 4))\nplt.xlim(0, 0.7)\nplt.ylim(0, 100000)\nplt.xticks(np.arange(0, 0.7, step=0.02))\nplt.hist(x=pd.Series(preds_start))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:46:21.391176Z","iopub.execute_input":"2023-06-08T06:46:21.392269Z","iopub.status.idle":"2023-06-08T06:46:23.186339Z","shell.execute_reply.started":"2023-06-08T06:46:21.392222Z","shell.execute_reply":"2023-06-08T06:46:23.184888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And now let's try to classify the received predictions and apply the precision metric to them.","metadata":{}},{"cell_type":"code","source":"pred_01 = pd.Series(np.where(preds_start<=0.1, 0, 1))\nprecision_score(y_valid, pred_01)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:46:23.188438Z","iopub.execute_input":"2023-06-08T06:46:23.189597Z","iopub.status.idle":"2023-06-08T06:46:23.765343Z","shell.execute_reply.started":"2023-06-08T06:46:23.189460Z","shell.execute_reply":"2023-06-08T06:46:23.763970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It's actually not that bad. In this situation, the most important thing is to choose a significance threshold for the predictions of the regression model. For this I wrote a function.","metadata":{}},{"cell_type":"code","source":"def binar(pred):\n    thresholds = np.arange(0, 1, step=0.2)\n    best_precision = 0\n    best_threshold = 0\n    for threshold in thresholds:\n        prediction = pd.Series(np.where(pred < threshold, 0, 1))\n        precision = average_precision_score(y_valid, prediction)\n        if precision > best_precision:\n            best_precision = precision\n            best_threshold = threshold\n    return best_precision, best_threshold","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:46:23.766794Z","iopub.execute_input":"2023-06-08T06:46:23.767197Z","iopub.status.idle":"2023-06-08T06:46:23.775395Z","shell.execute_reply.started":"2023-06-08T06:46:23.767165Z","shell.execute_reply":"2023-06-08T06:46:23.773921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And now I have already exceeded the metric obtained for many hours of the classification model, spending a couple of minutes on this)\nNow I will do the same for the other two targets.","metadata":{}},{"cell_type":"code","source":"print('Average_precision for start: ',  binar(preds_start)[0])","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:46:23.777014Z","iopub.execute_input":"2023-06-08T06:46:23.778339Z","iopub.status.idle":"2023-06-08T06:46:24.671431Z","shell.execute_reply.started":"2023-06-08T06:46:23.778285Z","shell.execute_reply":"2023-06-08T06:46:24.670322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold_start = binar(preds_start)[1]","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:46:24.672923Z","iopub.execute_input":"2023-06-08T06:46:24.673940Z","iopub.status.idle":"2023-06-08T06:46:25.546450Z","shell.execute_reply.started":"2023-06-08T06:46:24.673903Z","shell.execute_reply":"2023-06-08T06:46:25.544991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(X, turn, stratify=turn,\n                                                      test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:46:25.548019Z","iopub.execute_input":"2023-06-08T06:46:25.549131Z","iopub.status.idle":"2023-06-08T06:46:31.068763Z","shell.execute_reply.started":"2023-06-08T06:46:25.549088Z","shell.execute_reply":"2023-06-08T06:46:31.067151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = RandomForestRegressor(n_estimators=40, max_depth=10, random_state=42)\nrf.fit(X_train, y_train)\npreds_turn = rf.predict(X_valid)\nprint('MAE: ', mean_absolute_error(y_valid, preds_turn).round(3))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T06:46:31.070269Z","iopub.execute_input":"2023-06-08T06:46:31.070675Z","iopub.status.idle":"2023-06-08T07:08:39.731511Z","shell.execute_reply.started":"2023-06-08T06:46:31.070643Z","shell.execute_reply":"2023-06-08T07:08:39.729886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Predictions of the Random Forest model in regression by target - Turn')\nplt.figure(figsize=(25, 4))\nplt.xlim(0, 1)\nplt.ylim(0, 100000)\nplt.xticks(np.arange(0, 1, step=0.02))\nplt.hist(x=pd.Series(preds_turn))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:08:39.733367Z","iopub.execute_input":"2023-06-08T07:08:39.733749Z","iopub.status.idle":"2023-06-08T07:08:40.723151Z","shell.execute_reply.started":"2023-06-08T07:08:39.733719Z","shell.execute_reply":"2023-06-08T07:08:40.721845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Average_precision for turn: ',  binar(preds_turn)[0])","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:08:40.728166Z","iopub.execute_input":"2023-06-08T07:08:40.728567Z","iopub.status.idle":"2023-06-08T07:08:41.794712Z","shell.execute_reply.started":"2023-06-08T07:08:40.728536Z","shell.execute_reply":"2023-06-08T07:08:41.793179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold_turn = binar(preds_turn)[1]","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:08:41.796950Z","iopub.execute_input":"2023-06-08T07:08:41.797419Z","iopub.status.idle":"2023-06-08T07:08:42.847480Z","shell.execute_reply.started":"2023-06-08T07:08:41.797388Z","shell.execute_reply":"2023-06-08T07:08:42.846200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(X, walk, stratify=walk,\n                                                      test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:08:42.849427Z","iopub.execute_input":"2023-06-08T07:08:42.849968Z","iopub.status.idle":"2023-06-08T07:08:48.498406Z","shell.execute_reply.started":"2023-06-08T07:08:42.849919Z","shell.execute_reply":"2023-06-08T07:08:48.496905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = RandomForestRegressor(n_estimators=40, max_depth=10, random_state=42)\nrf.fit(X_train, y_train)\npreds_walk = rf.predict(X_valid)\nprint('MAE: ', mean_absolute_error(y_valid, preds_walk).round(3))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:08:48.500227Z","iopub.execute_input":"2023-06-08T07:08:48.500610Z","iopub.status.idle":"2023-06-08T07:30:57.761332Z","shell.execute_reply.started":"2023-06-08T07:08:48.500579Z","shell.execute_reply":"2023-06-08T07:30:57.759875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Predictions of the Random Forest model in regression by target - Walking')\nplt.figure(figsize=(20, 5))\nplt.xlim(0, 0.7)\nplt.ylim(0, 1000)\nplt.xticks(np.arange(0, 0.7, step=0.02))\nplt.hist(x=pd.Series(preds_walk))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:30:57.763116Z","iopub.execute_input":"2023-06-08T07:30:57.763530Z","iopub.status.idle":"2023-06-08T07:30:58.490189Z","shell.execute_reply.started":"2023-06-08T07:30:57.763494Z","shell.execute_reply":"2023-06-08T07:30:58.488795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Average_precision for walk: ',  binar(preds_walk)[0])","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:30:58.494117Z","iopub.execute_input":"2023-06-08T07:30:58.494516Z","iopub.status.idle":"2023-06-08T07:30:59.345067Z","shell.execute_reply.started":"2023-06-08T07:30:58.494484Z","shell.execute_reply":"2023-06-08T07:30:59.341108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold_walk = binar(preds_walk)[1]","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:30:59.347292Z","iopub.execute_input":"2023-06-08T07:30:59.347726Z","iopub.status.idle":"2023-06-08T07:31:00.268702Z","shell.execute_reply.started":"2023-06-08T07:30:59.347693Z","shell.execute_reply":"2023-06-08T07:31:00.267519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In total, we have that by setting the exact threshold of significance for classifying the predictions of the regression model, we may well achieve good performance for the metric we need.","metadata":{}},{"cell_type":"markdown","source":"Now let's move on to testing:","metadata":{}},{"cell_type":"code","source":"X_test = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog/003f117e14.csv\",\n    index_col=[0])","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:31:00.270421Z","iopub.execute_input":"2023-06-08T07:31:00.271180Z","iopub.status.idle":"2023-06-08T07:31:00.304964Z","shell.execute_reply.started":"2023-06-08T07:31:00.271132Z","shell.execute_reply":"2023-06-08T07:31:00.303394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = data[['AccV', 'AccML', 'AccAP']]\nstart = data.StartHesitation\nturn = data.Turn\nwalk = data.Walking","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:31:00.306867Z","iopub.execute_input":"2023-06-08T07:31:00.307308Z","iopub.status.idle":"2023-06-08T07:31:00.407810Z","shell.execute_reply.started":"2023-06-08T07:31:00.307271Z","shell.execute_reply":"2023-06-08T07:31:00.406114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = RandomForestRegressor(n_estimators=40, max_depth=10, random_state=42)\nrf.fit(X, start)\npreds_start = rf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:31:00.410934Z","iopub.execute_input":"2023-06-08T07:31:00.412171Z","iopub.status.idle":"2023-06-08T07:55:31.724011Z","shell.execute_reply.started":"2023-06-08T07:31:00.412122Z","shell.execute_reply":"2023-06-08T07:55:31.722662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_start = pd.Series(np.where(preds_start < threshold_start, 0, 1))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:55:31.725423Z","iopub.execute_input":"2023-06-08T07:55:31.725775Z","iopub.status.idle":"2023-06-08T07:55:31.732382Z","shell.execute_reply.started":"2023-06-08T07:55:31.725745Z","shell.execute_reply":"2023-06-08T07:55:31.731244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = RandomForestRegressor(n_estimators=40, max_depth=10, random_state=42)\nrf.fit(X, turn)\npreds_turn = rf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:55:31.733779Z","iopub.execute_input":"2023-06-08T07:55:31.734178Z","iopub.status.idle":"2023-06-08T08:20:08.235444Z","shell.execute_reply.started":"2023-06-08T07:55:31.734147Z","shell.execute_reply":"2023-06-08T08:20:08.233510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_turn = pd.Series(np.where(preds_turn < threshold_turn, 0, 1))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:20:08.237953Z","iopub.execute_input":"2023-06-08T08:20:08.238970Z","iopub.status.idle":"2023-06-08T08:20:08.246332Z","shell.execute_reply.started":"2023-06-08T08:20:08.238917Z","shell.execute_reply":"2023-06-08T08:20:08.244911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = RandomForestRegressor(n_estimators=40, max_depth=10, random_state=42)\nrf.fit(X, walk)\npreds_walk = rf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:20:08.248608Z","iopub.execute_input":"2023-06-08T08:20:08.249138Z","iopub.status.idle":"2023-06-08T08:44:55.271868Z","shell.execute_reply.started":"2023-06-08T08:20:08.249090Z","shell.execute_reply":"2023-06-08T08:44:55.270503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_walk = pd.Series(np.where(preds_walk < threshold_walk, 0, 1))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:44:55.273343Z","iopub.execute_input":"2023-06-08T08:44:55.273704Z","iopub.status.idle":"2023-06-08T08:44:55.280409Z","shell.execute_reply.started":"2023-06-08T08:44:55.273676Z","shell.execute_reply":"2023-06-08T08:44:55.279368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"time_index = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog/003f117e14.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:44:55.282158Z","iopub.execute_input":"2023-06-08T08:44:55.282483Z","iopub.status.idle":"2023-06-08T08:44:55.310431Z","shell.execute_reply.started":"2023-06-08T08:44:55.282458Z","shell.execute_reply":"2023-06-08T08:44:55.309161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"time_index = time_index.Time","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:44:55.317442Z","iopub.execute_input":"2023-06-08T08:44:55.317882Z","iopub.status.idle":"2023-06-08T08:44:55.323932Z","shell.execute_reply.started":"2023-06-08T08:44:55.317842Z","shell.execute_reply":"2023-06-08T08:44:55.322648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_td = pd.concat([time_index, prediction_start, prediction_turn,\n                         prediction_walk], axis=1)\npredictions_td = predictions_td.rename(columns={'Time': 'Id', 0: 'StartHesitation',\n                                          1: 'Turn', 2: 'Walking'})\npredictions_td.Id = '003f117e14_' + predictions_td.Id.astype(str)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:44:55.325680Z","iopub.execute_input":"2023-06-08T08:44:55.326187Z","iopub.status.idle":"2023-06-08T08:44:55.353640Z","shell.execute_reply.started":"2023-06-08T08:44:55.326143Z","shell.execute_reply":"2023-06-08T08:44:55.352316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predictions_td.to_csv('/kaggle/working/submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-06T17:25:45.203998Z","iopub.execute_input":"2023-06-06T17:25:45.204757Z","iopub.status.idle":"2023-06-06T17:25:45.224659Z","shell.execute_reply.started":"2023-06-06T17:25:45.204723Z","shell.execute_reply":"2023-06-06T17:25:45.223571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I realized that I still have time to look at the defog data, I will do it!","metadata":{}},{"cell_type":"code","source":"path = Path(r'/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog')","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:44:55.358102Z","iopub.execute_input":"2023-06-08T08:44:55.359270Z","iopub.status.idle":"2023-06-08T08:44:55.363720Z","shell.execute_reply.started":"2023-06-08T08:44:55.359232Z","shell.execute_reply":"2023-06-08T08:44:55.362810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_df=[]\nfor file in path.glob('*.csv'):\n    df = pd.read_csv(file, index_col=[0])\n    df['Id'] = file.name\n    all_df.append(df)\ndata_all = pd.DataFrame()\nfor x in all_df:\n    data_all = pd.concat([data_all, x])","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:44:55.365082Z","iopub.execute_input":"2023-06-08T08:44:55.365595Z","iopub.status.idle":"2023-06-08T08:45:51.423390Z","shell.execute_reply.started":"2023-06-08T08:44:55.365566Z","shell.execute_reply":"2023-06-08T08:45:51.422021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_all.Id = data_all.Id.str.replace('.csv', '')\ndata_all.Valid = np.where(data_all.Valid == False, 0, 1)\ndata_all.Task = np.where(data_all.Task == False, 0, 1)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:45:51.424735Z","iopub.execute_input":"2023-06-08T08:45:51.425124Z","iopub.status.idle":"2023-06-08T08:46:13.215482Z","shell.execute_reply.started":"2023-06-08T08:45:51.425093Z","shell.execute_reply":"2023-06-08T08:46:13.214316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Part of the data here is not validated, let's look at the ratio of all the data and this part. We also compare the number of features of interest to us.","metadata":{}},{"cell_type":"code","source":"data = pd.DataFrame(data_all.query('Valid==1 and Task==1'))\ndata = data.drop(['Valid', 'Task'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:46:13.217108Z","iopub.execute_input":"2023-06-08T08:46:13.217442Z","iopub.status.idle":"2023-06-08T08:46:15.527572Z","shell.execute_reply.started":"2023-06-08T08:46:13.217414Z","shell.execute_reply":"2023-06-08T08:46:15.526397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(data.shape[0]/data_all.shape[0]).__round__(2)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:46:15.529239Z","iopub.execute_input":"2023-06-08T08:46:15.529622Z","iopub.status.idle":"2023-06-08T08:46:15.537588Z","shell.execute_reply.started":"2023-06-08T08:46:15.529592Z","shell.execute_reply":"2023-06-08T08:46:15.536541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('The ratio of positive and negative reactions in the target (top - validated data / bottom - all data)')\nfig, axs = plt.subplots(ncols=3, nrows=2, figsize=(10, 3))\nautopct='%.0f'\nfont_size=15\ndata.StartHesitation.value_counts().plot.pie(autopct=autopct,fontsize=font_size, ax=axs[0, 0])\ndata.Turn.value_counts().plot.pie(autopct=autopct, fontsize=font_size, ax=axs[0, 1])\ndata.Walking.value_counts().plot.pie(autopct=autopct, fontsize=font_size, ax=axs[0, 2])\ndata_all.StartHesitation.value_counts().plot.pie(autopct=autopct,fontsize=font_size, ax=axs[1, 0])\ndata_all.Turn.value_counts().plot.pie(autopct=autopct, fontsize=font_size, ax=axs[1, 1])\ndata_all.Walking.value_counts().plot.pie(autopct=autopct, fontsize=font_size, ax=axs[1, 2])\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:46:15.539577Z","iopub.execute_input":"2023-06-08T08:46:15.539973Z","iopub.status.idle":"2023-06-08T08:46:16.888775Z","shell.execute_reply.started":"2023-06-08T08:46:15.539941Z","shell.execute_reply":"2023-06-08T08:46:16.887096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The ratio of the non-validated part to the total is 0.3, we also see that the StartHesitation feature is 0 in both cases, so it makes no sense to work with it in our regression problem) In a valid sample, the Turn feature in a positive binary response is represented more than in the entire dataset . I decided to work only with a valid sample.\nAlso, I will not combine this dataset with classifying features from other data, since I will also solve this problem through regression, although there are much more interesting cases for class weighting in this case than in the previous one.","metadata":{}},{"cell_type":"code","source":"X = data[['AccV', 'AccML', 'AccAP']]\nturn = data.Turn\nwalk = data.Walking","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:46:16.891458Z","iopub.execute_input":"2023-06-08T08:46:16.892536Z","iopub.status.idle":"2023-06-08T08:46:16.942480Z","shell.execute_reply.started":"2023-06-08T08:46:16.892474Z","shell.execute_reply":"2023-06-08T08:46:16.941003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(X, turn, stratify=turn,\n                                                      test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:46:16.944330Z","iopub.execute_input":"2023-06-08T08:46:16.945581Z","iopub.status.idle":"2023-06-08T08:46:19.646635Z","shell.execute_reply.started":"2023-06-08T08:46:16.945540Z","shell.execute_reply":"2023-06-08T08:46:19.645533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(X, height=1.5, aspect=1);","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:46:19.648299Z","iopub.execute_input":"2023-06-08T08:46:19.648891Z","iopub.status.idle":"2023-06-08T08:48:22.498856Z","shell.execute_reply.started":"2023-06-08T08:46:19.648849Z","shell.execute_reply":"2023-06-08T08:48:22.497573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = RandomForestRegressor(n_estimators=20, max_depth=25, random_state=42)\nrf.fit(X_train, y_train)\npreds_turn = rf.predict(X_valid)\nprint('MAE: ', mean_absolute_error(y_valid, preds_turn).round(3))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:48:22.500912Z","iopub.execute_input":"2023-06-08T08:48:22.501663Z","iopub.status.idle":"2023-06-08T08:57:23.545393Z","shell.execute_reply.started":"2023-06-08T08:48:22.501620Z","shell.execute_reply":"2023-06-08T08:57:23.544027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Predictions of the Random Forest model in regression by target - Turn')\nplt.figure(figsize=(25, 4))\nplt.xlim(0, 1)\nplt.ylim(0, 100000)\nplt.xticks(np.arange(0, 1, step=0.02))\nplt.hist(x=preds_turn)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:57:23.547184Z","iopub.execute_input":"2023-06-08T08:57:23.547670Z","iopub.status.idle":"2023-06-08T08:57:24.549620Z","shell.execute_reply.started":"2023-06-08T08:57:23.547636Z","shell.execute_reply":"2023-06-08T08:57:24.548360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Average_precision for turn defog: ', binar(pd.Series(preds_turn))[0])","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:57:24.551788Z","iopub.execute_input":"2023-06-08T08:57:24.552233Z","iopub.status.idle":"2023-06-08T08:57:25.092314Z","shell.execute_reply.started":"2023-06-08T08:57:24.552197Z","shell.execute_reply":"2023-06-08T08:57:25.090516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold_turn = binar(pd.Series(preds_turn))[1]","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:57:25.094517Z","iopub.execute_input":"2023-06-08T08:57:25.094974Z","iopub.status.idle":"2023-06-08T08:57:25.636336Z","shell.execute_reply.started":"2023-06-08T08:57:25.094938Z","shell.execute_reply":"2023-06-08T08:57:25.634925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(X, walk, test_size=0.2,\n                                                     stratify=walk)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:57:25.638430Z","iopub.execute_input":"2023-06-08T08:57:25.638874Z","iopub.status.idle":"2023-06-08T08:57:28.380452Z","shell.execute_reply.started":"2023-06-08T08:57:25.638839Z","shell.execute_reply":"2023-06-08T08:57:28.379094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = RandomForestRegressor(n_estimators=20, max_depth=25, random_state=42)\nrf.fit(X_train, y_train)\npreds_walk = rf.predict(X_valid)\nprint('MAE: ', mean_absolute_error(y_valid, preds_walk).round(3))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:57:28.382350Z","iopub.execute_input":"2023-06-08T08:57:28.382743Z","iopub.status.idle":"2023-06-08T09:06:52.717035Z","shell.execute_reply.started":"2023-06-08T08:57:28.382713Z","shell.execute_reply":"2023-06-08T09:06:52.715609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Predictions of the Random Forest model in regression by target - Walking')\nplt.figure(figsize=(20, 5))\nplt.xlim(0, 0.7)\nplt.ylim(0, 1000)\nplt.xticks(np.arange(0, 0.7, step=0.02))\nplt.hist(x=preds_walk)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T09:06:52.718834Z","iopub.execute_input":"2023-06-08T09:06:52.719270Z","iopub.status.idle":"2023-06-08T09:06:53.522255Z","shell.execute_reply.started":"2023-06-08T09:06:52.719238Z","shell.execute_reply":"2023-06-08T09:06:53.521046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Average_precision for walk defog: ', binar(pd.Series(preds_walk))[0])","metadata":{"execution":{"iopub.status.busy":"2023-06-08T09:06:53.524451Z","iopub.execute_input":"2023-06-08T09:06:53.527219Z","iopub.status.idle":"2023-06-08T09:06:53.974636Z","shell.execute_reply.started":"2023-06-08T09:06:53.527163Z","shell.execute_reply":"2023-06-08T09:06:53.973396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold_walk = binar(pd.Series(preds_walk))[1]","metadata":{"execution":{"iopub.status.busy":"2023-06-08T09:06:53.976093Z","iopub.execute_input":"2023-06-08T09:06:53.976434Z","iopub.status.idle":"2023-06-08T09:06:54.430764Z","shell.execute_reply.started":"2023-06-08T09:06:53.976406Z","shell.execute_reply":"2023-06-08T09:06:54.429533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog/02ab235146.csv\",\n    index_col=[0])","metadata":{"execution":{"iopub.status.busy":"2023-06-08T09:06:54.432278Z","iopub.execute_input":"2023-06-08T09:06:54.432651Z","iopub.status.idle":"2023-06-08T09:06:54.850042Z","shell.execute_reply.started":"2023-06-08T09:06:54.432620Z","shell.execute_reply":"2023-06-08T09:06:54.847974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = data[['AccV', 'AccML', 'AccAP']]\nturn = data.Turn\nwalk = data.Walking","metadata":{"execution":{"iopub.status.busy":"2023-06-08T09:06:54.852187Z","iopub.execute_input":"2023-06-08T09:06:54.853014Z","iopub.status.idle":"2023-06-08T09:06:54.984213Z","shell.execute_reply.started":"2023-06-08T09:06:54.852966Z","shell.execute_reply":"2023-06-08T09:06:54.982499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = RandomForestRegressor(n_estimators=20, max_depth=25, random_state=42)\nrf.fit(X, turn)\npreds_turn = rf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T09:06:54.986065Z","iopub.execute_input":"2023-06-08T09:06:54.986439Z","iopub.status.idle":"2023-06-08T09:16:52.211283Z","shell.execute_reply.started":"2023-06-08T09:06:54.986410Z","shell.execute_reply":"2023-06-08T09:16:52.209828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_turn = pd.Series(np.where(preds_turn < threshold_turn, 0, 1))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T09:16:52.213564Z","iopub.execute_input":"2023-06-08T09:16:52.214513Z","iopub.status.idle":"2023-06-08T09:16:52.222841Z","shell.execute_reply.started":"2023-06-08T09:16:52.214460Z","shell.execute_reply":"2023-06-08T09:16:52.221670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = RandomForestRegressor(n_estimators=20, max_depth=25, random_state=42)\nrf.fit(X, walk)\npreds_walk = rf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T09:16:52.224457Z","iopub.execute_input":"2023-06-08T09:16:52.225320Z","iopub.status.idle":"2023-06-08T09:26:57.770642Z","shell.execute_reply.started":"2023-06-08T09:16:52.225285Z","shell.execute_reply":"2023-06-08T09:26:57.769058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_walk = pd.Series(np.where(preds_walk < threshold_walk, 0, 1))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T09:26:57.772321Z","iopub.execute_input":"2023-06-08T09:26:57.772727Z","iopub.status.idle":"2023-06-08T09:26:57.780182Z","shell.execute_reply.started":"2023-06-08T09:26:57.772693Z","shell.execute_reply":"2023-06-08T09:26:57.779099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_start = pd.Series(np.zeros(281688, dtype=int))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T09:26:57.781793Z","iopub.execute_input":"2023-06-08T09:26:57.782390Z","iopub.status.idle":"2023-06-08T09:26:57.799620Z","shell.execute_reply.started":"2023-06-08T09:26:57.782357Z","shell.execute_reply":"2023-06-08T09:26:57.797929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"time_index = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog/02ab235146.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-06-08T09:26:57.802737Z","iopub.execute_input":"2023-06-08T09:26:57.803338Z","iopub.status.idle":"2023-06-08T09:26:58.072826Z","shell.execute_reply.started":"2023-06-08T09:26:57.803299Z","shell.execute_reply":"2023-06-08T09:26:58.071479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"time_index = time_index.Time","metadata":{"execution":{"iopub.status.busy":"2023-06-08T09:26:58.074549Z","iopub.execute_input":"2023-06-08T09:26:58.074971Z","iopub.status.idle":"2023-06-08T09:26:58.081210Z","shell.execute_reply.started":"2023-06-08T09:26:58.074935Z","shell.execute_reply":"2023-06-08T09:26:58.079916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_de = pd.concat([time_index, prediction_start, prediction_turn, prediction_walk], axis=1)\npredictions_de = predictions_de.rename(columns={'Time': 'Id', 0: 'StartHesitation', 1: 'Turn', 2: 'Walking'})\npredictions_de.Id = '02ab235146_' + predictions_de.Id.astype(str)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T09:26:58.082974Z","iopub.execute_input":"2023-06-08T09:26:58.083362Z","iopub.status.idle":"2023-06-08T09:26:58.416321Z","shell.execute_reply.started":"2023-06-08T09:26:58.083328Z","shell.execute_reply":"2023-06-08T09:26:58.415110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predictions = pd.concat([predictions_td, predictions_de], axis=0)","metadata":{"execution":{"iopub.status.busy":"2023-06-05T18:45:01.084566Z","iopub.execute_input":"2023-06-05T18:45:01.085124Z","iopub.status.idle":"2023-06-05T18:45:01.099085Z","shell.execute_reply.started":"2023-06-05T18:45:01.084961Z","shell.execute_reply":"2023-06-05T18:45:01.097936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_de.to_csv('/kaggle/working/submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-08T09:26:58.417943Z","iopub.execute_input":"2023-06-08T09:26:58.418328Z","iopub.status.idle":"2023-06-08T09:26:59.372465Z","shell.execute_reply.started":"2023-06-08T09:26:58.418299Z","shell.execute_reply":"2023-06-08T09:26:59.371336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So I implemented the process of solving the problem of binary classification through regression. It is clear that through careful work with all the available data on the subjects of the study, it is possible to give our features certain weights and predict classes with higher precision metrics, but with my work I wanted to show that one should not focus on only one method of solving problems, but look for what -something new! Well, or because I'm tired of waiting for the classification model to do its job ...","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}