{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"* My contribution to the Kagglen community is sincerely appreciated. Your upvote will be an encouragement to me.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport time\nimport optuna\nimport re\nimport sklearn\nimport math\nimport gc\nimport random\nimport cv2\nimport plotly.express as px\nimport tensorflow as tf\nimport xgboost as xgb\nimport lightgbm as lgb\nimport catboost as cb\nimport lightgbm as lgb\nimport scipy.stats\nfrom tqdm import tqdm\nfrom tqdm.notebook import tqdm\nfrom sklearn import ensemble\nimport optuna\nplt.style.use('fivethirtyeight') \n\nfrom tensorflow.keras import layers, models\n\nfrom sklearn.ensemble import *\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.linear_model import *\nfrom sklearn.model_selection import *\nfrom sklearn import metrics\nfrom sklearn.metrics import *\nfrom sklearn.model_selection import StratifiedKFold, GroupKFold, KFold, RandomizedSearchCV, cross_val_score, GridSearchCV, LeaveOneOut\nfrom sklearn.preprocessing import *\nfrom sklearn.svm import SVR\nfrom optuna.visualization import plot_slice\nfrom sklearn.calibration import CalibratedClassifierCV\nfrom sklearn.feature_selection import mutual_info_classif\nfrom sklearn.tree import *\nfrom sklearn.svm import LinearSVC\nfrom sklearn.neighbors import KNeighborsRegressor\n\n# Above is a special style template for matplotlib, highly useful for visualizing time series data\n%matplotlib inline\nfrom pylab import rcParams\nfrom plotly import tools\nimport plotly.subplots as sp\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.offline import init_notebook_mode, iplot\ninit_notebook_mode(connected=True)\nimport plotly.graph_objs as go\nimport plotly.figure_factory as ff\nimport plotly.graph_objects as go\nimport statsmodels.api as sm\nfrom numpy.random import normal, seed\nfrom scipy.stats import norm\nfrom scipy.stats import skew, kurtosis\nfrom scipy.stats import gaussian_kde\nfrom statsmodels.tsa.arima_model import ARMA\nfrom statsmodels.tsa.stattools import adfuller\nfrom statsmodels.graphics.tsaplots import plot_acf, plot_pacf\nfrom statsmodels.tsa.arima_process import ArmaProcess\nfrom statsmodels.tsa.arima_model import ARIMA\nfrom sklearn.metrics import mean_squared_error\nfrom optuna.integration import XGBoostPruningCallback\n\nimport warnings\nwarnings.filterwarnings(\"ignore\") # specify to ignore warning messages","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-05-07T11:08:27.482860Z","iopub.execute_input":"2023-05-07T11:08:27.483350Z","iopub.status.idle":"2023-05-07T11:08:41.056648Z","shell.execute_reply.started":"2023-05-07T11:08:27.483304Z","shell.execute_reply":"2023-05-07T11:08:41.055577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_PATH = '../input/tlvmc-parkinsons-freezing-gait-prediction/'\nfile_names = [\"sample_submission\", \"daily_metadata\", \"defog_metadata\", \"events\", \"subjects\", \"tasks\", \"tdcsfog_metadata\"]\n\nsample, daily_meta, defog_meta, events, subjects, tasks, tdcsfog_meta = [pd.read_csv(DATA_PATH + f + \".csv\") for f in file_names]\n\nunlabeled_example = pd.read_parquet(DATA_PATH + \"unlabeled/00c4c9313d.parquet\")\ntrain_tdcsfog_example = pd.read_csv(DATA_PATH + \"train/tdcsfog/003f117e14.csv\")\ntest_tdcsfog_example = pd.read_csv(DATA_PATH + \"test/tdcsfog/003f117e14.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:08:41.058998Z","iopub.execute_input":"2023-05-07T11:08:41.059807Z","iopub.status.idle":"2023-05-07T11:08:50.878643Z","shell.execute_reply.started":"2023-05-07T11:08:41.059776Z","shell.execute_reply":"2023-05-07T11:08:50.877609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def groupby(columns, object, data):\n    g = data.groupby(columns).describe()[object]\n    return print(g)\n\n# Define Display a Bar Graph Function\ndef display(x, y, data, figsize=(8, 7)):\n    plt.figure(figsize=figsize, dpi=20)\n    ax = sns.countplot(y=y, x=x, data=data)\n    ax.tick_params(rotation=90)\n    return plt.show()\n\n# Define Display Histogram Function\ndef histogram(df, col, figsize=(28, 26), KDE=True):\n    sns.set_style(\"whitegrid\")  # Set background to white grid\n    plt.figure(figsize=figsize, dpi=20)\n    \n    # Calculate the bin number using Sturges' formula\n    bin_number = int(1 + np.log2(len(df[col])))\n\n    ax = sns.histplot(df[col], color=\"b\", kde=True, bins=bin_number)  # Display KDE plot with calculated bin number\n    ax.xaxis.grid(False)\n    ax.set(ylabel=\"Frequency\")\n    ax.set(xlabel=str(col))\n    ax.set_title(\"Distribution of \" + str(col))\n    sns.despine(trim=True, left=True)\n    return plt.show()\n\ndef display_value_counts_bar(df, column, xaxis_title, yaxis_title, color='goldenrod'):\n    value_counts = df[column].value_counts()\n\n    fig = px.bar(x=value_counts.index, y=value_counts.values, color_discrete_sequence=[color])\n    fig.update_layout(xaxis_title=xaxis_title, yaxis_title=yaxis_title)\n    fig.show()\n\ndef display_histogram(df, col, kde=True, bins=None, color='goldenrod'):  \n    # Calculate the bin number using Sturges' formula\n    bin_number = int(1 + np.log2(len(df[col])))\n    \n    fig = px.histogram(df, x=col, nbins=bin_number, color_discrete_sequence=[color])\n    fig.show()\n\n# Define Display a Bar Graph Function\ndef jointplot(x, y, data, figsize=(8, 7)):\n    plt.figure(figsize=figsize, dpi=20)\n    sns.jointplot(y=y, x=x, data=data, kind='reg')\n    return plt.show()\n\ndef display_value_counts_bar(df, column, xaxis_title, yaxis_title):\n    value_counts = df[column].value_counts()\n\n    fig = px.bar(x=value_counts.index, y=value_counts.values, color_discrete_sequence=['darkslateblue'])\n    fig.update_layout(xaxis_title=xaxis_title, yaxis_title=yaxis_title)\n    fig.show()\n\n# Define Display a Bar Chart Function\ndef create_bar_chart(df, column, xaxis_title, yaxis_title):\n    counts = df[column].value_counts()\n    fig = px.bar(x=counts.index, y=counts.values, color_discrete_sequence=['darkgreen'])\n    fig.update_layout(xaxis_title=xaxis_title, yaxis_title=yaxis_title)\n    fig.show()\n\n# Define Display a scatter plots Function\ndef scatterplot(x, y, data, figsize=(48, 42), xlabel_fontsize=65, ylabel_fontsize=65, tick_fontsize=50):\n    plt.figure(figsize=figsize, dpi=20)\n    plt.scatter(data[x], data[y], alpha=0.8, s=1000, marker='o')\n    plt.xlabel(x, fontsize=xlabel_fontsize)\n    plt.ylabel(y, fontsize=ylabel_fontsize)\n    \n    # Set tick font size for x-axis and y-axis\n    plt.xticks(fontsize=tick_fontsize)\n    plt.yticks(fontsize=tick_fontsize)\n    \n    return plt.show()\n\n# Define Display a correlation heatmap Function\ndef display_correlation_heatmap(df, title):\n    corr_mat = np.round(df.corr(), 3)\n    \n    fig, ax = plt.subplots(figsize=(5, 5))\n    sns.heatmap(corr_mat, annot=True, fmt=\".3f\", cmap='coolwarm', cbar=False, square=True, linewidths=.5, annot_kws={\"size\": 12}, ax=ax)\n\n    ax.set_title(title, fontsize=16, pad=20, y=1.05)\n    ax.set_xticklabels(ax.get_xticklabels(), rotation=45, ha=\"right\", fontsize=12)\n    ax.set_yticklabels(ax.get_yticklabels(), rotation=0, fontsize=12)\n\n    plt.tight_layout()\n    plt.show()\n       \ndef plot_correlation_heatmap(df, column_name):\n    correlation_matrix = df.corr()\n    plt.figure(figsize=(12, 10))\n    sns.heatmap(correlation_matrix, annot=True, cmap='coolwarm')\n    plt.title(f'Correlation heatmap for {column_name}')\n    plt.show()\n\n# Define Display a pie charts\ndef plot_frequency_pie_charts(dataframes, column, titles, figsize=(8, 7)):\n    fig, axs = plt.subplots(nrows=len(dataframes), ncols=1, figsize=figsize, squeeze=False) # Add 'squeeze=False' here\n    axs = axs.flatten()\n\n    for i, (df, title) in enumerate(zip(dataframes, titles)):\n        counts = df[column].value_counts().to_dict()\n        \n        _ = axs[i].pie(\n            x=list(counts.values()),\n            autopct=lambda x: \"{:,.0f} = {:.2f}%\".format(x * sum(counts.values()) / 100, x),\n            explode=[0.05] * len(counts.keys()),  # Update this line to include explode parameter\n            labels=counts.keys(),\n            colors=sns.color_palette(\"Set2\")[0:len(counts.keys())],\n        )\n        _ = axs[i].set_title(f\"Frequency of Values in {title}\", fontsize=15)\n    \n    plt.tight_layout()\n    plt.show()\n    \n# Define a function to display a line chart of 3D accelerometer data     \ndef plot_line_graph(df, title):\n    plt.figure(figsize=(8, 5))\n    plt.plot(df[\"Time\"], df[\"AccV\"], label=\"AccV\", marker='o', linewidth=0.7, linestyle='--')\n    plt.plot(df[\"Time\"], df[\"AccML\"], label=\"AccML\", marker='o', linewidth=0.7, linestyle='--')\n    plt.plot(df[\"Time\"], df[\"AccAP\"], label=\"AccAP\", marker='o', linewidth=0.7, linestyle='--')\n    plt.xlabel(\"Time\")\n    plt.ylabel(\"AccV, AccML, AccAP\")\n    plt.legend()\n    plt.title(title)\n    plt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-05-07T11:08:50.880171Z","iopub.execute_input":"2023-05-07T11:08:50.880559Z","iopub.status.idle":"2023-05-07T11:08:50.908527Z","shell.execute_reply.started":"2023-05-07T11:08:50.880522Z","shell.execute_reply":"2023-05-07T11:08:50.907482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class color:\n   PURPLE = '\\033[95m'\n   CYAN = '\\033[96m'\n   DARKCYAN = '\\033[36m'\n   BLUE = '\\033[94m'\n   GREEN = '\\033[92m'\n   YELLOW = '\\033[93m'\n   RED = '\\033[91m'\n   BOLD = '\\033[1m'\n   UNDERLINE = '\\033[4m'\n   END = '\\033[0m'","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-05-07T11:08:50.911442Z","iopub.execute_input":"2023-05-07T11:08:50.911857Z","iopub.status.idle":"2023-05-07T11:08:50.924912Z","shell.execute_reply.started":"2023-05-07T11:08:50.911821Z","shell.execute_reply":"2023-05-07T11:08:50.923974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define dataframes\ndataframes = {\n    'daily_meta': daily_meta,\n    'defog_meta': defog_meta,\n    'events': events,\n    'subjects': subjects,\n    'tasks': tasks,\n    'tdcsfog_meta': tdcsfog_meta,\n    'train_tdcsfog_example': train_tdcsfog_example,\n    'test_tdcsfog_example': test_tdcsfog_example,\n    'unlabeled_example': unlabeled_example,\n    'sample': sample\n}","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-05-07T11:08:50.926418Z","iopub.execute_input":"2023-05-07T11:08:50.927283Z","iopub.status.idle":"2023-05-07T11:08:50.935377Z","shell.execute_reply.started":"2023-05-07T11:08:50.927247Z","shell.execute_reply":"2023-05-07T11:08:50.934340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get an overview of the data\nfor name, df in dataframes.items():\n    print(f'{color.BLUE}{name}{color.END} (head of the data is):')\n    print(df.head())\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:08:50.938846Z","iopub.execute_input":"2023-05-07T11:08:50.939106Z","iopub.status.idle":"2023-05-07T11:08:50.973349Z","shell.execute_reply.started":"2023-05-07T11:08:50.939081Z","shell.execute_reply":"2023-05-07T11:08:50.972470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Some data frames could be merged if the id is common. In that case, care should be taken about the duplicate darkening of the Visit columns and the data type.\n* Consideration of the proportion and meaning of categorical variables such as Type and Task will also be needed\n\n* [Dataset Description](https://www.kaggle.com/competitions/tlvmc-parkinsons-freezing-gait-prediction/data)　　explains that AccV, AccML, and AccAP Acceleration from a lower-back sensor on three axes: V - vertical, ML - mediolateral, AP - anteroposterior. Data is in units of m/s^2 for tdcsfog/ and g for defog/ and notype/.\n\n* And [thread for this discussion](https://www.kaggle.com/competitions/tlvmc-parkinsons-freezing-gait-prediction/discussion/393852) summarises the data from the three sensors as follows.\n * AccV - acceleration from vertical sensor\n * AccML - acceleration from mediolateral sensor\n * AccAP - acceleration from anteroposterior sensor\n \n* From the sample data, it can be understood that for each ID, the competition is predicted for column StartHesitation and Turn and Walking. In StartHesitation, Turn, and Walking are the class labels for each time step. The goal of the competition is to predict these class labels for the test set series.","metadata":{}},{"cell_type":"code","source":"# Get an overview of the data\nfor name, df in dataframes.items():\n    print(f'{color.BLUE}{name}{color.END} (overview of the data is):')\n    print(df.describe())\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:08:50.975008Z","iopub.execute_input":"2023-05-07T11:08:50.975329Z","iopub.status.idle":"2023-05-07T11:09:00.566905Z","shell.execute_reply.started":"2023-05-07T11:08:50.975294Z","shell.execute_reply":"2023-05-07T11:09:00.564857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#　Checking the shape of the dataframes\nfor df_name, df in dataframes.items():\n    print(f\"Shape of {df_name} : {color.BLUE}{df.shape}{color.END}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:09:00.568260Z","iopub.execute_input":"2023-05-07T11:09:00.568632Z","iopub.status.idle":"2023-05-07T11:09:00.578143Z","shell.execute_reply.started":"2023-05-07T11:09:00.568601Z","shell.execute_reply":"2023-05-07T11:09:00.573770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating gender-specific data frames.\nsubjects_m = subjects[subjects[\"Sex\"] == \"M\"]\nsubjects_f = subjects[subjects[\"Sex\"] == \"F\"]\n\nevents_zero_turn = events[(events[\"Kinetic\"] == 0) & (events[\"Type\"] == 'Turn')]\nevents_zero_walking = events[(events[\"Kinetic\"] == 0) & (events[\"Type\"] == 'Walking')]\nevents_zero_starthesitation = events[(events[\"Kinetic\"] == 0) & (events[\"Type\"] == 'StartHesitation')]\nevents_one_turn = events[(events[\"Kinetic\"] == 1) & (events[\"Type\"] == 'Turn')]\nevents_one_walking = events[(events[\"Kinetic\"] == 1) & (events[\"Type\"] == 'Walking')]\nevents_one_starthesitation = events[(events[\"Kinetic\"] == 1) & (events[\"Type\"] == 'StartHesitation')]\n\ntrain_tdcsfog_zero_example = train_tdcsfog_example[train_tdcsfog_example[\"Turn\"] == 0]\ntrain_tdcsfog_one_example = train_tdcsfog_example[train_tdcsfog_example[\"Turn\"] == 1]\n\ndefog_on_meta = defog_meta[defog_meta[\"Medication\"] == 'On']\ndefog_off_meta= defog_meta[defog_meta[\"Medication\"] == 'Off']","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:09:00.579890Z","iopub.execute_input":"2023-05-07T11:09:00.580688Z","iopub.status.idle":"2023-05-07T11:09:00.599513Z","shell.execute_reply.started":"2023-05-07T11:09:00.580651Z","shell.execute_reply":"2023-05-07T11:09:00.598472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a list of data frames\ndfs = [events_zero_turn, events_zero_walking, events_zero_starthesitation,\n       events_one_turn, events_one_walking, events_one_starthesitation]\n\n# Process each data frame sequentially\nfor df in dfs:\n    # Create column Duration and subtract Init from Completion\n    df['Duration'] = df['Completion'] - df['Init']\n\n    # Delete column Completion and column Init\n    df.drop(['Completion', 'Init'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:09:00.604693Z","iopub.execute_input":"2023-05-07T11:09:00.605417Z","iopub.status.idle":"2023-05-07T11:09:00.619903Z","shell.execute_reply.started":"2023-05-07T11:09:00.605390Z","shell.execute_reply":"2023-05-07T11:09:00.618909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define dataframes\ndataframes = {\n    'Data only for males in subjects': subjects_m,\n    'Data only for females in subjects': subjects_f,\n    'Data where the Kinetic of events is zero and type is turn': events_zero_turn,\n    'Data where the Kinetic of events is zero and type is walking': events_zero_walking,\n    'Data where the Kinetic of events is zero and type is starthesitation': events_zero_starthesitation,\n    'Data where the Kinetic of events is one and type is turn': events_one_turn,\n    'Data where the Kinetic of events is　one and type is walking': events_one_walking,\n    'Data where the Kinetic of events is one and type is starthesitation': events_one_starthesitation,  \n    'Data with zero Turn in train_tdcsfog_example': train_tdcsfog_zero_example,\n    'Data with one Turn in train_tdcsfog_example': train_tdcsfog_one_example,\n    'Data for which Medication in defog_meta is On': defog_on_meta,\n    'Data for which Medication in defog_meta is Off': defog_off_meta,\n}","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:09:00.621133Z","iopub.execute_input":"2023-05-07T11:09:00.621837Z","iopub.status.idle":"2023-05-07T11:09:00.628414Z","shell.execute_reply.started":"2023-05-07T11:09:00.621802Z","shell.execute_reply":"2023-05-07T11:09:00.627344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get an overview of the data\nfor name, df in dataframes.items():\n    print(f'{color.BLUE}{name}{color.END} (overview of the data is):')\n    print(df.describe())\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:09:00.629811Z","iopub.execute_input":"2023-05-07T11:09:00.630456Z","iopub.status.idle":"2023-05-07T11:09:00.747593Z","shell.execute_reply.started":"2023-05-07T11:09:00.630406Z","shell.execute_reply":"2023-05-07T11:09:00.746487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check missing values\nfor name, df in dataframes.items():\n    print(f'{color.BLUE}{name}{color.END} (percentage of missing values are):')\n    missing_values = df.isna().sum()\n    print(missing_values / len(df))\n    print()\n    \n    # Check for missing values in the training features data\n    cnl = df.isnull().sum()\n    \n    # If there are no missing values, skip the plot\n    if cnl.sum() == 0:\n        print(f'{color.BLUE}{name}{color.END} has no missing values, so no plot is displayed.')\n        continue\n\n    f, ax = plt.subplots(figsize=(8, 7))\n    sns.barplot(x=cnl, y=df.columns.values, orient='h')\n    ax.set_xlabel(\"null count\")\n    plt.tight_layout()\n    plt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-05-07T11:09:00.749270Z","iopub.execute_input":"2023-05-07T11:09:00.749909Z","iopub.status.idle":"2023-05-07T11:09:01.314005Z","shell.execute_reply.started":"2023-05-07T11:09:00.749872Z","shell.execute_reply":"2023-05-07T11:09:01.312470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* There is a small but significant amount of data with missing values. It is fortunate that there is little data with missing values.","metadata":{}},{"cell_type":"code","source":"groupby('Sex', 'UPDRSIII_On', subjects)","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:09:01.315505Z","iopub.execute_input":"2023-05-07T11:09:01.315844Z","iopub.status.idle":"2023-05-07T11:09:01.356784Z","shell.execute_reply.started":"2023-05-07T11:09:01.315806Z","shell.execute_reply":"2023-05-07T11:09:01.355688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* There appear to be marked gender differences in the total amount of UPDRSIII_On data and in the and average UPDRSIII_On data.","metadata":{}},{"cell_type":"code","source":"groupby('Sex', 'UPDRSIII_Off', subjects)","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:09:01.358202Z","iopub.execute_input":"2023-05-07T11:09:01.358666Z","iopub.status.idle":"2023-05-07T11:09:01.395956Z","shell.execute_reply.started":"2023-05-07T11:09:01.358625Z","shell.execute_reply":"2023-05-07T11:09:01.394893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* There appear to be marked gender differences in the total amount of UPDRSIII_Off data and in the and average UPDRSIII_Off data.","metadata":{}},{"cell_type":"code","source":"groupby('Sex', 'NFOGQ', subjects)","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:09:01.397391Z","iopub.execute_input":"2023-05-07T11:09:01.397830Z","iopub.status.idle":"2023-05-07T11:09:01.434835Z","shell.execute_reply.started":"2023-05-07T11:09:01.397793Z","shell.execute_reply":"2023-05-07T11:09:01.433751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* There appear to be marked gender differences in the total amount of NFOGQ data and in the and average NFOGQ data.","metadata":{}},{"cell_type":"code","source":"# Create column Duration and subtract Init from Completion\ndf = events\ndf['Duration'] = df['Completion'] - df['Init']\n\n# Delete column Completion and column Init\ndf.drop(['Completion', 'Init'], axis=1, inplace=True)\n\n# Confirmation\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:09:01.436602Z","iopub.execute_input":"2023-05-07T11:09:01.436964Z","iopub.status.idle":"2023-05-07T11:09:01.453387Z","shell.execute_reply.started":"2023-05-07T11:09:01.436929Z","shell.execute_reply":"2023-05-07T11:09:01.452280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns_to_check = [\n    ('events', events, 'Type'),\n    ('events', events, 'Kinetic'),\n    ('subjects', subjects, 'Visit'),\n    ('subjects', subjects, 'UPDRSIII_On'),\n    ('subjects', subjects, 'UPDRSIII_Off')\n]","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-05-07T11:09:01.455130Z","iopub.execute_input":"2023-05-07T11:09:01.455492Z","iopub.status.idle":"2023-05-07T11:09:01.461626Z","shell.execute_reply.started":"2023-05-07T11:09:01.455454Z","shell.execute_reply":"2023-05-07T11:09:01.460293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for columns of data frames with missing values\nfor df_name, df, column in columns_to_check:\n    unique_values = df[column].unique()\n    print(f\"Dataframe: {color.BLUE}{df_name}{color.END}, Column: {color.BLUE}{column}{color.END}\")\n    print(unique_values)\n    print(type(unique_values))\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:09:01.463111Z","iopub.execute_input":"2023-05-07T11:09:01.464205Z","iopub.status.idle":"2023-05-07T11:09:01.476125Z","shell.execute_reply.started":"2023-05-07T11:09:01.464169Z","shell.execute_reply":"2023-05-07T11:09:01.474911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#　Checking the shape of the dataframes\nfor df_name, df in dataframes.items():\n    print(f\"Shape of {df_name} : {color.BLUE}{df.shape}{color.END}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:09:01.477752Z","iopub.execute_input":"2023-05-07T11:09:01.478182Z","iopub.status.idle":"2023-05-07T11:09:01.484902Z","shell.execute_reply.started":"2023-05-07T11:09:01.478146Z","shell.execute_reply":"2023-05-07T11:09:01.483671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the number of characters in the data frame\nfor name, df in dataframes.items():\n    temp = len(df)\n    print(\n        f\"Length of the {name} file is: {color.BLUE}{temp}{color.END}\",\n    )","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:09:01.486708Z","iopub.execute_input":"2023-05-07T11:09:01.487192Z","iopub.status.idle":"2023-05-07T11:09:01.496629Z","shell.execute_reply.started":"2023-05-07T11:09:01.487156Z","shell.execute_reply":"2023-05-07T11:09:01.495164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# memory deallocation\ngc.collect()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-05-07T11:09:01.498053Z","iopub.execute_input":"2023-05-07T11:09:01.499059Z","iopub.status.idle":"2023-05-07T11:09:01.730802Z","shell.execute_reply.started":"2023-05-07T11:09:01.499023Z","shell.execute_reply":"2023-05-07T11:09:01.729588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = [\"UPDRSIII_On\", \"UPDRSIII_Off\", \"Age\", \"NFOGQ\"]\nlabels = [\"UPDRSIII_On\", \"UPDRSIII_Off\", \"Age\", \"NFOGQ\"]\n\nfig, axs = plt.subplots(nrows=4, ncols=1, figsize=(14, 13))\n\nsns.set_style('darkgrid')\n\naxs = axs.flatten()\n\nsns.set_style('darkgrid')\n\nfor x, feature in enumerate(features):\n    ax = axs[x]\n    _ = sns.histplot(data=subjects, x=feature, kde=True, ax=ax, element=\"step\")\n    _ = ax.set_title(\"{} Scores by Data Source\".format(labels[x]), fontsize=15)\n    _ = ax.set_ylabel(\"Count\")\n    _ = ax.set_xlabel(\"{} Score\".format(labels[x]))\n\n# Adjust the space between the subplots\nplt.subplots_adjust(hspace=0.4)\n\nplt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-05-07T11:09:01.732450Z","iopub.execute_input":"2023-05-07T11:09:01.732980Z","iopub.status.idle":"2023-05-07T11:09:02.361267Z","shell.execute_reply.started":"2023-05-07T11:09:01.732943Z","shell.execute_reply":"2023-05-07T11:09:02.360289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Distribution of men in the subjects\nfeatures = [\"UPDRSIII_On\", \"UPDRSIII_Off\", \"Age\", \"NFOGQ\"]\nlabels = [\"UPDRSIII_On of Men\", \"UPDRSIII_Off of Men\", \"Age of Men\", \"NFOGQ of Men\"]\n\nfig, axs = plt.subplots(nrows=2, ncols=2, figsize=(10, 8))\n\nsns.set_style('darkgrid')\n\naxs = axs.flatten()\n\nfor x, feature in enumerate(features):\n    ax = axs[x]\n    _ = sns.histplot(data=subjects_m, x=feature, kde=True, ax=ax, element=\"step\")\n    _ = ax.set_title(\"{} Scores by Data Source\".format(labels[x]), fontsize=12)\n    _ = ax.set_ylabel(\"Count\")\n    _ = ax.set_xlabel(\"{} Score\".format(labels[x]))\n\n# Adjust the space between the subplots\nplt.subplots_adjust(hspace=0.5, wspace=0.3)\n\nplt.show()\n\n# Distribution of women in the subjects\nfeatures = [\"UPDRSIII_On\", \"UPDRSIII_Off\", \"Age\", \"NFOGQ\"]\nlabels = [\"UPDRSIII_On of Women\", \"UPDRSIII_Off of Women\", \"Age of Women\", \"NFOGQ of Women\"]\n\nfig, axs = plt.subplots(nrows=2, ncols=2, figsize=(10, 8))\n\nsns.set_style('darkgrid')\n\naxs = axs.flatten()\n\nfor x, feature in enumerate(features):\n    ax = axs[x]\n    _ = sns.histplot(data=subjects_f, x=feature, kde=True, ax=ax, element=\"step\")\n    _ = ax.set_title(\"{} Scores by Data Source\".format(labels[x]), fontsize=12)\n    _ = ax.set_ylabel(\"Count\")\n    _ = ax.set_xlabel(\"{} Score\".format(labels[x]))\n\n# Adjust the space between the subplots\nplt.subplots_adjust(hspace=0.5, wspace=0.3)\n\nplt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-05-07T11:09:02.362393Z","iopub.execute_input":"2023-05-07T11:09:02.363517Z","iopub.status.idle":"2023-05-07T11:09:03.719686Z","shell.execute_reply.started":"2023-05-07T11:09:02.363478Z","shell.execute_reply":"2023-05-07T11:09:03.718687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The distribution of age appears to be slightly more skewed in the female data than in the male data; the distribution of UPDRSIII_Off is also more skewed in the female data than in the male data. Is there some relationship between age and UPDRSIII_Off?","metadata":{}},{"cell_type":"code","source":"# Pie chart of column Type in data frame events\nplot_frequency_pie_charts([events], 'Kinetic', ['Kinetic of events'])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:09:03.721162Z","iopub.execute_input":"2023-05-07T11:09:03.721803Z","iopub.status.idle":"2023-05-07T11:09:03.916735Z","shell.execute_reply.started":"2023-05-07T11:09:03.721766Z","shell.execute_reply":"2023-05-07T11:09:03.915699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* [Dataset Description](https://www.kaggle.com/competitions/tlvmc-parkinsons-freezing-gait-prediction/data)　explains that Kinetic Whether the event was kinetic (1) and involved movement, or akinetic (0) and static.","metadata":{}},{"cell_type":"code","source":"# Pie chart of column Type in data frame events\nplot_frequency_pie_charts([train_tdcsfog_example], 'StartHesitation', ['StartHesitation of train_tdcsfog_example'])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:09:03.918515Z","iopub.execute_input":"2023-05-07T11:09:03.919152Z","iopub.status.idle":"2023-05-07T11:09:04.114358Z","shell.execute_reply.started":"2023-05-07T11:09:03.919115Z","shell.execute_reply":"2023-05-07T11:09:04.113372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* In the train_tdcsfog_example data, StartHesitation label is all 0. Does StartHesitation make sense?　It is answered in the discussion thread [On StartHesitation label](https://www.kaggle.com/competitions/tlvmc-parkinsons-freezing-gait-prediction/discussion/395468).","metadata":{}},{"cell_type":"code","source":"# Pie chart of column Type in data frame events\nplot_frequency_pie_charts([subjects], 'Visit', ['Visit of subjects'])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:09:04.116063Z","iopub.execute_input":"2023-05-07T11:09:04.116743Z","iopub.status.idle":"2023-05-07T11:09:04.309356Z","shell.execute_reply.started":"2023-05-07T11:09:04.116703Z","shell.execute_reply":"2023-05-07T11:09:04.308421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pie chart of column Type in data frame events\nplot_frequency_pie_charts([train_tdcsfog_example], 'Turn', ['Turn of train_tdcsfog_example'])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:09:04.324412Z","iopub.execute_input":"2023-05-07T11:09:04.325158Z","iopub.status.idle":"2023-05-07T11:09:04.523805Z","shell.execute_reply.started":"2023-05-07T11:09:04.325120Z","shell.execute_reply":"2023-05-07T11:09:04.522828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# memory deallocation\ngc.collect()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-05-07T11:09:04.525344Z","iopub.execute_input":"2023-05-07T11:09:04.525964Z","iopub.status.idle":"2023-05-07T11:09:04.781741Z","shell.execute_reply.started":"2023-05-07T11:09:04.525926Z","shell.execute_reply":"2023-05-07T11:09:04.780374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the total value \nbar = tqdm(total = 100)\n# Add description\nbar.set_description('Progress rate')\nfor i in range(100):\n    # Set the progress\n    bar.update(5)\n    time.sleep(1)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-05-07T11:09:04.783657Z","iopub.execute_input":"2023-05-07T11:09:04.784304Z","iopub.status.idle":"2023-05-07T11:10:44.976926Z","shell.execute_reply.started":"2023-05-07T11:09:04.784264Z","shell.execute_reply":"2023-05-07T11:10:44.975867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define dataframes　with id\ndataframes = {\n    'daily_meta': daily_meta,\n    'defog_meta': defog_meta,\n    'events': events,\n    'tdcsfog_meta': tdcsfog_meta\n}","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-05-07T11:10:44.978000Z","iopub.execute_input":"2023-05-07T11:10:44.982682Z","iopub.status.idle":"2023-05-07T11:10:44.990997Z","shell.execute_reply.started":"2023-05-07T11:10:44.982636Z","shell.execute_reply":"2023-05-07T11:10:44.989984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compare the number of elements in id\nfor name, df in dataframes.items():\n    print(f'{color.BLUE}{name}{color.END} (The number of elements in column id of data is ):')\n    print((df['Id'].nunique()))","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:44.992552Z","iopub.execute_input":"2023-05-07T11:10:44.994402Z","iopub.status.idle":"2023-05-07T11:10:45.015904Z","shell.execute_reply.started":"2023-05-07T11:10:44.993207Z","shell.execute_reply":"2023-05-07T11:10:45.013085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pie chart of column Type in data frame events\nplot_frequency_pie_charts([events], 'Type', ['Type of events'])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:45.017118Z","iopub.execute_input":"2023-05-07T11:10:45.017851Z","iopub.status.idle":"2023-05-07T11:10:45.235482Z","shell.execute_reply.started":"2023-05-07T11:10:45.017815Z","shell.execute_reply":"2023-05-07T11:10:45.234492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pie chart of column Type in data frame events\nplot_frequency_pie_charts([defog_meta], 'Medication', ['Medication of defog_meta'])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:45.237063Z","iopub.execute_input":"2023-05-07T11:10:45.238222Z","iopub.status.idle":"2023-05-07T11:10:45.425559Z","shell.execute_reply.started":"2023-05-07T11:10:45.238181Z","shell.execute_reply":"2023-05-07T11:10:45.424491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display a line chart of 3D accelerometer data \nplot_line_graph(train_tdcsfog_example, 'Line graph of 3D accelerometer time series data')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:45.427372Z","iopub.execute_input":"2023-05-07T11:10:45.428550Z","iopub.status.idle":"2023-05-07T11:10:45.783597Z","shell.execute_reply.started":"2023-05-07T11:10:45.428509Z","shell.execute_reply":"2023-05-07T11:10:45.782401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display a line chart of 3D accelerometer data \nplot_line_graph(train_tdcsfog_zero_example, 'Line graph of 3D accelerometer time series data with Turn being zero')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:45.785195Z","iopub.execute_input":"2023-05-07T11:10:45.785591Z","iopub.status.idle":"2023-05-07T11:10:46.121330Z","shell.execute_reply.started":"2023-05-07T11:10:45.785551Z","shell.execute_reply":"2023-05-07T11:10:46.120405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display a line chart of 3D accelerometer data \nplot_line_graph(train_tdcsfog_one_example, 'Line graph of 3D accelerometer time series data with Turn being one')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:46.123000Z","iopub.execute_input":"2023-05-07T11:10:46.123712Z","iopub.status.idle":"2023-05-07T11:10:46.441365Z","shell.execute_reply.started":"2023-05-07T11:10:46.123673Z","shell.execute_reply":"2023-05-07T11:10:46.440340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# memory deallocation\ngc.collect()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-05-07T11:10:46.442878Z","iopub.execute_input":"2023-05-07T11:10:46.444025Z","iopub.status.idle":"2023-05-07T11:10:46.698524Z","shell.execute_reply.started":"2023-05-07T11:10:46.443975Z","shell.execute_reply":"2023-05-07T11:10:46.696216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use the function for the required dataframes\ndisplay_correlation_heatmap(train_tdcsfog_example.drop([\"StartHesitation\", \"Walking\"], axis=1), \"Correlation Matrix of tdcsfog train instance\")\ndisplay_correlation_heatmap(test_tdcsfog_example, \"Correlation Matrix of tdcsfog test instance\")\ndisplay_correlation_heatmap(unlabeled_example, \"Correlation Matrix of unlabeled instance\")","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:46.699988Z","iopub.execute_input":"2023-05-07T11:10:46.700859Z","iopub.status.idle":"2023-05-07T11:10:51.938455Z","shell.execute_reply.started":"2023-05-07T11:10:46.700800Z","shell.execute_reply":"2023-05-07T11:10:51.937366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* No existing features appear to be highly correlated.","metadata":{}},{"cell_type":"code","source":"unique_subject_id = \"13abfd\"\ntdcsfog_meta[tdcsfog_meta.Subject == unique_subject_id]","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:51.939881Z","iopub.execute_input":"2023-05-07T11:10:51.941833Z","iopub.status.idle":"2023-05-07T11:10:51.953646Z","shell.execute_reply.started":"2023-05-07T11:10:51.941792Z","shell.execute_reply":"2023-05-07T11:10:51.952497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_bar_chart(tdcsfog_meta, 'Visit', 'Visit', 'Count')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:51.955317Z","iopub.execute_input":"2023-05-07T11:10:51.956179Z","iopub.status.idle":"2023-05-07T11:10:52.898068Z","shell.execute_reply.started":"2023-05-07T11:10:51.956140Z","shell.execute_reply":"2023-05-07T11:10:52.896157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pie chart of column Test in data frame tdcsfog_meta\nplot_frequency_pie_charts([tdcsfog_meta], 'Test', ['Test of tdcsfog_meta'])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:52.899794Z","iopub.execute_input":"2023-05-07T11:10:52.900514Z","iopub.status.idle":"2023-05-07T11:10:53.109173Z","shell.execute_reply.started":"2023-05-07T11:10:52.900474Z","shell.execute_reply":"2023-05-07T11:10:53.104558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pie chart of column Medication in data frame tdcsfog_meta\nplot_frequency_pie_charts([tdcsfog_meta], 'Medication', ['Medication of tdcsfog_meta'])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:53.111037Z","iopub.execute_input":"2023-05-07T11:10:53.111602Z","iopub.status.idle":"2023-05-07T11:10:53.326038Z","shell.execute_reply.started":"2023-05-07T11:10:53.111555Z","shell.execute_reply":"2023-05-07T11:10:53.324644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_subject_id = \"bf608b\"\ndefog_meta[defog_meta.Subject == unique_subject_id]","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:53.332025Z","iopub.execute_input":"2023-05-07T11:10:53.335598Z","iopub.status.idle":"2023-05-07T11:10:53.353360Z","shell.execute_reply.started":"2023-05-07T11:10:53.335536Z","shell.execute_reply":"2023-05-07T11:10:53.352233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pie chart of column Visit in data frame defog_meta\nplot_frequency_pie_charts([defog_meta], 'Visit', ['Visit of defog_meta'])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:53.355487Z","iopub.execute_input":"2023-05-07T11:10:53.356527Z","iopub.status.idle":"2023-05-07T11:10:53.540991Z","shell.execute_reply.started":"2023-05-07T11:10:53.356488Z","shell.execute_reply":"2023-05-07T11:10:53.539996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pie chart of column Medication in data frame defog_meta\nplot_frequency_pie_charts([defog_meta], 'Medication', ['Medication of defog_meta'])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:53.542367Z","iopub.execute_input":"2023-05-07T11:10:53.543199Z","iopub.status.idle":"2023-05-07T11:10:53.735688Z","shell.execute_reply.started":"2023-05-07T11:10:53.543160Z","shell.execute_reply":"2023-05-07T11:10:53.734635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pie chart of column Visit in data frame daily_meta\nplot_frequency_pie_charts([train_tdcsfog_example], 'Walking', ['Walking of train_tdcsfog_example'])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:53.737126Z","iopub.execute_input":"2023-05-07T11:10:53.737688Z","iopub.status.idle":"2023-05-07T11:10:53.929412Z","shell.execute_reply.started":"2023-05-07T11:10:53.737648Z","shell.execute_reply":"2023-05-07T11:10:53.928234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pie chart of column Visit in data frame daily_meta\nplot_frequency_pie_charts([daily_meta], 'Visit', ['Visit of daily_meta'])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:53.931141Z","iopub.execute_input":"2023-05-07T11:10:53.931993Z","iopub.status.idle":"2023-05-07T11:10:54.153383Z","shell.execute_reply.started":"2023-05-07T11:10:53.931952Z","shell.execute_reply":"2023-05-07T11:10:54.152224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"daily_bor_counts = daily_meta[\"Beginning of recording [00:00-23:59]\"].value_counts()\n\nfig = px.bar(x=daily_bor_counts.index, y=daily_bor_counts.values, color_discrete_sequence=['cornflowerblue'])\nfig.update_layout(xaxis_title=\"Beginning of recording\", yaxis_title=\"Count\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:54.155038Z","iopub.execute_input":"2023-05-07T11:10:54.156229Z","iopub.status.idle":"2023-05-07T11:10:54.229402Z","shell.execute_reply.started":"2023-05-07T11:10:54.156185Z","shell.execute_reply":"2023-05-07T11:10:54.228408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use the functions for the required dataframes and columns\ndisplay_histogram(subjects, 'Age')\ndisplay_histogram(subjects, 'YearsSinceDx')\ndisplay_histogram(subjects, 'UPDRSIII_On')\ndisplay_histogram(subjects, 'NFOGQ')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:54.231134Z","iopub.execute_input":"2023-05-07T11:10:54.231874Z","iopub.status.idle":"2023-05-07T11:10:54.427858Z","shell.execute_reply.started":"2023-05-07T11:10:54.231834Z","shell.execute_reply":"2023-05-07T11:10:54.426745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pie chart of column Visit in data frame subjects\nplot_frequency_pie_charts([subjects], 'Visit', ['Visit of subjects'])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:54.429640Z","iopub.execute_input":"2023-05-07T11:10:54.430420Z","iopub.status.idle":"2023-05-07T11:10:54.643370Z","shell.execute_reply.started":"2023-05-07T11:10:54.430377Z","shell.execute_reply":"2023-05-07T11:10:54.642149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pie chart of column Sex in data frame subjects\nplot_frequency_pie_charts([subjects], 'Sex', ['Sex of subjects'])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:54.645392Z","iopub.execute_input":"2023-05-07T11:10:54.646296Z","iopub.status.idle":"2023-05-07T11:10:54.875290Z","shell.execute_reply.started":"2023-05-07T11:10:54.646252Z","shell.execute_reply":"2023-05-07T11:10:54.873893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the total value \nbar = tqdm(total = 100)\n# Add description\nbar.set_description('Progress rate')\nfor i in range(100):\n    # Set the progress\n    bar.update(5)\n    time.sleep(1)","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:10:54.881700Z","iopub.execute_input":"2023-05-07T11:10:54.885456Z","iopub.status.idle":"2023-05-07T11:12:35.088256Z","shell.execute_reply.started":"2023-05-07T11:10:54.885388Z","shell.execute_reply":"2023-05-07T11:12:35.087214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation Matrix of dataframes features\ndisplay_correlation_heatmap(subjects, \"Correlation Matrix of subjects features\")\ndisplay_correlation_heatmap(subjects_m, \"Correlation Matrix of men's subjects features\")\ndisplay_correlation_heatmap(subjects_f, \"Correlation Matrix of women's subjects features\")\n# display_correlation_heatmap(events, \"Correlation Matrix of events features\")","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:12:35.089619Z","iopub.execute_input":"2023-05-07T11:12:35.089981Z","iopub.status.idle":"2023-05-07T11:12:36.148320Z","shell.execute_reply.started":"2023-05-07T11:12:35.089945Z","shell.execute_reply":"2023-05-07T11:12:36.147366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the total value \nbar = tqdm(total = 100)\n# Add description\nbar.set_description('Progress rate')\nfor i in range(100):\n    # Set the progress\n    bar.update(5)\n    time.sleep(1)","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:12:36.149826Z","iopub.execute_input":"2023-05-07T11:12:36.150438Z","iopub.status.idle":"2023-05-07T11:14:16.347598Z","shell.execute_reply.started":"2023-05-07T11:12:36.150389Z","shell.execute_reply":"2023-05-07T11:14:16.346333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# memory deallocation\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:16.348972Z","iopub.execute_input":"2023-05-07T11:14:16.349368Z","iopub.status.idle":"2023-05-07T11:14:16.638914Z","shell.execute_reply.started":"2023-05-07T11:14:16.349320Z","shell.execute_reply":"2023-05-07T11:14:16.637483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation between age and NFOGQ\nsns.jointplot(x='Age', y='NFOGQ', data=subjects, kind='reg')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:16.640677Z","iopub.execute_input":"2023-05-07T11:14:16.641252Z","iopub.status.idle":"2023-05-07T11:14:17.365412Z","shell.execute_reply.started":"2023-05-07T11:14:16.641211Z","shell.execute_reply":"2023-05-07T11:14:17.364391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation between age and NFOGQ of Men's data\nsns.jointplot(x='Age', y='NFOGQ', data=subjects_m, kind='reg')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:17.366869Z","iopub.execute_input":"2023-05-07T11:14:17.367329Z","iopub.status.idle":"2023-05-07T11:14:18.101011Z","shell.execute_reply.started":"2023-05-07T11:14:17.367290Z","shell.execute_reply":"2023-05-07T11:14:18.099892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation between age and NFOGQ of Women's data\nsns.jointplot(x='Age', y='NFOGQ', data=subjects_f, kind='reg')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:18.102678Z","iopub.execute_input":"2023-05-07T11:14:18.103134Z","iopub.status.idle":"2023-05-07T11:14:18.883094Z","shell.execute_reply.started":"2023-05-07T11:14:18.103097Z","shell.execute_reply":"2023-05-07T11:14:18.881935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation between age and NFOGQ\nsns.jointplot(x='Age', y='UPDRSIII_On', data=subjects, kind='reg')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:18.884879Z","iopub.execute_input":"2023-05-07T11:14:18.885330Z","iopub.status.idle":"2023-05-07T11:14:19.849532Z","shell.execute_reply.started":"2023-05-07T11:14:18.885293Z","shell.execute_reply":"2023-05-07T11:14:19.848404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation between age and NFOGQ of Men's data\nsns.jointplot(x='Age', y='UPDRSIII_On', data=subjects_m, kind='reg')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:19.851149Z","iopub.execute_input":"2023-05-07T11:14:19.852348Z","iopub.status.idle":"2023-05-07T11:14:20.693656Z","shell.execute_reply.started":"2023-05-07T11:14:19.852303Z","shell.execute_reply":"2023-05-07T11:14:20.692353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation between age and NFOGQ of Women's data\nsns.jointplot(x='Age', y='UPDRSIII_On', data=subjects_f, kind='reg')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:20.695512Z","iopub.execute_input":"2023-05-07T11:14:20.696011Z","iopub.status.idle":"2023-05-07T11:14:21.417254Z","shell.execute_reply.started":"2023-05-07T11:14:20.695970Z","shell.execute_reply":"2023-05-07T11:14:21.416119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation between age and NFOGQ\nsns.jointplot(x='Age', y='UPDRSIII_Off', data=subjects, kind='reg')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:21.418740Z","iopub.execute_input":"2023-05-07T11:14:21.419191Z","iopub.status.idle":"2023-05-07T11:14:22.120542Z","shell.execute_reply.started":"2023-05-07T11:14:21.419152Z","shell.execute_reply":"2023-05-07T11:14:22.119382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation between age and NFOGQ of Men's data\nsns.jointplot(x='Age', y='UPDRSIII_Off', data=subjects_m, kind='reg')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:22.121988Z","iopub.execute_input":"2023-05-07T11:14:22.122482Z","iopub.status.idle":"2023-05-07T11:14:22.801504Z","shell.execute_reply.started":"2023-05-07T11:14:22.122422Z","shell.execute_reply":"2023-05-07T11:14:22.800380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation between age and NFOGQ of Women's data\nsns.jointplot(x='Age', y='UPDRSIII_Off', data=subjects_f, kind='reg')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:22.802948Z","iopub.execute_input":"2023-05-07T11:14:22.804060Z","iopub.status.idle":"2023-05-07T11:14:23.481909Z","shell.execute_reply.started":"2023-05-07T11:14:22.804016Z","shell.execute_reply":"2023-05-07T11:14:23.480383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# memory deallocation\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:23.483688Z","iopub.execute_input":"2023-05-07T11:14:23.484479Z","iopub.status.idle":"2023-05-07T11:14:23.776063Z","shell.execute_reply.started":"2023-05-07T11:14:23.484421Z","shell.execute_reply":"2023-05-07T11:14:23.774805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tasks[\"Duration\"] = tasks.End - tasks.Begin\nfig = px.histogram(x=tasks[\"Duration\"], nbins=30, color_discrete_sequence=['lightslategrey'])\nfig.update_layout(\n    xaxis_title=\"Task duration\",\n    title={\n        'text': \"Distribution of task durations\",\n        'y':0.95,\n        'x':0.5,\n        'xanchor': 'center',\n        'yanchor': 'top'\n    }\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:23.777822Z","iopub.execute_input":"2023-05-07T11:14:23.778234Z","iopub.status.idle":"2023-05-07T11:14:23.848705Z","shell.execute_reply.started":"2023-05-07T11:14:23.778196Z","shell.execute_reply":"2023-05-07T11:14:23.844093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tasks_task_counts = tasks.Task.value_counts()\n\nfig = px.bar(x=tasks_task_counts.index, y=tasks_task_counts.values, color_discrete_sequence=['lightslategrey'])\nfig.update_layout(xaxis_title=\"Task\", yaxis_title=\"Count\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:23.850253Z","iopub.execute_input":"2023-05-07T11:14:23.850735Z","iopub.status.idle":"2023-05-07T11:14:23.910100Z","shell.execute_reply.started":"2023-05-07T11:14:23.850697Z","shell.execute_reply":"2023-05-07T11:14:23.908905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pie chart of column Type in data frame events\nplot_frequency_pie_charts([tasks], 'Task', ['Task of tasks'])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:23.911962Z","iopub.execute_input":"2023-05-07T11:14:23.912319Z","iopub.status.idle":"2023-05-07T11:14:24.647275Z","shell.execute_reply.started":"2023-05-07T11:14:23.912282Z","shell.execute_reply":"2023-05-07T11:14:24.646064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.box(\n    tasks, \n    x='Duration', y='Task',\n    orientation='h', height=1000, \n    color_discrete_sequence=['lightslategrey']\n)\n\nfig.update_layout(\n    title={\n        'text': \"Task Duration Boxplot\",\n        'y':0.98,\n        'x':0.5,\n        'xanchor': 'center',\n        'yanchor': 'top'\n    },\n    margin=dict(l=50, r=0, t=50, b=50)\n)\nfig.show(autosize=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:24.648915Z","iopub.execute_input":"2023-05-07T11:14:24.650039Z","iopub.status.idle":"2023-05-07T11:14:24.761760Z","shell.execute_reply.started":"2023-05-07T11:14:24.649993Z","shell.execute_reply":"2023-05-07T11:14:24.760412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# memory deallocation\ngc.collect()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-05-07T11:14:24.763328Z","iopub.execute_input":"2023-05-07T11:14:24.764569Z","iopub.status.idle":"2023-05-07T11:14:25.171914Z","shell.execute_reply.started":"2023-05-07T11:14:24.764526Z","shell.execute_reply":"2023-05-07T11:14:25.170667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# View outliers\nscatterplot('Age', 'NFOGQ', subjects)","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:25.173977Z","iopub.execute_input":"2023-05-07T11:14:25.174401Z","iopub.status.idle":"2023-05-07T11:14:25.500685Z","shell.execute_reply.started":"2023-05-07T11:14:25.174359Z","shell.execute_reply":"2023-05-07T11:14:25.499496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* For ages 90+ and under 30, they may be regarded as outliers. However, outliers should be considered with caution as they are human data.","metadata":{}},{"cell_type":"code","source":"# Take the 'ID' column from data frames defog_meta and tdcsfog_meta and check for duplicate IDs.\ncommon_key = pd.merge(defog_meta['Id'], tdcsfog_meta['Id'], on='Id')\n\n# Show the number of duplicate IDs\nprint(\"Number of common IDs:\", len(common_key))\n\n# Show duplicate IDs\nprint(\"Common IDs:\", common_key['Id'].tolist())","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:25.502658Z","iopub.execute_input":"2023-05-07T11:14:25.503098Z","iopub.status.idle":"2023-05-07T11:14:25.520160Z","shell.execute_reply.started":"2023-05-07T11:14:25.503056Z","shell.execute_reply":"2023-05-07T11:14:25.519101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine data frames defog_meta and data frames tdcsfog_meta\nmetadata = pd.concat([defog_meta,tdcsfog_meta],axis = 0).reset_index(drop = True)\n\n# Check the results of combining defog_meta and tdcsfog_meta\nname = 'The data frame combining defog_meta and tdcsfog_meta'\ndf = metadata\nprint(f'{color.BLUE}{name}{color.END} (head of the data is):')\nprint(df.head())","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:25.521510Z","iopub.execute_input":"2023-05-07T11:14:25.522806Z","iopub.status.idle":"2023-05-07T11:14:25.537641Z","shell.execute_reply.started":"2023-05-07T11:14:25.522743Z","shell.execute_reply":"2023-05-07T11:14:25.536155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the number of unique Id's of events first\nname = events\nprint(events['Id'].nunique())","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:25.539625Z","iopub.execute_input":"2023-05-07T11:14:25.540842Z","iopub.status.idle":"2023-05-07T11:14:25.552784Z","shell.execute_reply.started":"2023-05-07T11:14:25.540798Z","shell.execute_reply":"2023-05-07T11:14:25.550536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the number of unique Id's of tdcsfog_meta first\nname = tdcsfog_meta\nprint(tdcsfog_meta['Id'].nunique())","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:25.554852Z","iopub.execute_input":"2023-05-07T11:14:25.555370Z","iopub.status.idle":"2023-05-07T11:14:25.563115Z","shell.execute_reply.started":"2023-05-07T11:14:25.555328Z","shell.execute_reply":"2023-05-07T11:14:25.561403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Take the 'ID' column from data frames events and tdcsfog_meta and check for duplicate IDs.\ncommon_key = pd.merge(events['Id'], tdcsfog_meta['Id'], on='Id')\n\n# Take the 'ID' column from data frames events and tdcsfog_meta and check for duplicate IDs.\ncommon_key = pd.merge(events['Id'], tdcsfog_meta['Id'], on='Id')\n\n# Calculate the total number of unique IDs in each data frame\ntotal_unique_ids = len(set(events['Id']).union(set(tdcsfog_meta['Id'])))\n\n# Calculate the number of common IDs\nnum_common_ids = len(common_key)\n\n# Calculate the duplication rate\nduplication_rate = num_common_ids / total_unique_ids\n\n# Show the number of duplicate IDs\nprint(\"Number of common IDs:\", num_common_ids)\n\n# Show the duplication rate\nprint(\"Duplication rate: {:.2f}%\".format(duplication_rate))","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:25.565356Z","iopub.execute_input":"2023-05-07T11:14:25.566525Z","iopub.status.idle":"2023-05-07T11:14:25.586708Z","shell.execute_reply.started":"2023-05-07T11:14:25.566483Z","shell.execute_reply":"2023-05-07T11:14:25.585478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine data frames metadata and events\npre_integrated_data_zero_turn = tdcsfog_meta.merge(events_zero_turn, how='inner', on='Id')\n\n# Check the results of combining defog_meta and tdcsfog_meta\nname = 'The data frame combining events and tdcsfog_meta'\ndf = pre_integrated_data_zero_turn\nprint(f'{color.BLUE}{name}{color.END} (head of the data is):')\nprint(df.head())","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:25.588556Z","iopub.execute_input":"2023-05-07T11:14:25.588975Z","iopub.status.idle":"2023-05-07T11:14:25.607259Z","shell.execute_reply.started":"2023-05-07T11:14:25.588934Z","shell.execute_reply":"2023-05-07T11:14:25.606363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine data frames metadata and events\npre_integrated_data_zero_walking = tdcsfog_meta.merge(events_zero_walking, how='inner', on='Id')\n\n# Check the results of combining defog_meta and tdcsfog_meta\nname = 'The data frame combining events and tdcsfog_meta'\ndf = pre_integrated_data_zero_walking\nprint(f'{color.BLUE}{name}{color.END} (head of the data is):')\nprint(df.head())","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:25.608812Z","iopub.execute_input":"2023-05-07T11:14:25.609972Z","iopub.status.idle":"2023-05-07T11:14:25.625934Z","shell.execute_reply.started":"2023-05-07T11:14:25.609928Z","shell.execute_reply":"2023-05-07T11:14:25.624324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine data frames metadata and events\npre_integrated_data_zero_starthesitation = tdcsfog_meta.merge(events_zero_starthesitation, how='inner', on='Id')\n\n# Check the results of combining defog_meta and tdcsfog_meta\nname = 'The data frame combining events and tdcsfog_meta'\ndf = pre_integrated_data_zero_starthesitation\nprint(f'{color.BLUE}{name}{color.END} (head of the data is):')\nprint(df.head())","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:25.627547Z","iopub.execute_input":"2023-05-07T11:14:25.628357Z","iopub.status.idle":"2023-05-07T11:14:25.647596Z","shell.execute_reply.started":"2023-05-07T11:14:25.628257Z","shell.execute_reply":"2023-05-07T11:14:25.646138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine data frames metadata and events\npre_integrated_data_events_one_turn = tdcsfog_meta.merge(events_one_turn, how='inner', on='Id')\n\n# Check the results of combining defog_meta and tdcsfog_meta\nname = 'The data frame combining events and tdcsfog_meta'\ndf = pre_integrated_data_events_one_turn\nprint(f'{color.BLUE}{name}{color.END} (head of the data is):')\nprint(df.head())","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:25.649263Z","iopub.execute_input":"2023-05-07T11:14:25.650307Z","iopub.status.idle":"2023-05-07T11:14:25.667965Z","shell.execute_reply.started":"2023-05-07T11:14:25.650265Z","shell.execute_reply":"2023-05-07T11:14:25.666596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine data frames metadata and events\npre_integrated_data_events_one_walking = tdcsfog_meta.merge(events_one_walking, how='inner', on='Id')\n\n# Check the results of combining defog_meta and tdcsfog_meta\nname = 'The data frame combining events and tdcsfog_meta'\ndf = pre_integrated_data_events_one_walking\nprint(f'{color.BLUE}{name}{color.END} (head of the data is):')\nprint(df.head())","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:25.681366Z","iopub.execute_input":"2023-05-07T11:14:25.682112Z","iopub.status.idle":"2023-05-07T11:14:25.699052Z","shell.execute_reply.started":"2023-05-07T11:14:25.682063Z","shell.execute_reply":"2023-05-07T11:14:25.697846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine data frames metadata and events\npre_integrated_data_one_starthesitation = tdcsfog_meta.merge(events_one_starthesitation, how='inner', on='Id')\n\n# Check the results of combining defog_meta and tdcsfog_meta\nname = 'The data frame combining events and tdcsfog_meta'\ndf = pre_integrated_data_one_starthesitation\nprint(f'{color.BLUE}{name}{color.END} (head of the data is):')\nprint(df.head())","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:25.700510Z","iopub.execute_input":"2023-05-07T11:14:25.701542Z","iopub.status.idle":"2023-05-07T11:14:25.716744Z","shell.execute_reply.started":"2023-05-07T11:14:25.701495Z","shell.execute_reply":"2023-05-07T11:14:25.715195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the total value \nbar = tqdm(total = 100)\n# Add description\nbar.set_description('Progress rate')\nfor i in range(100):\n    # Set the progress\n    bar.update(5)\n    time.sleep(1)","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:14:25.719024Z","iopub.execute_input":"2023-05-07T11:14:25.719486Z","iopub.status.idle":"2023-05-07T11:16:05.923616Z","shell.execute_reply.started":"2023-05-07T11:14:25.719423Z","shell.execute_reply":"2023-05-07T11:16:05.922485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine data frames metadata and pre_integrated_data\nintegrated_data_zero_turn = pre_integrated_data_zero_turn.merge(metadata, how='inner', on='Id')\nintegrated_data_zero_turn = pre_integrated_data_zero_turn.merge(metadata, how='inner', on='Id')\nintegrated_data_zero_walking = pre_integrated_data_zero_walking.merge(metadata, how='inner', on='Id')\nintegrated_data_zero_starthesitation = pre_integrated_data_zero_starthesitation.merge(metadata, how='inner', on='Id')\nintegrated_data_events_one_turn = pre_integrated_data_events_one_turn.merge(metadata, how='inner', on='Id')\nintegrated_data_one_starthesitation = pre_integrated_data_one_starthesitation.merge(metadata, how='inner', on='Id')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:16:05.925028Z","iopub.execute_input":"2023-05-07T11:16:05.926313Z","iopub.status.idle":"2023-05-07T11:16:05.957219Z","shell.execute_reply.started":"2023-05-07T11:16:05.926268Z","shell.execute_reply":"2023-05-07T11:16:05.956215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframes = [\n    integrated_data_zero_turn,\n    integrated_data_zero_turn,\n    integrated_data_zero_walking,\n    integrated_data_zero_starthesitation,\n    integrated_data_events_one_turn,\n    integrated_data_one_starthesitation\n]\n\nnames = [\n    'integrated_data_zero_turn',\n    'integrated_data_zero_turn',\n    'integrated_data_zero_walking',\n    'integrated_data_zero_starthesitation',\n    'integrated_data_events_one_turn',\n    'integrated_data_one_starthesitation'\n]\n\nclass color:\n    BLUE = '\\033[94m'\n    END = '\\033[0m'\n\nfor df, name in zip(dataframes, names):\n    # Check the results of combining defog_meta and tdcsfog_meta\n    print(f'{color.BLUE}{name}{color.END} (head of the data is):')\n    print(df.head())\n\n    # Check missing values\n    print(f'{color.BLUE}missing values of {name}{color.END} (percentage of missing values are):')\n    missing_values = df.isna().sum()\n    print(missing_values / len(df))\n    print()\n\n    # Check for missing values in the training features data\n    cnl = df.isnull().sum()\n\n    # If there are no missing values, skip the plot\n    if cnl.sum() == 0:\n        print(f'{color.BLUE}{name}{color.END} has no missing values, so no plot is displayed.')\n    else:\n        import matplotlib.pyplot as plt\n        import seaborn as sns\n\n        f, ax = plt.subplots(figsize=(8, 7))\n        sns.barplot(x=cnl, y=df.columns.values, orient='h')\n        ax.set_xlabel(\"null count\")\n        plt.tight_layout()\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:16:05.958657Z","iopub.execute_input":"2023-05-07T11:16:05.959595Z","iopub.status.idle":"2023-05-07T11:16:06.032638Z","shell.execute_reply.started":"2023-05-07T11:16:05.959550Z","shell.execute_reply":"2023-05-07T11:16:06.031473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Let us check the correlation coefficients for StartHesitation, Turn and Walking, as this is what the competition predicts in the StartHesitation and Turn and Walking columns, per ID.","metadata":{}},{"cell_type":"code","source":"# Statistics of Duration for data in data frame events whose column Kinetic is zero and column Type is Turn\nhistogram(integrated_data_zero_turn, 'Duration')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:16:06.034294Z","iopub.execute_input":"2023-05-07T11:16:06.034713Z","iopub.status.idle":"2023-05-07T11:16:06.296451Z","shell.execute_reply.started":"2023-05-07T11:16:06.034672Z","shell.execute_reply":"2023-05-07T11:16:06.295392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Statistics of Duration for data in data frame events whose column Kinetic is zero and column Type is Turn\nintegrated_data_zero_turn['Duration'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:16:06.297941Z","iopub.execute_input":"2023-05-07T11:16:06.298291Z","iopub.status.idle":"2023-05-07T11:16:06.308640Z","shell.execute_reply.started":"2023-05-07T11:16:06.298245Z","shell.execute_reply":"2023-05-07T11:16:06.307496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Histogram of Duration for data in data frame events whose column Kinetic is zero and column Type is Turn\nhistogram(integrated_data_zero_walking, 'Duration')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:16:06.310320Z","iopub.execute_input":"2023-05-07T11:16:06.311051Z","iopub.status.idle":"2023-05-07T11:16:06.563537Z","shell.execute_reply.started":"2023-05-07T11:16:06.311010Z","shell.execute_reply":"2023-05-07T11:16:06.562375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Statistics of Duration for data in data frame events whose column Kinetic is zero and column Type is Walking\nintegrated_data_zero_walking['Duration'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:16:06.564864Z","iopub.execute_input":"2023-05-07T11:16:06.565242Z","iopub.status.idle":"2023-05-07T11:16:06.580079Z","shell.execute_reply.started":"2023-05-07T11:16:06.565204Z","shell.execute_reply":"2023-05-07T11:16:06.578985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Histogram of Duration for data in data frame events whose column Kinetic is zero and column Type is StartHesitation\nhistogram(integrated_data_zero_starthesitation, 'Duration')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:16:06.581995Z","iopub.execute_input":"2023-05-07T11:16:06.582717Z","iopub.status.idle":"2023-05-07T11:16:06.849555Z","shell.execute_reply.started":"2023-05-07T11:16:06.582674Z","shell.execute_reply":"2023-05-07T11:16:06.848585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Statistics of Duration for data in data frame events whose column Kinetic is zero and column Type is StartHesitation\nintegrated_data_zero_starthesitation['Duration'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:16:06.851454Z","iopub.execute_input":"2023-05-07T11:16:06.852225Z","iopub.status.idle":"2023-05-07T11:16:06.864896Z","shell.execute_reply.started":"2023-05-07T11:16:06.852182Z","shell.execute_reply":"2023-05-07T11:16:06.863828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Histogram of Duration for data in data frame events whose column Kinetic is one and column Type is Turn\nhistogram(integrated_data_events_one_turn, 'Duration')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:16:06.866695Z","iopub.execute_input":"2023-05-07T11:16:06.867358Z","iopub.status.idle":"2023-05-07T11:16:07.136370Z","shell.execute_reply.started":"2023-05-07T11:16:06.867306Z","shell.execute_reply":"2023-05-07T11:16:07.135341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Statistics of Duration for data in data frame events whose column Kinetic is one and column Type is Turn\nintegrated_data_events_one_turn['Duration'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:16:07.137842Z","iopub.execute_input":"2023-05-07T11:16:07.138506Z","iopub.status.idle":"2023-05-07T11:16:07.149615Z","shell.execute_reply.started":"2023-05-07T11:16:07.138466Z","shell.execute_reply":"2023-05-07T11:16:07.148365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Histogram of Duration for data in data frame events whose column Kinetic is one and column Type is StartHesitation\nhistogram(integrated_data_one_starthesitation, 'Duration')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:16:07.151046Z","iopub.execute_input":"2023-05-07T11:16:07.152092Z","iopub.status.idle":"2023-05-07T11:16:07.412171Z","shell.execute_reply.started":"2023-05-07T11:16:07.152051Z","shell.execute_reply":"2023-05-07T11:16:07.411239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Statistics of Duration for data in data frame events whose column Kinetic is one and column Type is StartHesitation\nintegrated_data_one_starthesitation['Duration'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:16:07.414472Z","iopub.execute_input":"2023-05-07T11:16:07.415164Z","iopub.status.idle":"2023-05-07T11:16:07.426620Z","shell.execute_reply.started":"2023-05-07T11:16:07.415125Z","shell.execute_reply":"2023-05-07T11:16:07.425031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine data frames\npre_integrated_data = tdcsfog_meta.merge(events, how='inner', on='Id')\nintegrated_data = pre_integrated_data.merge(metadata, how='inner', on='Id')\nintegrated_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:16:07.428656Z","iopub.execute_input":"2023-05-07T11:16:07.429118Z","iopub.status.idle":"2023-05-07T11:16:07.457666Z","shell.execute_reply.started":"2023-05-07T11:16:07.429077Z","shell.execute_reply":"2023-05-07T11:16:07.456571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dummy variable conversion\ndf = integrated_data\ntype_dummies = pd.get_dummies(df['Type'], prefix='Type', drop_first=True)\n\n# Dummy variables are merged into the original data frame\nintegrated_data = pd.concat([df.drop('Type', axis=1), type_dummies], axis=1)\nintegrated_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:16:07.459342Z","iopub.execute_input":"2023-05-07T11:16:07.459766Z","iopub.status.idle":"2023-05-07T11:16:07.486075Z","shell.execute_reply.started":"2023-05-07T11:16:07.459728Z","shell.execute_reply":"2023-05-07T11:16:07.484982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = integrated_data\nsns.heatmap(df.corr(), annot=True, cmap='coolwarm')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T11:16:07.487984Z","iopub.execute_input":"2023-05-07T11:16:07.488395Z","iopub.status.idle":"2023-05-07T11:16:08.144090Z","shell.execute_reply.started":"2023-05-07T11:16:07.488355Z","shell.execute_reply":"2023-05-07T11:16:08.142929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the total value\nbar = tqdm(total=500)\n# Add description\nbar.set_description('Progress rate')\nfor i in range(100):\n    # Set the progress\n    bar.update(5)\n    time.sleep(1)\n    if bar.n >= 500:\n        print(\"finish!\")\n        break","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-05-07T11:16:08.145971Z","iopub.execute_input":"2023-05-07T11:16:08.146376Z","iopub.status.idle":"2023-05-07T11:17:48.373680Z","shell.execute_reply.started":"2023-05-07T11:16:08.146336Z","shell.execute_reply":"2023-05-07T11:17:48.372447Z"},"trusted":true},"execution_count":null,"outputs":[]}]}