{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport random\nimport cv2\nimport pandas as pd\nimport numpy as np\nimport plotly.express as px\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.cluster import KMeans\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2023-04-19T18:08:37.418348Z","iopub.execute_input":"2023-04-19T18:08:37.418938Z","iopub.status.idle":"2023-04-19T18:08:41.716885Z","shell.execute_reply.started":"2023-04-19T18:08:37.418896Z","shell.execute_reply":"2023-04-19T18:08:41.715513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class color:\n   PURPLE = '\\033[95m'\n   CYAN = '\\033[96m'\n   DARKCYAN = '\\033[36m'\n   BLUE = '\\033[94m'\n   GREEN = '\\033[92m'\n   YELLOW = '\\033[93m'\n   RED = '\\033[91m'\n   BOLD = '\\033[1m'\n   UNDERLINE = '\\033[4m'\n   END = '\\033[0m'","metadata":{"execution":{"iopub.status.busy":"2023-04-17T21:52:08.920514Z","iopub.execute_input":"2023-04-17T21:52:08.921351Z","iopub.status.idle":"2023-04-17T21:52:08.928243Z","shell.execute_reply.started":"2023-04-17T21:52:08.921308Z","shell.execute_reply":"2023-04-17T21:52:08.926556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I am interested in looking at the notype data. \n\n1. I first looked at a notype data example then looked at a defog example. \n2. Next, I looked at the features: AccV, AccML, and AccAP for each example.\n3. To go further into the notype data, I did a kmeans algorithm with just one notype data, then I did a clustering with all the notype csv files.\n4. I showed a visual representation of the kmeans clustering.\n5. I then showed how many labels were in each centroid.\n6. I then concatenated the file names for the notype data and looped through the defog_metadata to see if we have patients associated with the notype data.\n7. Then I made a list of the id's for the defog data.","metadata":{}},{"cell_type":"code","source":"#In Python, os.listdir() is a function in the os module that returns a list of \n# all files and directories in the specified directory. The function takes a \n# single argument, which is the path of the directory you want to list the contents of. \nos.listdir(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train\")","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:05:14.494587Z","iopub.execute_input":"2023-04-11T00:05:14.495024Z","iopub.status.idle":"2023-04-11T00:05:14.508049Z","shell.execute_reply.started":"2023-04-11T00:05:14.494990Z","shell.execute_reply":"2023-04-11T00:05:14.506595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"How many files are in defo train?","metadata":{}},{"cell_type":"code","source":"temp_defog = len(os.listdir(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog\"))\nprint(\n    f\"Number of files in folder defog/: {color.BLUE}{temp_defog}{color.END}\",\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T21:56:50.013226Z","iopub.execute_input":"2023-04-17T21:56:50.013727Z","iopub.status.idle":"2023-04-17T21:56:50.022137Z","shell.execute_reply.started":"2023-04-17T21:56:50.013682Z","shell.execute_reply":"2023-04-17T21:56:50.020683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"How many files are in folder notype/?","metadata":{}},{"cell_type":"code","source":"temp = len(os.listdir(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/notype\"))\nprint(\n    f\"Number of files in folder notype/: {color.BLUE}{temp}{color.END}\",\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T21:56:55.649272Z","iopub.execute_input":"2023-04-17T21:56:55.649727Z","iopub.status.idle":"2023-04-17T21:56:55.658569Z","shell.execute_reply.started":"2023-04-17T21:56:55.649688Z","shell.execute_reply":"2023-04-17T21:56:55.657050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"How the data looks:","metadata":{}},{"cell_type":"code","source":"train_notype_example_df = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/notype/0a900ed8a2.csv\")\ntrain_defog_example_df = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/02ea782681.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-04-19T15:48:52.173481Z","iopub.execute_input":"2023-04-19T15:48:52.174032Z","iopub.status.idle":"2023-04-19T15:48:53.011142Z","shell.execute_reply.started":"2023-04-19T15:48:52.173983Z","shell.execute_reply":"2023-04-19T15:48:53.009661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = len(train_notype_example_df)\nprint(\n    f\"Length of dataframe: {color.BLUE}{temp}{color.END}\",\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:05:38.200504Z","iopub.execute_input":"2023-04-11T00:05:38.202368Z","iopub.status.idle":"2023-04-11T00:05:38.210595Z","shell.execute_reply.started":"2023-04-11T00:05:38.202299Z","shell.execute_reply":"2023-04-11T00:05:38.208733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_2 = len(train_defog_example_df)\nprint(\n    f\"Length of dataframe: {color.BLUE}{temp_2}{color.END}\",\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:05:39.497566Z","iopub.execute_input":"2023-04-11T00:05:39.498118Z","iopub.status.idle":"2023-04-11T00:05:39.505573Z","shell.execute_reply.started":"2023-04-11T00:05:39.498060Z","shell.execute_reply":"2023-04-11T00:05:39.504011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_notype_example_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:05:53.452472Z","iopub.execute_input":"2023-04-11T00:05:53.452994Z","iopub.status.idle":"2023-04-11T00:05:53.486345Z","shell.execute_reply.started":"2023-04-11T00:05:53.452951Z","shell.execute_reply":"2023-04-11T00:05:53.484581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_defog_example_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:05:55.751953Z","iopub.execute_input":"2023-04-11T00:05:55.753123Z","iopub.status.idle":"2023-04-11T00:05:55.772094Z","shell.execute_reply.started":"2023-04-11T00:05:55.753044Z","shell.execute_reply":"2023-04-11T00:05:55.769774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_notype_example_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:06:02.133769Z","iopub.execute_input":"2023-04-11T00:06:02.134353Z","iopub.status.idle":"2023-04-11T00:06:02.218183Z","shell.execute_reply.started":"2023-04-11T00:06:02.134310Z","shell.execute_reply":"2023-04-11T00:06:02.216426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_defog_example_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:06:09.903213Z","iopub.execute_input":"2023-04-11T00:06:09.904357Z","iopub.status.idle":"2023-04-11T00:06:09.976318Z","shell.execute_reply.started":"2023-04-11T00:06:09.904285Z","shell.execute_reply":"2023-04-11T00:06:09.974933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Is there any NaN values in the dataframe?","metadata":{}},{"cell_type":"code","source":"train_notype_example_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:06:13.315511Z","iopub.execute_input":"2023-04-11T00:06:13.316002Z","iopub.status.idle":"2023-04-11T00:06:13.330193Z","shell.execute_reply.started":"2023-04-11T00:06:13.315963Z","shell.execute_reply":"2023-04-11T00:06:13.328534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_defog_example_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:06:16.066453Z","iopub.execute_input":"2023-04-11T00:06:16.067000Z","iopub.status.idle":"2023-04-11T00:06:16.082397Z","shell.execute_reply.started":"2023-04-11T00:06:16.066958Z","shell.execute_reply":"2023-04-11T00:06:16.080593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for column in ['AccV','AccML','AccAP', 'Event']:\n    fig = px.line(train_notype_example_df, x=\"Time\", y=column, color_discrete_sequence=['darkslateblue'])\n    fig.update_layout(\n        title={\n            'text': f\"{column} Time Series\",\n            'y':0.95,\n            'x':0.5,\n            'xanchor': 'center',\n            'yanchor': 'top'\n        }\n    )\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:07:14.319803Z","iopub.execute_input":"2023-04-11T00:07:14.320302Z","iopub.status.idle":"2023-04-11T00:07:14.950169Z","shell.execute_reply.started":"2023-04-11T00:07:14.320259Z","shell.execute_reply":"2023-04-11T00:07:14.948686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for column in ['AccV','AccML','AccAP']:\n    fig = px.line(train_defog_example_df, x=\"Time\", y=column, color_discrete_sequence=['darkslateblue'])\n    fig.update_layout(\n        title={\n            'text': f\"{column} Time Series\",\n            'y':0.95,\n            'x':0.5,\n            'xanchor': 'center',\n            'yanchor': 'top'\n        }\n    )\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:09:38.774093Z","iopub.execute_input":"2023-04-11T00:09:38.775269Z","iopub.status.idle":"2023-04-11T00:09:39.167629Z","shell.execute_reply.started":"2023-04-11T00:09:38.775207Z","shell.execute_reply":"2023-04-11T00:09:39.165071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## KMEANS?\nLet's try and make a kmeans algorithm. This is for one csv file.\n","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n# Load your data into a pandas dataframe or numpy array\ndata = train_notype_example_df\n\n# Create an instance of the KMeans class and set the number of clusters\nkmeans = KMeans(n_clusters=4)\n\n# Fit the KMeans model to your data\nkmeans.fit(data)\n\n# Retrieve the cluster assignments and centroids\nlabels = kmeans.labels_\ncentroids = kmeans.cluster_centers_\n\n# Visualize the clustering results (3D data)\nfig = plt.figure()\nax = fig.add_subplot(projection='3d') #1,2,3 are the columns I wanted from the df\nax.scatter(data.iloc[:, 1], data.iloc[:, 2], data.iloc[:, 3], c=labels, cmap='rainbow')\nax.scatter(centroids[:, 1], centroids[:,2], centroids[:, 3], marker='x', s=200, linewidths=3, color='black')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:15:12.391209Z","iopub.execute_input":"2023-04-11T00:15:12.391665Z","iopub.status.idle":"2023-04-11T00:15:20.287814Z","shell.execute_reply.started":"2023-04-11T00:15:12.391627Z","shell.execute_reply":"2023-04-11T00:15:20.286462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is just for me to make a list of the notype data instead of looking at each file name. Example0 and train_notype_example_df are the same file, but I was able to make the example be taken as an object so I can use it in a for loop later.","metadata":{}},{"cell_type":"code","source":"my_sequences = os.listdir(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/notype\")\ntrain_notype_example_df = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/notype/0a900ed8a2.csv\")\nexample0 = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/notype/\"+my_sequences[0])","metadata":{"execution":{"iopub.status.busy":"2023-04-12T01:40:44.918427Z","iopub.execute_input":"2023-04-12T01:40:44.919548Z","iopub.status.idle":"2023-04-12T01:40:45.670955Z","shell.execute_reply.started":"2023-04-12T01:40:44.919477Z","shell.execute_reply":"2023-04-12T01:40:45.669678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This is the list of notype data:\nlen(my_sequences)","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:15:39.756191Z","iopub.execute_input":"2023-04-11T00:15:39.756644Z","iopub.status.idle":"2023-04-11T00:15:39.765166Z","shell.execute_reply.started":"2023-04-11T00:15:39.756603Z","shell.execute_reply":"2023-04-11T00:15:39.763784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is the kmeans algorithm for all the 46 files. I concatenated the data and selected teh features I wanted from each file.","metadata":{}},{"cell_type":"code","source":"# Create an empty DataFrame to store the concatenated data\nconcatenated_data = pd.DataFrame()\n\n# Loop through each CSV file and concatenate the data to the DataFrame\nfor i in range(0,45): # 46 CSV files with names '... + my_sequences[i]\n    my_sequences = os.listdir(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/notype\")\n    #filename = 'data' + str(i) + '.csv'\n    data= pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/notype/\"+my_sequences[i])\n    #data = pd.read_csv(filename)\n    concatenated_data = pd.concat([concatenated_data, data], axis=0, ignore_index=True)\n\n# Select the three features you want to use in clustering\nfeatures = ['AccV', 'AccML', 'AccAP']\ndata_for_clustering = concatenated_data[features]\n\n# Create an instance of the KMeans class and set the number of clusters\nkmeans = KMeans(n_clusters=4)\n\n# Fit the KMeans model to your data\nkmeans.fit(data_for_clustering)\n\n# Retrieve the cluster assignments and centroids\nlabels = kmeans.labels_\ncentroids = kmeans.cluster_centers_\n","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:16:00.525548Z","iopub.execute_input":"2023-04-11T00:16:00.526933Z","iopub.status.idle":"2023-04-11T00:17:26.512971Z","shell.execute_reply.started":"2023-04-11T00:16:00.526875Z","shell.execute_reply":"2023-04-11T00:17:26.511532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data_for_clustering.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:17:32.974918Z","iopub.execute_input":"2023-04-11T00:17:32.976612Z","iopub.status.idle":"2023-04-11T00:17:32.984312Z","shell.execute_reply.started":"2023-04-11T00:17:32.976534Z","shell.execute_reply":"2023-04-11T00:17:32.983282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize the clustering results (3D data)\nfig = plt.figure()\nax = fig.add_subplot(projection='3d')\nax.scatter(data_for_clustering.iloc[:, 0], data_for_clustering.iloc[:, 1], data_for_clustering.iloc[:, 2], c=labels, cmap='rainbow')\nax.scatter(centroids[:, 0], centroids[:, 1], centroids[:, 2], marker='x', s=200, linewidths=3, color='black')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:17:35.778912Z","iopub.execute_input":"2023-04-11T00:17:35.779384Z","iopub.status.idle":"2023-04-11T00:21:23.746557Z","shell.execute_reply.started":"2023-04-11T00:17:35.779342Z","shell.execute_reply":"2023-04-11T00:21:23.745180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(labels.shape)\nprint(centroids)\nprint(data_for_clustering.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:21:34.563433Z","iopub.execute_input":"2023-04-11T00:21:34.563851Z","iopub.status.idle":"2023-04-11T00:21:34.571866Z","shell.execute_reply.started":"2023-04-11T00:21:34.563817Z","shell.execute_reply":"2023-04-11T00:21:34.570096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_labels = np.unique(labels)\ncounts = np.bincount(labels)\n\n# Print the cluster assignments and counts\nfor i, label in enumerate(unique_labels):\n    print(f\"Cluster {label}: {counts[i]} data points\")\n    print(f\"Centroid coordinates: {centroids[label]}\")","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:21:43.542643Z","iopub.execute_input":"2023-04-11T00:21:43.543221Z","iopub.status.idle":"2023-04-11T00:21:43.895203Z","shell.execute_reply.started":"2023-04-11T00:21:43.543171Z","shell.execute_reply":"2023-04-11T00:21:43.893655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize the clustering results (3D data)\nfig = plt.figure()\nax = fig.add_subplot(projection='3d')\n#ax.scatter(data_for_clustering.iloc[:, 0], data_for_clustering.iloc[:, 1], data_for_clustering.iloc[:, 2], c=labels, cmap='rainbow')\nax.scatter(centroids[:, 0], centroids[:, 1], centroids[:, 2], marker='x', s=200, linewidths=3, color='black')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T00:23:00.451734Z","iopub.execute_input":"2023-04-11T00:23:00.452201Z","iopub.status.idle":"2023-04-11T00:23:00.712506Z","shell.execute_reply.started":"2023-04-11T00:23:00.452160Z","shell.execute_reply":"2023-04-11T00:23:00.711148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Id's for the notype?","metadata":{}},{"cell_type":"markdown","source":"Next, I'd like to see if the filenames in notype are also listed in the metadata to see if I can get more information on the patients.\n\n1. First, I will concantenate all the csv files in one df for notype \n2. I will then loop through the file names in noytpe and see if they are the in the defog metadata.","metadata":{}},{"cell_type":"code","source":"# Create an empty DataFrame to store the concatenated data for notype\nconcatenated_data_notype = pd.DataFrame()\n\n\n# concatenate the notype data into a data frame\nfor i in range(0,45): # 46 CSV files with names '... + my_sequences[i]\n    my_sequences = os.listdir(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/notype\")\n    data = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/notype/\"+my_sequences[i])\n    concatenated_data_notype = pd.concat([concatenated_data_notype, data], axis=0, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T15:49:50.358091Z","iopub.execute_input":"2023-04-19T15:49:50.358692Z","iopub.status.idle":"2023-04-19T15:50:17.321219Z","shell.execute_reply.started":"2023-04-19T15:49:50.358644Z","shell.execute_reply":"2023-04-19T15:50:17.320117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(concatenated_data_notype.shape)\nprint(concatenated_data_notype.head())\nprint(concatenated_data_notype.describe())","metadata":{"execution":{"iopub.status.busy":"2023-04-19T03:49:54.412341Z","iopub.execute_input":"2023-04-19T03:49:54.412801Z","iopub.status.idle":"2023-04-19T03:49:56.375254Z","shell.execute_reply.started":"2023-04-19T03:49:54.412761Z","shell.execute_reply":"2023-04-19T03:49:56.374087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_metadata = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv\")\ndefog_metadata","metadata":{"execution":{"iopub.status.busy":"2023-04-19T03:49:56.376707Z","iopub.execute_input":"2023-04-19T03:49:56.377573Z","iopub.status.idle":"2023-04-19T03:49:56.405700Z","shell.execute_reply.started":"2023-04-19T03:49:56.377534Z","shell.execute_reply":"2023-04-19T03:49:56.404482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# here is my list of Id's from my notype csv files.\nmy_sequences = os.listdir(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/notype\")\nprint(my_sequences)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T16:40:50.221059Z","iopub.execute_input":"2023-04-19T16:40:50.221818Z","iopub.status.idle":"2023-04-19T16:40:50.232766Z","shell.execute_reply.started":"2023-04-19T16:40:50.221770Z","shell.execute_reply":"2023-04-19T16:40:50.231098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loop through each file name in the list and remove the last '.csv' extension\nfor i in range(len(my_sequences)):\n    my_sequences[i] = my_sequences[i].rsplit('.csv', 1)[0]\n\nprint(my_sequences)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T16:40:56.830420Z","iopub.execute_input":"2023-04-19T16:40:56.831454Z","iopub.status.idle":"2023-04-19T16:40:56.840237Z","shell.execute_reply.started":"2023-04-19T16:40:56.831398Z","shell.execute_reply":"2023-04-19T16:40:56.838486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loop through each value in the list and check if it's in the defog_metadata DataFrame\nfor value in my_sequences:\n    if value in defog_metadata['Id'].values:\n        print(f'{value} found in the defog_metadata DataFrame')\n    else:\n        print(f'{value} not found in the defog_metadata DataFrame')","metadata":{"execution":{"iopub.status.busy":"2023-04-19T03:49:56.430755Z","iopub.execute_input":"2023-04-19T03:49:56.431316Z","iopub.status.idle":"2023-04-19T03:49:56.441821Z","shell.execute_reply.started":"2023-04-19T03:49:56.431262Z","shell.execute_reply":"2023-04-19T03:49:56.440482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"YAY! I have metadata for my noytpe folder! So that means I have a Subject for each file.","metadata":{}},{"cell_type":"code","source":"defog_metadata.query(\"Id == '1e8d55d48d'\")","metadata":{"execution":{"iopub.status.busy":"2023-04-19T03:49:56.443131Z","iopub.execute_input":"2023-04-19T03:49:56.443623Z","iopub.status.idle":"2023-04-19T03:49:56.460599Z","shell.execute_reply.started":"2023-04-19T03:49:56.443590Z","shell.execute_reply":"2023-04-19T03:49:56.459322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Here I look for duplicates in the metadata \nduplicates_defog_metadata = defog_metadata['Subject'].duplicated()\nprint(defog_metadata[duplicates_defog_metadata])","metadata":{"execution":{"iopub.status.busy":"2023-04-19T03:50:11.130192Z","iopub.execute_input":"2023-04-19T03:50:11.131439Z","iopub.status.idle":"2023-04-19T03:50:11.143918Z","shell.execute_reply.started":"2023-04-19T03:50:11.131389Z","shell.execute_reply":"2023-04-19T03:50:11.142451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_metadata.query(\"Subject == '3dd575'\")","metadata":{"execution":{"iopub.status.busy":"2023-04-19T03:50:15.783862Z","iopub.execute_input":"2023-04-19T03:50:15.785157Z","iopub.status.idle":"2023-04-19T03:50:15.801218Z","shell.execute_reply.started":"2023-04-19T03:50:15.785100Z","shell.execute_reply":"2023-04-19T03:50:15.799674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'c816aa3562' in my_sequences:\n    print(\"c816aa3562 is in the notype list\")\nelse:\n    print(\"c816aa3562 is not in the notype list\")","metadata":{"execution":{"iopub.status.busy":"2023-04-19T03:50:19.050035Z","iopub.execute_input":"2023-04-19T03:50:19.050510Z","iopub.status.idle":"2023-04-19T03:50:19.056125Z","shell.execute_reply.started":"2023-04-19T03:50:19.050473Z","shell.execute_reply":"2023-04-19T03:50:19.055122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check if defog metadata Id is in my_sequences so we would have duplicate Subjects\n\n# create a df as my duplicates\ndf = defog_metadata[duplicates_defog_metadata]\n\n# here is my list of no_type files\nmy_sequences\n\n# loop through each component in 'Id' and check if it's in 'my_Sequences'\nfor trial in df['Id']:\n    if trial in my_sequences:\n        print(f\"{trial} is in the no_type list.\")\n    else:\n        print(f\"{trial} is not in no_type the list.\")\n\n#if '0ec76d2d8e' in my_sequences:\n#    print(\"06414383cf is in the notype list\")\n#else:\n#    print(\"06414383cf is not in the notype list\")","metadata":{"execution":{"iopub.status.busy":"2023-04-19T03:50:25.314390Z","iopub.execute_input":"2023-04-19T03:50:25.315596Z","iopub.status.idle":"2023-04-19T03:50:25.324884Z","shell.execute_reply.started":"2023-04-19T03:50:25.315547Z","shell.execute_reply":"2023-04-19T03:50:25.323383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Ok, so there are duplicates in my notype data.  I started looking at the defog test set: ","metadata":{}},{"cell_type":"code","source":"my_sequences_defog = os.listdir(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog\")\nmy_sequences_defog[:10]","metadata":{"execution":{"iopub.status.busy":"2023-04-19T03:50:37.352984Z","iopub.execute_input":"2023-04-19T03:50:37.353505Z","iopub.status.idle":"2023-04-19T03:50:37.364426Z","shell.execute_reply.started":"2023-04-19T03:50:37.353461Z","shell.execute_reply":"2023-04-19T03:50:37.362941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loop through each file name in the list and remove the last '.csv' extension\nfor i in range(len(my_sequences_defog)):\n    my_sequences_defog[i] = my_sequences_defog[i].rsplit('.csv', 1)[0]\n\nprint(my_sequences_defog[:10])","metadata":{"execution":{"iopub.status.busy":"2023-04-19T03:50:42.371930Z","iopub.execute_input":"2023-04-19T03:50:42.372477Z","iopub.status.idle":"2023-04-19T03:50:42.380052Z","shell.execute_reply.started":"2023-04-19T03:50:42.372424Z","shell.execute_reply":"2023-04-19T03:50:42.378637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loop through each value in the list and check if it's in the defog_metadata DataFrame\nfor value in my_sequences_defog:\n    if value in defog_metadata['Id'].values:\n        print(f'{value} found in the defog_metadata DataFrame')\n    else:\n        print(f'{value} not found in the defog_metadata DataFrame')","metadata":{"execution":{"iopub.status.busy":"2023-04-19T03:50:44.961680Z","iopub.execute_input":"2023-04-19T03:50:44.962753Z","iopub.status.idle":"2023-04-19T03:50:44.971287Z","shell.execute_reply.started":"2023-04-19T03:50:44.962704Z","shell.execute_reply":"2023-04-19T03:50:44.969984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create an empty DataFrame to store the concatenated data for defog train\nconcatenated_data_defog = pd.DataFrame()\n\n\n# concatenate the notype data into a data frame\nfor i in range(0,91): # 91 CSV files with names '... + my_sequences[i]\n    my_sequences_defog = os.listdir(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog\")\n    data_defog = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/\"+my_sequences_defog[i])\n    concatenated_data_defog = pd.concat([concatenated_data_defog, data_defog], axis=0, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:48:50.326346Z","iopub.execute_input":"2023-04-19T04:48:50.327377Z","iopub.status.idle":"2023-04-19T04:49:16.437530Z","shell.execute_reply.started":"2023-04-19T04:48:50.327301Z","shell.execute_reply":"2023-04-19T04:49:16.435946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(concatenated_data_defog)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:49:32.765727Z","iopub.execute_input":"2023-04-19T04:49:32.766267Z","iopub.status.idle":"2023-04-19T04:49:32.774543Z","shell.execute_reply.started":"2023-04-19T04:49:32.766228Z","shell.execute_reply":"2023-04-19T04:49:32.773072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"concatenated_data_defog[200000:210000]","metadata":{"execution":{"iopub.status.busy":"2023-04-19T03:53:34.870074Z","iopub.execute_input":"2023-04-19T03:53:34.870592Z","iopub.status.idle":"2023-04-19T03:53:34.891388Z","shell.execute_reply.started":"2023-04-19T03:53:34.870548Z","shell.execute_reply":"2023-04-19T03:53:34.889873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max(concatenated_data_defog['Time'])","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:15:55.609073Z","iopub.execute_input":"2023-04-19T04:15:55.609529Z","iopub.status.idle":"2023-04-19T04:15:56.977181Z","shell.execute_reply.started":"2023-04-19T04:15:55.609493Z","shell.execute_reply":"2023-04-19T04:15:56.976257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"concatenated_data_defog['Time_frac']=(concatenated_data_defog.Time/concatenated_data_defog.Time.max()).values","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:17:13.320886Z","iopub.execute_input":"2023-04-19T04:17:13.321400Z","iopub.status.idle":"2023-04-19T04:17:13.424597Z","shell.execute_reply.started":"2023-04-19T04:17:13.321363Z","shell.execute_reply":"2023-04-19T04:17:13.423554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"concatenated_data_defog[200000:210000]","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:17:44.181793Z","iopub.execute_input":"2023-04-19T04:17:44.183254Z","iopub.status.idle":"2023-04-19T04:17:44.205134Z","shell.execute_reply.started":"2023-04-19T04:17:44.183198Z","shell.execute_reply":"2023-04-19T04:17:44.203612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Histogram\n\n# Added a new colum to df: divide the time by max time \nconcatenated_data_defog['Time_frac']=(concatenated_data_defog.Time/concatenated_data_defog.Time.max()).values\n\nfig,ax=plt.subplots(4,1,figsize=(15,10))\nconcatenated_data_defog.loc[concatenated_data_defog['Valid']==True,'Time_frac'].hist(ax=ax[0],bins=100)\nconcatenated_data_defog.loc[concatenated_data_defog['StartHesitation']==1,'Time_frac'].hist(ax=ax[1],bins=100)\nconcatenated_data_defog.loc[concatenated_data_defog['Turn']==1,'Time_frac'].hist(ax=ax[2],bins=100)\nconcatenated_data_defog.loc[concatenated_data_defog['Walking']==1,'Time_frac'].hist(ax=ax[3],bins=100)\n\nplt.subplots_adjust(hspace=0.5)\n\nax[0].set_xlabel('Defog Train')\nax[0].set_ylabel('Frequency')\nax[0].set_title('Histogram of Defog Data for Valid')\n\nax[1].set_xlabel('Defog Train')\nax[1].set_ylabel('Frequency')\nax[1].set_title('Histogram of Defog Data for Start Hesitatioj')\n\n\nax[2].set_xlabel('Defog Train')\nax[2].set_ylabel('Frequency')\nax[2].set_title('Histogram of Defog Data for Turn')\n\n\nax[3].set_xlabel('Defog Train')\nax[3].set_ylabel('Frequency')\nax[3].set_title('Histogram of Defog Data for Walking')\n","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:21:39.816898Z","iopub.execute_input":"2023-04-19T05:21:39.818291Z","iopub.status.idle":"2023-04-19T05:21:41.541661Z","shell.execute_reply.started":"2023-04-19T05:21:39.818239Z","shell.execute_reply":"2023-04-19T05:21:41.540214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_defog_True = concatenated_data_defog.drop(concatenated_data_defog[concatenated_data_defog['Valid'] == False].index)\n#df = df.drop(df[df['name'] == 'Charlie'].index)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:49:40.683179Z","iopub.execute_input":"2023-04-19T04:49:40.683692Z","iopub.status.idle":"2023-04-19T04:49:43.965106Z","shell.execute_reply.started":"2023-04-19T04:49:40.683651Z","shell.execute_reply":"2023-04-19T04:49:43.963950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(data_defog_True)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:49:52.327236Z","iopub.execute_input":"2023-04-19T04:49:52.327678Z","iopub.status.idle":"2023-04-19T04:49:52.335731Z","shell.execute_reply.started":"2023-04-19T04:49:52.327645Z","shell.execute_reply":"2023-04-19T04:49:52.334378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Histogram\n\n# Added a new colum to df: divide the time by max time \ndata_defog_True['Time_frac']=(data_defog_True.Time/data_defog_True.Time.max()).values\n\nfig,ax=plt.subplots(3,1,figsize=(15,10))\ndata_defog_True.loc[data_defog_True['StartHesitation']==1,'Time_frac'].hist(ax=ax[0],bins=100)\ndata_defog_True.loc[data_defog_True['Turn']==1,'Time_frac'].hist(ax=ax[1],bins=100)\ndata_defog_True.loc[data_defog_True['Walking']==1,'Time_frac'].hist(ax=ax[2],bins=100)\n\n#give me space betwen plots\nplt.subplots_adjust(hspace=0.5)\n\nax[0].set_xlabel('Defog Train Data')\nax[0].set_ylabel('Frequency')\nax[0].set_title('Histogram of Defog Data for Start hesitation')\n\nax[1].set_xlabel('Defog Train')\nax[1].set_ylabel('Frequency')\nax[1].set_title('Histogram of Defog Data for Turn')\n\nax[2].set_xlabel('Defog Train')\nax[2].set_ylabel('Frequency')\nax[2].set_title('Histogram of Defog Data for Walking')","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:11:10.473202Z","iopub.execute_input":"2023-04-19T05:11:10.473969Z","iopub.status.idle":"2023-04-19T05:11:11.745553Z","shell.execute_reply.started":"2023-04-19T05:11:10.473904Z","shell.execute_reply":"2023-04-19T05:11:11.744287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"concatenated_data_notype.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T15:51:18.335732Z","iopub.execute_input":"2023-04-19T15:51:18.336546Z","iopub.status.idle":"2023-04-19T15:51:18.373736Z","shell.execute_reply.started":"2023-04-19T15:51:18.336480Z","shell.execute_reply":"2023-04-19T15:51:18.372369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Histogram\n\n# Added a new colum to df: divide the time by max time \nconcatenated_data_notype['Time_frac']=(concatenated_data_notype.Time/concatenated_data_notype.Time.max()).values\n\nfig,ax=plt.subplots(3,1,figsize=(15,10))\nconcatenated_data_notype.loc[concatenated_data_notype['Valid']==True,'Time_frac'].hist(ax=ax[0],bins=100)\nconcatenated_data_notype.loc[concatenated_data_notype['Event']==1,'Time_frac'].hist(ax=ax[1],bins=100)\n\nax[1].set_xlabel('Notype Train Data')\nax[1].set_ylabel('Frequency')\nax[1].set_title('Histogram of Notype Data for any Fog')","metadata":{"execution":{"iopub.status.busy":"2023-04-19T15:59:31.409600Z","iopub.execute_input":"2023-04-19T15:59:31.410018Z","iopub.status.idle":"2023-04-19T15:59:32.649589Z","shell.execute_reply.started":"2023-04-19T15:59:31.409979Z","shell.execute_reply":"2023-04-19T15:59:32.648286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_notype_True = concatenated_data_notype.drop(concatenated_data_notype[concatenated_data_notype['Valid'] == False].index)\n#df = df.drop(df[df['name'] == 'Charlie'].index)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T16:02:09.316098Z","iopub.execute_input":"2023-04-19T16:02:09.317232Z","iopub.status.idle":"2023-04-19T16:02:11.937907Z","shell.execute_reply.started":"2023-04-19T16:02:09.317177Z","shell.execute_reply":"2023-04-19T16:02:11.936360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_notype_True['Time_frac']=(data_notype_True.Time/data_notype_True.Time.max()).values\n\nfig,ax=plt.subplots(3,1,figsize=(15,10))\ndata_notype_True.loc[data_notype_True['Event']==1,'Time_frac'].hist(ax=ax[0],bins=100)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-19T16:03:53.935626Z","iopub.execute_input":"2023-04-19T16:03:53.936091Z","iopub.status.idle":"2023-04-19T16:03:54.608643Z","shell.execute_reply.started":"2023-04-19T16:03:53.936042Z","shell.execute_reply":"2023-04-19T16:03:54.607344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom tqdm.auto import tqdm\n\nimport glob\n\np = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/'\n\ntrain = glob.glob(p+'train/**/**')\ndefog_train = glob.glob(p+'train/defog/**ˇ')","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:26:00.002057Z","iopub.execute_input":"2023-04-19T04:26:00.003475Z","iopub.status.idle":"2023-04-19T04:26:00.019058Z","shell.execute_reply.started":"2023-04-19T04:26:00.003407Z","shell.execute_reply":"2023-04-19T04:26:00.017582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"https://www.kaggle.com/code/xzj19013742/simple-eda-on-time-for-targets\nThis notebook provides a good description of labeling for all training data. Perhaps we can use this timing to change our notype data to have an event.","metadata":{}},{"cell_type":"code","source":"len(defog_train)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:30:47.124876Z","iopub.execute_input":"2023-04-19T04:30:47.125392Z","iopub.status.idle":"2023-04-19T04:30:47.133379Z","shell.execute_reply.started":"2023-04-19T04:30:47.125349Z","shell.execute_reply":"2023-04-19T04:30:47.132034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-04-19T03:54:13.350577Z","iopub.execute_input":"2023-04-19T03:54:13.351101Z","iopub.status.idle":"2023-04-19T03:54:13.360582Z","shell.execute_reply.started":"2023-04-19T03:54:13.351056Z","shell.execute_reply":"2023-04-19T03:54:13.359102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}